Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. Β See raw diff
- evals/correctness_ab/glean_keep25_nogold_s1224_step100_chat.json +0 -0
- evals/correctness_ab/glean_keep25_nogold_s1224_step100_chat.json.server.log +131 -0
- evals/correctness_ab/glean_keep25_nogold_s1225_step100_chat.json +0 -0
- evals/correctness_ab/glean_keep25_nogold_s1225_step100_chat.json.server.log +130 -0
- evals/correctness_ab/glean_keep25_nogold_s1226_step100_chat.json +0 -0
- evals/correctness_ab/glean_keep25_nogold_s1226_step100_chat.json.server.log +132 -0
- evals/general_suite/base_full.run.log +89 -0
- evals/general_suite/base_full_v2.log +73 -0
- evals/general_suite/gsm8k_protocol.log +39 -0
- evals/general_suite/math_fix.log +27 -0
- evals/general_suite/math_full.log +26 -0
- evals/general_suite/policy_confirm.combo_on25_seed1224.eval.log +0 -0
- evals/general_suite/policy_confirm.combo_on25_seed1225.eval.log +0 -0
- evals/general_suite/policy_confirm.combo_on25_seed1226.eval.log +0 -0
- evals/general_suite/policy_confirm.off_forward_seed1224.eval.log +0 -0
- evals/general_suite/policy_confirm.off_forward_seed1225.eval.log +0 -0
- evals/general_suite/policy_confirm.off_forward_seed1226.eval.log +0 -0
- evals/general_suite/policy_confirm.on_reverse_seed1224.eval.log +0 -0
- evals/general_suite/policy_confirm.on_reverse_seed1225.eval.log +0 -0
- evals/general_suite/policy_confirm.on_reverse_seed1226.eval.log +0 -0
- evals/general_suite/pruneval.log +23 -0
- evals/general_suite/ragged_smoke.log +74 -0
- evals/general_suite/smoke_full.log +134 -0
- evals/general_suite/smoke_full2.log +73 -0
- evals/general_suite/smoke_risky.log +55 -0
- evals/grid_math/glean_keep25_s1224_step100_chat.json +0 -0
- evals/grid_math/glean_keep25_s1224_step100_chat.json.server.log +131 -0
- evals/grid_math/glean_keep25_s1224_step150_chat.json +0 -0
- evals/grid_math/glean_keep25_s1224_step150_chat.json.server.log +130 -0
- evals/grid_math/glean_keep25_s1225_step100_chat.json +0 -0
- evals/grid_math/glean_keep25_s1225_step100_chat.json.server.log +130 -0
- evals/grid_math/glean_keep25_s1225_step150_chat.json +0 -0
- evals/grid_math/glean_keep25_s1225_step150_chat.json.server.log +130 -0
- evals/grid_math/glean_keep25_s1226_step100_chat.json +0 -0
- evals/grid_math/glean_keep25_s1226_step100_chat.json.server.log +130 -0
- evals/grid_math/glean_keep25_s1226_step150_chat.json +0 -0
- evals/grid_math/glean_keep25_s1226_step150_chat.json.server.log +130 -0
- evals/grid_math/glean_keep50_s1224_step100_chat.json +0 -0
- evals/grid_math/glean_keep50_s1224_step100_chat.json.server.log +130 -0
- evals/grid_math/glean_keep50_s1224_step150_chat.json +0 -0
- evals/grid_math/glean_keep50_s1224_step150_chat.json.server.log +130 -0
- evals/grid_math/glean_keep50_s1225_step100_chat.json +0 -0
- evals/grid_math/glean_keep50_s1225_step100_chat.json.server.log +129 -0
- evals/grid_math/glean_keep50_s1225_step150_chat.json +0 -0
- evals/grid_math/glean_keep50_s1225_step150_chat.json.server.log +129 -0
- evals/grid_math/glean_keep50_s1226_step100_chat.json +0 -0
- evals/grid_math/glean_keep50_s1226_step100_chat.json.server.log +129 -0
- evals/grid_math/glean_keep50_s1226_step150_chat.json +0 -0
- evals/grid_math/glean_keep50_s1226_step150_chat.json.server.log +130 -0
- evals/grid_math/glean_keep75_s1224_step100_chat.json +0 -0
evals/correctness_ab/glean_keep25_nogold_s1224_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/correctness_ab/glean_keep25_nogold_s1224_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339] ββββ β β β β model outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100
|
| 5 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100', 'host': '127.0.0.1', 'port': 8395, 'model': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=101781) WARNING 07-17 05:33:32 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=101781) WARNING 07-17 05:33:32 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=101781) INFO 07-17 05:33:32 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=101896) INFO 07-17 05:33:39 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100', speculative_config=None, tokenizer='outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=101896) INFO 07-17 05:33:39 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:53257 backend=nccl
|
| 18 |
+
(EngineCore pid=101896) INFO 07-17 05:33:40 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=101896) INFO 07-17 05:33:40 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=101896) INFO 07-17 05:33:40 [gpu_model_runner.py:5209] Starting to load model outputs/healed/correctness_ab/glean_keep25_nogold_s1224/step0100...
|
| 21 |
+
(EngineCore pid=101896) INFO 07-17 05:33:41 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=101896) INFO 07-17 05:33:41 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=101896) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=101896) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=101896) INFO 07-17 05:33:41 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 99.20 GiB.
|
| 26 |
+
(EngineCore pid=101896) INFO 07-17 05:33:41 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=101896)
|
| 28 |
+
(EngineCore pid=101896)
|
| 29 |
+
(EngineCore pid=101896)
|
| 30 |
+
(EngineCore pid=101896)
|
| 31 |
+
(EngineCore pid=101896) INFO 07-17 05:33:43 [default_loader.py:430] Loading weights took 2.43 seconds
|
| 32 |
+
(EngineCore pid=101896) INFO 07-17 05:33:44 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.615814 seconds
|
| 33 |
+
(EngineCore pid=101896) INFO 07-17 05:33:45 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=101896) INFO 07-17 05:33:45 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=101896) INFO 07-17 05:33:45 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=101896) INFO 07-17 05:33:45 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.05 s
|
| 39 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=101896) WARNING 07-17 05:33:46 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=101896) WARNING 07-17 05:33:46 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=101896) INFO 07-17 05:33:46 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=101781) WARNING 07-17 05:33:46 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8395
|
| 49 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 51 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=101781) INFO 07-17 05:33:46 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=101781) INFO: Started server process [101781]
|
| 97 |
+
(APIServer pid=101781) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=101781) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=101781) INFO: 127.0.0.1:46152 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=101896) WARNING 07-17 05:33:48 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=101781) INFO 07-17 05:33:57 [loggers.py:273] Engine 000: Avg prompt throughput: 2257.9 tokens/s, Avg generation throughput: 2090.2 tokens/s, Running: 256 reqs, Waiting: 1029 reqs, GPU KV cache usage: 33.5%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=101781) INFO 07-17 05:34:07 [loggers.py:273] Engine 000: Avg prompt throughput: 1162.5 tokens/s, Avg generation throughput: 2928.5 tokens/s, Running: 255 reqs, Waiting: 883 reqs, GPU KV cache usage: 41.7%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=101781) INFO 07-17 05:34:17 [loggers.py:273] Engine 000: Avg prompt throughput: 1094.5 tokens/s, Avg generation throughput: 2929.4 tokens/s, Running: 253 reqs, Waiting: 740 reqs, GPU KV cache usage: 44.2%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=101781) INFO 07-17 05:34:27 [loggers.py:273] Engine 000: Avg prompt throughput: 1081.5 tokens/s, Avg generation throughput: 2878.2 tokens/s, Running: 252 reqs, Waiting: 597 reqs, GPU KV cache usage: 44.3%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=101781) INFO 07-17 05:34:37 [loggers.py:273] Engine 000: Avg prompt throughput: 1285.2 tokens/s, Avg generation throughput: 2902.8 tokens/s, Running: 253 reqs, Waiting: 440 reqs, GPU KV cache usage: 40.6%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=101781) INFO 07-17 05:34:47 [loggers.py:273] Engine 000: Avg prompt throughput: 1322.8 tokens/s, Avg generation throughput: 2927.5 tokens/s, Running: 255 reqs, Waiting: 278 reqs, GPU KV cache usage: 42.4%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=101781) INFO 07-17 05:34:57 [loggers.py:273] Engine 000: Avg prompt throughput: 1029.6 tokens/s, Avg generation throughput: 2903.8 tokens/s, Running: 256 reqs, Waiting: 143 reqs, GPU KV cache usage: 44.6%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=101781) INFO 07-17 05:35:07 [loggers.py:273] Engine 000: Avg prompt throughput: 1081.8 tokens/s, Avg generation throughput: 2854.0 tokens/s, Running: 254 reqs, Waiting: 14 reqs, GPU KV cache usage: 47.5%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=101781) INFO 07-17 05:35:17 [loggers.py:273] Engine 000: Avg prompt throughput: 116.1 tokens/s, Avg generation throughput: 2719.9 tokens/s, Running: 116 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.6%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=101781) INFO 07-17 05:35:27 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1278.7 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 93.4%
|
| 111 |
+
(APIServer pid=101781) INFO: 127.0.0.1:46166 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 112 |
+
(EngineCore pid=101896) INFO 07-17 05:35:29 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 113 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 114 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=101896) INFO 07-17 05:35:29 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 116 |
+
(EngineCore pid=101896) INFO 07-17 05:35:29 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 117 |
+
(EngineCore pid=101896) INFO 07-17 05:35:29 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 118 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 119 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 120 |
+
(APIServer pid=101781) WARNING 07-17 05:35:29 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 121 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 122 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 123 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [core_client.py:662] [shutdown] MPClient: complete
|
| 124 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 125 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 126 |
+
(APIServer pid=101781) INFO 07-17 05:35:29 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 127 |
+
(APIServer pid=101781) INFO: Shutting down
|
| 128 |
+
(APIServer pid=101781) INFO: Waiting for application shutdown.
|
| 129 |
+
(APIServer pid=101781) INFO: Application shutdown complete.
|
| 130 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 131 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/correctness_ab/glean_keep25_nogold_s1225_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/correctness_ab/glean_keep25_nogold_s1225_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339] ββββ β β β β model outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100
|
| 5 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100', 'host': '127.0.0.1', 'port': 8396, 'model': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=102611) WARNING 07-17 05:35:23 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=102611) WARNING 07-17 05:35:23 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=102611) INFO 07-17 05:35:23 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=102728) INFO 07-17 05:35:30 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100', speculative_config=None, tokenizer='outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=102728) INFO 07-17 05:35:31 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:53281 backend=nccl
|
| 18 |
+
(EngineCore pid=102728) INFO 07-17 05:35:31 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [gpu_model_runner.py:5209] Starting to load model outputs/healed/correctness_ab/glean_keep25_nogold_s1225/step0100...
|
| 21 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=102728) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=102728) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 108.96 GiB.
|
| 26 |
+
(EngineCore pid=102728) INFO 07-17 05:35:32 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=102728)
|
| 28 |
+
(EngineCore pid=102728)
|
| 29 |
+
(EngineCore pid=102728)
|
| 30 |
+
(EngineCore pid=102728)
|
| 31 |
+
(EngineCore pid=102728) INFO 07-17 05:35:35 [default_loader.py:430] Loading weights took 2.40 seconds
|
| 32 |
+
(EngineCore pid=102728) INFO 07-17 05:35:35 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.577566 seconds
|
| 33 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.02 s
|
| 39 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=102728) WARNING 07-17 05:35:37 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=102728) WARNING 07-17 05:35:37 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=102728) INFO 07-17 05:35:37 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=102611) INFO 07-17 05:35:37 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=102611) WARNING 07-17 05:35:37 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=102611) INFO 07-17 05:35:37 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8396
|
| 49 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 51 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=102611) INFO 07-17 05:35:38 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=102611) INFO: Started server process [102611]
|
| 97 |
+
(APIServer pid=102611) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=102611) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=102611) INFO: 127.0.0.1:55570 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=102728) WARNING 07-17 05:35:40 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=102611) INFO 07-17 05:35:48 [loggers.py:273] Engine 000: Avg prompt throughput: 2291.7 tokens/s, Avg generation throughput: 2048.0 tokens/s, Running: 254 reqs, Waiting: 1024 reqs, GPU KV cache usage: 32.6%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=102611) INFO 07-17 05:35:58 [loggers.py:273] Engine 000: Avg prompt throughput: 1233.6 tokens/s, Avg generation throughput: 2953.9 tokens/s, Running: 253 reqs, Waiting: 871 reqs, GPU KV cache usage: 41.0%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=102611) INFO 07-17 05:36:08 [loggers.py:273] Engine 000: Avg prompt throughput: 1193.3 tokens/s, Avg generation throughput: 2901.7 tokens/s, Running: 256 reqs, Waiting: 714 reqs, GPU KV cache usage: 42.4%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=102611) INFO 07-17 05:36:18 [loggers.py:273] Engine 000: Avg prompt throughput: 1067.7 tokens/s, Avg generation throughput: 2929.9 tokens/s, Running: 253 reqs, Waiting: 576 reqs, GPU KV cache usage: 46.1%, Prefix cache hit rate: 93.4%
|
| 105 |
+
(APIServer pid=102611) INFO 07-17 05:36:28 [loggers.py:273] Engine 000: Avg prompt throughput: 1354.5 tokens/s, Avg generation throughput: 2875.3 tokens/s, Running: 255 reqs, Waiting: 404 reqs, GPU KV cache usage: 41.3%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=102611) INFO 07-17 05:36:38 [loggers.py:273] Engine 000: Avg prompt throughput: 1204.0 tokens/s, Avg generation throughput: 2903.7 tokens/s, Running: 256 reqs, Waiting: 260 reqs, GPU KV cache usage: 44.6%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=102611) INFO 07-17 05:36:48 [loggers.py:273] Engine 000: Avg prompt throughput: 1087.4 tokens/s, Avg generation throughput: 2929.5 tokens/s, Running: 254 reqs, Waiting: 118 reqs, GPU KV cache usage: 45.0%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=102611) INFO 07-17 05:36:58 [loggers.py:273] Engine 000: Avg prompt throughput: 1001.3 tokens/s, Avg generation throughput: 2902.1 tokens/s, Running: 231 reqs, Waiting: 0 reqs, GPU KV cache usage: 44.9%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=102611) INFO 07-17 05:37:08 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2416.3 tokens/s, Running: 66 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.6%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=102611) INFO: 127.0.0.1:55584 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 111 |
+
(EngineCore pid=102728) INFO 07-17 05:37:18 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=102728) INFO 07-17 05:37:18 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=102728) INFO 07-17 05:37:18 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=102728) INFO 07-17 05:37:18 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=102611) WARNING 07-17 05:37:18 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=102611) INFO 07-17 05:37:18 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=102611) INFO: Shutting down
|
| 127 |
+
(APIServer pid=102611) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=102611) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/correctness_ab/glean_keep25_nogold_s1226_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/correctness_ab/glean_keep25_nogold_s1226_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339] ββββ β β β β model outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100
|
| 5 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100', 'host': '127.0.0.1', 'port': 8395, 'model': 'outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=108453) WARNING 07-17 06:28:19 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=108453) WARNING 07-17 06:28:19 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=108453) INFO 07-17 06:28:19 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=108568) INFO 07-17 06:28:26 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100', speculative_config=None, tokenizer='outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=108568) INFO 07-17 06:28:27 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:34491 backend=nccl
|
| 18 |
+
(EngineCore pid=108568) INFO 07-17 06:28:27 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [gpu_model_runner.py:5209] Starting to load model outputs/healed/correctness_ab/glean_keep25_nogold_s1226/step0100...
|
| 21 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=108568) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=108568) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 109.58 GiB.
|
| 26 |
+
(EngineCore pid=108568) INFO 07-17 06:28:28 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=108568)
|
| 28 |
+
(EngineCore pid=108568)
|
| 29 |
+
(EngineCore pid=108568)
|
| 30 |
+
(EngineCore pid=108568)
|
| 31 |
+
(EngineCore pid=108568) INFO 07-17 06:28:31 [default_loader.py:430] Loading weights took 2.44 seconds
|
| 32 |
+
(EngineCore pid=108568) INFO 07-17 06:28:31 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.620238 seconds
|
| 33 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.05 s
|
| 39 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=108568) WARNING 07-17 06:28:33 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=108568) WARNING 07-17 06:28:33 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=108568) INFO 07-17 06:28:33 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=108453) WARNING 07-17 06:28:34 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8395
|
| 49 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=108453) INFO 07-17 06:28:34 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=108453) INFO: Started server process [108453]
|
| 97 |
+
(APIServer pid=108453) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=108453) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=108453) INFO: 127.0.0.1:51658 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=108568) WARNING 07-17 06:28:36 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=108453) INFO 07-17 06:28:44 [loggers.py:273] Engine 000: Avg prompt throughput: 2300.6 tokens/s, Avg generation throughput: 2197.9 tokens/s, Running: 254 reqs, Waiting: 1022 reqs, GPU KV cache usage: 33.7%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=108453) INFO 07-17 06:28:54 [loggers.py:273] Engine 000: Avg prompt throughput: 1099.2 tokens/s, Avg generation throughput: 2930.0 tokens/s, Running: 256 reqs, Waiting: 884 reqs, GPU KV cache usage: 42.7%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=108453) INFO 07-17 06:29:04 [loggers.py:273] Engine 000: Avg prompt throughput: 1158.2 tokens/s, Avg generation throughput: 2928.4 tokens/s, Running: 255 reqs, Waiting: 735 reqs, GPU KV cache usage: 44.0%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=108453) INFO 07-17 06:29:14 [loggers.py:273] Engine 000: Avg prompt throughput: 1035.2 tokens/s, Avg generation throughput: 2904.9 tokens/s, Running: 255 reqs, Waiting: 600 reqs, GPU KV cache usage: 46.3%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=108453) INFO 07-17 06:29:24 [loggers.py:273] Engine 000: Avg prompt throughput: 1195.9 tokens/s, Avg generation throughput: 2903.0 tokens/s, Running: 254 reqs, Waiting: 450 reqs, GPU KV cache usage: 43.6%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=108453) INFO 07-17 06:29:34 [loggers.py:273] Engine 000: Avg prompt throughput: 1280.2 tokens/s, Avg generation throughput: 2902.9 tokens/s, Running: 255 reqs, Waiting: 296 reqs, GPU KV cache usage: 43.6%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=108453) INFO 07-17 06:29:44 [loggers.py:273] Engine 000: Avg prompt throughput: 1039.6 tokens/s, Avg generation throughput: 2956.0 tokens/s, Running: 255 reqs, Waiting: 162 reqs, GPU KV cache usage: 44.8%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=108453) INFO 07-17 06:29:54 [loggers.py:273] Engine 000: Avg prompt throughput: 1040.4 tokens/s, Avg generation throughput: 2905.3 tokens/s, Running: 256 reqs, Waiting: 33 reqs, GPU KV cache usage: 47.3%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=108453) INFO 07-17 06:30:04 [loggers.py:273] Engine 000: Avg prompt throughput: 285.7 tokens/s, Avg generation throughput: 2772.9 tokens/s, Running: 135 reqs, Waiting: 0 reqs, GPU KV cache usage: 34.9%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=108453) INFO 07-17 06:30:14 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1356.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 93.4%
|
| 111 |
+
(APIServer pid=108453) INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 112 |
+
(EngineCore pid=108568) INFO 07-17 06:30:17 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 113 |
+
(APIServer pid=108453) INFO 07-17 06:30:17 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 114 |
+
(APIServer pid=108453) INFO 07-17 06:30:17 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=108568) INFO 07-17 06:30:17 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 116 |
+
(EngineCore pid=108568) INFO 07-17 06:30:17 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 117 |
+
(EngineCore pid=108568) INFO 07-17 06:30:17 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 118 |
+
(APIServer pid=108453) INFO 07-17 06:30:17 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 119 |
+
(APIServer pid=108453) INFO 07-17 06:30:17 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 120 |
+
(APIServer pid=108453) WARNING 07-17 06:30:17 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 121 |
+
(APIServer pid=108453) INFO: Shutting down
|
| 122 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 123 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 124 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [core_client.py:662] [shutdown] MPClient: complete
|
| 125 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 126 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 127 |
+
(APIServer pid=108453) INFO 07-17 06:30:18 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 128 |
+
(APIServer pid=108453) INFO: Shutting down
|
| 129 |
+
(APIServer pid=108453) INFO: Waiting for application shutdown.
|
| 130 |
+
(APIServer pid=108453) INFO: Application shutdown complete.
|
| 131 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 132 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/general_suite/base_full.run.log
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T10:13:10-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T10:13:10-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T10:13:40-07:00 server up; running lm_eval [gsm8k,minerva_math500,ifeval,humaneval_instruct,mbpp_instruct]
|
| 4 |
+
2026-07-17:10:13:48 INFO [_cli.run:388] Selected Tasks: ['gsm8k', 'minerva_math500', 'ifeval', 'humaneval_instruct', 'mbpp_instruct']
|
| 5 |
+
2026-07-17:10:13:49 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 6 |
+
2026-07-17:10:13:49 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 32, 'tokenized_requests': False, 'max_retries': 3}
|
| 7 |
+
2026-07-17:10:13:49 INFO [models.api_models:179] Using max length 2048 - 1
|
| 8 |
+
2026-07-17:10:13:49 INFO [models.api_models:200] Using tokenizer None
|
| 9 |
+
Traceback (most recent call last):
|
| 10 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/utils.py", line 14, in <module>
|
| 11 |
+
import antlr4
|
| 12 |
+
ModuleNotFoundError: No module named 'antlr4'
|
| 13 |
+
|
| 14 |
+
The above exception was the direct cause of the following exception:
|
| 15 |
+
|
| 16 |
+
Traceback (most recent call last):
|
| 17 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/bin/lm_eval", line 10, in <module>
|
| 18 |
+
sys.exit(cli_evaluate())
|
| 19 |
+
^^^^^^^^^^^^^^
|
| 20 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate
|
| 21 |
+
parser.execute(args)
|
| 22 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/_cli/harness.py", line 60, in execute
|
| 23 |
+
args.func(args)
|
| 24 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/_cli/run.py", line 391, in _execute
|
| 25 |
+
results = simple_evaluate(
|
| 26 |
+
^^^^^^^^^^^^^^^^
|
| 27 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/utils.py", line 575, in _wrapper
|
| 28 |
+
return fn(*args, **kwargs)
|
| 29 |
+
^^^^^^^^^^^^^^^^^^^
|
| 30 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/evaluator.py", line 302, in simple_evaluate
|
| 31 |
+
loaded = task_manager.load(tasks)
|
| 32 |
+
^^^^^^^^^^^^^^^^^^^^^^^^
|
| 33 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/manager.py", line 208, in load
|
| 34 |
+
obj = self._load_spec(spec) if not isinstance(spec, (Task, Group)) else spec # type:ignore[invalid-argument-type]
|
| 35 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 36 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/manager.py", line 154, in _load_spec
|
| 37 |
+
return self._factory.build(
|
| 38 |
+
^^^^^^^^^^^^^^^^^^^^
|
| 39 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_factory.py", line 61, in build
|
| 40 |
+
return self._build_task(entry, overrides)
|
| 41 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 42 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_factory.py", line 67, in _build_task
|
| 43 |
+
cfg = self._load_full_config(entry, overrides)
|
| 44 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 45 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_factory.py", line 261, in _load_full_config
|
| 46 |
+
cfg = deepcopy(load_yaml(entry.yaml_path, resolve_func=True))
|
| 47 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 48 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_yaml_loader.py", line 198, in load_yaml
|
| 49 |
+
inc_cfg = load_yaml(
|
| 50 |
+
^^^^^^^^^^
|
| 51 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_yaml_loader.py", line 185, in load_yaml
|
| 52 |
+
cfg = yaml.load(fh, Loader=loader_cls) # noqa: S506
|
| 53 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 54 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/__init__.py", line 81, in load
|
| 55 |
+
return loader.get_single_data()
|
| 56 |
+
^^^^^^^^^^^^^^^^^^^^^^^^
|
| 57 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 51, in get_single_data
|
| 58 |
+
return self.construct_document(node)
|
| 59 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 60 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 60, in construct_document
|
| 61 |
+
for dummy in generator:
|
| 62 |
+
^^^^^^^^^
|
| 63 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 413, in construct_yaml_map
|
| 64 |
+
value = self.construct_mapping(node)
|
| 65 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 66 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 218, in construct_mapping
|
| 67 |
+
return super().construct_mapping(node, deep=deep)
|
| 68 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 69 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 143, in construct_mapping
|
| 70 |
+
value = self.construct_object(value_node, deep=deep)
|
| 71 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 72 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/yaml/constructor.py", line 100, in construct_object
|
| 73 |
+
data = constructor(self, node)
|
| 74 |
+
^^^^^^^^^^^^^^^^^^^^^^^
|
| 75 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_yaml_loader.py", line 22, in ctor
|
| 76 |
+
return _import_func_in_yml(spec, base_dir)
|
| 77 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 78 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_yaml_loader.py", line 109, in _import_func_in_yml
|
| 79 |
+
module = _load_module_with_cache(rel)
|
| 80 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 81 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/_yaml_loader.py", line 88, in _load_module_with_cache
|
| 82 |
+
spec.loader.exec_module(module) # type: ignore[arg-type]
|
| 83 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 84 |
+
File "<frozen importlib._bootstrap_external>", line 999, in exec_module
|
| 85 |
+
File "<frozen importlib._bootstrap>", line 488, in _call_with_frames_removed
|
| 86 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/tasks/minerva_math/utils.py", line 21, in <module>
|
| 87 |
+
raise type(e)(
|
| 88 |
+
ModuleNotFoundError: `sympy`, `math_verify` and `antlr4-python3-runtime==4.11` are required for generating translation task prompt templates. Please install the required packages via pip install lm-eval[math] or pip install -e .[math]
|
| 89 |
+
2026-07-17T10:13:51-07:00 lm_eval exit=1 -> outputs/evals/general_suite/base_full
|
evals/general_suite/base_full_v2.log
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/1319 [00:00<?, ?it/s]
|
| 1 |
15%|ββ | 193/1319 [00:00<00:00, 1923.92it/s]
|
| 2 |
29%|βββ | 388/1319 [00:00<00:00, 1939.00it/s]
|
| 3 |
44%|βββββ | 583/1319 [00:00<00:00, 1939.69it/s]
|
| 4 |
59%|ββββββ | 779/1319 [00:00<00:00, 1945.97it/s]
|
| 5 |
74%|ββββββββ | 977/1319 [00:00<00:00, 1953.91it/s]
|
| 6 |
89%|βββββββββ | 1175/1319 [00:00<00:00, 1958.62it/s]
|
|
|
|
|
|
|
| 7 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 8 |
8%|β | 42/500 [00:00<00:01, 412.52it/s]
|
| 9 |
17%|ββ | 85/500 [00:00<00:00, 418.64it/s]
|
| 10 |
26%|βββ | 128/500 [00:00<00:00, 422.16it/s]
|
| 11 |
34%|ββββ | 171/500 [00:00<00:00, 424.56it/s]
|
| 12 |
43%|βββββ | 214/500 [00:00<00:00, 425.65it/s]
|
| 13 |
51%|ββββββ | 257/500 [00:00<00:00, 426.65it/s]
|
| 14 |
60%|ββββββ | 300/500 [00:00<00:00, 426.76it/s]
|
| 15 |
69%|βββββββ | 343/500 [00:00<00:00, 426.27it/s]
|
| 16 |
77%|ββββββββ | 386/500 [00:00<00:00, 426.73it/s]
|
| 17 |
86%|βββββββββ | 429/500 [00:01<00:00, 427.04it/s]
|
| 18 |
94%|ββββββββββ| 472/500 [00:01<00:00, 427.16it/s]
|
|
|
|
|
|
|
| 19 |
0%| | 0/541 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 21 |
89%|βββββββββ | 146/164 [00:00<00:00, 1454.03it/s]
|
|
|
|
|
|
|
| 22 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 23 |
3%|β | 17/500 [00:00<00:02, 162.63it/s]
|
| 24 |
7%|β | 34/500 [00:00<00:02, 163.74it/s]
|
| 25 |
10%|β | 51/500 [00:00<00:02, 164.28it/s]
|
| 26 |
14%|ββ | 68/500 [00:00<00:02, 164.70it/s]
|
| 27 |
17%|ββ | 85/500 [00:00<00:02, 165.02it/s]
|
| 28 |
20%|ββ | 102/500 [00:00<00:02, 165.26it/s]
|
| 29 |
24%|βββ | 119/500 [00:00<00:02, 165.44it/s]
|
| 30 |
27%|βββ | 136/500 [00:00<00:02, 165.44it/s]
|
| 31 |
31%|βββ | 153/500 [00:00<00:02, 165.47it/s]
|
| 32 |
34%|ββββ | 170/500 [00:01<00:01, 165.61it/s]
|
| 33 |
37%|ββββ | 187/500 [00:01<00:01, 165.85it/s]
|
| 34 |
41%|ββββ | 204/500 [00:01<00:01, 164.85it/s]
|
| 35 |
44%|βββββ | 221/500 [00:01<00:01, 165.24it/s]
|
| 36 |
48%|βββββ | 238/500 [00:01<00:01, 165.61it/s]
|
| 37 |
51%|βββββ | 255/500 [00:01<00:01, 165.86it/s]
|
| 38 |
54%|ββββββ | 272/500 [00:01<00:01, 166.24it/s]
|
| 39 |
58%|ββββββ | 289/500 [00:01<00:01, 166.44it/s]
|
| 40 |
61%|ββββββ | 306/500 [00:01<00:01, 166.55it/s]
|
| 41 |
65%|βββββββ | 323/500 [00:01<00:01, 166.91it/s]
|
| 42 |
68%|βββββββ | 340/500 [00:02<00:00, 166.95it/s]
|
| 43 |
71%|ββββββββ | 357/500 [00:02<00:00, 166.93it/s]
|
| 44 |
75%|ββββββββ | 374/500 [00:02<00:00, 167.11it/s]
|
| 45 |
78%|ββββββββ | 391/500 [00:02<00:00, 167.10it/s]
|
| 46 |
82%|βββββββββ | 408/500 [00:02<00:00, 167.15it/s]
|
| 47 |
85%|βββββββββ | 425/500 [00:02<00:00, 167.19it/s]
|
| 48 |
88%|βββββββββ | 442/500 [00:02<00:00, 167.47it/s]
|
| 49 |
92%|ββββββββββ| 459/500 [00:02<00:00, 167.38it/s]
|
| 50 |
95%|ββββββββββ| 476/500 [00:02<00:00, 167.34it/s]
|
| 51 |
99%|ββββββββββ| 493/500 [00:02<00:00, 167.50it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:05:01-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:05:01-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:05:42-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval]
|
| 4 |
+
2026-07-17:11:05:49 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval']
|
| 5 |
+
2026-07-17:11:05:50 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 6 |
+
2026-07-17:11:05:50 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 7 |
+
2026-07-17:11:05:50 INFO [models.api_models:179] Using max length 2048 - 1
|
| 8 |
+
2026-07-17:11:05:50 INFO [models.api_models:200] Using tokenizer None
|
| 9 |
+
Using the latest cached version of the dataset since google/IFEval couldn't be found on the Hugging Face Hub
|
| 10 |
+
Found the latest cached dataset configuration 'default' at /home/henry/.cache/huggingface/datasets/google___if_eval/default/0.0.0/966cd89545d6b6acfd7638bc708b98261ca58e84 (last modified on Wed Feb 4 07:51:11 2026).
|
| 11 |
+
2026-07-17:11:06:06 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:11:06:06 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 13 |
+
2026-07-17:11:06:06 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml)
|
| 14 |
+
2026-07-17:11:06:06 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 15 |
+
2026-07-17:11:06:06 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False}
|
| 16 |
+
2026-07-17:11:06:06 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0}
|
| 17 |
+
2026-07-17:11:06:06 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 18 |
+
2026-07-17:11:06:06 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 19 |
+
|
| 20 |
0%| | 0/1319 [00:00<?, ?it/s]
|
| 21 |
15%|ββ | 193/1319 [00:00<00:00, 1923.92it/s]
|
| 22 |
29%|βββ | 388/1319 [00:00<00:00, 1939.00it/s]
|
| 23 |
44%|βββββ | 583/1319 [00:00<00:00, 1939.69it/s]
|
| 24 |
59%|ββββββ | 779/1319 [00:00<00:00, 1945.97it/s]
|
| 25 |
74%|ββββββββ | 977/1319 [00:00<00:00, 1953.91it/s]
|
| 26 |
89%|βββββββββ | 1175/1319 [00:00<00:00, 1958.62it/s]
|
| 27 |
+
2026-07-17:11:06:07 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 28 |
+
|
| 29 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 30 |
8%|β | 42/500 [00:00<00:01, 412.52it/s]
|
| 31 |
17%|ββ | 85/500 [00:00<00:00, 418.64it/s]
|
| 32 |
26%|βββ | 128/500 [00:00<00:00, 422.16it/s]
|
| 33 |
34%|ββββ | 171/500 [00:00<00:00, 424.56it/s]
|
| 34 |
43%|βββββ | 214/500 [00:00<00:00, 425.65it/s]
|
| 35 |
51%|ββββββ | 257/500 [00:00<00:00, 426.65it/s]
|
| 36 |
60%|ββββββ | 300/500 [00:00<00:00, 426.76it/s]
|
| 37 |
69%|βββββββ | 343/500 [00:00<00:00, 426.27it/s]
|
| 38 |
77%|ββββββββ | 386/500 [00:00<00:00, 426.73it/s]
|
| 39 |
86%|βββββββββ | 429/500 [00:01<00:00, 427.04it/s]
|
| 40 |
94%|ββββββββββ| 472/500 [00:01<00:00, 427.16it/s]
|
| 41 |
+
2026-07-17:11:06:08 INFO [api.task:312] Building contexts for ifeval on rank 0...
|
| 42 |
+
|
| 43 |
0%| | 0/541 [00:00<?, ?it/s]
|
| 44 |
+
2026-07-17:11:06:08 INFO [evaluator:585] Running generate_until requests
|
| 45 |
+
2026-07-17:11:06:08 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
2026-07-17:11:10:28 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 50 |
+
2026-07-17:11:10:28 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/base_full_v2/student/*.jsonl
|
| 51 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
|
| 52 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 53 |
+
|------------------|------:|----------------|-----:|-----------------------|---|-----:|---|------|
|
| 54 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match |β |0.6945|Β± |0.0127|
|
| 55 |
+
| | |strict-match | 0|exact_match |β |0.0000|Β± | 0|
|
| 56 |
+
|ifeval | 4|none | 0|inst_level_loose_acc |β |0.7758|Β± | N/A|
|
| 57 |
+
| | |none | 0|inst_level_strict_acc |β |0.7218|Β± | N/A|
|
| 58 |
+
| | |none | 0|prompt_level_loose_acc |β |0.6765|Β± |0.0201|
|
| 59 |
+
| | |none | 0|prompt_level_strict_acc|β |0.6155|Β± |0.0209|
|
| 60 |
+
|minerva_math500 | 3|none | 4|exact_match |β |0.1000|Β± |0.0134|
|
| 61 |
+
| | |none | 4|math_verify |β |0.1740|Β± |0.0170|
|
| 62 |
+
|
| 63 |
+
2026-07-17T11:10:30-07:00 code pass [humaneval,mbpp] via /v1/completions (function-continuation)
|
| 64 |
+
2026-07-17:11:10:37 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 65 |
+
2026-07-17:11:10:37 WARNING [evaluator:184] pretrained=None appears to be an instruct or chat variant but chat template is not applied. Recommend setting `apply_chat_template`
|
| 66 |
+
(optionally `fewshot_as_multiturn`).
|
| 67 |
+
2026-07-17:11:10:38 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 68 |
+
2026-07-17:11:10:38 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'allenai/OLMoE-1B-7B-0125-Instruct', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 69 |
+
2026-07-17:11:10:38 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 70 |
+
2026-07-17:11:10:38 INFO [models.api_models:179] Using max length 2048 - 1
|
| 71 |
+
2026-07-17:11:10:38 INFO [models.api_models:200] Using tokenizer huggingface
|
| 72 |
+
2026-07-17:11:10:45 INFO [evaluator_utils:446] Selected tasks:
|
| 73 |
+
2026-07-17:11:10:45 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
|
| 74 |
+
2026-07-17:11:10:45 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
|
| 75 |
+
2026-07-17:11:10:45 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 76 |
+
2026-07-17:11:10:45 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
|
| 77 |
+
2026-07-17:11:10:45 INFO [api.task:312] Building contexts for humaneval on rank 0...
|
| 78 |
+
|
| 79 |
0%| | 0/164 [00:00<?, ?it/s]
|
| 80 |
89%|βββββββββ | 146/164 [00:00<00:00, 1454.03it/s]
|
| 81 |
+
2026-07-17:11:10:45 INFO [api.task:312] Building contexts for mbpp on rank 0...
|
| 82 |
+
|
| 83 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 84 |
3%|β | 17/500 [00:00<00:02, 162.63it/s]
|
| 85 |
7%|β | 34/500 [00:00<00:02, 163.74it/s]
|
| 86 |
10%|β | 51/500 [00:00<00:02, 164.28it/s]
|
| 87 |
14%|ββ | 68/500 [00:00<00:02, 164.70it/s]
|
| 88 |
17%|ββ | 85/500 [00:00<00:02, 165.02it/s]
|
| 89 |
20%|ββ | 102/500 [00:00<00:02, 165.26it/s]
|
| 90 |
24%|βββ | 119/500 [00:00<00:02, 165.44it/s]
|
| 91 |
27%|βββ | 136/500 [00:00<00:02, 165.44it/s]
|
| 92 |
31%|βββ | 153/500 [00:00<00:02, 165.47it/s]
|
| 93 |
34%|ββββ | 170/500 [00:01<00:01, 165.61it/s]
|
| 94 |
37%|ββββ | 187/500 [00:01<00:01, 165.85it/s]
|
| 95 |
41%|ββββ | 204/500 [00:01<00:01, 164.85it/s]
|
| 96 |
44%|βββββ | 221/500 [00:01<00:01, 165.24it/s]
|
| 97 |
48%|βββββ | 238/500 [00:01<00:01, 165.61it/s]
|
| 98 |
51%|βββββ | 255/500 [00:01<00:01, 165.86it/s]
|
| 99 |
54%|ββββββ | 272/500 [00:01<00:01, 166.24it/s]
|
| 100 |
58%|ββββββ | 289/500 [00:01<00:01, 166.44it/s]
|
| 101 |
61%|ββββββ | 306/500 [00:01<00:01, 166.55it/s]
|
| 102 |
65%|βββββββ | 323/500 [00:01<00:01, 166.91it/s]
|
| 103 |
68%|βββββββ | 340/500 [00:02<00:00, 166.95it/s]
|
| 104 |
71%|ββββββββ | 357/500 [00:02<00:00, 166.93it/s]
|
| 105 |
75%|ββββββββ | 374/500 [00:02<00:00, 167.11it/s]
|
| 106 |
78%|ββββββββ | 391/500 [00:02<00:00, 167.10it/s]
|
| 107 |
82%|βββββββββ | 408/500 [00:02<00:00, 167.15it/s]
|
| 108 |
85%|βββββββββ | 425/500 [00:02<00:00, 167.19it/s]
|
| 109 |
88%|βββββββββ | 442/500 [00:02<00:00, 167.47it/s]
|
| 110 |
92%|ββββββββββ| 459/500 [00:02<00:00, 167.38it/s]
|
| 111 |
95%|ββββββββββ| 476/500 [00:02<00:00, 167.34it/s]
|
| 112 |
99%|ββββββββββ| 493/500 [00:02<00:00, 167.50it/s]
|
| 113 |
+
2026-07-17:11:10:48 INFO [evaluator:585] Running generate_until requests
|
| 114 |
+
2026-07-17:11:10:48 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
2026-07-17:11:13:41 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 118 |
+
2026-07-17:11:13:41 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/base_full_v2/student/*.jsonl
|
| 119 |
+
local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'allenai/OLMoE-1B-7B-0125-Instruct', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: None, num_fewshot: None, batch_size: 1
|
| 120 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 121 |
+
|---------|------:|-----------|-----:|---------|---|-----:|---|-----:|
|
| 122 |
+
|humaneval| 1|create_test| 0|pass@1 |β |0.3476|Β± |0.0373|
|
| 123 |
+
|mbpp | 1|none | 3|pass_at_1|β |0.3020|Β± |0.0206|
|
| 124 |
+
|
| 125 |
+
2026-07-17T11:13:42-07:00 lm_eval exit=0 -> outputs/evals/general_suite/base_full_v2
|
evals/general_suite/gsm8k_protocol.log
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 1 |
26%|βββ | 26/100 [00:00<00:00, 257.58it/s]
|
| 2 |
61%|ββββββ | 61/100 [00:00<00:00, 307.90it/s]
|
| 3 |
96%|ββββββββββ| 96/100 [00:00<00:00, 324.16it/s]
|
|
|
|
|
|
|
| 4 |
0%| | 0/100 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 5 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 6 |
11%|β | 11/100 [00:00<00:00, 103.53it/s]
|
| 7 |
22%|βββ | 22/100 [00:00<00:00, 104.15it/s]
|
| 8 |
33%|ββββ | 33/100 [00:00<00:00, 104.38it/s]
|
| 9 |
44%|βββββ | 44/100 [00:00<00:00, 104.53it/s]
|
| 10 |
55%|ββββββ | 55/100 [00:00<00:00, 104.65it/s]
|
| 11 |
66%|βββββββ | 66/100 [00:00<00:00, 104.85it/s]
|
| 12 |
77%|ββββββββ | 77/100 [00:00<00:00, 104.98it/s]
|
| 13 |
88%|βββββββββ | 88/100 [00:00<00:00, 104.96it/s]
|
| 14 |
99%|ββββββββββ| 99/100 [00:00<00:00, 105.01it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T10:24:04-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T10:24:04-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T10:24:34-07:00 server up; running lm_eval [gsm8k,gsm8k_cot_zeroshot,gsm8k_cot]
|
| 4 |
+
2026-07-17:10:24:34 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:10:24:41 INFO [_cli.run:388] Selected Tasks: ['gsm8k', 'gsm8k_cot_zeroshot', 'gsm8k_cot']
|
| 6 |
+
2026-07-17:10:24:43 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:10:24:43 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 32, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:10:24:43 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:10:24:43 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
2026-07-17:10:24:45 INFO [evaluator_utils:446] Selected tasks:
|
| 11 |
+
2026-07-17:10:24:45 INFO [evaluator_utils:480] Task: gsm8k (gsm8k/gsm8k.yaml)
|
| 12 |
+
2026-07-17:10:24:45 INFO [evaluator_utils:480] Task: gsm8k_cot (gsm8k/gsm8k-cot.yaml)
|
| 13 |
+
2026-07-17:10:24:45 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 14 |
+
2026-07-17:10:24:45 INFO [evaluator:314] gsm8k: Using gen_kwargs: {'until': ['Question:', '</s>', '<|im_end|>'], 'do_sample': False, 'temperature': 0.0}
|
| 15 |
+
2026-07-17:10:24:45 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False}
|
| 16 |
+
2026-07-17:10:24:45 INFO [evaluator:314] gsm8k_cot: Using gen_kwargs: {'do_sample': False, 'until': ['Q:', '</s>', '<|im_end|>']}
|
| 17 |
+
2026-07-17:10:24:45 INFO [api.task:312] Building contexts for gsm8k on rank 0...
|
| 18 |
+
|
| 19 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 20 |
26%|βββ | 26/100 [00:00<00:00, 257.58it/s]
|
| 21 |
61%|ββββββ | 61/100 [00:00<00:00, 307.90it/s]
|
| 22 |
96%|ββββββββββ| 96/100 [00:00<00:00, 324.16it/s]
|
| 23 |
+
2026-07-17:10:24:46 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 24 |
+
|
| 25 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 26 |
+
2026-07-17:10:24:46 INFO [api.task:312] Building contexts for gsm8k_cot on rank 0...
|
| 27 |
+
|
| 28 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 29 |
11%|β | 11/100 [00:00<00:00, 103.53it/s]
|
| 30 |
22%|βββ | 22/100 [00:00<00:00, 104.15it/s]
|
| 31 |
33%|ββββ | 33/100 [00:00<00:00, 104.38it/s]
|
| 32 |
44%|βββββ | 44/100 [00:00<00:00, 104.53it/s]
|
| 33 |
55%|ββββββ | 55/100 [00:00<00:00, 104.65it/s]
|
| 34 |
66%|βββββββ | 66/100 [00:00<00:00, 104.85it/s]
|
| 35 |
77%|ββββββββ | 77/100 [00:00<00:00, 104.98it/s]
|
| 36 |
88%|βββββββββ | 88/100 [00:00<00:00, 104.96it/s]
|
| 37 |
99%|ββββββββββ| 99/100 [00:00<00:00, 105.01it/s]
|
| 38 |
+
2026-07-17:10:24:47 INFO [evaluator:585] Running generate_until requests
|
| 39 |
+
2026-07-17:10:24:47 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
2026-07-17:10:25:18 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 43 |
+
2026-07-17:10:25:18 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/gsm8k_protocol/student/*.jsonl
|
| 44 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 32, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 100.0, num_fewshot: None, batch_size: 1
|
| 45 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 46 |
+
|------------------|------:|----------------|-----:|-----------|---|----:|---|-----:|
|
| 47 |
+
|gsm8k | 3|flexible-extract| 5|exact_match|β | 0.74|Β± |0.0441|
|
| 48 |
+
| | |strict-match | 5|exact_match|β | 0.71|Β± |0.0456|
|
| 49 |
+
|gsm8k_cot | 3|flexible-extract| 8|exact_match|β | 0.70|Β± |0.0461|
|
| 50 |
+
| | |strict-match | 8|exact_match|β | 0.69|Β± |0.0465|
|
| 51 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match|β | 0.70|Β± |0.0461|
|
| 52 |
+
| | |strict-match | 0|exact_match|β | 0.00|Β± |0.0000|
|
| 53 |
+
|
| 54 |
+
2026-07-17T10:25:19-07:00 lm_eval exit=0 -> outputs/evals/general_suite/gsm8k_protocol
|
evals/general_suite/math_fix.log
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 1 |
40%|ββββ | 40/100 [00:00<00:00, 396.25it/s]
|
| 2 |
82%|βββββββββ | 82/100 [00:00<00:00, 409.76it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:15:57-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:15:57-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:16:27-07:00 server up; chat pass [minerva_math500]
|
| 4 |
+
2026-07-17:11:16:27 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:11:16:34 INFO [_cli.run:388] Selected Tasks: ['minerva_math500']
|
| 6 |
+
2026-07-17:11:16:35 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:11:16:35 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 8 |
+
2026-07-17:11:16:35 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 32, 'tokenized_requests': False, 'max_retries': 3}
|
| 9 |
+
2026-07-17:11:16:35 INFO [models.api_models:179] Using max length 2048 - 1
|
| 10 |
+
2026-07-17:11:16:35 INFO [models.api_models:200] Using tokenizer None
|
| 11 |
+
2026-07-17:11:16:37 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:11:16:37 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 13 |
+
2026-07-17:11:16:37 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 14 |
+
2026-07-17:11:16:37 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 15 |
+
|
| 16 |
0%| | 0/100 [00:00<?, ?it/s]
|
| 17 |
40%|ββββ | 40/100 [00:00<00:00, 396.25it/s]
|
| 18 |
82%|βββββββββ | 82/100 [00:00<00:00, 409.76it/s]
|
| 19 |
+
2026-07-17:11:16:37 INFO [evaluator:585] Running generate_until requests
|
| 20 |
+
2026-07-17:11:16:37 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 21 |
+
|
| 22 |
+
2026-07-17:11:17:27 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 23 |
+
2026-07-17:11:17:27 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/math_fix/student/*.jsonl
|
| 24 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 32, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 100.0, num_fewshot: None, batch_size: 1
|
| 25 |
+
| Tasks |Version|Filter|n-shot| Metric | |Value| |Stderr|
|
| 26 |
+
|---------------|------:|------|-----:|-----------|---|----:|---|-----:|
|
| 27 |
+
|minerva_math500| 3|none | 4|exact_match|β | 0.17|Β± |0.0378|
|
| 28 |
+
| | |none | 4|math_verify|β | 0.33|Β± |0.0473|
|
| 29 |
+
|
| 30 |
+
2026-07-17T11:17:28-07:00 lm_eval exit=0 -> outputs/evals/general_suite/math_fix
|
evals/general_suite/math_full.log
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 1 |
8%|β | 41/500 [00:00<00:01, 403.43it/s]
|
| 2 |
17%|ββ | 83/500 [00:00<00:01, 412.27it/s]
|
| 3 |
25%|βββ | 126/500 [00:00<00:00, 416.62it/s]
|
| 4 |
34%|ββββ | 169/500 [00:00<00:00, 419.75it/s]
|
| 5 |
42%|βββββ | 212/500 [00:00<00:00, 421.80it/s]
|
| 6 |
51%|βββββ | 255/500 [00:00<00:00, 423.00it/s]
|
| 7 |
60%|ββββββ | 298/500 [00:00<00:00, 424.13it/s]
|
| 8 |
68%|βββββββ | 341/500 [00:00<00:00, 425.46it/s]
|
| 9 |
77%|ββββββββ | 384/500 [00:00<00:00, 426.17it/s]
|
| 10 |
85%|βββββββββ | 427/500 [00:01<00:00, 427.17it/s]
|
| 11 |
94%|ββββββββββ| 471/500 [00:01<00:00, 427.83it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:18:30-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:18:30-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:19:00-07:00 server up; chat pass [minerva_math500]
|
| 4 |
+
2026-07-17:11:19:07 INFO [_cli.run:388] Selected Tasks: ['minerva_math500']
|
| 5 |
+
2026-07-17:11:19:08 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 6 |
+
2026-07-17:11:19:08 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 7 |
+
2026-07-17:11:19:08 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:11:19:08 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:11:19:08 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
2026-07-17:11:19:10 INFO [evaluator_utils:446] Selected tasks:
|
| 11 |
+
2026-07-17:11:19:10 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 12 |
+
2026-07-17:11:19:10 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 13 |
+
2026-07-17:11:19:10 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 14 |
+
|
| 15 |
0%| | 0/500 [00:00<?, ?it/s]
|
| 16 |
8%|β | 41/500 [00:00<00:01, 403.43it/s]
|
| 17 |
17%|ββ | 83/500 [00:00<00:01, 412.27it/s]
|
| 18 |
25%|βββ | 126/500 [00:00<00:00, 416.62it/s]
|
| 19 |
34%|ββββ | 169/500 [00:00<00:00, 419.75it/s]
|
| 20 |
42%|βββββ | 212/500 [00:00<00:00, 421.80it/s]
|
| 21 |
51%|βββββ | 255/500 [00:00<00:00, 423.00it/s]
|
| 22 |
60%|ββββββ | 298/500 [00:00<00:00, 424.13it/s]
|
| 23 |
68%|βββββββ | 341/500 [00:00<00:00, 425.46it/s]
|
| 24 |
77%|ββββββββ | 384/500 [00:00<00:00, 426.17it/s]
|
| 25 |
85%|βββββββββ | 427/500 [00:01<00:00, 427.17it/s]
|
| 26 |
94%|ββββββββββ| 471/500 [00:01<00:00, 427.83it/s]
|
| 27 |
+
2026-07-17:11:19:11 INFO [evaluator:585] Running generate_until requests
|
| 28 |
+
2026-07-17:11:19:11 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 29 |
+
|
| 30 |
+
2026-07-17:11:22:08 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 31 |
+
2026-07-17:11:22:08 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/math_full/student/*.jsonl
|
| 32 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 48, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: None, num_fewshot: None, batch_size: 1
|
| 33 |
+
| Tasks |Version|Filter|n-shot| Metric | |Value| |Stderr|
|
| 34 |
+
|---------------|------:|------|-----:|-----------|---|----:|---|-----:|
|
| 35 |
+
|minerva_math500| 3|none | 4|exact_match|β |0.156|Β± |0.0162|
|
| 36 |
+
| | |none | 4|math_verify|β |0.254|Β± |0.0195|
|
| 37 |
+
|
| 38 |
+
2026-07-17T11:22:09-07:00 lm_eval exit=0 -> outputs/evals/general_suite/math_full
|
evals/general_suite/policy_confirm.combo_on25_seed1224.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.combo_on25_seed1225.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.combo_on25_seed1226.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.off_forward_seed1224.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.off_forward_seed1225.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.off_forward_seed1226.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.on_reverse_seed1224.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.on_reverse_seed1225.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/policy_confirm.on_reverse_seed1226.eval.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/general_suite/pruneval.log
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:36:56-07:00 === eval 9 general prunes, 3 staggered lanes ===
|
| 2 |
+
2026-07-17T11:36:56-07:00 EVAL glean_keep75 on GPU 0 (port 8410)
|
| 3 |
+
2026-07-17T11:38:56-07:00 EVAL reap_keep75 on GPU 1 (port 8411)
|
| 4 |
+
2026-07-17T11:40:56-07:00 EVAL uniform_keep75 on GPU 2 (port 8412)
|
| 5 |
+
2026-07-17T11:47:53-07:00 reap_keep75 done
|
| 6 |
+
2026-07-17T11:47:53-07:00 EVAL reap_keep50 on GPU 1 (port 8411)
|
| 7 |
+
2026-07-17T11:50:42-07:00 glean_keep75 done
|
| 8 |
+
2026-07-17T11:50:42-07:00 EVAL glean_keep50 on GPU 0 (port 8410)
|
| 9 |
+
2026-07-17T11:54:36-07:00 uniform_keep75 done
|
| 10 |
+
2026-07-17T11:54:36-07:00 EVAL uniform_keep50 on GPU 2 (port 8412)
|
| 11 |
+
2026-07-17T12:04:54-07:00 glean_keep50 done
|
| 12 |
+
2026-07-17T12:04:54-07:00 EVAL glean_keep25 on GPU 0 (port 8410)
|
| 13 |
+
2026-07-17T12:04:58-07:00 reap_keep50 done
|
| 14 |
+
2026-07-17T12:04:58-07:00 EVAL reap_keep25 on GPU 1 (port 8411)
|
| 15 |
+
2026-07-17T12:07:22-07:00 uniform_keep50 done
|
| 16 |
+
2026-07-17T12:07:22-07:00 EVAL uniform_keep25 on GPU 2 (port 8412)
|
| 17 |
+
2026-07-17T12:22:04-07:00 glean_keep25 done
|
| 18 |
+
2026-07-17T12:22:04-07:00 LANE glean (GPU 0) complete
|
| 19 |
+
2026-07-17T12:25:54-07:00 reap_keep25 done
|
| 20 |
+
2026-07-17T12:25:54-07:00 LANE reap (GPU 1) complete
|
| 21 |
+
2026-07-17T12:30:45-07:00 uniform_keep25 done
|
| 22 |
+
2026-07-17T12:30:45-07:00 LANE uniform (GPU 2) complete
|
| 23 |
+
2026-07-17T12:30:45-07:00 === ALL 9 PRUNE EVALS COMPLETE ===
|
evals/general_suite/ragged_smoke.log
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/8 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 1 |
0%| | 0/8 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/8 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
0%| | 0/8 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 4 |
0%| | 0/8 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:34:22-07:00 serving outputs/pruned/glean-0125inst-general-keep75 on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:34:22-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:34:58-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval]
|
| 4 |
+
2026-07-17:11:34:58 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:11:35:05 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval']
|
| 6 |
+
2026-07-17:11:35:06 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:11:35:06 WARNING [evaluator:226] generation_kwargs: {'max_gen_toks': 1280} specified through cli, these settings will update set parameters in yaml tasks. Ensure 'do_sample=True' for non-greedy decoding!
|
| 8 |
+
2026-07-17:11:35:06 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 9 |
+
2026-07-17:11:35:06 INFO [models.api_models:179] Using max length 2048 - 1
|
| 10 |
+
2026-07-17:11:35:06 INFO [models.api_models:200] Using tokenizer None
|
| 11 |
+
2026-07-17:11:35:11 INFO [evaluator_utils:446] Selected tasks:
|
| 12 |
+
2026-07-17:11:35:11 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 13 |
+
2026-07-17:11:35:11 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml)
|
| 14 |
+
2026-07-17:11:35:11 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 15 |
+
2026-07-17:11:35:11 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False, 'max_gen_toks': 1280}
|
| 16 |
+
2026-07-17:11:35:11 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 17 |
+
2026-07-17:11:35:11 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 18 |
+
2026-07-17:11:35:11 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 19 |
+
|
| 20 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 21 |
+
2026-07-17:11:35:11 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 22 |
+
|
| 23 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 24 |
+
2026-07-17:11:35:11 INFO [api.task:312] Building contexts for ifeval on rank 0...
|
| 25 |
+
|
| 26 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 27 |
+
2026-07-17:11:35:11 INFO [evaluator:585] Running generate_until requests
|
| 28 |
+
2026-07-17:11:35:11 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
2026-07-17:11:35:56 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 33 |
+
2026-07-17:11:35:56 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/ragged_smoke/student/*.jsonl
|
| 34 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({'max_gen_toks': 1280}), limit: 8.0, num_fewshot: None, batch_size: 1
|
| 35 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 36 |
+
|------------------|------:|----------------|-----:|-----------------------|---|-----:|---|------|
|
| 37 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match |β |0.5000|Β± | 0.189|
|
| 38 |
+
| | |strict-match | 0|exact_match |β |0.0000|Β± | 0|
|
| 39 |
+
|ifeval | 4|none | 0|inst_level_loose_acc |β |0.7857|Β± | N/A|
|
| 40 |
+
| | |none | 0|inst_level_strict_acc |β |0.6429|Β± | N/A|
|
| 41 |
+
| | |none | 0|prompt_level_loose_acc |β |0.6250|Β± |0.1830|
|
| 42 |
+
| | |none | 0|prompt_level_strict_acc|β |0.5000|Β± |0.1890|
|
| 43 |
+
|minerva_math500 | 3|none | 4|exact_match |β |0.1250|Β± |0.1250|
|
| 44 |
+
| | |none | 4|math_verify |β |0.1250|Β± |0.1250|
|
| 45 |
+
|
| 46 |
+
2026-07-17T11:35:57-07:00 code pass [humaneval,mbpp] via /v1/completions (function-continuation)
|
| 47 |
+
2026-07-17:11:35:58 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 48 |
+
2026-07-17:11:36:05 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 49 |
+
2026-07-17:11:36:05 WARNING [evaluator:184] pretrained=None appears to be an instruct or chat variant but chat template is not applied. Recommend setting `apply_chat_template`
|
| 50 |
+
(optionally `fewshot_as_multiturn`).
|
| 51 |
+
2026-07-17:11:36:06 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 52 |
+
2026-07-17:11:36:06 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'outputs/pruned/glean-0125inst-general-keep75', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 53 |
+
2026-07-17:11:36:06 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 54 |
+
2026-07-17:11:36:06 INFO [models.api_models:179] Using max length 2048 - 1
|
| 55 |
+
2026-07-17:11:36:06 INFO [models.api_models:200] Using tokenizer huggingface
|
| 56 |
+
2026-07-17:11:36:13 INFO [evaluator_utils:446] Selected tasks:
|
| 57 |
+
2026-07-17:11:36:13 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
|
| 58 |
+
2026-07-17:11:36:13 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
|
| 59 |
+
2026-07-17:11:36:13 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 60 |
+
2026-07-17:11:36:13 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
|
| 61 |
+
2026-07-17:11:36:13 INFO [api.task:312] Building contexts for humaneval on rank 0...
|
| 62 |
+
|
| 63 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 64 |
+
2026-07-17:11:36:13 INFO [api.task:312] Building contexts for mbpp on rank 0...
|
| 65 |
+
|
| 66 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 67 |
+
2026-07-17:11:36:13 INFO [evaluator:585] Running generate_until requests
|
| 68 |
+
2026-07-17:11:36:13 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
2026-07-17:11:36:22 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 72 |
+
2026-07-17:11:36:22 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/ragged_smoke/student/*.jsonl
|
| 73 |
+
local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'outputs/pruned/glean-0125inst-general-keep75', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 8.0, num_fewshot: None, batch_size: 1
|
| 74 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 75 |
+
|---------|------:|-----------|-----:|---------|---|----:|---|-----:|
|
| 76 |
+
|humaneval| 1|create_test| 0|pass@1 |β |0.500|Β± | 0.189|
|
| 77 |
+
|mbpp | 1|none | 3|pass_at_1|β |0.375|Β± | 0.183|
|
| 78 |
+
|
| 79 |
+
2026-07-17T11:36:24-07:00 lm_eval exit=0 -> outputs/evals/general_suite/ragged_smoke
|
evals/general_suite/smoke_full.log
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 1 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:00:26-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:00:26-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:00:56-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval]
|
| 4 |
+
2026-07-17:11:00:56 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:11:01:03 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval']
|
| 6 |
+
2026-07-17:11:01:04 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:11:01:04 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:11:01:04 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:11:01:04 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
2026-07-17:11:01:09 INFO [evaluator_utils:446] Selected tasks:
|
| 11 |
+
2026-07-17:11:01:09 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 12 |
+
2026-07-17:11:01:09 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml)
|
| 13 |
+
2026-07-17:11:01:09 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 14 |
+
2026-07-17:11:01:09 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False}
|
| 15 |
+
2026-07-17:11:01:09 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0}
|
| 16 |
+
2026-07-17:11:01:09 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 17 |
+
2026-07-17:11:01:09 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 18 |
+
|
| 19 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 20 |
+
2026-07-17:11:01:09 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 21 |
+
|
| 22 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 23 |
+
2026-07-17:11:01:09 INFO [api.task:312] Building contexts for ifeval on rank 0...
|
| 24 |
+
|
| 25 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 26 |
+
2026-07-17:11:01:09 INFO [evaluator:585] Running generate_until requests
|
| 27 |
+
2026-07-17:11:01:09 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
2026-07-17:11:01:29 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 32 |
+
2026-07-17:11:01:29 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/smoke_full/student/*.jsonl
|
| 33 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 10.0, num_fewshot: None, batch_size: 1
|
| 34 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 35 |
+
|------------------|------:|----------------|-----:|-----------------------|---|-----:|---|------|
|
| 36 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match |β |0.7000|Β± |0.1528|
|
| 37 |
+
| | |strict-match | 0|exact_match |β |0.0000|Β± | 0|
|
| 38 |
+
|ifeval | 4|none | 0|inst_level_loose_acc |β |0.7222|Β± | N/A|
|
| 39 |
+
| | |none | 0|inst_level_strict_acc |β |0.7222|Β± | N/A|
|
| 40 |
+
| | |none | 0|prompt_level_loose_acc |β |0.6000|Β± |0.1633|
|
| 41 |
+
| | |none | 0|prompt_level_strict_acc|β |0.6000|Β± |0.1633|
|
| 42 |
+
|minerva_math500 | 3|none | 4|exact_match |β |0.0000|Β± |0.0000|
|
| 43 |
+
| | |none | 4|math_verify |β |0.1000|Β± |0.1000|
|
| 44 |
+
|
| 45 |
+
2026-07-17T11:01:31-07:00 code pass [humaneval,mbpp] via /v1/completions (function-continuation)
|
| 46 |
+
2026-07-17:11:01:31 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 47 |
+
2026-07-17:11:01:38 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 48 |
+
2026-07-17:11:01:39 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 49 |
+
2026-07-17:11:01:39 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 50 |
+
2026-07-17:11:01:39 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 51 |
+
2026-07-17:11:01:39 INFO [models.api_models:179] Using max length 2048 - 1
|
| 52 |
+
2026-07-17:11:01:39 INFO [models.api_models:200] Using tokenizer huggingface
|
| 53 |
+
Traceback (most recent call last):
|
| 54 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/utils/_http.py", line 403, in hf_raise_for_status
|
| 55 |
+
response.raise_for_status()
|
| 56 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/requests/models.py", line 1167, in raise_for_status
|
| 57 |
+
raise HTTPError(http_error_msg, response=self)
|
| 58 |
+
requests.exceptions.HTTPError: 404 Client Error: Not Found for url: https://huggingface.co/student/resolve/main/tokenizer_config.json
|
| 59 |
+
|
| 60 |
+
The above exception was the direct cause of the following exception:
|
| 61 |
+
|
| 62 |
+
Traceback (most recent call last):
|
| 63 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/utils/hub.py", line 479, in cached_files
|
| 64 |
+
hf_hub_download(
|
| 65 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/utils/_validators.py", line 114, in _inner_fn
|
| 66 |
+
return fn(*args, **kwargs)
|
| 67 |
+
^^^^^^^^^^^^^^^^^^^
|
| 68 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 1014, in hf_hub_download
|
| 69 |
+
return _hf_hub_download_to_cache_dir(
|
| 70 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 71 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 1121, in _hf_hub_download_to_cache_dir
|
| 72 |
+
_raise_on_head_call_error(head_call_error, force_download, local_files_only)
|
| 73 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 1662, in _raise_on_head_call_error
|
| 74 |
+
raise head_call_error
|
| 75 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 1550, in _get_metadata_or_catch_error
|
| 76 |
+
metadata = get_hf_file_metadata(
|
| 77 |
+
^^^^^^^^^^^^^^^^^^^^^
|
| 78 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/utils/_validators.py", line 114, in _inner_fn
|
| 79 |
+
return fn(*args, **kwargs)
|
| 80 |
+
^^^^^^^^^^^^^^^^^^^
|
| 81 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 1467, in get_hf_file_metadata
|
| 82 |
+
r = _request_wrapper(
|
| 83 |
+
^^^^^^^^^^^^^^^^^
|
| 84 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 283, in _request_wrapper
|
| 85 |
+
response = _request_wrapper(
|
| 86 |
+
^^^^^^^^^^^^^^^^^
|
| 87 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/file_download.py", line 307, in _request_wrapper
|
| 88 |
+
hf_raise_for_status(response)
|
| 89 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/huggingface_hub/utils/_http.py", line 453, in hf_raise_for_status
|
| 90 |
+
raise _format(RepositoryNotFoundError, message, response) from e
|
| 91 |
+
huggingface_hub.errors.RepositoryNotFoundError: 404 Client Error. (Request ID: Root=1-6a5a6e06-7ca3e69563bb61cc36929c63;ccf20dcb-0c04-4cc0-a2fd-ac8935f5a941)
|
| 92 |
+
|
| 93 |
+
Repository Not Found for url: https://huggingface.co/student/resolve/main/tokenizer_config.json.
|
| 94 |
+
Please make sure you specified the correct `repo_id` and `repo_type`.
|
| 95 |
+
If you are trying to access a private or gated repo, make sure you are authenticated. For more details, see https://huggingface.co/docs/huggingface_hub/authentication
|
| 96 |
+
|
| 97 |
+
The above exception was the direct cause of the following exception:
|
| 98 |
+
|
| 99 |
+
Traceback (most recent call last):
|
| 100 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/bin/lm_eval", line 10, in <module>
|
| 101 |
+
sys.exit(cli_evaluate())
|
| 102 |
+
^^^^^^^^^^^^^^
|
| 103 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/__main__.py", line 10, in cli_evaluate
|
| 104 |
+
parser.execute(args)
|
| 105 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/_cli/harness.py", line 60, in execute
|
| 106 |
+
args.func(args)
|
| 107 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/_cli/run.py", line 391, in _execute
|
| 108 |
+
results = simple_evaluate(
|
| 109 |
+
^^^^^^^^^^^^^^^^
|
| 110 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/utils.py", line 575, in _wrapper
|
| 111 |
+
return fn(*args, **kwargs)
|
| 112 |
+
^^^^^^^^^^^^^^^^^^^
|
| 113 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/evaluator.py", line 242, in simple_evaluate
|
| 114 |
+
lm = lm_eval.api.registry.get_model(model).create_from_arg_obj(
|
| 115 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 116 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/api/model.py", line 169, in create_from_arg_obj
|
| 117 |
+
return cls(**arg_dict, **additional_config)
|
| 118 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 119 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/models/openai_completions.py", line 52, in __init__
|
| 120 |
+
super().__init__(
|
| 121 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/lm_eval/models/api_models.py", line 209, in __init__
|
| 122 |
+
self.tokenizer = transformers.AutoTokenizer.from_pretrained(
|
| 123 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 124 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/models/auto/tokenization_auto.py", line 1089, in from_pretrained
|
| 125 |
+
tokenizer_config = get_tokenizer_config(pretrained_model_name_or_path, **kwargs)
|
| 126 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 127 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/models/auto/tokenization_auto.py", line 921, in get_tokenizer_config
|
| 128 |
+
resolved_config_file = cached_file(
|
| 129 |
+
^^^^^^^^^^^^
|
| 130 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/utils/hub.py", line 322, in cached_file
|
| 131 |
+
file = cached_files(path_or_repo_id=path_or_repo_id, filenames=[filename], **kwargs)
|
| 132 |
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
| 133 |
+
File "/home/henry/Documents/PythonProjects/variable-reap/.venv/lib/python3.12/site-packages/transformers/utils/hub.py", line 511, in cached_files
|
| 134 |
+
raise OSError(
|
| 135 |
+
OSError: student is not a local folder and is not a valid model identifier listed on 'https://huggingface.co/models'
|
| 136 |
+
If this is a private repository, make sure to pass a token having permission to this repo either by logging in with `hf auth login` or by passing `token=<your_token>`
|
| 137 |
+
2026-07-17T11:01:43-07:00 lm_eval exit=1 -> outputs/evals/general_suite/smoke_full
|
evals/general_suite/smoke_full2.log
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 1 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 4 |
0%| | 0/10 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T11:02:49-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T11:02:49-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T11:03:20-07:00 server up; chat pass [gsm8k_cot_zeroshot,minerva_math500,ifeval]
|
| 4 |
+
2026-07-17:11:03:20 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:11:03:27 INFO [_cli.run:388] Selected Tasks: ['gsm8k_cot_zeroshot', 'minerva_math500', 'ifeval']
|
| 6 |
+
2026-07-17:11:03:28 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:11:03:28 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:11:03:28 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:11:03:28 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
2026-07-17:11:03:33 INFO [evaluator_utils:446] Selected tasks:
|
| 11 |
+
2026-07-17:11:03:33 INFO [evaluator_utils:480] Task: gsm8k_cot_zeroshot (gsm8k/gsm8k-cot-zeroshot.yaml)
|
| 12 |
+
2026-07-17:11:03:33 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml)
|
| 13 |
+
2026-07-17:11:03:33 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 14 |
+
2026-07-17:11:03:33 INFO [evaluator:314] gsm8k_cot_zeroshot: Using gen_kwargs: {'until': ['Q:', '</s>', '<|im_end|>'], 'do_sample': False}
|
| 15 |
+
2026-07-17:11:03:33 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0}
|
| 16 |
+
2026-07-17:11:03:33 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 17 |
+
2026-07-17:11:03:33 INFO [api.task:312] Building contexts for gsm8k_cot_zeroshot on rank 0...
|
| 18 |
+
|
| 19 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 20 |
+
2026-07-17:11:03:33 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 21 |
+
|
| 22 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 23 |
+
2026-07-17:11:03:33 INFO [api.task:312] Building contexts for ifeval on rank 0...
|
| 24 |
+
|
| 25 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 26 |
+
2026-07-17:11:03:33 INFO [evaluator:585] Running generate_until requests
|
| 27 |
+
2026-07-17:11:03:33 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
2026-07-17:11:03:53 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 32 |
+
2026-07-17:11:03:53 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/smoke_full2/student/*.jsonl
|
| 33 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 10.0, num_fewshot: None, batch_size: 1
|
| 34 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value | |Stderr|
|
| 35 |
+
|------------------|------:|----------------|-----:|-----------------------|---|-----:|---|------|
|
| 36 |
+
|gsm8k_cot_zeroshot| 3|flexible-extract| 0|exact_match |β |0.7000|Β± |0.1528|
|
| 37 |
+
| | |strict-match | 0|exact_match |β |0.0000|Β± | 0|
|
| 38 |
+
|ifeval | 4|none | 0|inst_level_loose_acc |β |0.7222|Β± | N/A|
|
| 39 |
+
| | |none | 0|inst_level_strict_acc |β |0.7222|Β± | N/A|
|
| 40 |
+
| | |none | 0|prompt_level_loose_acc |β |0.6000|Β± |0.1633|
|
| 41 |
+
| | |none | 0|prompt_level_strict_acc|β |0.6000|Β± |0.1633|
|
| 42 |
+
|minerva_math500 | 3|none | 4|exact_match |β |0.0000|Β± |0.0000|
|
| 43 |
+
| | |none | 4|math_verify |β |0.1000|Β± |0.1000|
|
| 44 |
+
|
| 45 |
+
2026-07-17T11:03:54-07:00 code pass [humaneval,mbpp] via /v1/completions (function-continuation)
|
| 46 |
+
2026-07-17:11:03:54 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 47 |
+
2026-07-17:11:04:01 INFO [_cli.run:388] Selected Tasks: ['humaneval', 'mbpp']
|
| 48 |
+
2026-07-17:11:04:01 WARNING [evaluator:184] pretrained=None appears to be an instruct or chat variant but chat template is not applied. Recommend setting `apply_chat_template`
|
| 49 |
+
(optionally `fewshot_as_multiturn`).
|
| 50 |
+
2026-07-17:11:04:03 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 51 |
+
2026-07-17:11:04:03 INFO [evaluator:239] Initializing local-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'allenai/OLMoE-1B-7B-0125-Instruct', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}
|
| 52 |
+
2026-07-17:11:04:03 INFO [models.openai_completions:42] Remote tokenizer not supported. Using huggingface tokenizer backend.
|
| 53 |
+
2026-07-17:11:04:03 INFO [models.api_models:179] Using max length 2048 - 1
|
| 54 |
+
2026-07-17:11:04:03 INFO [models.api_models:200] Using tokenizer huggingface
|
| 55 |
+
2026-07-17:11:04:09 INFO [evaluator_utils:446] Selected tasks:
|
| 56 |
+
2026-07-17:11:04:09 INFO [evaluator_utils:480] Task: humaneval (humaneval/humaneval.yaml)
|
| 57 |
+
2026-07-17:11:04:09 INFO [evaluator_utils:480] Task: mbpp (mbpp/mbpp.yaml)
|
| 58 |
+
2026-07-17:11:04:09 INFO [evaluator:314] humaneval: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 59 |
+
2026-07-17:11:04:09 INFO [evaluator:314] mbpp: Using gen_kwargs: {'until': ['[DONE]'], 'do_sample': False}
|
| 60 |
+
2026-07-17:11:04:09 INFO [api.task:312] Building contexts for humaneval on rank 0...
|
| 61 |
+
|
| 62 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 63 |
+
2026-07-17:11:04:09 INFO [api.task:312] Building contexts for mbpp on rank 0...
|
| 64 |
+
|
| 65 |
0%| | 0/10 [00:00<?, ?it/s]
|
| 66 |
+
2026-07-17:11:04:09 INFO [evaluator:585] Running generate_until requests
|
| 67 |
+
2026-07-17:11:04:09 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
2026-07-17:11:04:16 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 71 |
+
2026-07-17:11:04:16 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/smoke_full2/student/*.jsonl
|
| 72 |
+
local-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/completions', 'tokenizer': 'allenai/OLMoE-1B-7B-0125-Instruct', 'num_concurrent': 16, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 10.0, num_fewshot: None, batch_size: 1
|
| 73 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 74 |
+
|---------|------:|-----------|-----:|---------|---|----:|---|-----:|
|
| 75 |
+
|humaneval| 1|create_test| 0|pass@1 |β | 0.5|Β± |0.1667|
|
| 76 |
+
|mbpp | 1|none | 3|pass_at_1|β | 0.3|Β± |0.1528|
|
| 77 |
+
|
| 78 |
+
2026-07-17T11:04:18-07:00 lm_eval exit=0 -> outputs/evals/general_suite/smoke_full2
|
evals/general_suite/smoke_risky.log
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/5 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 1 |
0%| | 0/5 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 2 |
0%| | 0/5 [00:00<?, ?it/s]
|
|
|
|
|
|
|
| 3 |
0%| | 0/5 [00:00<?, ?it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-07-17T10:15:58-07:00 serving allenai/OLMoE-1B-7B-0125-Instruct on GPU 0 port 8399
|
| 2 |
+
2026-07-17T10:15:58-07:00 waiting for server /health ...
|
| 3 |
+
2026-07-17T10:16:28-07:00 server up; running lm_eval [minerva_math500,ifeval,humaneval_instruct,mbpp_instruct]
|
| 4 |
+
2026-07-17:10:16:29 WARNING [config.evaluate_config:287] --limit SHOULD ONLY BE USED FOR TESTING. REAL METRICS SHOULD NOT BE COMPUTED USING LIMIT.
|
| 5 |
+
2026-07-17:10:16:36 INFO [_cli.run:388] Selected Tasks: ['minerva_math500', 'ifeval', 'humaneval_instruct', 'mbpp_instruct']
|
| 6 |
+
2026-07-17:10:16:37 INFO [evaluator:214] Setting random seed to 0 | Setting numpy seed to 1234 | Setting torch manual seed to 1234 | Setting fewshot manual seed to 1234
|
| 7 |
+
2026-07-17:10:16:37 INFO [evaluator:239] Initializing local-chat-completions model, with arguments: {'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 8, 'tokenized_requests': False, 'max_retries': 3}
|
| 8 |
+
2026-07-17:10:16:37 INFO [models.api_models:179] Using max length 2048 - 1
|
| 9 |
+
2026-07-17:10:16:37 INFO [models.api_models:200] Using tokenizer None
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
2026-07-17:10:16:50 INFO [evaluator_utils:446] Selected tasks:
|
| 19 |
+
2026-07-17:10:16:50 INFO [evaluator_utils:480] Task: humaneval_instruct (humaneval/humaneval_instruct.yaml)
|
| 20 |
+
2026-07-17:10:16:50 INFO [evaluator_utils:480] Task: ifeval (ifeval/ifeval.yaml)
|
| 21 |
+
2026-07-17:10:16:50 INFO [evaluator_utils:480] Task: mbpp_instruct (mbpp/mbpp_instruct.yaml)
|
| 22 |
+
2026-07-17:10:16:50 INFO [evaluator_utils:480] Task: minerva_math500 (minerva_math/minerva_math500.yaml)
|
| 23 |
+
2026-07-17:10:16:50 INFO [evaluator:314] minerva_math500: Using gen_kwargs: {'until': ['Problem:'], 'do_sample': False, 'temperature': 0.0}
|
| 24 |
+
2026-07-17:10:16:50 INFO [evaluator:314] ifeval: Using gen_kwargs: {'until': [], 'do_sample': False, 'temperature': 0.0, 'max_gen_toks': 1280}
|
| 25 |
+
2026-07-17:10:16:50 INFO [evaluator:314] humaneval_instruct: Using gen_kwargs: {'until': ['\nclass', '\ndef', '\n#', '\nif', '\nprint'], 'max_gen_toks': 1024, 'do_sample': False}
|
| 26 |
+
2026-07-17:10:16:50 INFO [evaluator:314] mbpp_instruct: Using gen_kwargs: {'max_gen_toks': 256, 'until': [], 'do_sample': False}
|
| 27 |
+
2026-07-17:10:16:50 INFO [api.task:312] Building contexts for minerva_math500 on rank 0...
|
| 28 |
+
|
| 29 |
0%| | 0/5 [00:00<?, ?it/s]
|
| 30 |
+
2026-07-17:10:16:50 INFO [api.task:312] Building contexts for ifeval on rank 0...
|
| 31 |
+
|
| 32 |
0%| | 0/5 [00:00<?, ?it/s]
|
| 33 |
+
2026-07-17:10:16:50 INFO [api.task:312] Building contexts for humaneval_instruct on rank 0...
|
| 34 |
+
|
| 35 |
0%| | 0/5 [00:00<?, ?it/s]
|
| 36 |
+
2026-07-17:10:16:50 INFO [api.task:312] Building contexts for mbpp_instruct on rank 0...
|
| 37 |
+
|
| 38 |
0%| | 0/5 [00:00<?, ?it/s]
|
| 39 |
+
2026-07-17:10:16:50 INFO [evaluator:585] Running generate_until requests
|
| 40 |
+
2026-07-17:10:16:50 INFO [models.api_models:747] Tokenized requests are disabled. Context + generation length is not checked.
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
2026-07-17:10:17:18 INFO [loggers.evaluation_tracker:247] Saving results aggregated
|
| 46 |
+
2026-07-17:10:17:18 INFO [loggers.evaluation_tracker:119] Saving per-task samples to outputs/evals/general_suite/smoke_risky/student/*.jsonl
|
| 47 |
+
local-chat-completions ({'model': 'student', 'base_url': 'http://127.0.0.1:8399/v1/chat/completions', 'num_concurrent': 8, 'tokenized_requests': False, 'max_retries': 3}), gen_kwargs: ({}), limit: 5.0, num_fewshot: None, batch_size: 1
|
| 48 |
+
| Tasks |Version| Filter |n-shot| Metric | |Value| |Stderr|
|
| 49 |
+
|------------------|------:|------------|-----:|-----------------------|---|----:|---|------|
|
| 50 |
+
|humaneval_instruct| 4|create_test | 0|pass@1 |β |0.000|Β± | 0|
|
| 51 |
+
|ifeval | 4|none | 0|inst_level_loose_acc |β |0.875|Β± | N/A|
|
| 52 |
+
| | |none | 0|inst_level_strict_acc |β |0.875|Β± | N/A|
|
| 53 |
+
| | |none | 0|prompt_level_loose_acc |β |0.800|Β± |0.2000|
|
| 54 |
+
| | |none | 0|prompt_level_strict_acc|β |0.800|Β± |0.2000|
|
| 55 |
+
|mbpp_instruct | 1|extract_code| 3|pass_at_1 |β |0.000|Β± |0.0000|
|
| 56 |
+
|minerva_math500 | 3|none | 4|exact_match |β |0.000|Β± |0.0000|
|
| 57 |
+
| | |none | 4|math_verify |β |0.200|Β± |0.2000|
|
| 58 |
+
|
| 59 |
+
2026-07-17T10:17:20-07:00 lm_eval exit=0 -> outputs/evals/general_suite/smoke_risky
|
evals/grid_math/glean_keep25_s1224_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1224_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1224/step0100
|
| 5 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1224/step0100', 'host': '127.0.0.1', 'port': 8380, 'model': 'outputs/healed/grid_math/glean_keep25_s1224/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=389651) WARNING 07-16 05:38:25 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=389651) WARNING 07-16 05:38:25 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=389651) INFO 07-16 05:38:25 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=389770) INFO 07-16 05:38:32 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1224/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1224/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=389770) INFO 07-16 05:38:33 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:43393 backend=nccl
|
| 18 |
+
(EngineCore pid=389770) INFO 07-16 05:38:33 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1224/step0100...
|
| 21 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=389770) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=389770) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 100.09 GiB.
|
| 26 |
+
(EngineCore pid=389770) INFO 07-16 05:38:34 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=389770)
|
| 28 |
+
(EngineCore pid=389770)
|
| 29 |
+
(EngineCore pid=389770)
|
| 30 |
+
(EngineCore pid=389770)
|
| 31 |
+
(EngineCore pid=389770) INFO 07-16 05:38:37 [default_loader.py:430] Loading weights took 2.44 seconds
|
| 32 |
+
(EngineCore pid=389770) INFO 07-16 05:38:37 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.624341 seconds
|
| 33 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=389770) INFO 07-16 05:38:39 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.08 s
|
| 39 |
+
(EngineCore pid=389770) INFO 07-16 05:38:40 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=389770) WARNING 07-16 05:38:40 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=389770) WARNING 07-16 05:38:40 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=389770) INFO 07-16 05:38:40 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=389770) INFO 07-16 05:38:40 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=389770) INFO 07-16 05:38:40 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=389651) WARNING 07-16 05:38:40 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8380
|
| 49 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 51 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=389651) INFO 07-16 05:38:40 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=389651) INFO: Started server process [389651]
|
| 97 |
+
(APIServer pid=389651) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=389651) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=389651) INFO: 127.0.0.1:47006 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=389770) WARNING 07-16 05:38:42 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=389651) INFO 07-16 05:38:50 [loggers.py:273] Engine 000: Avg prompt throughput: 2211.1 tokens/s, Avg generation throughput: 2168.6 tokens/s, Running: 254 reqs, Waiting: 1035 reqs, GPU KV cache usage: 34.2%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=389651) INFO 07-16 05:39:00 [loggers.py:273] Engine 000: Avg prompt throughput: 1142.0 tokens/s, Avg generation throughput: 2929.7 tokens/s, Running: 256 reqs, Waiting: 893 reqs, GPU KV cache usage: 42.3%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=389651) INFO 07-16 05:39:10 [loggers.py:273] Engine 000: Avg prompt throughput: 1121.6 tokens/s, Avg generation throughput: 2928.7 tokens/s, Running: 254 reqs, Waiting: 745 reqs, GPU KV cache usage: 42.5%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=389651) INFO 07-16 05:39:20 [loggers.py:273] Engine 000: Avg prompt throughput: 1175.0 tokens/s, Avg generation throughput: 2903.1 tokens/s, Running: 256 reqs, Waiting: 593 reqs, GPU KV cache usage: 42.0%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=389651) INFO 07-16 05:39:30 [loggers.py:273] Engine 000: Avg prompt throughput: 1206.1 tokens/s, Avg generation throughput: 2901.9 tokens/s, Running: 256 reqs, Waiting: 442 reqs, GPU KV cache usage: 41.2%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=389651) INFO 07-16 05:39:40 [loggers.py:273] Engine 000: Avg prompt throughput: 1229.7 tokens/s, Avg generation throughput: 2928.1 tokens/s, Running: 256 reqs, Waiting: 293 reqs, GPU KV cache usage: 42.7%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=389651) INFO 07-16 05:39:50 [loggers.py:273] Engine 000: Avg prompt throughput: 1027.4 tokens/s, Avg generation throughput: 2930.4 tokens/s, Running: 253 reqs, Waiting: 159 reqs, GPU KV cache usage: 43.8%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=389651) INFO 07-16 05:40:00 [loggers.py:273] Engine 000: Avg prompt throughput: 1026.3 tokens/s, Avg generation throughput: 2906.0 tokens/s, Running: 253 reqs, Waiting: 38 reqs, GPU KV cache usage: 47.6%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=389651) INFO 07-16 05:40:10 [loggers.py:273] Engine 000: Avg prompt throughput: 299.8 tokens/s, Avg generation throughput: 2748.2 tokens/s, Running: 123 reqs, Waiting: 0 reqs, GPU KV cache usage: 30.3%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=389651) INFO 07-16 05:40:20 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1154.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 93.4%
|
| 111 |
+
(APIServer pid=389651) INFO: 127.0.0.1:47018 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 112 |
+
(EngineCore pid=389770) INFO 07-16 05:40:23 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 113 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 114 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=389770) INFO 07-16 05:40:23 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 116 |
+
(EngineCore pid=389770) INFO 07-16 05:40:23 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 117 |
+
(EngineCore pid=389770) INFO 07-16 05:40:23 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 118 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 119 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 120 |
+
(APIServer pid=389651) WARNING 07-16 05:40:23 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 121 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 122 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 123 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [core_client.py:662] [shutdown] MPClient: complete
|
| 124 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 125 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 126 |
+
(APIServer pid=389651) INFO 07-16 05:40:23 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 127 |
+
(APIServer pid=389651) INFO: Shutting down
|
| 128 |
+
(APIServer pid=389651) INFO: Waiting for application shutdown.
|
| 129 |
+
(APIServer pid=389651) INFO: Application shutdown complete.
|
| 130 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 131 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep25_s1224_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1224_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1224/step0150
|
| 5 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1224/step0150', 'host': '127.0.0.1', 'port': 8380, 'model': 'outputs/healed/grid_math/glean_keep25_s1224/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=390462) WARNING 07-16 05:40:48 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=390462) WARNING 07-16 05:40:48 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=390462) INFO 07-16 05:40:48 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=390583) INFO 07-16 05:40:55 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1224/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1224/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=390583) INFO 07-16 05:40:56 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:33389 backend=nccl
|
| 18 |
+
(EngineCore pid=390583) INFO 07-16 05:40:56 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=390583) INFO 07-16 05:40:56 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=390583) INFO 07-16 05:40:57 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1224/step0150...
|
| 21 |
+
(EngineCore pid=390583) INFO 07-16 05:40:57 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=390583) INFO 07-16 05:40:57 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=390583) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=390583) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=390583) INFO 07-16 05:40:57 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 98.33 GiB.
|
| 26 |
+
(EngineCore pid=390583) INFO 07-16 05:40:57 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=390583)
|
| 28 |
+
(EngineCore pid=390583)
|
| 29 |
+
(EngineCore pid=390583)
|
| 30 |
+
(EngineCore pid=390583)
|
| 31 |
+
(EngineCore pid=390583) INFO 07-16 05:40:59 [default_loader.py:430] Loading weights took 2.36 seconds
|
| 32 |
+
(EngineCore pid=390583) INFO 07-16 05:41:00 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.538811 seconds
|
| 33 |
+
(EngineCore pid=390583) INFO 07-16 05:41:01 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=390583) INFO 07-16 05:41:01 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=390583) INFO 07-16 05:41:01 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.10 s
|
| 39 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=390583) WARNING 07-16 05:41:02 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=390583) WARNING 07-16 05:41:02 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=390583) INFO 07-16 05:41:02 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=390462) WARNING 07-16 05:41:02 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8380
|
| 49 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=390462) INFO 07-16 05:41:02 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=390462) INFO: Started server process [390462]
|
| 97 |
+
(APIServer pid=390462) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=390462) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=390462) INFO: 127.0.0.1:60960 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=390583) WARNING 07-16 05:41:04 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=390462) INFO 07-16 05:41:13 [loggers.py:273] Engine 000: Avg prompt throughput: 2264.9 tokens/s, Avg generation throughput: 2170.9 tokens/s, Running: 255 reqs, Waiting: 1027 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=390462) INFO 07-16 05:41:23 [loggers.py:273] Engine 000: Avg prompt throughput: 1396.3 tokens/s, Avg generation throughput: 2925.3 tokens/s, Running: 256 reqs, Waiting: 852 reqs, GPU KV cache usage: 40.3%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=390462) INFO 07-16 05:41:33 [loggers.py:273] Engine 000: Avg prompt throughput: 1331.2 tokens/s, Avg generation throughput: 2900.3 tokens/s, Running: 254 reqs, Waiting: 678 reqs, GPU KV cache usage: 39.7%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=390462) INFO 07-16 05:41:43 [loggers.py:273] Engine 000: Avg prompt throughput: 1292.1 tokens/s, Avg generation throughput: 2900.7 tokens/s, Running: 256 reqs, Waiting: 514 reqs, GPU KV cache usage: 41.3%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=390462) INFO 07-16 05:41:53 [loggers.py:273] Engine 000: Avg prompt throughput: 1459.5 tokens/s, Avg generation throughput: 2899.8 tokens/s, Running: 253 reqs, Waiting: 330 reqs, GPU KV cache usage: 38.4%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=390462) INFO 07-16 05:42:03 [loggers.py:273] Engine 000: Avg prompt throughput: 1318.1 tokens/s, Avg generation throughput: 2902.3 tokens/s, Running: 255 reqs, Waiting: 169 reqs, GPU KV cache usage: 41.7%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=390462) INFO 07-16 05:42:13 [loggers.py:273] Engine 000: Avg prompt throughput: 1191.5 tokens/s, Avg generation throughput: 2852.4 tokens/s, Running: 254 reqs, Waiting: 22 reqs, GPU KV cache usage: 43.4%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=390462) INFO 07-16 05:42:23 [loggers.py:273] Engine 000: Avg prompt throughput: 183.1 tokens/s, Avg generation throughput: 2646.4 tokens/s, Running: 96 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.3%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=390462) INFO 07-16 05:42:33 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 910.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=390462) INFO: 127.0.0.1:60972 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 111 |
+
(EngineCore pid=390583) INFO 07-16 05:42:34 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=390583) INFO 07-16 05:42:34 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=390583) INFO 07-16 05:42:34 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=390583) INFO 07-16 05:42:34 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=390462) WARNING 07-16 05:42:34 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=390462) INFO 07-16 05:42:34 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=390462) INFO 07-16 05:42:35 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=390462) INFO: Shutting down
|
| 127 |
+
(APIServer pid=390462) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=390462) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep25_s1225_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1225_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1225/step0100
|
| 5 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1225/step0100', 'host': '127.0.0.1', 'port': 8381, 'model': 'outputs/healed/grid_math/glean_keep25_s1225/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=385267) WARNING 07-16 05:13:18 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=385267) WARNING 07-16 05:13:18 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=385267) INFO 07-16 05:13:18 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=385384) INFO 07-16 05:13:25 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1225/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1225/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=385384) INFO 07-16 05:13:25 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:36119 backend=nccl
|
| 18 |
+
(EngineCore pid=385384) INFO 07-16 05:13:25 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=385384) INFO 07-16 05:13:26 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=385384) INFO 07-16 05:13:26 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1225/step0100...
|
| 21 |
+
(EngineCore pid=385384) INFO 07-16 05:13:26 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=385384) INFO 07-16 05:13:26 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=385384) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=385384) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=385384) INFO 07-16 05:13:27 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 96.03 GiB.
|
| 26 |
+
(EngineCore pid=385384) INFO 07-16 05:13:27 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=385384)
|
| 28 |
+
(EngineCore pid=385384)
|
| 29 |
+
(EngineCore pid=385384)
|
| 30 |
+
(EngineCore pid=385384)
|
| 31 |
+
(EngineCore pid=385384) INFO 07-16 05:13:29 [default_loader.py:430] Loading weights took 2.46 seconds
|
| 32 |
+
(EngineCore pid=385384) INFO 07-16 05:13:30 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.636416 seconds
|
| 33 |
+
(EngineCore pid=385384) INFO 07-16 05:13:31 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=385384) INFO 07-16 05:13:31 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=385384) INFO 07-16 05:13:31 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=385384) INFO 07-16 05:13:31 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=385384) INFO 07-16 05:13:31 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=385384) INFO 07-16 05:13:32 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.04 s
|
| 39 |
+
(EngineCore pid=385384) INFO 07-16 05:13:32 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=385384) WARNING 07-16 05:13:32 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=385384) WARNING 07-16 05:13:32 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=385384) INFO 07-16 05:13:32 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=385384) INFO 07-16 05:13:32 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=385384) INFO 07-16 05:13:32 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=385267) WARNING 07-16 05:13:32 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8381
|
| 49 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=385267) INFO 07-16 05:13:32 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=385267) INFO: Started server process [385267]
|
| 97 |
+
(APIServer pid=385267) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=385267) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=385267) INFO: 127.0.0.1:46926 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=385384) WARNING 07-16 05:13:34 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=385267) INFO 07-16 05:13:42 [loggers.py:273] Engine 000: Avg prompt throughput: 2283.8 tokens/s, Avg generation throughput: 2069.9 tokens/s, Running: 255 reqs, Waiting: 1025 reqs, GPU KV cache usage: 32.9%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=385267) INFO 07-16 05:13:52 [loggers.py:273] Engine 000: Avg prompt throughput: 1403.8 tokens/s, Avg generation throughput: 2926.3 tokens/s, Running: 256 reqs, Waiting: 849 reqs, GPU KV cache usage: 39.4%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=385267) INFO 07-16 05:14:02 [loggers.py:273] Engine 000: Avg prompt throughput: 1316.8 tokens/s, Avg generation throughput: 2925.8 tokens/s, Running: 256 reqs, Waiting: 675 reqs, GPU KV cache usage: 40.7%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=385267) INFO 07-16 05:14:12 [loggers.py:273] Engine 000: Avg prompt throughput: 1317.4 tokens/s, Avg generation throughput: 2901.8 tokens/s, Running: 255 reqs, Waiting: 511 reqs, GPU KV cache usage: 43.3%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=385267) INFO 07-16 05:14:22 [loggers.py:273] Engine 000: Avg prompt throughput: 1500.9 tokens/s, Avg generation throughput: 2872.7 tokens/s, Running: 256 reqs, Waiting: 321 reqs, GPU KV cache usage: 38.3%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=385267) INFO 07-16 05:14:32 [loggers.py:273] Engine 000: Avg prompt throughput: 1331.7 tokens/s, Avg generation throughput: 2927.1 tokens/s, Running: 254 reqs, Waiting: 154 reqs, GPU KV cache usage: 40.1%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=385267) INFO 07-16 05:14:42 [loggers.py:273] Engine 000: Avg prompt throughput: 1279.2 tokens/s, Avg generation throughput: 2901.2 tokens/s, Running: 245 reqs, Waiting: 0 reqs, GPU KV cache usage: 41.4%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=385267) INFO 07-16 05:14:52 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2498.1 tokens/s, Running: 70 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.0%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=385267) INFO: 127.0.0.1:46930 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(APIServer pid=385267) INFO 07-16 05:15:02 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 748.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 93.4%
|
| 111 |
+
(EngineCore pid=385384) INFO 07-16 05:15:03 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=385384) INFO 07-16 05:15:03 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=385384) INFO 07-16 05:15:03 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=385384) INFO 07-16 05:15:03 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=385267) WARNING 07-16 05:15:03 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=385267) INFO 07-16 05:15:03 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=385267) INFO: Shutting down
|
| 127 |
+
(APIServer pid=385267) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=385267) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep25_s1225_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1225_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1225/step0150
|
| 5 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1225/step0150', 'host': '127.0.0.1', 'port': 8381, 'model': 'outputs/healed/grid_math/glean_keep25_s1225/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=386637) WARNING 07-16 05:15:27 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=386637) WARNING 07-16 05:15:27 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=386637) INFO 07-16 05:15:27 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=386752) INFO 07-16 05:15:34 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1225/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1225/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=386752) INFO 07-16 05:15:35 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:47883 backend=nccl
|
| 18 |
+
(EngineCore pid=386752) INFO 07-16 05:15:35 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1225/step0150...
|
| 21 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=386752) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=386752) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 95.94 GiB.
|
| 26 |
+
(EngineCore pid=386752) INFO 07-16 05:15:36 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=386752)
|
| 28 |
+
(EngineCore pid=386752)
|
| 29 |
+
(EngineCore pid=386752)
|
| 30 |
+
(EngineCore pid=386752)
|
| 31 |
+
(EngineCore pid=386752) INFO 07-16 05:15:39 [default_loader.py:430] Loading weights took 2.48 seconds
|
| 32 |
+
(EngineCore pid=386752) INFO 07-16 05:15:39 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.655688 seconds
|
| 33 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.03 s
|
| 39 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=386752) WARNING 07-16 05:15:41 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=386752) WARNING 07-16 05:15:41 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=386752) INFO 07-16 05:15:41 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=386637) INFO 07-16 05:15:41 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=386637) WARNING 07-16 05:15:41 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8381
|
| 49 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=386637) INFO 07-16 05:15:42 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=386637) INFO: Started server process [386637]
|
| 97 |
+
(APIServer pid=386637) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=386637) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=386637) INFO: 127.0.0.1:48752 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=386752) WARNING 07-16 05:15:44 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=386637) INFO 07-16 05:15:52 [loggers.py:273] Engine 000: Avg prompt throughput: 2292.7 tokens/s, Avg generation throughput: 2093.8 tokens/s, Running: 254 reqs, Waiting: 1024 reqs, GPU KV cache usage: 32.9%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=386637) INFO 07-16 05:16:02 [loggers.py:273] Engine 000: Avg prompt throughput: 1373.8 tokens/s, Avg generation throughput: 2926.9 tokens/s, Running: 255 reqs, Waiting: 852 reqs, GPU KV cache usage: 39.9%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=386637) INFO 07-16 05:16:12 [loggers.py:273] Engine 000: Avg prompt throughput: 1359.8 tokens/s, Avg generation throughput: 2900.2 tokens/s, Running: 255 reqs, Waiting: 673 reqs, GPU KV cache usage: 39.7%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=386637) INFO 07-16 05:16:22 [loggers.py:273] Engine 000: Avg prompt throughput: 1312.2 tokens/s, Avg generation throughput: 2901.0 tokens/s, Running: 254 reqs, Waiting: 504 reqs, GPU KV cache usage: 40.8%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=386637) INFO 07-16 05:16:32 [loggers.py:273] Engine 000: Avg prompt throughput: 1472.3 tokens/s, Avg generation throughput: 2900.3 tokens/s, Running: 256 reqs, Waiting: 324 reqs, GPU KV cache usage: 39.1%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=386637) INFO 07-16 05:16:42 [loggers.py:273] Engine 000: Avg prompt throughput: 1366.4 tokens/s, Avg generation throughput: 2926.5 tokens/s, Running: 251 reqs, Waiting: 152 reqs, GPU KV cache usage: 38.8%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=386637) INFO 07-16 05:16:52 [loggers.py:273] Engine 000: Avg prompt throughput: 1249.6 tokens/s, Avg generation throughput: 2928.3 tokens/s, Running: 251 reqs, Waiting: 0 reqs, GPU KV cache usage: 42.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=386637) INFO 07-16 05:17:02 [loggers.py:273] Engine 000: Avg prompt throughput: 6.7 tokens/s, Avg generation throughput: 2526.9 tokens/s, Running: 72 reqs, Waiting: 0 reqs, GPU KV cache usage: 22.0%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=386637) INFO: 127.0.0.1:48756 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 612.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 93.4%
|
| 111 |
+
(EngineCore pid=386752) INFO 07-16 05:17:12 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=386752) INFO 07-16 05:17:12 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=386752) INFO 07-16 05:17:12 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=386752) INFO 07-16 05:17:12 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=386637) WARNING 07-16 05:17:12 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=386637) INFO 07-16 05:17:12 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=386637) INFO: Shutting down
|
| 127 |
+
(APIServer pid=386637) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=386637) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep25_s1226_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1226_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1226/step0100
|
| 5 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1226/step0100', 'host': '127.0.0.1', 'port': 8382, 'model': 'outputs/healed/grid_math/glean_keep25_s1226/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=384507) WARNING 07-16 05:11:54 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=384507) WARNING 07-16 05:11:54 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=384507) INFO 07-16 05:11:54 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=384624) INFO 07-16 05:12:01 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1226/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1226/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=384624) INFO 07-16 05:12:01 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:56865 backend=nccl
|
| 18 |
+
(EngineCore pid=384624) INFO 07-16 05:12:01 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=384624) INFO 07-16 05:12:02 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=384624) INFO 07-16 05:12:02 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1226/step0100...
|
| 21 |
+
(EngineCore pid=384624) INFO 07-16 05:12:03 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=384624) INFO 07-16 05:12:03 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=384624) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=384624) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=384624) INFO 07-16 05:12:03 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 88.83 GiB.
|
| 26 |
+
(EngineCore pid=384624) INFO 07-16 05:12:03 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=384624)
|
| 28 |
+
(EngineCore pid=384624)
|
| 29 |
+
(EngineCore pid=384624)
|
| 30 |
+
(EngineCore pid=384624)
|
| 31 |
+
(EngineCore pid=384624) INFO 07-16 05:12:05 [default_loader.py:430] Loading weights took 2.57 seconds
|
| 32 |
+
(EngineCore pid=384624) INFO 07-16 05:12:06 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.750867 seconds
|
| 33 |
+
(EngineCore pid=384624) INFO 07-16 05:12:07 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=384624) INFO 07-16 05:12:07 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=384624) INFO 07-16 05:12:07 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.06 s
|
| 39 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=384624) WARNING 07-16 05:12:08 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=384624) WARNING 07-16 05:12:08 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=384624) INFO 07-16 05:12:08 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=384507) WARNING 07-16 05:12:08 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8382
|
| 49 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=384507) INFO 07-16 05:12:08 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=384507) INFO: Started server process [384507]
|
| 97 |
+
(APIServer pid=384507) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=384507) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=384507) INFO: 127.0.0.1:35974 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=384624) WARNING 07-16 05:12:10 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=384507) INFO 07-16 05:12:19 [loggers.py:273] Engine 000: Avg prompt throughput: 2227.6 tokens/s, Avg generation throughput: 2126.0 tokens/s, Running: 255 reqs, Waiting: 1034 reqs, GPU KV cache usage: 33.9%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=384507) INFO 07-16 05:12:29 [loggers.py:273] Engine 000: Avg prompt throughput: 1232.7 tokens/s, Avg generation throughput: 2927.6 tokens/s, Running: 256 reqs, Waiting: 879 reqs, GPU KV cache usage: 41.2%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=384507) INFO 07-16 05:12:39 [loggers.py:273] Engine 000: Avg prompt throughput: 1046.2 tokens/s, Avg generation throughput: 2904.4 tokens/s, Running: 255 reqs, Waiting: 740 reqs, GPU KV cache usage: 44.0%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=384507) INFO 07-16 05:12:49 [loggers.py:273] Engine 000: Avg prompt throughput: 1169.3 tokens/s, Avg generation throughput: 2902.9 tokens/s, Running: 255 reqs, Waiting: 590 reqs, GPU KV cache usage: 44.5%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=384507) INFO 07-16 05:12:59 [loggers.py:273] Engine 000: Avg prompt throughput: 1340.2 tokens/s, Avg generation throughput: 2850.2 tokens/s, Running: 253 reqs, Waiting: 423 reqs, GPU KV cache usage: 39.5%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=384507) INFO 07-16 05:13:09 [loggers.py:273] Engine 000: Avg prompt throughput: 1325.6 tokens/s, Avg generation throughput: 2902.1 tokens/s, Running: 254 reqs, Waiting: 262 reqs, GPU KV cache usage: 40.3%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=384507) INFO 07-16 05:13:19 [loggers.py:273] Engine 000: Avg prompt throughput: 1235.5 tokens/s, Avg generation throughput: 2928.0 tokens/s, Running: 254 reqs, Waiting: 103 reqs, GPU KV cache usage: 39.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=384507) INFO 07-16 05:13:29 [loggers.py:273] Engine 000: Avg prompt throughput: 867.3 tokens/s, Avg generation throughput: 2947.3 tokens/s, Running: 224 reqs, Waiting: 0 reqs, GPU KV cache usage: 42.8%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=384507) INFO 07-16 05:13:39 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2304.9 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.4%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=384507) INFO: 127.0.0.1:35988 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 111 |
+
(EngineCore pid=384624) INFO 07-16 05:13:47 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=384624) INFO 07-16 05:13:47 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=384624) INFO 07-16 05:13:47 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=384624) INFO 07-16 05:13:47 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=384507) WARNING 07-16 05:13:47 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=384507) INFO 07-16 05:13:47 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=384507) INFO: Shutting down
|
| 127 |
+
(APIServer pid=384507) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=384507) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep25_s1226_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep25_s1226_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep25_s1226/step0150
|
| 5 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep25_s1226/step0150', 'host': '127.0.0.1', 'port': 8382, 'model': 'outputs/healed/grid_math/glean_keep25_s1226/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=385941) WARNING 07-16 05:14:11 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=385941) WARNING 07-16 05:14:11 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=385941) INFO 07-16 05:14:11 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=386059) INFO 07-16 05:14:18 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep25_s1226/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep25_s1226/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=386059) INFO 07-16 05:14:19 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:46003 backend=nccl
|
| 18 |
+
(EngineCore pid=386059) INFO 07-16 05:14:19 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep25_s1226/step0150...
|
| 21 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=386059) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=386059) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 3.89 GiB. Available RAM: 95.97 GiB.
|
| 26 |
+
(EngineCore pid=386059) INFO 07-16 05:14:20 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=386059)
|
| 28 |
+
(EngineCore pid=386059)
|
| 29 |
+
(EngineCore pid=386059)
|
| 30 |
+
(EngineCore pid=386059)
|
| 31 |
+
(EngineCore pid=386059) INFO 07-16 05:14:23 [default_loader.py:430] Loading weights took 2.34 seconds
|
| 32 |
+
(EngineCore pid=386059) INFO 07-16 05:14:23 [gpu_model_runner.py:5306] Model loading took 3.89 GiB memory and 2.521557 seconds
|
| 33 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [gpu_worker.py:538] Available KV cache memory: 15.82 GiB
|
| 34 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [kv_cache_utils.py:2146] GPU KV cache size: 129,584 tokens
|
| 35 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 63.27x
|
| 36 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 37 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 38 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.03 s
|
| 39 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 40 |
+
(EngineCore pid=386059) WARNING 07-16 05:14:25 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 41 |
+
(EngineCore pid=386059) WARNING 07-16 05:14:25 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 42 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 43 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 44 |
+
(EngineCore pid=386059) INFO 07-16 05:14:25 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 45 |
+
(APIServer pid=385941) INFO 07-16 05:14:25 [api_server.py:612] Supported tasks: ['generate']
|
| 46 |
+
(APIServer pid=385941) WARNING 07-16 05:14:26 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 47 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 48 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8382
|
| 49 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:37] Available routes are:
|
| 50 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 51 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /load, Methods: GET
|
| 55 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /version, Methods: GET
|
| 56 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /health, Methods: GET
|
| 57 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /metrics, Methods: GET
|
| 58 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 59 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 60 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 61 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /ping, Methods: GET
|
| 62 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /ping, Methods: POST
|
| 63 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /invocations, Methods: POST
|
| 64 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 65 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 66 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 67 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /pause, Methods: POST
|
| 68 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /resume, Methods: POST
|
| 69 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 70 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 71 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 72 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 73 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 74 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 75 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 76 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /server_info, Methods: GET
|
| 77 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /sleep, Methods: POST
|
| 78 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 79 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 80 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 81 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 82 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 83 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 84 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 85 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 86 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 87 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 88 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 89 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 90 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 92 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 94 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=385941) INFO 07-16 05:14:26 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 96 |
+
(APIServer pid=385941) INFO: Started server process [385941]
|
| 97 |
+
(APIServer pid=385941) INFO: Waiting for application startup.
|
| 98 |
+
(APIServer pid=385941) INFO: Application startup complete.
|
| 99 |
+
(APIServer pid=385941) INFO: 127.0.0.1:41948 - "GET /health HTTP/1.1" 200 OK
|
| 100 |
+
(EngineCore pid=386059) WARNING 07-16 05:14:28 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 101 |
+
(APIServer pid=385941) INFO 07-16 05:14:36 [loggers.py:273] Engine 000: Avg prompt throughput: 2196.9 tokens/s, Avg generation throughput: 2075.4 tokens/s, Running: 254 reqs, Waiting: 1037 reqs, GPU KV cache usage: 33.4%, Prefix cache hit rate: 93.1%
|
| 102 |
+
(APIServer pid=385941) INFO 07-16 05:14:46 [loggers.py:273] Engine 000: Avg prompt throughput: 1213.0 tokens/s, Avg generation throughput: 2928.4 tokens/s, Running: 254 reqs, Waiting: 886 reqs, GPU KV cache usage: 41.5%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=385941) INFO 07-16 05:14:56 [loggers.py:273] Engine 000: Avg prompt throughput: 1207.8 tokens/s, Avg generation throughput: 2902.1 tokens/s, Running: 255 reqs, Waiting: 728 reqs, GPU KV cache usage: 41.3%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=385941) INFO 07-16 05:15:06 [loggers.py:273] Engine 000: Avg prompt throughput: 1186.6 tokens/s, Avg generation throughput: 2902.8 tokens/s, Running: 255 reqs, Waiting: 576 reqs, GPU KV cache usage: 42.9%, Prefix cache hit rate: 93.4%
|
| 105 |
+
(APIServer pid=385941) INFO 07-16 05:15:16 [loggers.py:273] Engine 000: Avg prompt throughput: 1331.4 tokens/s, Avg generation throughput: 2901.3 tokens/s, Running: 255 reqs, Waiting: 406 reqs, GPU KV cache usage: 39.8%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=385941) INFO 07-16 05:15:26 [loggers.py:273] Engine 000: Avg prompt throughput: 1200.6 tokens/s, Avg generation throughput: 2929.3 tokens/s, Running: 255 reqs, Waiting: 262 reqs, GPU KV cache usage: 43.3%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=385941) INFO 07-16 05:15:36 [loggers.py:273] Engine 000: Avg prompt throughput: 1162.1 tokens/s, Avg generation throughput: 2903.5 tokens/s, Running: 254 reqs, Waiting: 116 reqs, GPU KV cache usage: 43.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=385941) INFO 07-16 05:15:46 [loggers.py:273] Engine 000: Avg prompt throughput: 946.7 tokens/s, Avg generation throughput: 2913.1 tokens/s, Running: 235 reqs, Waiting: 0 reqs, GPU KV cache usage: 45.3%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=385941) INFO 07-16 05:15:56 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2414.4 tokens/s, Running: 73 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.7%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(APIServer pid=385941) INFO: 127.0.0.1:41960 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 111 |
+
(EngineCore pid=386059) INFO 07-16 05:16:06 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 113 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=386059) INFO 07-16 05:16:06 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 115 |
+
(EngineCore pid=386059) INFO 07-16 05:16:06 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 116 |
+
(EngineCore pid=386059) INFO 07-16 05:16:06 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 117 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 118 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 119 |
+
(APIServer pid=385941) WARNING 07-16 05:16:06 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 120 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=385941) INFO 07-16 05:16:06 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=385941) INFO: Shutting down
|
| 127 |
+
(APIServer pid=385941) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=385941) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1224_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1224_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1224/step0100
|
| 5 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1224/step0100', 'host': '127.0.0.1', 'port': 8380, 'model': 'outputs/healed/grid_math/glean_keep50_s1224/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=342175) WARNING 07-15 23:03:31 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=342175) WARNING 07-15 23:03:31 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=342175) INFO 07-15 23:03:31 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=342292) INFO 07-15 23:03:38 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1224/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1224/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=342292) INFO 07-15 23:03:39 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:59935 backend=nccl
|
| 18 |
+
(EngineCore pid=342292) INFO 07-15 23:03:39 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1224/step0100...
|
| 21 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=342292) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=342292) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 94.10 GiB.
|
| 26 |
+
(EngineCore pid=342292) INFO 07-15 23:03:40 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=342292)
|
| 28 |
+
(EngineCore pid=342292)
|
| 29 |
+
(EngineCore pid=342292)
|
| 30 |
+
(EngineCore pid=342292)
|
| 31 |
+
(EngineCore pid=342292)
|
| 32 |
+
(EngineCore pid=342292) INFO 07-15 23:03:52 [default_loader.py:430] Loading weights took 11.38 seconds
|
| 33 |
+
(EngineCore pid=342292) INFO 07-15 23:03:52 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 11.572880 seconds
|
| 34 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=342292) INFO 07-15 23:03:54 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.08 s
|
| 40 |
+
(EngineCore pid=342292) INFO 07-15 23:03:55 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=342292) WARNING 07-15 23:03:55 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=342292) WARNING 07-15 23:03:55 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=342292) INFO 07-15 23:03:55 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=342292) INFO 07-15 23:03:55 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=342292) INFO 07-15 23:03:55 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=342175) WARNING 07-15 23:03:55 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8380
|
| 50 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 55 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=342175) INFO 07-15 23:03:55 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=342175) INFO: Started server process [342175]
|
| 98 |
+
(APIServer pid=342175) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=342175) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=342175) INFO: 127.0.0.1:57932 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=342292) WARNING 07-15 23:03:58 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=342175) INFO 07-15 23:04:05 [loggers.py:273] Engine 000: Avg prompt throughput: 2357.1 tokens/s, Avg generation throughput: 1728.4 tokens/s, Running: 255 reqs, Waiting: 1013 reqs, GPU KV cache usage: 37.1%, Prefix cache hit rate: 93.1%
|
| 103 |
+
(APIServer pid=342175) INFO 07-15 23:04:15 [loggers.py:273] Engine 000: Avg prompt throughput: 1816.0 tokens/s, Avg generation throughput: 2689.7 tokens/s, Running: 254 reqs, Waiting: 780 reqs, GPU KV cache usage: 39.2%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=342175) INFO 07-15 23:04:25 [loggers.py:273] Engine 000: Avg prompt throughput: 1704.9 tokens/s, Avg generation throughput: 2691.2 tokens/s, Running: 255 reqs, Waiting: 562 reqs, GPU KV cache usage: 41.0%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=342175) INFO 07-15 23:04:35 [loggers.py:273] Engine 000: Avg prompt throughput: 1855.3 tokens/s, Avg generation throughput: 2664.6 tokens/s, Running: 253 reqs, Waiting: 332 reqs, GPU KV cache usage: 41.2%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=342175) INFO 07-15 23:04:45 [loggers.py:273] Engine 000: Avg prompt throughput: 1764.6 tokens/s, Avg generation throughput: 2691.4 tokens/s, Running: 254 reqs, Waiting: 113 reqs, GPU KV cache usage: 42.3%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=342175) INFO 07-15 23:04:55 [loggers.py:273] Engine 000: Avg prompt throughput: 929.5 tokens/s, Avg generation throughput: 2587.5 tokens/s, Running: 133 reqs, Waiting: 0 reqs, GPU KV cache usage: 29.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=342175) INFO 07-15 23:05:05 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 847.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=342175) INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(EngineCore pid=342292) INFO 07-15 23:05:07 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 111 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 112 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=342292) INFO 07-15 23:05:07 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=342292) INFO 07-15 23:05:07 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=342292) INFO 07-15 23:05:07 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=342175) WARNING 07-15 23:05:07 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=342175) INFO: Shutting down
|
| 120 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=342175) INFO 07-15 23:05:07 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=342175) INFO: Shutting down
|
| 127 |
+
(APIServer pid=342175) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=342175) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1224_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1224_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1224/step0150
|
| 5 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1224/step0150', 'host': '127.0.0.1', 'port': 8380, 'model': 'outputs/healed/grid_math/glean_keep50_s1224/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=342925) WARNING 07-15 23:05:32 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=342925) WARNING 07-15 23:05:32 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=342925) INFO 07-15 23:05:32 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=343040) INFO 07-15 23:05:39 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1224/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1224/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=343040) INFO 07-15 23:05:39 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:38977 backend=nccl
|
| 18 |
+
(EngineCore pid=343040) INFO 07-15 23:05:39 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=343040) INFO 07-15 23:05:40 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=343040) INFO 07-15 23:05:40 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1224/step0150...
|
| 21 |
+
(EngineCore pid=343040) INFO 07-15 23:05:41 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=343040) INFO 07-15 23:05:41 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=343040) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=343040) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=343040) INFO 07-15 23:05:41 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 93.74 GiB.
|
| 26 |
+
(EngineCore pid=343040) INFO 07-15 23:05:41 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=343040)
|
| 28 |
+
(EngineCore pid=343040)
|
| 29 |
+
(EngineCore pid=343040)
|
| 30 |
+
(EngineCore pid=343040)
|
| 31 |
+
(EngineCore pid=343040)
|
| 32 |
+
(EngineCore pid=343040) INFO 07-15 23:05:46 [default_loader.py:430] Loading weights took 5.05 seconds
|
| 33 |
+
(EngineCore pid=343040) INFO 07-15 23:05:46 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 5.238363 seconds
|
| 34 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=343040) INFO 07-15 23:05:48 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.05 s
|
| 40 |
+
(EngineCore pid=343040) INFO 07-15 23:05:49 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=343040) WARNING 07-15 23:05:49 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=343040) WARNING 07-15 23:05:49 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=343040) INFO 07-15 23:05:49 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=343040) INFO 07-15 23:05:49 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=343040) INFO 07-15 23:05:49 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=342925) WARNING 07-15 23:05:49 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8380
|
| 50 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 55 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=342925) INFO 07-15 23:05:49 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=342925) INFO: Started server process [342925]
|
| 98 |
+
(APIServer pid=342925) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=342925) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=342925) INFO: 127.0.0.1:44038 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=343040) WARNING 07-15 23:05:50 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=342925) INFO 07-15 23:05:59 [loggers.py:273] Engine 000: Avg prompt throughput: 2582.2 tokens/s, Avg generation throughput: 2015.9 tokens/s, Running: 256 reqs, Waiting: 983 reqs, GPU KV cache usage: 37.8%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=342925) INFO 07-15 23:06:09 [loggers.py:273] Engine 000: Avg prompt throughput: 1784.6 tokens/s, Avg generation throughput: 2690.2 tokens/s, Running: 254 reqs, Waiting: 753 reqs, GPU KV cache usage: 38.5%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=342925) INFO 07-15 23:06:19 [loggers.py:273] Engine 000: Avg prompt throughput: 1839.5 tokens/s, Avg generation throughput: 2689.8 tokens/s, Running: 254 reqs, Waiting: 520 reqs, GPU KV cache usage: 38.6%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=342925) INFO 07-15 23:06:29 [loggers.py:273] Engine 000: Avg prompt throughput: 1845.5 tokens/s, Avg generation throughput: 2665.3 tokens/s, Running: 253 reqs, Waiting: 293 reqs, GPU KV cache usage: 39.5%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=342925) INFO 07-15 23:06:39 [loggers.py:273] Engine 000: Avg prompt throughput: 1702.1 tokens/s, Avg generation throughput: 2717.9 tokens/s, Running: 254 reqs, Waiting: 80 reqs, GPU KV cache usage: 42.6%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=342925) INFO 07-15 23:06:49 [loggers.py:273] Engine 000: Avg prompt throughput: 663.7 tokens/s, Avg generation throughput: 2476.2 tokens/s, Running: 76 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.8%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=342925) INFO 07-15 23:06:59 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 542.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=342925) INFO: 127.0.0.1:44042 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(EngineCore pid=343040) INFO 07-15 23:07:00 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 111 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 112 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=343040) INFO 07-15 23:07:00 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=343040) INFO 07-15 23:07:00 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=343040) INFO 07-15 23:07:00 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=342925) WARNING 07-15 23:07:00 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=342925) INFO: Shutting down
|
| 120 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=342925) INFO 07-15 23:07:00 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=342925) INFO: Shutting down
|
| 127 |
+
(APIServer pid=342925) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=342925) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1225_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1225_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1225/step0100
|
| 5 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1225/step0100', 'host': '127.0.0.1', 'port': 8381, 'model': 'outputs/healed/grid_math/glean_keep50_s1225/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=338938) WARNING 07-15 22:54:14 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=338938) WARNING 07-15 22:54:14 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=338938) INFO 07-15 22:54:14 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=339108) INFO 07-15 22:54:21 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1225/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1225/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=339108) INFO 07-15 22:54:22 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:38469 backend=nccl
|
| 18 |
+
(EngineCore pid=339108) INFO 07-15 22:54:22 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1225/step0100...
|
| 21 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=339108) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=339108) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 91.88 GiB.
|
| 26 |
+
(EngineCore pid=339108) INFO 07-15 22:54:23 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=339108)
|
| 28 |
+
(EngineCore pid=339108)
|
| 29 |
+
(EngineCore pid=339108)
|
| 30 |
+
(EngineCore pid=339108)
|
| 31 |
+
(EngineCore pid=339108)
|
| 32 |
+
(EngineCore pid=339108) INFO 07-15 22:54:35 [default_loader.py:430] Loading weights took 11.79 seconds
|
| 33 |
+
(EngineCore pid=339108) INFO 07-15 22:54:36 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 11.980934 seconds
|
| 34 |
+
(EngineCore pid=339108) INFO 07-15 22:54:37 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=339108) INFO 07-15 22:54:37 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=339108) INFO 07-15 22:54:37 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=339108) INFO 07-15 22:54:37 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=339108) INFO 07-15 22:54:37 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=339108) INFO 07-15 22:54:38 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.03 s
|
| 40 |
+
(EngineCore pid=339108) INFO 07-15 22:54:38 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=339108) WARNING 07-15 22:54:38 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=339108) WARNING 07-15 22:54:38 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=339108) INFO 07-15 22:54:38 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=339108) INFO 07-15 22:54:38 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=339108) INFO 07-15 22:54:38 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=338938) WARNING 07-15 22:54:38 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8381
|
| 50 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 55 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=338938) INFO 07-15 22:54:38 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=338938) INFO: Started server process [338938]
|
| 98 |
+
(APIServer pid=338938) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=338938) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=338938) INFO: 127.0.0.1:37588 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=339108) WARNING 07-15 22:54:43 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=338938) INFO 07-15 22:54:48 [loggers.py:273] Engine 000: Avg prompt throughput: 2315.2 tokens/s, Avg generation throughput: 1714.9 tokens/s, Running: 253 reqs, Waiting: 1018 reqs, GPU KV cache usage: 37.0%, Prefix cache hit rate: 93.1%
|
| 103 |
+
(APIServer pid=338938) INFO 07-15 22:54:58 [loggers.py:273] Engine 000: Avg prompt throughput: 1875.4 tokens/s, Avg generation throughput: 2664.3 tokens/s, Running: 253 reqs, Waiting: 782 reqs, GPU KV cache usage: 39.1%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=338938) INFO 07-15 22:55:08 [loggers.py:273] Engine 000: Avg prompt throughput: 1719.9 tokens/s, Avg generation throughput: 2690.9 tokens/s, Running: 252 reqs, Waiting: 559 reqs, GPU KV cache usage: 39.8%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=338938) INFO 07-15 22:55:18 [loggers.py:273] Engine 000: Avg prompt throughput: 1853.3 tokens/s, Avg generation throughput: 2664.8 tokens/s, Running: 252 reqs, Waiting: 329 reqs, GPU KV cache usage: 39.7%, Prefix cache hit rate: 93.3%
|
| 106 |
+
(APIServer pid=338938) INFO 07-15 22:55:28 [loggers.py:273] Engine 000: Avg prompt throughput: 1765.6 tokens/s, Avg generation throughput: 2691.7 tokens/s, Running: 254 reqs, Waiting: 110 reqs, GPU KV cache usage: 41.6%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=338938) INFO 07-15 22:55:38 [loggers.py:273] Engine 000: Avg prompt throughput: 903.9 tokens/s, Avg generation throughput: 2595.9 tokens/s, Running: 129 reqs, Waiting: 0 reqs, GPU KV cache usage: 29.4%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=338938) INFO 07-15 22:55:48 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 855.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=338938) INFO: 127.0.0.1:37594 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(EngineCore pid=339108) INFO 07-15 22:55:52 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 111 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 112 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=339108) INFO 07-15 22:55:52 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=339108) INFO 07-15 22:55:52 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=339108) INFO 07-15 22:55:52 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=338938) WARNING 07-15 22:55:52 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 120 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 121 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [core_client.py:662] [shutdown] MPClient: complete
|
| 122 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 123 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 124 |
+
(APIServer pid=338938) INFO 07-15 22:55:52 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 125 |
+
(APIServer pid=338938) INFO: Shutting down
|
| 126 |
+
(APIServer pid=338938) INFO: Waiting for application shutdown.
|
| 127 |
+
(APIServer pid=338938) INFO: Application shutdown complete.
|
| 128 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 129 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1225_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1225_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1225/step0150
|
| 5 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1225/step0150', 'host': '127.0.0.1', 'port': 8381, 'model': 'outputs/healed/grid_math/glean_keep50_s1225/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=340347) WARNING 07-15 22:56:16 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=340347) WARNING 07-15 22:56:16 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=340347) INFO 07-15 22:56:16 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=340525) INFO 07-15 22:56:23 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1225/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1225/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=340525) INFO 07-15 22:56:24 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:54709 backend=nccl
|
| 18 |
+
(EngineCore pid=340525) INFO 07-15 22:56:24 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1225/step0150...
|
| 21 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=340525) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=340525) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 91.48 GiB.
|
| 26 |
+
(EngineCore pid=340525) INFO 07-15 22:56:25 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=340525)
|
| 28 |
+
(EngineCore pid=340525)
|
| 29 |
+
(EngineCore pid=340525)
|
| 30 |
+
(EngineCore pid=340525)
|
| 31 |
+
(EngineCore pid=340525)
|
| 32 |
+
(EngineCore pid=340525) INFO 07-15 22:56:30 [default_loader.py:430] Loading weights took 4.99 seconds
|
| 33 |
+
(EngineCore pid=340525) INFO 07-15 22:56:31 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 5.178779 seconds
|
| 34 |
+
(EngineCore pid=340525) INFO 07-15 22:56:32 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=340525) INFO 07-15 22:56:32 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=340525) INFO 07-15 22:56:32 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=340525) INFO 07-15 22:56:32 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=340525) INFO 07-15 22:56:32 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=340525) INFO 07-15 22:56:33 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.04 s
|
| 40 |
+
(EngineCore pid=340525) INFO 07-15 22:56:33 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=340525) WARNING 07-15 22:56:33 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=340525) WARNING 07-15 22:56:33 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=340525) INFO 07-15 22:56:33 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=340525) INFO 07-15 22:56:33 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=340525) INFO 07-15 22:56:33 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=340347) WARNING 07-15 22:56:33 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8381
|
| 50 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 55 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=340347) INFO 07-15 22:56:33 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=340347) INFO: Started server process [340347]
|
| 98 |
+
(APIServer pid=340347) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=340347) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=340347) INFO: 127.0.0.1:58248 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=340525) WARNING 07-15 22:56:35 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=340347) INFO 07-15 22:56:43 [loggers.py:273] Engine 000: Avg prompt throughput: 2519.1 tokens/s, Avg generation throughput: 1973.2 tokens/s, Running: 255 reqs, Waiting: 988 reqs, GPU KV cache usage: 37.7%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=340347) INFO 07-15 22:56:53 [loggers.py:273] Engine 000: Avg prompt throughput: 1798.9 tokens/s, Avg generation throughput: 2690.8 tokens/s, Running: 255 reqs, Waiting: 763 reqs, GPU KV cache usage: 40.0%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=340347) INFO 07-15 22:57:03 [loggers.py:273] Engine 000: Avg prompt throughput: 1779.7 tokens/s, Avg generation throughput: 2665.1 tokens/s, Running: 253 reqs, Waiting: 535 reqs, GPU KV cache usage: 39.0%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=340347) INFO 07-15 22:57:13 [loggers.py:273] Engine 000: Avg prompt throughput: 1818.4 tokens/s, Avg generation throughput: 2690.4 tokens/s, Running: 255 reqs, Waiting: 306 reqs, GPU KV cache usage: 40.8%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=340347) INFO 07-15 22:57:23 [loggers.py:273] Engine 000: Avg prompt throughput: 1763.9 tokens/s, Avg generation throughput: 2691.3 tokens/s, Running: 256 reqs, Waiting: 92 reqs, GPU KV cache usage: 43.1%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=340347) INFO 07-15 22:57:33 [loggers.py:273] Engine 000: Avg prompt throughput: 741.7 tokens/s, Avg generation throughput: 2511.5 tokens/s, Running: 93 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=340347) INFO 07-15 22:57:43 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 619.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=340347) INFO: 127.0.0.1:58258 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 111 |
+
(EngineCore pid=340525) INFO 07-15 22:57:45 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 112 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=340525) INFO 07-15 22:57:45 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=340525) INFO 07-15 22:57:45 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=340525) INFO 07-15 22:57:45 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=340347) WARNING 07-15 22:57:45 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 120 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 121 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [core_client.py:662] [shutdown] MPClient: complete
|
| 122 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 123 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 124 |
+
(APIServer pid=340347) INFO 07-15 22:57:45 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 125 |
+
(APIServer pid=340347) INFO: Shutting down
|
| 126 |
+
(APIServer pid=340347) INFO: Waiting for application shutdown.
|
| 127 |
+
(APIServer pid=340347) INFO: Application shutdown complete.
|
| 128 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 129 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1226_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1226_step100_chat.json.server.log
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1226/step0100
|
| 5 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1226/step0100', 'host': '127.0.0.1', 'port': 8382, 'model': 'outputs/healed/grid_math/glean_keep50_s1226/step0100', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=338408) WARNING 07-15 22:53:58 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=338408) WARNING 07-15 22:53:58 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=338408) INFO 07-15 22:53:58 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=338584) INFO 07-15 22:54:05 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1226/step0100', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1226/step0100', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=338584) INFO 07-15 22:54:05 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:57925 backend=nccl
|
| 18 |
+
(EngineCore pid=338584) INFO 07-15 22:54:06 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=338584) INFO 07-15 22:54:06 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=338584) INFO 07-15 22:54:06 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1226/step0100...
|
| 21 |
+
(EngineCore pid=338584) INFO 07-15 22:54:07 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=338584) INFO 07-15 22:54:07 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=338584) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=338584) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=338584) INFO 07-15 22:54:07 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 93.70 GiB.
|
| 26 |
+
(EngineCore pid=338584) INFO 07-15 22:54:07 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=338584)
|
| 28 |
+
(EngineCore pid=338584)
|
| 29 |
+
(EngineCore pid=338584)
|
| 30 |
+
(EngineCore pid=338584)
|
| 31 |
+
(EngineCore pid=338584)
|
| 32 |
+
(EngineCore pid=338584) INFO 07-15 22:54:19 [default_loader.py:430] Loading weights took 12.22 seconds
|
| 33 |
+
(EngineCore pid=338584) INFO 07-15 22:54:20 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 12.426588 seconds
|
| 34 |
+
(EngineCore pid=338584) INFO 07-15 22:54:21 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=338584) INFO 07-15 22:54:21 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=338584) INFO 07-15 22:54:21 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=338584) INFO 07-15 22:54:21 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=338584) INFO 07-15 22:54:21 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=338584) INFO 07-15 22:54:22 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.06 s
|
| 40 |
+
(EngineCore pid=338584) INFO 07-15 22:54:22 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=338584) WARNING 07-15 22:54:22 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=338584) WARNING 07-15 22:54:22 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=338584) INFO 07-15 22:54:22 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=338584) INFO 07-15 22:54:22 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=338584) INFO 07-15 22:54:22 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=338408) WARNING 07-15 22:54:22 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8382
|
| 50 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
|
| 52 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /docs, Methods: HEAD, GET
|
| 53 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
| 54 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
|
| 55 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=338408) INFO 07-15 22:54:22 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=338408) INFO: Started server process [338408]
|
| 98 |
+
(APIServer pid=338408) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=338408) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=338408) INFO: 127.0.0.1:39310 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=338584) WARNING 07-15 22:54:24 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=338408) INFO 07-15 22:54:32 [loggers.py:273] Engine 000: Avg prompt throughput: 2455.6 tokens/s, Avg generation throughput: 1866.0 tokens/s, Running: 253 reqs, Waiting: 997 reqs, GPU KV cache usage: 36.9%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=338408) INFO 07-15 22:54:42 [loggers.py:273] Engine 000: Avg prompt throughput: 1830.8 tokens/s, Avg generation throughput: 2664.6 tokens/s, Running: 255 reqs, Waiting: 765 reqs, GPU KV cache usage: 39.3%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=338408) INFO 07-15 22:54:52 [loggers.py:273] Engine 000: Avg prompt throughput: 1751.8 tokens/s, Avg generation throughput: 2691.0 tokens/s, Running: 255 reqs, Waiting: 542 reqs, GPU KV cache usage: 39.5%, Prefix cache hit rate: 93.4%
|
| 105 |
+
(APIServer pid=338408) INFO 07-15 22:55:02 [loggers.py:273] Engine 000: Avg prompt throughput: 1859.3 tokens/s, Avg generation throughput: 2664.8 tokens/s, Running: 255 reqs, Waiting: 311 reqs, GPU KV cache usage: 41.2%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=338408) INFO 07-15 22:55:12 [loggers.py:273] Engine 000: Avg prompt throughput: 1719.2 tokens/s, Avg generation throughput: 2666.6 tokens/s, Running: 254 reqs, Waiting: 96 reqs, GPU KV cache usage: 42.9%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=338408) INFO 07-15 22:55:22 [loggers.py:273] Engine 000: Avg prompt throughput: 799.5 tokens/s, Avg generation throughput: 2534.8 tokens/s, Running: 100 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.8%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=338408) INFO: 127.0.0.1:39314 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 109 |
+
(APIServer pid=338408) INFO 07-15 22:55:32 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 578.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 93.4%
|
| 110 |
+
(EngineCore pid=338584) INFO 07-15 22:55:33 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 111 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 112 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=338584) INFO 07-15 22:55:33 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=338584) INFO 07-15 22:55:33 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=338584) INFO 07-15 22:55:33 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=338408) WARNING 07-15 22:55:33 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 120 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 121 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [core_client.py:662] [shutdown] MPClient: complete
|
| 122 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 123 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 124 |
+
(APIServer pid=338408) INFO 07-15 22:55:33 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 125 |
+
(APIServer pid=338408) INFO: Shutting down
|
| 126 |
+
(APIServer pid=338408) INFO: Waiting for application shutdown.
|
| 127 |
+
(APIServer pid=338408) INFO: Application shutdown complete.
|
| 128 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 129 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep50_s1226_step150_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evals/grid_math/glean_keep50_s1226_step150_chat.json.server.log
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339]
|
| 2 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339] β β ββ ββ
|
| 3 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339] ββ ββ β β β βββ β version 0.25.0
|
| 4 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339] ββββ β β β β model outputs/healed/grid_math/glean_keep50_s1226/step0150
|
| 5 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339] ββ βββββ βββββ β β
|
| 6 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:339]
|
| 7 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [api_utils.py:273] non-default args: {'model_tag': 'outputs/healed/grid_math/glean_keep50_s1226/step0150', 'host': '127.0.0.1', 'port': 8382, 'model': 'outputs/healed/grid_math/glean_keep50_s1226/step0150', 'max_model_len': 2048, 'enforce_eager': True, 'served_model_name': ['student'], 'gpu_memory_utilization': 0.85}
|
| 8 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [model.py:619] Resolved architecture: PrunedOlmoeForCausalLM
|
| 9 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [model.py:1776] Using max model len 2048
|
| 10 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 11 |
+
(APIServer pid=339780) WARNING 07-15 22:55:57 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 12 |
+
(APIServer pid=339780) WARNING 07-15 22:55:57 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 13 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 14 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 15 |
+
(APIServer pid=339780) INFO 07-15 22:55:57 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 16 |
+
(EngineCore pid=339897) INFO 07-15 22:56:04 [core.py:114] Initializing a V1 LLM engine (v0.25.0) with config: model='outputs/healed/grid_math/glean_keep50_s1226/step0150', speculative_config=None, tokenizer='outputs/healed/grid_math/glean_keep50_s1226/step0150', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=2048, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=None, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=student, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, moe_backend='auto', linear_backend='auto')
|
| 17 |
+
(EngineCore pid=339897) INFO 07-15 22:56:05 [parallel_state.py:1607] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.0.15:56135 backend=nccl
|
| 18 |
+
(EngineCore pid=339897) INFO 07-15 22:56:05 [parallel_state.py:1942] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0, EPLB rank N/A
|
| 19 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
|
| 20 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [gpu_model_runner.py:5209] Starting to load model outputs/healed/grid_math/glean_keep50_s1226/step0150...
|
| 21 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [cuda.py:476] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
|
| 22 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [flash_attn.py:718] Using FlashAttention version 2
|
| 23 |
+
(EngineCore pid=339897) /home/henry/.cache/glean/megablocks-variable-93a1479bc15b/megablocks/grouped_gemm_util.py:10: UserWarning: Grouped GEMM not available.
|
| 24 |
+
(EngineCore pid=339897) warnings.warn('Grouped GEMM not available.')
|
| 25 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [weight_utils.py:849] Filesystem type for checkpoints: EXT4. Checkpoint size: 6.89 GiB. Available RAM: 93.57 GiB.
|
| 26 |
+
(EngineCore pid=339897) INFO 07-15 22:56:06 [weight_utils.py:872] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
| 27 |
+
(EngineCore pid=339897)
|
| 28 |
+
(EngineCore pid=339897)
|
| 29 |
+
(EngineCore pid=339897)
|
| 30 |
+
(EngineCore pid=339897)
|
| 31 |
+
(EngineCore pid=339897)
|
| 32 |
+
(EngineCore pid=339897) INFO 07-15 22:56:11 [default_loader.py:430] Loading weights took 5.01 seconds
|
| 33 |
+
(EngineCore pid=339897) INFO 07-15 22:56:12 [gpu_model_runner.py:5306] Model loading took 6.89 GiB memory and 5.190667 seconds
|
| 34 |
+
(EngineCore pid=339897) INFO 07-15 22:56:13 [gpu_worker.py:538] Available KV cache memory: 12.81 GiB
|
| 35 |
+
(EngineCore pid=339897) INFO 07-15 22:56:13 [kv_cache_utils.py:2146] GPU KV cache size: 104,960 tokens
|
| 36 |
+
(EngineCore pid=339897) INFO 07-15 22:56:13 [kv_cache_utils.py:2147] Maximum concurrency for 2,048 tokens per request: 51.25x
|
| 37 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [cutedsl_warmup.py:97] Skipping CuTeDSL warmup because no compile units were requested.
|
| 38 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [jit_monitor.py:73] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
| 39 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [core.py:344] init engine (profile, create kv cache, warmup model) took 2.04 s
|
| 40 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [vllm.py:1042] Asynchronous scheduling is enabled.
|
| 41 |
+
(EngineCore pid=339897) WARNING 07-15 22:56:14 [vllm.py:1096] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
|
| 42 |
+
(EngineCore pid=339897) WARNING 07-15 22:56:14 [vllm.py:1144] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
|
| 43 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [kernel.py:292] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
|
| 44 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [vllm.py:1322] Cudagraph is disabled under eager mode
|
| 45 |
+
(EngineCore pid=339897) INFO 07-15 22:56:14 [compilation.py:312] Enabled custom fusions: norm_quant, act_quant
|
| 46 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [api_server.py:612] Supported tasks: ['generate']
|
| 47 |
+
(APIServer pid=339780) WARNING 07-15 22:56:14 [__init__.py:36] SECURITY WARNING: Development endpoints are enabled! This should NOT be used in production!
|
| 48 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [hf.py:548] Detected the chat template content format to be 'string'. You can set `--chat-template-content-format` to override this.
|
| 49 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [api_server.py:616] Starting vLLM server on http://127.0.0.1:8382
|
| 50 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:37] Available routes are:
|
| 51 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
|
| 52 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /docs, Methods: GET, HEAD
|
| 53 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
| 54 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
|
| 55 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /load, Methods: GET
|
| 56 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /version, Methods: GET
|
| 57 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /health, Methods: GET
|
| 58 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /metrics, Methods: GET
|
| 59 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /tokenize, Methods: POST
|
| 60 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /detokenize, Methods: POST
|
| 61 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/models, Methods: GET
|
| 62 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /ping, Methods: GET
|
| 63 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /ping, Methods: POST
|
| 64 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /invocations, Methods: POST
|
| 65 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /reset_prefix_cache, Methods: POST
|
| 66 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /reset_mm_cache, Methods: POST
|
| 67 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /reset_encoder_cache, Methods: POST
|
| 68 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /pause, Methods: POST
|
| 69 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /resume, Methods: POST
|
| 70 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /is_paused, Methods: GET
|
| 71 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /init_weight_transfer_engine, Methods: POST
|
| 72 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /start_weight_update, Methods: POST
|
| 73 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /update_weights, Methods: POST
|
| 74 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /finish_weight_update, Methods: POST
|
| 75 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /get_world_size, Methods: GET
|
| 76 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /collective_rpc, Methods: POST
|
| 77 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /server_info, Methods: GET
|
| 78 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /sleep, Methods: POST
|
| 79 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /wake_up, Methods: POST
|
| 80 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /is_sleeping, Methods: GET
|
| 81 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
|
| 82 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/chat/completions/batch, Methods: POST
|
| 83 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/responses, Methods: POST
|
| 84 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
|
| 85 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
| 86 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/completions, Methods: POST
|
| 87 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/messages, Methods: POST
|
| 88 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/messages/count_tokens, Methods: POST
|
| 89 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /generative_scoring, Methods: POST
|
| 90 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
|
| 91 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
|
| 92 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/chat/completions/render, Methods: POST
|
| 93 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/completions/render, Methods: POST
|
| 94 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/chat/completions/derender, Methods: POST
|
| 95 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /v1/completions/derender, Methods: POST
|
| 96 |
+
(APIServer pid=339780) INFO 07-15 22:56:14 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
|
| 97 |
+
(APIServer pid=339780) INFO: Started server process [339780]
|
| 98 |
+
(APIServer pid=339780) INFO: Waiting for application startup.
|
| 99 |
+
(APIServer pid=339780) INFO: Application startup complete.
|
| 100 |
+
(APIServer pid=339780) INFO: 127.0.0.1:42148 - "GET /health HTTP/1.1" 200 OK
|
| 101 |
+
(EngineCore pid=339897) WARNING 07-15 22:56:16 [jit_monitor.py:129] Triton kernel JIT compilation during inference: _build_route_rows. This causes a latency spike; consider extending warmup to cover this shape/config.
|
| 102 |
+
(APIServer pid=339780) INFO 07-15 22:56:25 [loggers.py:273] Engine 000: Avg prompt throughput: 2545.3 tokens/s, Avg generation throughput: 1993.3 tokens/s, Running: 252 reqs, Waiting: 989 reqs, GPU KV cache usage: 37.4%, Prefix cache hit rate: 93.2%
|
| 103 |
+
(APIServer pid=339780) INFO 07-15 22:56:35 [loggers.py:273] Engine 000: Avg prompt throughput: 1819.3 tokens/s, Avg generation throughput: 2664.3 tokens/s, Running: 255 reqs, Waiting: 755 reqs, GPU KV cache usage: 39.2%, Prefix cache hit rate: 93.3%
|
| 104 |
+
(APIServer pid=339780) INFO 07-15 22:56:45 [loggers.py:273] Engine 000: Avg prompt throughput: 1725.5 tokens/s, Avg generation throughput: 2691.4 tokens/s, Running: 252 reqs, Waiting: 536 reqs, GPU KV cache usage: 39.8%, Prefix cache hit rate: 93.3%
|
| 105 |
+
(APIServer pid=339780) INFO 07-15 22:56:55 [loggers.py:273] Engine 000: Avg prompt throughput: 1825.1 tokens/s, Avg generation throughput: 2665.1 tokens/s, Running: 250 reqs, Waiting: 309 reqs, GPU KV cache usage: 41.3%, Prefix cache hit rate: 93.4%
|
| 106 |
+
(APIServer pid=339780) INFO 07-15 22:57:05 [loggers.py:273] Engine 000: Avg prompt throughput: 1728.5 tokens/s, Avg generation throughput: 2692.0 tokens/s, Running: 254 reqs, Waiting: 96 reqs, GPU KV cache usage: 44.4%, Prefix cache hit rate: 93.4%
|
| 107 |
+
(APIServer pid=339780) INFO 07-15 22:57:15 [loggers.py:273] Engine 000: Avg prompt throughput: 777.4 tokens/s, Avg generation throughput: 2563.2 tokens/s, Running: 110 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.7%, Prefix cache hit rate: 93.4%
|
| 108 |
+
(APIServer pid=339780) INFO 07-15 22:57:25 [loggers.py:273] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 702.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 93.4%
|
| 109 |
+
(APIServer pid=339780) INFO: 127.0.0.1:42158 - "POST /v1/completions HTTP/1.1" 200 OK
|
| 110 |
+
(EngineCore pid=339897) INFO 07-15 22:57:26 [core.py:1214] [shutdown] EngineCore: trigger received signal=SIGTERM
|
| 111 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [launcher.py:100] [shutdown] API server: shutdown triggered
|
| 112 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [launcher.py:116] [shutdown] API server: stopping engine client mode=abort timeout=0s
|
| 113 |
+
(EngineCore pid=339897) INFO 07-15 22:57:26 [core.py:1333] [shutdown] EngineCore: start mode=abort timeout=0s
|
| 114 |
+
(EngineCore pid=339897) INFO 07-15 22:57:26 [core.py:1364] [shutdown] EngineCore: request processing complete; starting resource teardown
|
| 115 |
+
(EngineCore pid=339897) INFO 07-15 22:57:26 [core.py:1227] [shutdown] EngineCore: exiting busy loop
|
| 116 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [core_client.py:655] [shutdown] MPClient: start timeout=0s
|
| 117 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [core_client.py:657] [shutdown] MPClient: stopping engine manager
|
| 118 |
+
(APIServer pid=339780) WARNING 07-15 22:57:26 [utils.py:626] [shutdown] Process manager: force killing remaining processes count=1
|
| 119 |
+
(APIServer pid=339780) INFO: Shutting down
|
| 120 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [core_client.py:659] [shutdown] MPClient: engine manager stopped
|
| 121 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [core_client.py:660] [shutdown] MPClient: cleaning up background resources
|
| 122 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [core_client.py:662] [shutdown] MPClient: complete
|
| 123 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [launcher.py:125] [shutdown] API server: engine client stopped
|
| 124 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [launcher.py:128] [shutdown] API server: signalling HTTP server shutdown
|
| 125 |
+
(APIServer pid=339780) INFO 07-15 22:57:26 [launcher.py:149] [shutdown] API server: shutting down FastAPI HTTP server
|
| 126 |
+
(APIServer pid=339780) INFO: Shutting down
|
| 127 |
+
(APIServer pid=339780) INFO: Waiting for application shutdown.
|
| 128 |
+
(APIServer pid=339780) INFO: Application shutdown complete.
|
| 129 |
+
/home/henry/.local/share/uv/python/cpython-3.12.12-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
|
| 130 |
+
warnings.warn('resource_tracker: There appear to be %d '
|
evals/grid_math/glean_keep75_s1224_step100_chat.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|