ANNOUNCE box benchbox · cards 0+1 (both 3090s, 250 W each) · arm 2 · vLLM serve openjev/openjev-FP8 TP=2 · start 2026-09-21T18:22:51Z · expect ~10 min to ready, then ~45 min of task arms (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] █ █ █▄ ▄█ (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.29.0 (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] █▄█▀ █ █ █ █ model /workshop/hf-cache/hub/models--openjev--openjev-FP8/snapshots/4ec320f267401e67c9be04d5df1be4d2b6b64f10 (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:347] (APIServer pid=24092) INFO 09-21 18:23:00 [api_utils.py:286] non-default args: {'model_tag': '/workshop/hf-cache/hub/models--openjev--openjev-FP8/snapshots/4ec320f267401e67c9be04d5df1be4d2b6b64f10', 'host': '127.0.0.1', 'model': '/workshop/hf-cache/hub/models--openjev--openjev-FP8/snapshots/4ec320f267401e67c9be04d5df1be4d2b6b64f10', 'trust_remote_code': True, 'max_model_len': 16384, 'quantization': 'fp8', 'max_logprobs': 64, 'served_model_name': ['openjev-fp8'], 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.9, 'enable_prefix_caching': True, 'limit_mm_per_prompt': {'image': 0}, 'max_num_seqs': 1, 'gdn_prefill_backend': 'triton'} (APIServer pid=24092) INFO 09-21 18:23:06 [model.py:684] Resolved architecture: Qwen3_5ForConditionalGeneration (APIServer pid=24092) INFO 09-21 18:23:06 [model.py:2021] Using max model len 16384 (APIServer pid=24092) INFO 09-21 18:23:08 [config.py:625] Mamba cache mode is set to 'align' for Qwen3_5ForConditionalGeneration by default when prefix caching is enabled (APIServer pid=24092) INFO 09-21 18:23:08 [kernel.py:369] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']) (APIServer pid=24092) INFO 09-21 18:23:08 [compilation.py:329] Enabled custom fusions: norm_quant, act_quant (APIServer pid=24092) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. (EngineCore pid=24326) INFO 09-21 18:23:22 [core.py:123] Initializing a V1 LLM engine (v0.29.0) with config: model='/workshop/hf-cache/hub/models--openjev--openjev-FP8/snapshots/4ec320f267401e67c9be04d5df1be4d2b6b64f10', speculative_config=None, tokenizer='/workshop/hf-cache/hub/models--openjev--openjev-FP8/snapshots/4ec320f267401e67c9be04d5df1be4d2b6b64f10', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=16384, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=fp8, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=openjev-fp8, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'none', '+quant_fp8'], 'ir_enable_torch_wrap': True, 'splitting_ops': ['vllm::unified_attention_with_output', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::qwen4_exp_compute_ple_ngram_ids', 'vllm::qwen4_exp_ple_short_conv', 'vllm::qwen4_exp_qsa_with_output', 'vllm::linear_attention', 'vllm::qwen_gdn_attention_core', 'vllm::qwen_gdn_attention_core_fused_norm_packed', 'vllm::gdn_attention_core_xpu', 'vllm::olmo_hybrid_gdn_full_forward', 'vllm::sparse_attn_indexer', 'vllm::rocm_aiter_sparse_attn_indexer', 'vllm::deepseek_v4_attention', 'vllm::hpc_rope_norm_forward', 'vllm::unified_kv_cache_update', 'vllm::unified_mla_kv_cache_update'], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 2, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['native'], fused_add_rms_norm=['native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, enable_jit_warmup=True, enable_bf16x3_router_gemm=False, moe_backend='auto', linear_backend='auto') (EngineCore pid=24326) INFO 09-21 18:23:22 [multiproc_executor.py:153] DP group leader: node_rank=0, node_rank_within_dp=0, master_addr=127.0.0.1, mq_connect_ip= (local), world_size=2, local_world_size=2 (Worker pid=24379) INFO 09-21 18:23:30 [parallel_state.py:1775] world_size=2 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_c180a2bb8fa14a6c83c6948316aa6191 backend=nccl (Worker pid=24380) INFO 09-21 18:23:30 [parallel_state.py:1775] world_size=2 rank=1 local_rank=1 distributed_init_method=file:///tmp/vllm_dist_c180a2bb8fa14a6c83c6948316aa6191 backend=nccl (Worker pid=24379) INFO 09-21 18:23:32 [pynccl.py:113] vLLM is using nccl==2.29.7 (Worker pid=24379) WARNING 09-21 18:23:32 [symm_mem.py:67] SymmMemCommunicator: Device capability 8.6 not supported, communicator is not available. (Worker pid=24380) WARNING 09-21 18:23:32 [symm_mem.py:67] SymmMemCommunicator: Device capability 8.6 not supported, communicator is not available. (Worker pid=24379) WARNING 09-21 18:23:32 [flashinfer_all_reduce.py:383] FlashInfer All Reduce is disabled because it is not supported for world_size=2. (Worker pid=24380) WARNING 09-21 18:23:32 [flashinfer_all_reduce.py:383] FlashInfer All Reduce is disabled because it is not supported for world_size=2. (Worker pid=24380) WARNING 09-21 18:23:32 [custom_all_reduce.py:252] Custom allreduce is disabled because your platform lacks GPU P2P capability or P2P test failed. To silence this warning, specify disable_custom_all_reduce=True explicitly. (Worker pid=24379) WARNING 09-21 18:23:32 [custom_all_reduce.py:252] Custom allreduce is disabled because your platform lacks GPU P2P capability or P2P test failed. To silence this warning, specify disable_custom_all_reduce=True explicitly. (Worker pid=24379) INFO 09-21 18:23:32 [cuda_communicator.py:269] Using ['PYNCCL'] all-reduce backends (in dispatch order) for group 'tp:0' out of potential backends: ['FLASHINFER', 'NCCL_SYMM_MEM', 'QUICK_REDUCE', 'AITER_CUSTOM', 'CUSTOM', 'SYMM_MEM', 'PYNCCL']. (Worker pid=24379) INFO 09-21 18:23:32 [parallel_state.py:2119] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank N/A, EPLB rank N/A (Worker pid=24379) INFO 09-21 18:23:32 [gpu_worker.py:429] Using V2 Model Runner (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [model_runner.py:382] Loading model from scratch... (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [cuda.py:551] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention. (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] Module vllm.third_party.deep_gemm was found but failed to import (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] Traceback (most recent call last): (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/utils/import_utils.py", line 406, in _has_module (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] importlib.import_module(module_name) (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/importlib/__init__.py", line 90, in import_module (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] return _bootstrap._gcd_import(name[level:], package, level) (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1387, in _gcd_import (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1360, in _find_and_load (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1331, in _find_and_load_unlocked (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 935, in _load_unlocked (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 999, in exec_module (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 488, in _call_with_frames_removed (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/third_party/deep_gemm/__init__.py", line 126, in (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] _find_cuda_home() # CUDA home (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/third_party/deep_gemm/__init__.py", line 120, in _find_cuda_home (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] assert cuda_home is not None (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) WARNING 09-21 18:23:32 [import_utils.py:408] AssertionError (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] Module vllm.third_party.deep_gemm was found but failed to import (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] Traceback (most recent call last): (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/utils/import_utils.py", line 406, in _has_module (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] importlib.import_module(module_name) (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/importlib/__init__.py", line 90, in import_module (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] return _bootstrap._gcd_import(name[level:], package, level) (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1387, in _gcd_import (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1360, in _find_and_load (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 1331, in _find_and_load_unlocked (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 935, in _load_unlocked (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 999, in exec_module (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "", line 488, in _call_with_frames_removed (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/third_party/deep_gemm/__init__.py", line 126, in (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] _find_cuda_home() # CUDA home (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/third_party/deep_gemm/__init__.py", line 120, in _find_cuda_home (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] assert cuda_home is not None (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) WARNING 09-21 18:23:32 [import_utils.py:408] AssertionError (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [__init__.py:695] Selected MarlinFP8ScaledMMLinearKernel for Fp8LinearMethod (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [qwen_gdn_linear_attn.py:167] Using Triton/FLA GDN prefill kernel (requested=triton, head_k_dim=128). (Worker_TP0 pid=24379) INFO 09-21 18:23:32 [qwen_gdn_linear_attn.py:519] GDN decode kernel: cuda (Worker_TP0 pid=24379) INFO 09-21 18:23:33 [cuda.py:492] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']. (Worker_TP0 pid=24379) INFO 09-21 18:23:33 [flash_attn.py:897] Using FlashAttention version 2 (Worker_TP0 pid=24379) INFO 09-21 18:23:33 [weight_utils.py:863] Filesystem type for checkpoints: EXT4. Checkpoint size: 28.30 GiB. Available RAM: 23.67 GiB. (Worker_TP0 pid=24379) INFO 09-21 18:23:33 [weight_utils.py:893] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre) and the checkpoint size (28.30 GiB) exceeds 90% of available RAM (23.67 GiB). (Worker_TP0 pid=24379) Loading safetensors checkpoint shards: 0% Completed | 0/12 [00:00= mamba page size. (Worker_TP1 pid=24380) INFO 09-21 18:24:04 [interface.py:942] Padding mamba page size by 0.13% to ensure that mamba page size and attention page size are exactly equal. (Worker_TP0 pid=24379) INFO 09-21 18:24:04 [topk_topp_sampler.py:62] Using FlashInfer for top-p & top-k sampling. (Worker_TP0 pid=24379) INFO 09-21 18:24:04 [interface.py:918] Setting attention block size to 784 tokens to ensure that attention page size is >= mamba page size. (Worker_TP0 pid=24379) INFO 09-21 18:24:04 [interface.py:942] Padding mamba page size by 0.13% to ensure that mamba page size and attention page size are exactly equal. (EngineCore pid=24326) INFO 09-21 18:24:04 [torch_utils.py:277] Reducing Torch threads from 8 to 1 for serving. Set OMP_NUM_THREADS in the external environment to override. (EngineCore pid=24326) INFO 09-21 18:24:05 [utils.py:306] Using LBNHC KV cache layout. (Worker_TP1 pid=24380) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. (Worker_TP0 pid=24379) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. (Worker_TP1 pid=24380) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. (Worker_TP0 pid=24379) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. (Worker_TP0 pid=24379) INFO 09-21 18:24:13 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 12288 tokens, and profiled with 1 video items of the maximum feature size. (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] WorkerProc hit an exception. (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] Traceback (most recent call last): (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 1047, in _execute_worker_rpc (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] output = func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu_worker.py", line 557, in determine_available_memory (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_runner.profile_run() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/model_runner.py", line 864, in profile_run (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_state.encoder_runner.profile_encoder_cache( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 139, in profile_encoder_cache (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] dummy_encoder_outputs = self.execute_mm_encoder(dummy_mm_inputs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 168, in execute_mm_encoder (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] else self.model.embed_multimodal(**mm_kwargs_batch) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2911, in embed_multimodal (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeddings = self._process_video_input(multimodal_input) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2337, in _process_video_input (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeds = self.visual(pixel_values_videos, grid_thw=grid_thw) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return self._call_impl(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return forward_call(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 864, in forward (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] encoder_metadata = self.prepare_encoder_metadata(grid_thw_list) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 786, in prepare_encoder_metadata (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] metadata["pos_embeds"] = self.fast_pos_embed_interpolate(grid_thw_list) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 741, in fast_pos_embed_interpolate (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] interpolate_fn( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 280, in triton_pos_embed_interpolate (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] _bilinear_pos_embed_kernel[(total_out,)]( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 370, in (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 713, in run (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] device = driver.active.get_current_device() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 39, in active (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._active = self.default (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 33, in default (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._default = _create_driver() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 21, in _create_driver (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return active_drivers[0]() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 336, in __init__ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.utils = CudaUtils() # TODO: make static (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 66, in __init__ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] mod = compile_module_from_src( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 93, in compile_module_from_src (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] so = _build(name, src_path, tmpdir, library_dirs or [], include_dirs or [], libraries or [], ccflags or []) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 32, in _build (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] raise RuntimeError( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] RuntimeError: Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl. (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] Traceback (most recent call last): (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 1047, in _execute_worker_rpc (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] output = func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu_worker.py", line 557, in determine_available_memory (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_runner.profile_run() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/model_runner.py", line 864, in profile_run (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_state.encoder_runner.profile_encoder_cache( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 139, in profile_encoder_cache (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] dummy_encoder_outputs = self.execute_mm_encoder(dummy_mm_inputs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 168, in execute_mm_encoder (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] else self.model.embed_multimodal(**mm_kwargs_batch) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2911, in embed_multimodal (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeddings = self._process_video_input(multimodal_input) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2337, in _process_video_input (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeds = self.visual(pixel_values_videos, grid_thw=grid_thw) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return self._call_impl(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return forward_call(*args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 864, in forward (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] encoder_metadata = self.prepare_encoder_metadata(grid_thw_list) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 786, in prepare_encoder_metadata (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] metadata["pos_embeds"] = self.fast_pos_embed_interpolate(grid_thw_list) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 741, in fast_pos_embed_interpolate (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] interpolate_fn( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 280, in triton_pos_embed_interpolate (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] _bilinear_pos_embed_kernel[(total_out,)]( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 370, in (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 713, in run (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] device = driver.active.get_current_device() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 39, in active (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._active = self.default (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 33, in default (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._default = _create_driver() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 21, in _create_driver (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return active_drivers[0]() (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 336, in __init__ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.utils = CudaUtils() # TODO: make static (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 66, in __init__ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] mod = compile_module_from_src( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 93, in compile_module_from_src (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] so = _build(name, src_path, tmpdir, library_dirs or [], include_dirs or [], libraries or [], ccflags or []) (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 32, in _build (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] raise RuntimeError( (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] RuntimeError: Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl. (Worker_TP1 pid=24380) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] WorkerProc hit an exception. (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] Traceback (most recent call last): (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 1047, in _execute_worker_rpc (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] output = func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu_worker.py", line 557, in determine_available_memory (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_runner.profile_run() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/model_runner.py", line 864, in profile_run (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_state.encoder_runner.profile_encoder_cache( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 139, in profile_encoder_cache (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] dummy_encoder_outputs = self.execute_mm_encoder(dummy_mm_inputs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 168, in execute_mm_encoder (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] else self.model.embed_multimodal(**mm_kwargs_batch) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2911, in embed_multimodal (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeddings = self._process_video_input(multimodal_input) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2337, in _process_video_input (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeds = self.visual(pixel_values_videos, grid_thw=grid_thw) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return self._call_impl(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return forward_call(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 864, in forward (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] encoder_metadata = self.prepare_encoder_metadata(grid_thw_list) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 786, in prepare_encoder_metadata (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] metadata["pos_embeds"] = self.fast_pos_embed_interpolate(grid_thw_list) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 741, in fast_pos_embed_interpolate (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] interpolate_fn( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 280, in triton_pos_embed_interpolate (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] _bilinear_pos_embed_kernel[(total_out,)]( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 370, in (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 713, in run (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] device = driver.active.get_current_device() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 39, in active (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._active = self.default (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 33, in default (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._default = _create_driver() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 21, in _create_driver (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return active_drivers[0]() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 336, in __init__ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.utils = CudaUtils() # TODO: make static (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 66, in __init__ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] mod = compile_module_from_src( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 93, in compile_module_from_src (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] so = _build(name, src_path, tmpdir, library_dirs or [], include_dirs or [], libraries or [], ccflags or []) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 32, in _build (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] raise RuntimeError( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] RuntimeError: Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl. (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] Traceback (most recent call last): (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 1047, in _execute_worker_rpc (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] output = func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu_worker.py", line 557, in determine_available_memory (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_runner.profile_run() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/model_runner.py", line 864, in profile_run (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.model_state.encoder_runner.profile_encoder_cache( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 139, in profile_encoder_cache (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] dummy_encoder_outputs = self.execute_mm_encoder(dummy_mm_inputs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return func(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/worker/gpu/mm/encoder_runner.py", line 168, in execute_mm_encoder (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] else self.model.embed_multimodal(**mm_kwargs_batch) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2911, in embed_multimodal (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeddings = self._process_video_input(multimodal_input) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 2337, in _process_video_input (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] video_embeds = self.visual(pixel_values_videos, grid_thw=grid_thw) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1778, in _wrapped_call_impl (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return self._call_impl(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1789, in _call_impl (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return forward_call(*args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 864, in forward (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] encoder_metadata = self.prepare_encoder_metadata(grid_thw_list) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 786, in prepare_encoder_metadata (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] metadata["pos_embeds"] = self.fast_pos_embed_interpolate(grid_thw_list) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 741, in fast_pos_embed_interpolate (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] interpolate_fn( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/model_executor/models/qwen3_vl.py", line 280, in triton_pos_embed_interpolate (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] _bilinear_pos_embed_kernel[(total_out,)]( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 370, in (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/jit.py", line 713, in run (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] device = driver.active.get_current_device() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 39, in active (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._active = self.default (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 33, in default (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self._default = _create_driver() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/driver.py", line 21, in _create_driver (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] return active_drivers[0]() (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 336, in __init__ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] self.utils = CudaUtils() # TODO: make static (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/backends/nvidia/driver.py", line 66, in __init__ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] mod = compile_module_from_src( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 93, in compile_module_from_src (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] so = _build(name, src_path, tmpdir, library_dirs or [], include_dirs or [], libraries or [], ccflags or []) (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/triton/runtime/build.py", line 32, in _build (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] raise RuntimeError( (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] RuntimeError: Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl. (Worker_TP0 pid=24379) ERROR 09-21 18:24:13 [multiproc_executor.py:1055] (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] EngineCore failed to start. (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] Traceback (most recent call last): (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 1336, in run_engine_core (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return func(*args, **kwargs) (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 1093, in __init__ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] super().__init__( (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 145, in __init__ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] kv_cache_config = self._initialize_kv_caches(vllm_config) (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return func(*args, **kwargs) (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 308, in _initialize_kv_caches (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] available_gpu_memory = self.model_executor.determine_available_memory() (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/abstract.py", line 149, in determine_available_memory (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return self.collective_rpc("determine_available_memory") (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 448, in collective_rpc (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return future if non_block else future.result() (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 99, in result (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return super().result() (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/concurrent/futures/_base.py", line 449, in result (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] return self.__get_result() (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/concurrent/futures/_base.py", line 401, in __get_result (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] raise self._exception (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 103, in _wait_for_response (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] response = self.aggregate(self.get_response()) (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] ^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 437, in get_response (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] raise RuntimeError( (EngineCore pid=24326) ERROR 09-21 18:24:13 [core.py:1374] RuntimeError: Worker failed with error 'Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl.', please check the stack trace above for the root cause (EngineCore pid=24326) ERROR 09-21 18:24:17 [multiproc_executor.py:314] Worker proc VllmWorker-0 died unexpectedly (exit code: None), shutting down executor. (EngineCore pid=24326) INFO 09-21 18:24:17 [multiproc_executor.py:472] [shutdown] Executor: waiting for worker exit count=2 (EngineCore pid=24326) Process EngineCore: (EngineCore pid=24326) Traceback (most recent call last): (EngineCore pid=24326) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/multiprocessing/process.py", line 314, in _bootstrap (EngineCore pid=24326) self.run() (EngineCore pid=24326) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/multiprocessing/process.py", line 108, in run (EngineCore pid=24326) self._target(*self._args, **self._kwargs) (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 1378, in run_engine_core (EngineCore pid=24326) raise e (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 1336, in run_engine_core (EngineCore pid=24326) engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs) (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (EngineCore pid=24326) return func(*args, **kwargs) (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 1093, in __init__ (EngineCore pid=24326) super().__init__( (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 145, in __init__ (EngineCore pid=24326) kv_cache_config = self._initialize_kv_caches(vllm_config) (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (EngineCore pid=24326) return func(*args, **kwargs) (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core.py", line 308, in _initialize_kv_caches (EngineCore pid=24326) available_gpu_memory = self.model_executor.determine_available_memory() (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/abstract.py", line 149, in determine_available_memory (EngineCore pid=24326) return self.collective_rpc("determine_available_memory") (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 448, in collective_rpc (EngineCore pid=24326) return future if non_block else future.result() (EngineCore pid=24326) ^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 99, in result (EngineCore pid=24326) return super().result() (EngineCore pid=24326) ^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/concurrent/futures/_base.py", line 449, in result (EngineCore pid=24326) return self.__get_result() (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/concurrent/futures/_base.py", line 401, in __get_result (EngineCore pid=24326) raise self._exception (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 103, in _wait_for_response (EngineCore pid=24326) response = self.aggregate(self.get_response()) (EngineCore pid=24326) ^^^^^^^^^^^^^^^^^^^ (EngineCore pid=24326) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/executor/multiproc_executor.py", line 437, in get_response (EngineCore pid=24326) raise RuntimeError( (EngineCore pid=24326) RuntimeError: Worker failed with error 'Failed to find C compiler. Please specify via CC environment variable or set triton.knobs.build.impl.', please check the stack trace above for the root cause (EngineCore pid=24326) INFO 09-21 18:24:17 [multiproc_executor.py:479] [shutdown] Executor: all workers exited gracefully (APIServer pid=24092) INFO 09-21 18:24:21 [utils.py:620] [shutdown] Process manager: send sigterm to process EngineCore (APIServer pid=24092) Traceback (most recent call last): (APIServer pid=24092) File "/workshop/bench-vllm-venv/bin/vllm", line 10, in (APIServer pid=24092) sys.exit(main()) (APIServer pid=24092) ^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/cli/main.py", line 97, in main (APIServer pid=24092) args.dispatch_function(args) (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/cli/serve.py", line 153, in cmd (APIServer pid=24092) uvloop.run(run_server(args)) (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/uvloop/__init__.py", line 96, in run (APIServer pid=24092) return __asyncio.run( (APIServer pid=24092) ^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/asyncio/runners.py", line 195, in run (APIServer pid=24092) return runner.run(main) (APIServer pid=24092) ^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/asyncio/runners.py", line 118, in run (APIServer pid=24092) return self._loop.run_until_complete(task) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/uvloop/__init__.py", line 48, in wrapper (APIServer pid=24092) return await main (APIServer pid=24092) ^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/launchers/api_server/entry.py", line 176, in run_server (APIServer pid=24092) await run_server_worker(listen_address, sock, args, **uvicorn_kwargs) (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/launchers/api_server/entry.py", line 190, in run_server_worker (APIServer pid=24092) async with build_async_engine_client( (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/contextlib.py", line 210, in __aenter__ (APIServer pid=24092) return await anext(self.gen) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/launchers/api_server/entry.py", line 58, in build_async_engine_client (APIServer pid=24092) async with build_async_engine_client_from_engine_args( (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/contextlib.py", line 210, in __aenter__ (APIServer pid=24092) return await anext(self.gen) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/entrypoints/launchers/api_server/entry.py", line 94, in build_async_engine_client_from_engine_args (APIServer pid=24092) async_llm = AsyncLLM.from_vllm_config( (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/async_llm.py", line 227, in from_vllm_config (APIServer pid=24092) return cls( (APIServer pid=24092) ^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/async_llm.py", line 156, in __init__ (APIServer pid=24092) self.engine_core = EngineCoreClient.make_async_mp_client( (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (APIServer pid=24092) return func(*args, **kwargs) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core_client.py", line 139, in make_async_mp_client (APIServer pid=24092) return AsyncMPClient(*client_args) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/tracing/otel.py", line 178, in sync_wrapper (APIServer pid=24092) return func(*args, **kwargs) (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core_client.py", line 991, in __init__ (APIServer pid=24092) super().__init__( (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/core_client.py", line 609, in __init__ (APIServer pid=24092) with launch_core_engines( (APIServer pid=24092) ^^^^^^^^^^^^^^^^^^^^ (APIServer pid=24092) File "/workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/contextlib.py", line 144, in __exit__ (APIServer pid=24092) next(self.gen) (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/utils.py", line 1240, in launch_core_engines (APIServer pid=24092) wait_for_engine_startup( (APIServer pid=24092) File "/workshop/bench-vllm-venv/lib/python3.12/site-packages/vllm/v1/engine/utils.py", line 1320, in wait_for_engine_startup (APIServer pid=24092) raise RuntimeError( (APIServer pid=24092) RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {} /workshop/.local/share/uv/python/cpython-3.12.14-linux-x86_64-gnu/lib/python3.12/multiprocessing/resource_tracker.py:279: UserWarning: resource_tracker: There appear to be 2 leaked shared_memory objects to clean up at shutdown warnings.warn('resource_tracker: There appear to be %d '