mcgyvr-vllm Exited (1) 12 minutes ago vllm/vllm-openai:v0.26.0
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345] 
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345]        █     █     █▄   ▄█
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345]  ▄▄ ▄█ █     █     █ ▀▄▀ █  version 0.26.0
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345]   █▄█▀ █     █     █     █  model   Qwen/Qwen2.5-Coder-14B-Instruct-AWQ
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345]    ▀▀  ▀▀▀▀▀ ▀▀▀▀▀ ▀     ▀
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:345] 
(APIServer pid=1) INFO 08-23 06:32:21 [api_utils.py:273] non-default args: {'model_tag': 'Qwen/Qwen2.5-Coder-14B-Instruct-AWQ', 'model': 'Qwen/Qwen2.5-Coder-14B-Instruct-AWQ', 'max_model_len': 8192, 'enforce_eager': True, 'kv_cache_memory_bytes': 12884901888, 'max_num_seqs': 8}
(APIServer pid=1) INFO 08-23 06:32:42 [model.py:623] Resolved architecture: Qwen2ForCausalLM
(APIServer pid=1) INFO 08-23 06:32:42 [model.py:1788] Using max model len 8192
(APIServer pid=1) 
Parse safetensors files:   0%|          | 0/3 [00:00<?, ?it/s]
Parse safetensors files:  33%|███▎      | 1/3 [00:00<00:01,  1.34it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00,  3.94it/s]
(APIServer pid=1) INFO 08-23 06:32:45 [vllm.py:1109] Asynchronous scheduling is enabled.
(APIServer pid=1) WARNING 08-23 06:32:45 [vllm.py:1163] Enforce eager set, disabling torch.compile and CUDAGraphs. This is equivalent to setting -cc.mode=none -cc.cudagraph_mode=none
(APIServer pid=1) WARNING 08-23 06:32:45 [vllm.py:1213] Inductor compilation was disabled by user settings, optimizations settings that are only active during inductor compilation will be ignored.
(APIServer pid=1) INFO 08-23 06:32:45 [kernel.py:295] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'])
(APIServer pid=1) INFO 08-23 06:32:45 [vllm.py:1392] Cudagraph is disabled under eager mode
(APIServer pid=1) INFO 08-23 06:32:45 [compilation.py:329] Enabled custom fusions: norm_quant, act_quant
(EngineCore pid=208) INFO 08-23 06:33:00 [core.py:116] Initializing a V1 LLM engine (v0.26.0) with config: model='Qwen/Qwen2.5-Coder-14B-Instruct-AWQ', speculative_config=None, tokenizer='Qwen/Qwen2.5-Coder-14B-Instruct-AWQ', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.float16, max_seq_len=8192, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=auto_awq, quantization_config=None, enforce_eager=True, enable_return_routed_experts=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=Qwen/Qwen2.5-Coder-14B-Instruct-AWQ, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['all'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'size_asserts': False, 'alignment_asserts': False, 'scalar_asserts': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.NONE: 0>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 0, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native']), enable_flashinfer_autotune=True, enable_cutedsl_warmup=True, enable_bf16x3_router_gemm=False, moe_backend='auto', linear_backend='auto')
(EngineCore pid=208) INFO 08-23 06:33:00 [parallel_state.py:1615] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:42951 backend=nccl
(EngineCore pid=208) INFO 08-23 06:33:00 [parallel_state.py:1946] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank N/A, EPLB rank N/A
(EngineCore pid=208) INFO 08-23 06:33:00 [gpu_worker.py:378] Using V2 Model Runner
(EngineCore pid=208) INFO 08-23 06:33:01 [model_runner.py:284] Loading model from scratch...
(EngineCore pid=208) INFO 08-23 06:33:01 [auto_awq.py:473] Using MarlinLinearKernel for AutoAWQMarlinLinearMethod
(EngineCore pid=208) INFO 08-23 06:33:01 [cuda.py:482] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION'].
(EngineCore pid=208) INFO 08-23 06:33:01 [flash_attn.py:776] Using FlashAttention version 2
(EngineCore pid=208) INFO 08-23 06:33:03 [weight_utils.py:869] Filesystem type for checkpoints: EXT4. Checkpoint size: 9.29 GiB. Available RAM: 26.84 GiB.
(EngineCore pid=208) INFO 08-23 06:33:03 [weight_utils.py:892] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
(EngineCore pid=208) 
Loading safetensors checkpoint shards:   0% Completed | 0/3 [00:00<?, ?it/s]
(EngineCore pid=208) 
Loading safetensors checkpoint shards:  33% Completed | 1/3 [00:03<00:07,  3.62s/it]
(EngineCore pid=208) 
Loading safetensors checkpoint shards:  67% Completed | 2/3 [00:07<00:03,  3.72s/it]
(EngineCore pid=208) 
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:09<00:00,  2.86s/it]
(EngineCore pid=208) 
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:09<00:00,  3.08s/it]
(EngineCore pid=208) 
(EngineCore pid=208) INFO 08-23 06:33:12 [default_loader.py:430] Loading weights took 9.26 seconds
(EngineCore pid=208) INFO 08-23 06:33:16 [model_runner.py:305] Model loading took 9.38 GiB and 15.423580 seconds
(EngineCore pid=208) INFO 08-23 06:33:16 [topk_topp_sampler.py:55] Using FlashInfer for top-p & top-k sampling.
(EngineCore pid=208) INFO 08-23 06:33:22 [gpu_worker.py:479] Initial free memory 11.52 GiB, reserved 12.0 GiB memory for KV Cache as specified by kv_cache_memory_bytes config and skipped memory profiling. This does not respect the gpu_memory_utilization config. Only use kv_cache_memory_bytes config when you want manual control of KV cache memory size. If OOM'ed, check the difference of initial free memory between the current run and the previous run where kv_cache_memory_bytes is suggested and update it correspondingly.
(EngineCore pid=208) INFO 08-23 06:33:22 [kv_cache_utils.py:2177] GPU KV cache size: 65,536 tokens
(EngineCore pid=208) INFO 08-23 06:33:22 [kv_cache_utils.py:2178] Maximum concurrency for 8,192 tokens per request: 8.00x
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330] EngineCore failed to start.
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330] Traceback (most recent call last):
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 1299, in run_engine_core
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]                   ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     return func(*args, **kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 1065, in __init__
(EngineCore pid=208) Process EngineCore:
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     super().__init__(
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 136, in __init__
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     kv_cache_config = self._initialize_kv_caches(vllm_config)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]                       ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     return func(*args, **kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 324, in _initialize_kv_caches
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     self.model_executor.initialize_from_config(kv_cache_configs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/executor/abstract.py", line 123, in initialize_from_config
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     self.collective_rpc("initialize_from_config", args=(kv_cache_configs,))
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/executor/uniproc_executor.py", line 92, in collective_rpc
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     result = run_method(self.driver_worker, method, args, kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/serial_utils.py", line 510, in run_method
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     return func(*args, **kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/worker_base.py", line 325, in initialize_from_config
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     self.worker.initialize_from_config(kv_cache_config)  # type: ignore
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     return func(*args, **kwargs)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu_worker.py", line 732, in initialize_from_config
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     self.model_runner.initialize_kv_cache(kv_cache_config)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/model_runner.py", line 504, in initialize_kv_cache
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     kv_caches_dict = init_kv_cache(
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]                      ^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/attn_utils.py", line 531, in init_kv_cache
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     kv_cache_raw_tensors = _allocate_kv_cache(
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]                            ^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/attn_utils.py", line 186, in _allocate_kv_cache
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]     tensor = torch.zeros(kv_cache_tensor.size, dtype=torch.int8, device=device)
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330]              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) ERROR 08-23 06:33:22 [core.py:1330] torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 256.00 MiB. GPU 0 has a total capacity of 11.63 GiB of which 44.12 MiB is free. Including non-PyTorch memory, this process has 11.58 GiB memory in use. Of the allocated memory 11.40 GiB is allocated by PyTorch, and 33.13 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation.  See documentation for Memory Management  (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
(EngineCore pid=208) Traceback (most recent call last):
(EngineCore pid=208)   File "/usr/lib/python3.12/multiprocessing/process.py", line 314, in _bootstrap
(EngineCore pid=208)     self.run()
(EngineCore pid=208)   File "/usr/lib/python3.12/multiprocessing/process.py", line 108, in run
(EngineCore pid=208)     self._target(*self._args, **self._kwargs)
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 1334, in run_engine_core
(EngineCore pid=208)     raise e
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 1299, in run_engine_core
(EngineCore pid=208)     engine_core = EngineCoreProc(*args, engine_index=dp_rank, **kwargs)
(EngineCore pid=208)                   ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208)     return func(*args, **kwargs)
(EngineCore pid=208)            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 1065, in __init__
(EngineCore pid=208)     super().__init__(
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 136, in __init__
(EngineCore pid=208)     kv_cache_config = self._initialize_kv_caches(vllm_config)
(EngineCore pid=208)                       ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208)     return func(*args, **kwargs)
(EngineCore pid=208)            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core.py", line 324, in _initialize_kv_caches
(EngineCore pid=208)     self.model_executor.initialize_from_config(kv_cache_configs)
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/executor/abstract.py", line 123, in initialize_from_config
(EngineCore pid=208)     self.collective_rpc("initialize_from_config", args=(kv_cache_configs,))
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/executor/uniproc_executor.py", line 92, in collective_rpc
(EngineCore pid=208)     result = run_method(self.driver_worker, method, args, kwargs)
(EngineCore pid=208)              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/serial_utils.py", line 510, in run_method
(EngineCore pid=208)     return func(*args, **kwargs)
(EngineCore pid=208)            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/worker_base.py", line 325, in initialize_from_config
(EngineCore pid=208)     self.worker.initialize_from_config(kv_cache_config)  # type: ignore
(EngineCore pid=208)     ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(EngineCore pid=208)     return func(*args, **kwargs)
(EngineCore pid=208)            ^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu_worker.py", line 732, in initialize_from_config
(EngineCore pid=208)     self.model_runner.initialize_kv_cache(kv_cache_config)
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/model_runner.py", line 504, in initialize_kv_cache
(EngineCore pid=208)     kv_caches_dict = init_kv_cache(
(EngineCore pid=208)                      ^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/attn_utils.py", line 531, in init_kv_cache
(EngineCore pid=208)     kv_cache_raw_tensors = _allocate_kv_cache(
(EngineCore pid=208)                            ^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/worker/gpu/attn_utils.py", line 186, in _allocate_kv_cache
(EngineCore pid=208)     tensor = torch.zeros(kv_cache_tensor.size, dtype=torch.int8, device=device)
(EngineCore pid=208)              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(EngineCore pid=208) torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 256.00 MiB. GPU 0 has a total capacity of 11.63 GiB of which 44.12 MiB is free. Including non-PyTorch memory, this process has 11.58 GiB memory in use. Of the allocated memory 11.40 GiB is allocated by PyTorch, and 33.13 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation.  See documentation for Memory Management  (https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage-with-pytorch-cuda-alloc-conf)
[rank0]:[W823 06:33:22.192893590 ProcessGroupNCCL.cpp:1575] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
(APIServer pid=1) Traceback (most recent call last):
(APIServer pid=1)   File "/usr/local/bin/vllm", line 10, in <module>
(APIServer pid=1)     sys.exit(main())
(APIServer pid=1)              ^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/cli/main.py", line 95, in main
(APIServer pid=1)     args.dispatch_function(args)
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/cli/serve.py", line 148, in cmd
(APIServer pid=1)     uvloop.run(run_server(args))
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/uvloop/__init__.py", line 96, in run
(APIServer pid=1)     return __asyncio.run(
(APIServer pid=1)            ^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/lib/python3.12/asyncio/runners.py", line 195, in run
(APIServer pid=1)     return runner.run(main)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/lib/python3.12/asyncio/runners.py", line 118, in run
(APIServer pid=1)     return self._loop.run_until_complete(task)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/uvloop/__init__.py", line 48, in wrapper
(APIServer pid=1)     return await main
(APIServer pid=1)            ^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/openai/api_server.py", line 759, in run_server
(APIServer pid=1)     await run_server_worker(listen_address, sock, args, **uvicorn_kwargs)
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/openai/api_server.py", line 773, in run_server_worker
(APIServer pid=1)     async with build_async_engine_client(
(APIServer pid=1)                ^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/lib/python3.12/contextlib.py", line 210, in __aenter__
(APIServer pid=1)     return await anext(self.gen)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/openai/api_server.py", line 139, in build_async_engine_client
(APIServer pid=1)     async with build_async_engine_client_from_engine_args(
(APIServer pid=1)                ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/lib/python3.12/contextlib.py", line 210, in __aenter__
(APIServer pid=1)     return await anext(self.gen)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/openai/api_server.py", line 175, in build_async_engine_client_from_engine_args
(APIServer pid=1)     async_llm = AsyncLLM.from_vllm_config(
(APIServer pid=1)                 ^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/async_llm.py", line 217, in from_vllm_config
(APIServer pid=1)     return cls(
(APIServer pid=1)            ^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/async_llm.py", line 146, in __init__
(APIServer pid=1)     self.engine_core = EngineCoreClient.make_async_mp_client(
(APIServer pid=1)                        ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(APIServer pid=1)     return func(*args, **kwargs)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core_client.py", line 132, in make_async_mp_client
(APIServer pid=1)     return AsyncMPClient(*client_args)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/tracing/otel.py", line 178, in sync_wrapper
(APIServer pid=1)     return func(*args, **kwargs)
(APIServer pid=1)            ^^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core_client.py", line 963, in __init__
(APIServer pid=1)     super().__init__(
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/core_client.py", line 573, in __init__
(APIServer pid=1)     with launch_core_engines(
(APIServer pid=1)          ^^^^^^^^^^^^^^^^^^^^
(APIServer pid=1)   File "/usr/lib/python3.12/contextlib.py", line 144, in __exit__
(APIServer pid=1)     next(self.gen)
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/utils.py", line 1213, in launch_core_engines
(APIServer pid=1)     wait_for_engine_startup(
(APIServer pid=1)   File "/usr/local/lib/python3.12/dist-packages/vllm/v1/engine/utils.py", line 1272, in wait_for_engine_startup
(APIServer pid=1)     raise RuntimeError(
(APIServer pid=1) RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {}
