Qualify and enable MTP 2 with paired workload evidence and fallback
This commit is contained in:
@@ -0,0 +1,540 @@
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347]
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347] █ █ █▄ ▄█
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:347]
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.96, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
|
||||
(APIServer pid=1) INFO 09-18 02:59:48 [model.py:2024] Using max model len 131072
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
|
||||
(APIServer pid=1) INFO 09-18 02:59:52 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) INFO 09-18 03:00:04 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=None, tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 1, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) INFO 09-18 03:00:05 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_1ab23a515b4248119daf657c8672946f backend=nccl
|
||||
(EngineCore pid=113) INFO 09-18 03:00:05 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
|
||||
(EngineCore pid=113) INFO 09-18 03:00:05 [gpu_worker.py:441] Using V2 Model Runner
|
||||
(EngineCore pid=113) INFO 09-18 03:00:06 [model_runner.py:387] Loading model from scratch...
|
||||
(EngineCore pid=113) INFO 09-18 03:00:06 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
|
||||
(EngineCore pid=113) INFO 09-18 03:00:06 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
|
||||
(EngineCore pid=113) INFO 09-18 03:00:06 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
|
||||
(EngineCore pid=113) INFO 09-18 03:00:06 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
|
||||
(EngineCore pid=113) INFO 09-18 03:00:08 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
|
||||
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
|
||||
(APIServer pid=1) INFO 09-18 03:00:10 [base.py:261] Multi-modal warmup completed in 12.284s
|
||||
(APIServer pid=1) INFO 09-18 03:00:12 [base.py:261] Readonly multi-modal warmup completed in 1.318s
|
||||
(EngineCore pid=113) INFO 09-18 03:00:42 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
|
||||
(EngineCore pid=113) INFO 09-18 03:00:42 [flash_attn.py:1115] Using FlashAttention version 2
|
||||
(EngineCore pid=113) WARNING 09-18 03:00:42 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
|
||||
(EngineCore pid=113) INFO 09-18 03:00:42 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.30 GiB.
|
||||
(EngineCore pid=113) INFO 09-18 03:00:42 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:16, 1.63s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:06<00:33, 3.75s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:12<00:36, 4.54s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:17<00:34, 4.91s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:23<00:30, 5.15s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:29<00:26, 5.35s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:34<00:21, 5.40s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:40<00:16, 5.48s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:41<00:08, 4.21s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 2.40s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 3.86s/it]
|
||||
(EngineCore pid=113)
|
||||
(EngineCore pid=113) INFO 09-18 03:01:25 [default_loader.py:430] Loading weights took 42.62 seconds
|
||||
(EngineCore pid=113) INFO 09-18 03:01:25 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
|
||||
(EngineCore pid=113) INFO 09-18 03:01:26 [model_runner.py:419] Model loading took 73.77 GiB memory and 80.192933 seconds
|
||||
(EngineCore pid=113) INFO 09-18 03:01:26 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:26 [interface.py:918] Setting attention block size to 3136 tokens to ensure that attention page size is >= mamba page size.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:26 [interface.py:942] Padding mamba page size by 0.13% to ensure that mamba page size and attention page size are exactly equal.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:26 [utils.py:320] Using BLNHC KV cache layout.
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) INFO 09-18 03:01:30 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
|
||||
(EngineCore pid=113) WARNING 09-18 03:01:52 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
|
||||
(EngineCore pid=113)
|
||||
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:02<00:00, 2.12s/it]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:02<00:00, 2.12s/it]
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.09 GiB
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [gpu_worker.py:641] Available KV cache memory: 3.27 GiB
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9600 is equivalent to --gpu-memory-utilization=0.9590 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9610. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [kv_cache_utils.py:2404] GPU KV cache size: 218,453 tokens, Maximum concurrency for 131,072 tokens per request: 1.67x
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [kernel_warmup.py:171] JIT kernel warmup starting.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:55 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
|
||||
(EngineCore pid=113) INFO 09-18 03:01:56 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:56 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
|
||||
(EngineCore pid=113) INFO 09-18 03:01:56 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1),).
|
||||
(EngineCore pid=113) INFO 09-18 03:02:01 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
|
||||
(EngineCore pid=113) INFO 09-18 03:02:01 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
|
||||
(EngineCore pid=113)
|
||||
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.77it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.76it/s]
|
||||
(EngineCore pid=113) INFO 09-18 03:02:06 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.05 GiB
|
||||
(EngineCore pid=113) INFO 09-18 03:02:06 [gpu_worker.py:824] CUDA graph pool memory: 0.05 GiB (actual), 0.09 GiB (estimated), difference: 0.04 GiB (76.0%).
|
||||
(EngineCore pid=113) INFO 09-18 03:02:06 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.96, 79.73 GiB). Actual usage is 74.5 GiB for consumed memory (weights + non-torch), 1.95 GiB for peak activation, and 0.05 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=3303888446` (3.08 GiB) to fit into requested memory, or `--kv-cache-memory=6379883520` (5.94 GiB) to fully utilize gpu memory. Current kv cache memory in use is 3.27 GiB.
|
||||
(EngineCore pid=113) INFO 09-18 03:02:07 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
||||
(EngineCore pid=113) WARNING 09-18 03:02:08 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [core.py:380] init engine (profile, create kv cache, warmup model) took 41.59 s
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [kv_cache_utils.py:762] kv cache group sizes [3136, 3136, 3136, 3136, 4, 3136]
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [kv_cache_utils.py:763] kv lcm block sizes 3136
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
|
||||
(EngineCore pid=113) INFO 09-18 03:02:08 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [entry.py:132] Supported tasks: ['generate']
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [parser_manager.py:34] "auto" tool choice has been enabled.
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
|
||||
(APIServer pid=1) WARNING 09-18 03:02:08 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:60] Available routes are:
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /openapi.json, Methods: GET, HEAD
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /docs, Methods: GET, HEAD
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /redoc, Methods: GET, HEAD
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /load, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /version, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /health, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /metrics, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /tokenize, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /detokenize, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/models, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /ping, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /ping, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /invocations, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/responses, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/completions, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/messages, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /generative_scoring, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 03:02:08 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
|
||||
(APIServer pid=1) INFO: Started server process [1]
|
||||
(APIServer pid=1) INFO: Waiting for application startup.
|
||||
(APIServer pid=1) INFO: Application startup complete.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33126 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33128 - "GET /v1/models HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33134 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(EngineCore pid=113) WARNING 09-18 03:02:09 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33136 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33152 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:45876 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:02:19 [loggers.py:323] Engine 000: Avg prompt throughput: 36.8 tokens/s, Avg generation throughput: 13.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
|
||||
(APIServer pid=1) INFO 09-18 03:02:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:48246 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:45366 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:47264 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:40622 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:49122 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:46812 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:43410 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:49034 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:57140 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:39204 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:36896 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:40524 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:42630 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:47942 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:35640 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44096 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44108 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44120 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44134 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44140 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44146 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38816 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38820 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38832 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38848 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38862 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:09:59 [loggers.py:323] Engine 000: Avg prompt throughput: 59.0 tokens/s, Avg generation throughput: 67.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38878 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38882 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38896 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38900 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46990 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:10:09 [loggers.py:323] Engine 000: Avg prompt throughput: 1547.3 tokens/s, Avg generation throughput: 49.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46992 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47000 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47012 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47018 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47032 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47044 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47058 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47074 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47084 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47094 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47108 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47116 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47124 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) WARNING 09-18 03:10:12 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47140 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47142 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47154 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47168 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47176 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47184 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47196 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47206 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47214 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47226 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:59622 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47232 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47244 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47252 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47266 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47276 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44820 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:10:19 [loggers.py:323] Engine 000: Avg prompt throughput: 1235.6 tokens/s, Avg generation throughput: 14.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 03:10:29 [loggers.py:323] Engine 000: Avg prompt throughput: 12598.2 tokens/s, Avg generation throughput: 9.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 54.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54018 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54020 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54032 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:10:39 [loggers.py:323] Engine 000: Avg prompt throughput: 111.0 tokens/s, Avg generation throughput: 27.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 31.0%, Prefix cache hit rate: 48.4%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:57240 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) WARNING 09-18 03:10:45 [hf.py:879] Chat template rejected the request: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38526 - "POST /v1/chat/completions HTTP/1.1" 400 Bad Request
|
||||
(APIServer pid=1) INFO 09-18 03:10:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 48.4%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35840 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:11:09 [loggers.py:323] Engine 000: Avg prompt throughput: 5.9 tokens/s, Avg generation throughput: 3.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 48.4%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35848 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35864 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35872 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35888 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35896 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35910 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:53726 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45488 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45496 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45510 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45520 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:11:19 [loggers.py:323] Engine 000: Avg prompt throughput: 59.1 tokens/s, Avg generation throughput: 71.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 48.3%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45536 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45552 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45566 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:45576 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52076 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52080 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:11:29 [loggers.py:323] Engine 000: Avg prompt throughput: 2776.9 tokens/s, Avg generation throughput: 50.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52090 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52094 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:11:39 [loggers.py:323] Engine 000: Avg prompt throughput: 15.4 tokens/s, Avg generation throughput: 76.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:49718 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:11:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:11:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35380 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:12:09 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:41060 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:12:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.9%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35078 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:12:29 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:12:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:38006 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:12:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60488 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43418 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:12:59 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 76.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:13:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:38036 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:13:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43076 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:13:29 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:13:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:59088 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:13:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43398 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:13:59 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:14:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:37614 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:50104 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:14:19 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:50108 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:14:29 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:14:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:49780 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36156 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:14:49 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:14:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:15:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33486 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:41522 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:15:19 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:15:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:15:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46554 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:54702 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:15:49 [loggers.py:323] Engine 000: Avg prompt throughput: 7.5 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:15:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:16:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:55592 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:16:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:16:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:16:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:41374 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:16:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:16:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:17:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:38054 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:17:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:17:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:17:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:46862 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:17:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:17:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:18:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:43844 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:18:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:18:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:18:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:34270 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:18:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:18:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO 09-18 03:19:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.1%, Prefix cache hit rate: 45.8%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36216 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36232 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36246 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36254 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36270 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53790 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53794 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53802 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53814 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53826 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53834 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53838 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53846 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53852 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53866 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:55542 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53876 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:19:19 [loggers.py:323] Engine 000: Avg prompt throughput: 1937.1 tokens/s, Avg generation throughput: 46.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 53.0%, MM cache hit rate: 84.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53888 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53900 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59136 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59146 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59154 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59170 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:19:29 [loggers.py:323] Engine 000: Avg prompt throughput: 637.6 tokens/s, Avg generation throughput: 67.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 68.9%, MM cache hit rate: 85.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59186 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59190 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:59206 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44972 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:19:39 [loggers.py:323] Engine 000: Avg prompt throughput: 643.5 tokens/s, Avg generation throughput: 68.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 72.1%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:58578 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:19:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 72.1%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO 09-18 03:19:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 72.1%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:57826 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:20:09 [loggers.py:323] Engine 000: Avg prompt throughput: 8.6 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 72.1%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:43346 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:20:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 72.1%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:51208 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:51212 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:20:29 [loggers.py:323] Engine 000: Avg prompt throughput: 55.5 tokens/s, Avg generation throughput: 74.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 74.7%, MM cache hit rate: 87.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:51220 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:51222 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:20:39 [loggers.py:323] Engine 000: Avg prompt throughput: 588.2 tokens/s, Avg generation throughput: 71.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 74.6%, MM cache hit rate: 88.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:42298 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:20:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 74.6%, MM cache hit rate: 88.9%
|
||||
(APIServer pid=1) INFO 09-18 03:20:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 74.6%, MM cache hit rate: 88.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:42934 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:09 [loggers.py:323] Engine 000: Avg prompt throughput: 8.4 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 74.6%, MM cache hit rate: 88.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:58604 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 74.6%, MM cache hit rate: 88.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:57912 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:57914 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:29 [loggers.py:323] Engine 000: Avg prompt throughput: 369.0 tokens/s, Avg generation throughput: 72.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 76.6%, MM cache hit rate: 89.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:57928 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:57934 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:39 [loggers.py:323] Engine 000: Avg prompt throughput: 274.8 tokens/s, Avg generation throughput: 73.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 76.6%, MM cache hit rate: 90.0%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:60802 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 76.6%, MM cache hit rate: 90.0%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46352 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:21:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 76.6%, MM cache hit rate: 90.0%
|
||||
(APIServer pid=1) INFO 09-18 03:22:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 76.6%, MM cache hit rate: 90.0%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:36974 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:22:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 76.6%, MM cache hit rate: 90.0%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53112 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53114 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:22:29 [loggers.py:323] Engine 000: Avg prompt throughput: 369.0 tokens/s, Avg generation throughput: 72.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 78.3%, MM cache hit rate: 90.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53116 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:53120 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:22:39 [loggers.py:323] Engine 000: Avg prompt throughput: 274.6 tokens/s, Avg generation throughput: 73.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 78.2%, MM cache hit rate: 90.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:54350 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:22:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 78.2%, MM cache hit rate: 90.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60880 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:22:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 78.2%, MM cache hit rate: 90.9%
|
||||
(APIServer pid=1) INFO 09-18 03:23:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 78.2%, MM cache hit rate: 90.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:51446 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:23:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 78.2%, MM cache hit rate: 90.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47600 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47602 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47618 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:23:29 [loggers.py:323] Engine 000: Avg prompt throughput: 635.1 tokens/s, Avg generation throughput: 69.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47626 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:23:39 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:50808 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:23:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33200 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:23:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO 09-18 03:24:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:53902 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:24:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 79.6%, MM cache hit rate: 91.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35818 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60464 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60480 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:24:29 [loggers.py:323] Engine 000: Avg prompt throughput: 635.1 tokens/s, Avg generation throughput: 69.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60486 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:24:39 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:36330 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:24:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47132 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:24:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO 09-18 03:25:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:60010 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:25:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 80.8%, MM cache hit rate: 92.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:37630 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60520 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60522 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:25:29 [loggers.py:323] Engine 000: Avg prompt throughput: 635.1 tokens/s, Avg generation throughput: 69.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60526 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:25:39 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:43716 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:25:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52900 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:25:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO 09-18 03:26:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:41628 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:26:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 81.8%, MM cache hit rate: 92.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52472 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60460 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60464 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60476 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:26:29 [loggers.py:323] Engine 000: Avg prompt throughput: 643.8 tokens/s, Avg generation throughput: 68.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO 09-18 03:26:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:55028 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:26:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:34742 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:26:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO 09-18 03:27:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:49266 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:27:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 82.7%, MM cache hit rate: 93.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44380 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:44394 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:55470 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:55480 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:27:29 [loggers.py:323] Engine 000: Avg prompt throughput: 643.8 tokens/s, Avg generation throughput: 68.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO 09-18 03:27:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:47762 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:27:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:32878 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:27:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO 09-18 03:28:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:54118 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:28:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 83.5%, MM cache hit rate: 93.8%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:37126 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:37132 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38250 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38262 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:28:29 [loggers.py:323] Engine 000: Avg prompt throughput: 643.6 tokens/s, Avg generation throughput: 68.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO 09-18 03:28:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:47208 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:28:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:37546 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:28:59 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO 09-18 03:29:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:40912 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:29:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.2%, MM cache hit rate: 94.1%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38312 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:38320 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33350 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33362 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:29:29 [loggers.py:323] Engine 000: Avg prompt throughput: 643.8 tokens/s, Avg generation throughput: 68.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:29:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:53680 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:29:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:29:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 28.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46494 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:46498 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54326 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:30:09 [loggers.py:323] Engine 000: Avg prompt throughput: 51.0 tokens/s, Avg generation throughput: 62.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54342 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 127.0.0.1:54520 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:30:19 [loggers.py:323] Engine 000: Avg prompt throughput: 12.3 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:30:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:30:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 77.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:41504 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:30:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:30:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:31:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:45670 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:31:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:31:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO 09-18 03:31:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:37016 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:31:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:54348 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 03:31:59 [loggers.py:323] Engine 000: Avg prompt throughput: 13.4 tokens/s, Avg generation throughput: 76.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.1%, Prefix cache hit rate: 84.8%, MM cache hit rate: 94.4%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33524 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33196 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
Reference in New Issue
Block a user