Compare MTP steps and prefill batches; retain qualified MTP 2 baseline

This commit is contained in:
2026-09-18 23:22:45 +08:00
parent 918b78a3b8
commit c8bfe6b9ab
45 changed files with 3785 additions and 3 deletions
@@ -0,0 +1,46 @@
name: qwen38-flash-rtx6000d
services:
vllm:
image: local/qwen38-flash-6000d:0bfc7a15
build: .
container_name: qwen38-flash-6000d
gpus: all
shm_size: 8g
mem_limit: 210g
memswap_limit: 210g
restart: "no"
ports:
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
environment:
HF_HUB_OFFLINE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_WORKER_MULTIPROC_METHOD: spawn
OMP_NUM_THREADS: "8"
TORCHINDUCTOR_COMPILE_THREADS: "2"
CUTE_DSL_ARCH: sm_120a
volumes:
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
- /data/flash-next/cache:/root/.cache
- ./secrets/api-key:/run/secrets/api-key:ro
entrypoint: ["/bin/bash", "-lc"]
command:
- >-
exec vllm serve /model
--served-model-name qwen3.8-flash-next
--host 0.0.0.0 --port 8000
--tensor-parallel-size 1 --dtype bfloat16
--engram-config '{"cpu_offload":true}'
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
--enable-chunked-prefill --enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":3}'
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,4]}' --no-enable-flashinfer-autotune
--load-format safetensors
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
--api-key "$$(cat /run/secrets/api-key)"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30m
@@ -0,0 +1,28 @@
# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_total counter
vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 8872.0
# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_created gauge
vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577659e+09
# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_total counter
vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 26616.0
# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_created gauge
vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897438345776758e+09
# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_total counter
vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 14825.0
# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_created gauge
vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577687e+09
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 6682.0
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 4815.0
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="2"} 3328.0
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.789743834577702e+09
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.7897438345777066e+09
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="2"} 1.7897438345777097e+09
@@ -0,0 +1,7 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228, "response": {"id": "chatcmpl-bbb7d55702a41460", "object": "chat.completion", "created": 1789744068, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236, "response": {"id": "chatcmpl-8083ab34aa888310", "object": "chat.completion", "created": 1789744069, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427, "response": {"id": "chatcmpl-ae45cca384883fdb", "object": "chat.completion", "created": 1789744071, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661, "response": {"id": "chatcmpl-af2a24bebab98e35", "object": "chat.completion", "created": 1789744073, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-9c2f3c0c1afec9f2", "object": "chat.completion", "created": 1789744074, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-a61f6a3fb52eef25", "object": "chat.completion", "created": 1789744080, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-9b2fb34a4ff8498a", "object": "chat.completion", "created": 1789744082, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-b78447dfbb2c5061", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
@@ -0,0 +1,8 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236}
{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "tool-arguments", "passed": true}
QUALITY_COMPLETE
File diff suppressed because one or more lines are too long
@@ -0,0 +1,38 @@
{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6857045710057719, "elapsed_s": 1.1168529220012715, "decode_tps": 187.87036947485734, "max_stream_gap_s": 0.021299945998180192, "events": 23, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"}
{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.09021790199767565, "elapsed_s": 0.30122662099893205, "decode_tps": 170.60906378842878, "max_stream_gap_s": 0.0215804309991654, "events": 10, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"}
{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0900614569982281, "elapsed_s": 0.43906909199722577, "decode_tps": 186.2423439538412, "max_stream_gap_s": 0.021113369999511633, "events": 18, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"}
{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.09134426800301298, "elapsed_s": 0.293823849002365, "decode_tps": 187.67324493881486, "max_stream_gap_s": 0.020979296998120844, "events": 11, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"}
{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08995563200005563, "elapsed_s": 0.4795566350003355, "decode_tps": 182.23772385911695, "max_stream_gap_s": 0.021156058995984495, "events": 20, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"}
{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08942628499789862, "elapsed_s": 0.3126287199993385, "decode_tps": 183.68975230819282, "max_stream_gap_s": 0.02103689500654582, "events": 12, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"}
{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08937776899983874, "elapsed_s": 0.4788391099937144, "decode_tps": 179.73542591253175, "max_stream_gap_s": 0.02123644899984356, "events": 20, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"}
{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.09002831100224284, "elapsed_s": 0.6215811549991486, "decode_tps": 174.95908647709422, "max_stream_gap_s": 0.021011450000514742, "events": 26, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"}
{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08917911599564832, "elapsed_s": 0.5392506789939944, "decode_tps": 186.6369859948443, "max_stream_gap_s": 0.021216852001089137, "events": 23, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"}
{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.09073446100228466, "elapsed_s": 0.539050845996826, "decode_tps": 185.13711025977918, "max_stream_gap_s": 0.020977412998036016, "events": 23, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"}
{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08987670400529169, "elapsed_s": 0.5182491330051562, "decode_tps": 191.42221685799936, "max_stream_gap_s": 0.02107875300134765, "events": 22, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"}
{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09002829199744156, "elapsed_s": 0.5601620039960835, "decode_tps": 189.3078452545796, "max_stream_gap_s": 0.021013982994190883, "events": 24, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"}
{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.1072728509971057, "elapsed_s": 0.40980977100116434, "decode_tps": 138.82603154496502, "max_stream_gap_s": 0.020985293995181564, "events": 16, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"}
{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.09010298399516614, "elapsed_s": 0.5101531999971485, "decode_tps": 147.60140011380784, "max_stream_gap_s": 0.020926092998706736, "events": 22, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"}
{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3285, "completion_tokens": 150, "completion_tokens_details": {"reasoning_tokens": 111}}, "ttft_s": 0.7640111340006115, "elapsed_s": 1.6888792590034427, "decode_tps": 161.10404929302098, "max_stream_gap_s": 0.026646995996998157, "events": 45, "sha256": "59acfe36a1bac3e2c16c79d5c3712068c6d77030f2161a2297c21a795d2a2f9c"}
{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12191, "completion_tokens": 122, "completion_tokens_details": {"reasoning_tokens": 108}}, "ttft_s": 2.5409359719997155, "elapsed_s": 3.243877524000709, "decode_tps": 172.1337992548049, "max_stream_gap_s": 0.03231711799890036, "events": 35, "sha256": "34f1ae68e14f16d3cbbeed12b616b538f07ebec97e9ad601ced414e453c29f65"}
{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.463653916005569, "elapsed_s": 1.8863289360015187, "decode_tps": 165.61186890266376, "max_stream_gap_s": 0.020653829000366386, "events": 22, "sha256": "c4d4ac93f59330deca1f3c5fe9842a60a2ac31170fdbc40f311385c571ed9024"}
{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08850599199649878, "elapsed_s": 2.1877699620017665, "decode_tps": 121.47114590803943, "max_stream_gap_s": 0.0206061229982879, "events": 106, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"}
{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08851322199916467, "elapsed_s": 16.576275200000964, "decode_tps": 124.152689900008, "max_stream_gap_s": 0.021029635005106684, "events": 824, "sha256": "9e8927825d4c68bc3bb71bc5d5dc66a8a35e5961a52ff04a44c5211082f2d4c9"}
{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.09001792799972463, "elapsed_s": 16.722888370000874, "decode_tps": 123.06955718424507, "max_stream_gap_s": 0.021017879000282846, "events": 831, "sha256": "649658539a14ad95cd3aba4a031955cf646f9c2cef5e606a2f3f25be99a3962b"}
{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07350860699807527, "elapsed_s": 16.641791551002825, "decode_tps": 123.54931448950835, "max_stream_gap_s": 0.021029645002272446, "events": 828, "sha256": "4af2092c8a16885a6236f0c3adc0227eb9dd8079ec421ba9c2eb5cc404b3440b"}
{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07293527999718208, "elapsed_s": 1.9857562760007568, "decode_tps": 133.31095828243588, "max_stream_gap_s": 0.020934283995302394, "events": 97, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"}
{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08870576600020286, "elapsed_s": 15.336975530000927, "decode_tps": 134.24473934955645, "max_stream_gap_s": 0.020986846000596415, "events": 765, "sha256": "d4b632309eb489d41b5803e52e0ae083ffe5c0f801ac927326c6037eb95c6474"}
{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08885356999962823, "elapsed_s": 15.44073144599679, "decode_tps": 133.33873657244945, "max_stream_gap_s": 0.02105036300054053, "events": 770, "sha256": "795e37bad007369fe0ce6ce658b4ebf659222130ea24f338f6f342b4ce6aff47"}
{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07399053900007857, "elapsed_s": 15.320730624000134, "decode_tps": 134.25820789152598, "max_stream_gap_s": 0.020978505002858583, "events": 765, "sha256": "ece641cb2f77db4659a64c75f36366c11b83095a29fc22cb320a1b306b6da1f5"}
{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.09028758799831849, "elapsed_s": 2.1541953229971114, "decode_tps": 123.55203465534235, "max_stream_gap_s": 0.02097705299820518, "events": 105, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"}
{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0739625900023384, "elapsed_s": 16.64171508599975, "decode_tps": 123.55327015505169, "max_stream_gap_s": 0.021100942998600658, "events": 828, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"}
{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07516271799977403, "elapsed_s": 16.646220769005595, "decode_tps": 123.52862404436223, "max_stream_gap_s": 0.021613906996208243, "events": 828, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07455741699959617, "elapsed_s": 16.66266934000305, "decode_tps": 123.40162699055197, "max_stream_gap_s": 0.021173682995140553, "events": 829, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126125, "completion_tokens": 130, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.516383726004278, "elapsed_s": 15.218214432999957, "decode_tps": 183.80500983237002, "max_stream_gap_s": 0.02093203100230312, "events": 36, "sha256": "9a2e65aa6f54431f5ea3bfa29bcdd76256bef03427fca286f07cc3232bfdc529"}
{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126119, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.510828671998752, "elapsed_s": 15.23376362700219, "decode_tps": 170.13978802479537, "max_stream_gap_s": 0.021043553999334108, "events": 37, "sha256": "1df90f428d94a645b711abef54df465c7fa0db783e3b3602ea5b4aa73f29aaab"}
{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 14.506387795998307, "elapsed_s": 15.205817506997846, "decode_tps": 168.7088754513573, "max_stream_gap_s": 0.020960481000656728, "events": 36, "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"}
{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 12.83541851100017, "elapsed_s": 13.968540412002767, "decode_tps": 167.67834054913806, "max_stream_gap_s": 0.021041211002739146, "events": 57, "sha256": "86bc2cd86b8219571bdb62f3c793782b1b6de944ea376483b0595f97b9bb55aa"}
{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08649174699530704, "elapsed_s": 0.44836226699408144, "decode_tps": 174.09541954457458, "max_stream_gap_s": 0.020703647001937497, "events": 19, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"}
{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.5161374739982421, "elapsed_s": 0.8586695900012273, "decode_tps": 183.9243593714609, "max_stream_gap_s": 0.021041070998762734, "events": 18, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"}
{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.9265989400009857, "elapsed_s": 1.1674528340008692, "decode_tps": 174.37957635852183, "max_stream_gap_s": 0.02082027800497599, "events": 13, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"}
{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.2341738940012874, "elapsed_s": 1.5963426840025932, "decode_tps": 173.95204042781506, "max_stream_gap_s": 0.02091294499405194, "events": 19, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"}
SUITE_COMPLETE
@@ -0,0 +1,345 @@
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █ █ █▄ ▄█
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 3}, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:22 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
(APIServer pid=1) INFO 09-18 15:01:22 [model.py:2024] Using max model len 131072
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:25 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:25 [model.py:691] Resolved architecture: Qwen4ExpMTP
(APIServer pid=1) INFO 09-18 15:01:25 [model.py:2024] Using max model len 262144
(APIServer pid=1) WARNING 09-18 15:01:25 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate
(APIServer pid=1) INFO 09-18 15:01:25 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072
(APIServer pid=1) INFO 09-18 15:01:25 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
(APIServer pid=1) INFO 09-18 15:01:25 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) WARNING 09-18 15:01:25 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(APIServer pid=1) INFO 09-18 15:01:25 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:01:38 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=3), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 4, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_f7efa1f40cb14f80867628d23379f752 backend=nccl
(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
(EngineCore pid=112) INFO 09-18 15:01:39 [gpu_worker.py:441] Using V2 Model Runner
(EngineCore pid=112) INFO 09-18 15:01:40 [model_runner.py:387] Loading model from scratch...
(EngineCore pid=112) INFO 09-18 15:01:40 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
(EngineCore pid=112) INFO 09-18 15:01:40 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
(EngineCore pid=112) INFO 09-18 15:01:42 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
(APIServer pid=1) INFO 09-18 15:01:44 [base.py:261] Multi-modal warmup completed in 12.297s
(APIServer pid=1) INFO 09-18 15:01:45 [base.py:261] Readonly multi-modal warmup completed in 1.333s
(EngineCore pid=112) INFO 09-18 15:02:16 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
(EngineCore pid=112) INFO 09-18 15:02:16 [flash_attn.py:1115] Using FlashAttention version 2
(EngineCore pid=112) WARNING 09-18 15:02:16 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.27 GiB.
(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:16, 1.63s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:06<00:33, 3.75s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:12<00:36, 4.53s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:17<00:34, 4.89s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:23<00:30, 5.14s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:28<00:26, 5.29s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:34<00:21, 5.38s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:40<00:16, 5.48s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:41<00:08, 4.21s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 2.44s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 3.86s/it]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 15:02:59 [default_loader.py:430] Loading weights took 42.67 seconds
(EngineCore pid=112) INFO 09-18 15:02:59 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) INFO 09-18 15:02:59 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(EngineCore pid=112) INFO 09-18 15:02:59 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(EngineCore pid=112) WARNING 09-18 15:02:59 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(EngineCore pid=112) INFO 09-18 15:02:59 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(EngineCore pid=112) INFO 09-18 15:03:00 [fp8.py:433] Using DEEPGEMM Fp8 MoE backend out of potential backends: ['AITER', 'FLASHINFER_TRTLLM', 'FLASHINFER_CUTLASS', 'DEEPGEMM', 'TRITON', 'MARLIN', 'HUMMING', 'BATCHED_DEEPGEMM', 'BATCHED_TRITON', 'XPU', 'CPU', 'HPC'].
(EngineCore pid=112) INFO 09-18 15:03:00 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.14 GiB.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:00<00:01, 5.20it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:00<00:01, 4.09it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:01<00:01, 3.69it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:01<00:01, 3.47it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:01<00:01, 3.33it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:01<00:01, 3.26it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:02<00:00, 3.19it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:02<00:00, 3.69it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.89it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.70it/s]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 15:03:03 [default_loader.py:430] Loading weights took 2.98 seconds
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:196] deep_gemm not found in site-packages, trying vendored vllm.third_party.deep_gemm
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:223] DeepGEMM PDL enabled on vllm.third_party.deep_gemm.
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:136] DeepGEMM E8M0 enabled on current platform.
(EngineCore pid=112) INFO 09-18 15:03:06 [fp8.py:733] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) WARNING 09-18 15:03:06 [speculator.py:235] Draft model Qwen4ExpMTP does not support external multimodal embeddings. Embeddings from the target model will not be passed to the drafter; using text-only draft inputs instead.
(EngineCore pid=112) INFO 09-18 15:03:06 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.716523 seconds
(EngineCore pid=112) INFO 09-18 15:03:06 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
(EngineCore pid=112) INFO 09-18 15:03:06 [interface.py:918] Setting attention block size to 3200 tokens to ensure that attention page size is >= mamba page size.
(EngineCore pid=112) INFO 09-18 15:03:06 [interface.py:942] Padding mamba page size by 0.25% to ensure that mamba page size and attention page size are exactly equal.
(EngineCore pid=112) INFO 09-18 15:03:06 [utils.py:320] Using BLNHC KV cache layout.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:03:11 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
(EngineCore pid=112) WARNING 09-18 15:03:33 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) WARNING 09-18 15:03:33 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
(EngineCore pid=112) INFO 09-18 15:03:33 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps.
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.85s/it]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.85s/it]
(EngineCore pid=112) INFO 09-18 15:03:35 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.16it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.16it/s]
(EngineCore pid=112)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.71it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.71it/s]
(EngineCore pid=112) INFO 09-18 15:03:36 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.15 GiB
(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB
(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9850 is equivalent to --gpu-memory-utilization=0.9832 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9868. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
(EngineCore pid=112) WARNING 09-18 15:03:37 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) INFO 09-18 15:03:37 [kv_cache_utils.py:2404] GPU KV cache size: 137,414 tokens, Maximum concurrency for 131,072 tokens per request: 1.05x
(EngineCore pid=112) INFO 09-18 15:03:37 [kernel_warmup.py:171] JIT kernel warmup starting.
(EngineCore pid=112) INFO 09-18 15:03:37 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
(EngineCore pid=112) INFO 09-18 15:03:37 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
(EngineCore pid=112) INFO 09-18 15:03:38 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
(EngineCore pid=112) INFO 09-18 15:03:38 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
(EngineCore pid=112) INFO 09-18 15:03:38 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1), (2, 1), (3, 1), (4, 1)).
(EngineCore pid=112) INFO 09-18 15:03:44 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
(EngineCore pid=112) INFO 09-18 15:03:44 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.71it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.70it/s]
(EngineCore pid=112) INFO 09-18 15:03:52 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 117.82it/s]
(EngineCore pid=112)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 175.37it/s]
(EngineCore pid=112) INFO 09-18 15:03:53 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.14 GiB
(EngineCore pid=112) INFO 09-18 15:03:53 [gpu_worker.py:824] CUDA graph pool memory: 0.14 GiB (actual), 0.15 GiB (estimated), difference: 0.01 GiB (5.6%).
(EngineCore pid=112) INFO 09-18 15:03:53 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.985, 81.8 GiB). Actual usage is 77.85 GiB for consumed memory (weights + non-torch), 1.17 GiB for peak activation, and 0.14 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=2676895704` (2.49 GiB) to fit into requested memory, or `--kv-cache-memory=3523562496` (3.28 GiB) to fully utilize gpu memory. Current kv cache memory in use is 2.78 GiB.
(EngineCore pid=112) INFO 09-18 15:03:54 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
(EngineCore pid=112) WARNING 09-18 15:03:54 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
(EngineCore pid=112) INFO 09-18 15:03:54 [core.py:380] init engine (profile, create kv cache, warmup model) took 47.63 s
(EngineCore pid=112) INFO 09-18 15:03:54 [kv_cache_utils.py:762] kv cache group sizes [3200, 3200, 3200, 3200, 8, 3200]
(EngineCore pid=112) INFO 09-18 15:03:54 [kv_cache_utils.py:763] kv lcm block sizes 3200
(EngineCore pid=112) INFO 09-18 15:03:54 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) INFO 09-18 15:03:54 [entry.py:132] Supported tasks: ['generate']
(APIServer pid=1) INFO 09-18 15:03:54 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
(APIServer pid=1) INFO 09-18 15:03:54 [parser_manager.py:34] "auto" tool choice has been enabled.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:03:55 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
(APIServer pid=1) WARNING 09-18 15:03:55 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
(APIServer pid=1) INFO 09-18 15:03:55 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:60] Available routes are:
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /openapi.json, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /docs, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /redoc, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /load, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /version, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /health, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /metrics, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /tokenize, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /detokenize, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/models, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /ping, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /ping, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /invocations, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/completions, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/messages, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /generative_scoring, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
(APIServer pid=1) INFO: Started server process [1]
(APIServer pid=1) INFO: Waiting for application startup.
(APIServer pid=1) INFO: Application startup complete.
(APIServer pid=1) INFO: 127.0.0.1:37726 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42266 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42282 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _count_expert_num_tokens. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _compute_local_logits_stats_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _rejection_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _resample_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(APIServer pid=1) INFO: 172.21.0.1:42298 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42314 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42316 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42324 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42344 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42350 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42356 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42358 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51442 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51456 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51468 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:05 [loggers.py:323] Engine 000: Avg prompt throughput: 70.7 tokens/s, Avg generation throughput: 78.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%
(APIServer pid=1) INFO 09-18 15:04:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.81, Accepted throughput: 57.96 tokens/s, Drafted throughput: 61.97 tokens/s, Accepted: 637 tokens, Drafted: 681 tokens, Per-position acceptance rate: 0.996, 0.934, 0.877, Avg Draft acceptance rate: 93.5%
(APIServer pid=1) INFO: 172.21.0.1:51480 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51482 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51486 - "POST /v1/chat/completions HTTP/1.1" 200 OK
[rank0]:[W918 15:04:08.952733466 CUDACachingAllocator.cpp:3933] memory allocation failed with OOM on device 0 while trying to allocate 377487360 bytes (free: 92209152, total: 89173131264).
(APIServer pid=1) INFO: 172.21.0.1:51502 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51512 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50416 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:15 [loggers.py:323] Engine 000: Avg prompt throughput: 2779.5 tokens/s, Avg generation throughput: 72.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.95, Accepted throughput: 47.90 tokens/s, Drafted throughput: 73.80 tokens/s, Accepted: 479 tokens, Drafted: 738 tokens, Per-position acceptance rate: 0.821, 0.646, 0.480, Avg Draft acceptance rate: 64.9%
(APIServer pid=1) INFO 09-18 15:04:25 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 129.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.59, Accepted throughput: 79.59 tokens/s, Drafted throughput: 150.28 tokens/s, Accepted: 796 tokens, Drafted: 1503 tokens, Per-position acceptance rate: 0.743, 0.511, 0.335, Avg Draft acceptance rate: 53.0%
(APIServer pid=1) INFO: 127.0.0.1:48946 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59844 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:35 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 116.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.36, Accepted throughput: 66.99 tokens/s, Drafted throughput: 147.89 tokens/s, Accepted: 670 tokens, Drafted: 1479 tokens, Per-position acceptance rate: 0.684, 0.422, 0.254, Avg Draft acceptance rate: 45.3%
(APIServer pid=1) INFO 09-18 15:04:45 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.60, Accepted throughput: 80.09 tokens/s, Drafted throughput: 149.99 tokens/s, Accepted: 801 tokens, Drafted: 1500 tokens, Per-position acceptance rate: 0.750, 0.506, 0.346, Avg Draft acceptance rate: 53.4%
(APIServer pid=1) INFO: 172.21.0.1:41026 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:55 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 121.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.44, Accepted throughput: 71.39 tokens/s, Drafted throughput: 148.79 tokens/s, Accepted: 714 tokens, Drafted: 1488 tokens, Per-position acceptance rate: 0.692, 0.456, 0.292, Avg Draft acceptance rate: 48.0%
(APIServer pid=1) INFO: 127.0.0.1:57870 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33434 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:05 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 118.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 69.00 tokens/s, Drafted throughput: 148.19 tokens/s, Accepted: 690 tokens, Drafted: 1482 tokens, Per-position acceptance rate: 0.704, 0.433, 0.259, Avg Draft acceptance rate: 46.6%
(APIServer pid=1) INFO: 172.21.0.1:33450 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:15 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 132.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 82.20 tokens/s, Drafted throughput: 149.41 tokens/s, Accepted: 822 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.759, 0.534, 0.357, Avg Draft acceptance rate: 55.0%
(APIServer pid=1) INFO: 172.21.0.1:59598 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:25 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 131.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 81.60 tokens/s, Drafted throughput: 148.49 tokens/s, Accepted: 816 tokens, Drafted: 1485 tokens, Per-position acceptance rate: 0.731, 0.560, 0.358, Avg Draft acceptance rate: 54.9%
(APIServer pid=1) INFO: 127.0.0.1:37860 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:35 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 136.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.73, Accepted throughput: 86.80 tokens/s, Drafted throughput: 150.30 tokens/s, Accepted: 868 tokens, Drafted: 1503 tokens, Per-position acceptance rate: 0.774, 0.563, 0.395, Avg Draft acceptance rate: 57.8%
(APIServer pid=1) INFO: 172.21.0.1:57216 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:45 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 131.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.63, Accepted throughput: 81.10 tokens/s, Drafted throughput: 149.40 tokens/s, Accepted: 811 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.739, 0.528, 0.361, Avg Draft acceptance rate: 54.3%
(APIServer pid=1) INFO: 172.21.0.1:38916 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50652 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:55 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 130.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 81.50 tokens/s, Drafted throughput: 147.89 tokens/s, Accepted: 815 tokens, Drafted: 1479 tokens, Per-position acceptance rate: 0.740, 0.550, 0.363, Avg Draft acceptance rate: 55.1%
(APIServer pid=1) INFO: 127.0.0.1:49510 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:05 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 119.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 70.00 tokens/s, Drafted throughput: 149.71 tokens/s, Accepted: 700 tokens, Drafted: 1497 tokens, Per-position acceptance rate: 0.697, 0.437, 0.269, Avg Draft acceptance rate: 46.8%
(APIServer pid=1) INFO: 172.21.0.1:34316 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:15 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 126.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.55, Accepted throughput: 77.00 tokens/s, Drafted throughput: 149.10 tokens/s, Accepted: 770 tokens, Drafted: 1491 tokens, Per-position acceptance rate: 0.714, 0.519, 0.316, Avg Draft acceptance rate: 51.6%
(APIServer pid=1) INFO 09-18 15:06:25 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 122.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.45, Accepted throughput: 72.29 tokens/s, Drafted throughput: 149.39 tokens/s, Accepted: 723 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.709, 0.454, 0.289, Avg Draft acceptance rate: 48.4%
(APIServer pid=1) INFO: 127.0.0.1:54572 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50616 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:35 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 123.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.49, Accepted throughput: 73.70 tokens/s, Drafted throughput: 148.49 tokens/s, Accepted: 737 tokens, Drafted: 1485 tokens, Per-position acceptance rate: 0.715, 0.477, 0.297, Avg Draft acceptance rate: 49.6%
(APIServer pid=1) INFO 09-18 15:06:45 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.49, Accepted throughput: 73.10 tokens/s, Drafted throughput: 147.29 tokens/s, Accepted: 731 tokens, Drafted: 1473 tokens, Per-position acceptance rate: 0.701, 0.491, 0.297, Avg Draft acceptance rate: 49.6%
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
(APIServer pid=1) INFO: 172.21.0.1:47280 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47290 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47300 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47314 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47322 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47324 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47340 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47356 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47372 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47378 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47382 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47384 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47398 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47414 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47418 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:55 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 70.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO: 127.0.0.1:40192 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:41722 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:05 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 13.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 48.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:07:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.69, Accepted throughput: 4.70 tokens/s, Drafted throughput: 5.25 tokens/s, Accepted: 94 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.943, 0.943, 0.800, Avg Draft acceptance rate: 89.5%
(APIServer pid=1) INFO 09-18 15:07:15 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 90.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO: 172.21.0.1:45834 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:25 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.9 tokens/s, Avg generation throughput: 12.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 68.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:07:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.47, Accepted throughput: 4.45 tokens/s, Drafted throughput: 5.40 tokens/s, Accepted: 89 tokens, Drafted: 108 tokens, Per-position acceptance rate: 0.917, 0.833, 0.722, Avg Draft acceptance rate: 82.4%
(APIServer pid=1) INFO: 127.0.0.1:36836 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36034 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:35 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.43, Accepted throughput: 8.50 tokens/s, Drafted throughput: 10.50 tokens/s, Accepted: 85 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.886, 0.857, 0.686, Avg Draft acceptance rate: 81.0%
(APIServer pid=1) INFO 09-18 15:07:45 [loggers.py:323] Engine 000: Avg prompt throughput: 11167.1 tokens/s, Avg generation throughput: 2.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.33, Accepted throughput: 1.40 tokens/s, Drafted throughput: 1.80 tokens/s, Accepted: 14 tokens, Drafted: 18 tokens, Per-position acceptance rate: 0.833, 0.833, 0.667, Avg Draft acceptance rate: 77.8%
(APIServer pid=1) INFO: 172.21.0.1:48276 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48278 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48294 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48308 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48318 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48332 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48342 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:55 [loggers.py:323] Engine 000: Avg prompt throughput: 64.3 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.42, Accepted throughput: 110.00 tokens/s, Drafted throughput: 136.50 tokens/s, Accepted: 1100 tokens, Drafted: 1365 tokens, Per-position acceptance rate: 0.923, 0.804, 0.690, Avg Draft acceptance rate: 80.6%
(APIServer pid=1) INFO: 127.0.0.1:49226 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37666 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37670 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37676 - "POST /v1/chat/completions HTTP/1.1" 200 OK
@@ -0,0 +1,5 @@
{
"startup": true,
"records": 44,
"failures": []
}