Compare MTP steps and prefill batches; retain qualified MTP 2 baseline

This commit is contained in:
2026-09-18 23:22:45 +08:00
parent 918b78a3b8
commit c8bfe6b9ab
45 changed files with 3785 additions and 3 deletions
@@ -0,0 +1,46 @@
name: qwen38-flash-rtx6000d
services:
vllm:
image: local/qwen38-flash-6000d:0bfc7a15
build: .
container_name: qwen38-flash-6000d
gpus: all
shm_size: 8g
mem_limit: 210g
memswap_limit: 210g
restart: "no"
ports:
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
environment:
HF_HUB_OFFLINE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_WORKER_MULTIPROC_METHOD: spawn
OMP_NUM_THREADS: "8"
TORCHINDUCTOR_COMPILE_THREADS: "2"
CUTE_DSL_ARCH: sm_120a
volumes:
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
- /data/flash-next/cache:/root/.cache
- ./secrets/api-key:/run/secrets/api-key:ro
entrypoint: ["/bin/bash", "-lc"]
command:
- >-
exec vllm serve /model
--served-model-name qwen3.8-flash-next
--host 0.0.0.0 --port 8000
--tensor-parallel-size 1 --dtype bfloat16
--engram-config '{"cpu_offload":true}'
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
--enable-chunked-prefill --enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":1}'
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,2]}' --no-enable-flashinfer-autotune
--load-format safetensors
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
--api-key "$$(cat /run/secrets/api-key)"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30m
@@ -0,0 +1,24 @@
# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_total counter
vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 13297.0
# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_created gauge
vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897433982360141e+09
# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_total counter
vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 13297.0
# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_created gauge
vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897433982360315e+09
# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_total counter
vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 10374.0
# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_created gauge
vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743398236043e+09
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 10374.0
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897433982360573e+09
@@ -0,0 +1,7 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.6113444969960256, "response": {"id": "chatcmpl-b16e6f7979109685", "object": "chat.completion", "created": 1789743630, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "deduplicate", "passed": true, "elapsed_s": 2.2101684480003314, "response": {"id": "chatcmpl-b7a2d1e2b74737cb", "object": "chat.completion", "created": 1789743632, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "schedule", "passed": true, "elapsed_s": 2.493778939002368, "response": {"id": "chatcmpl-94ebf857ca1ee68c", "object": "chat.completion", "created": 1789743634, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.6490989489975618, "response": {"id": "chatcmpl-9c57abdc32a627c6", "object": "chat.completion", "created": 1789743637, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-b3de4e6b44566b44", "object": "chat.completion", "created": 1789743638, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-b6b995fb183886f7", "object": "chat.completion", "created": 1789743647, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-a63e6f2654b5a582", "object": "chat.completion", "created": 1789743650, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-8666af43eb56d793", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
@@ -0,0 +1,8 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.6113444969960256}
{"name": "deduplicate", "passed": true, "elapsed_s": 2.2101684480003314}
{"name": "schedule", "passed": true, "elapsed_s": 2.493778939002368}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.6490989489975618}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "tool-arguments", "passed": true}
QUALITY_COMPLETE
File diff suppressed because one or more lines are too long
@@ -0,0 +1,38 @@
{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6788992599977064, "elapsed_s": 1.330652390999603, "decode_tps": 124.28018546759276, "max_stream_gap_s": 0.016711308002413716, "events": 42, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"}
{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08442727399960859, "elapsed_s": 0.38878584299527574, "decode_tps": 118.28153916873126, "max_stream_gap_s": 0.017301366999163292, "events": 19, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"}
{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08499549300177023, "elapsed_s": 0.6222111509996466, "decode_tps": 120.99423952430095, "max_stream_gap_s": 0.03257976900204085, "events": 32, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"}
{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08491974999924423, "elapsed_s": 0.39251107800373575, "decode_tps": 123.54054402809794, "max_stream_gap_s": 0.03264573799970094, "events": 19, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"}
{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.085520082997391, "elapsed_s": 0.6701685270018061, "decode_tps": 121.4405010876311, "max_stream_gap_s": 0.032602148996375035, "events": 36, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"}
{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08477247699920554, "elapsed_s": 0.4252711029985221, "decode_tps": 120.41164594914487, "max_stream_gap_s": 0.032577335005044006, "events": 21, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"}
{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08490634099871386, "elapsed_s": 0.6701965460015344, "decode_tps": 119.59878945806493, "max_stream_gap_s": 0.03245343499293085, "events": 36, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"}
{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.0855920120011433, "elapsed_s": 0.8476354470039951, "decode_tps": 122.04028763747826, "max_stream_gap_s": 0.03259085400350159, "events": 47, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"}
{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08551327399618458, "elapsed_s": 0.7655381610020413, "decode_tps": 123.52489093428804, "max_stream_gap_s": 0.03254275799554307, "events": 42, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"}
{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08536207699944498, "elapsed_s": 0.7670097699956386, "decode_tps": 121.7637803997724, "max_stream_gap_s": 0.016597622001427226, "events": 42, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"}
{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08568318800098496, "elapsed_s": 0.7488038730007247, "decode_tps": 123.65773207637488, "max_stream_gap_s": 0.03235569100070279, "events": 41, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"}
{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08726995399774751, "elapsed_s": 0.8155018589968677, "decode_tps": 122.21381594110125, "max_stream_gap_s": 0.016604981996351853, "events": 45, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"}
{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.10169535600289237, "elapsed_s": 0.4396858489999431, "decode_tps": 124.26385022719111, "max_stream_gap_s": 0.016457880003144965, "events": 22, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"}
{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08512602300470462, "elapsed_s": 0.6315631280012894, "decode_tps": 113.46228034860022, "max_stream_gap_s": 0.0164907750004204, "events": 34, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"}
{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3263, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.7587772789993323, "elapsed_s": 1.8307137399970088, "decode_tps": 118.4771715683578, "max_stream_gap_s": 0.017939804994966835, "events": 67, "sha256": "8921e4d6739375a20a9acdb345ebda84acb52f498071a986927bd0a4e61a3779"}
{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12197, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 2.490573155002494, "elapsed_s": 3.5669434599985834, "decode_tps": 117.98913386082441, "max_stream_gap_s": 0.03252299100131495, "events": 66, "sha256": "3c4546e143b0d70a31d19afcd37f4462f9f4bc75d2ed102649b295b696577524"}
{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.4522779820035794, "elapsed_s": 2.0256833830062533, "decode_tps": 122.07767816207503, "max_stream_gap_s": 0.01650294200226199, "events": 37, "sha256": "81142145348f20b3b8803d7c6d9426ccc5641f864e2bd78da3e927825dc764b3"}
{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08386934799636947, "elapsed_s": 2.4533141759966384, "decode_tps": 107.62014670550964, "max_stream_gap_s": 0.016294626002490986, "events": 150, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"}
{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0830643839944969, "elapsed_s": 18.754658626996388, "decode_tps": 109.6317739856207, "max_stream_gap_s": 0.016825594000692945, "events": 1176, "sha256": "4906c37a2e701fd6633345dd53b182e28095d6b5da93f478eac8d410d6af322a"}
{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07255172300210688, "elapsed_s": 18.743193971000437, "decode_tps": 109.63736398620448, "max_stream_gap_s": 0.01642427300248528, "events": 1176, "sha256": "bfcbf92932af2de82cc8324af42ba2d393695f14eb500ea32fbcad6f781dcfaf"}
{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07399964099749923, "elapsed_s": 18.77744392999739, "decode_tps": 109.44508232656955, "max_stream_gap_s": 0.01670600000215927, "events": 1178, "sha256": "685a932841c3272e4da5b4f12e188bdb7fd6b50b4b564bc426e3e139793f068c"}
{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08481309299531858, "elapsed_s": 2.3875851429984323, "decode_tps": 110.73610173427943, "max_stream_gap_s": 0.03143068100325763, "events": 145, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"}
{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08440493299713125, "elapsed_s": 18.277994324002066, "decode_tps": 112.5121577720147, "max_stream_gap_s": 0.03122073299891781, "events": 1147, "sha256": "5e28251162bb5f8117e86f4f2cb747928f5bb5287e6cec1443d2d64deefb8fb3"}
{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07261836499674246, "elapsed_s": 18.21824221299903, "decode_tps": 112.80956869528413, "max_stream_gap_s": 0.03128572199784685, "events": 1144, "sha256": "314d5c03936def50095096c59aab4bfc5743e49e546edd449ec7db880db067df"}
{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07247121300315484, "elapsed_s": 18.250864649002324, "decode_tps": 112.60621062069599, "max_stream_gap_s": 0.03126876999886008, "events": 1146, "sha256": "18736c57814a1186850c8f28964837b22549363a2ab0e0e00f2332ebe1aeaa88"}
{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.0831944410019787, "elapsed_s": 2.406979079998564, "decode_tps": 109.7347816663895, "max_stream_gap_s": 0.01621198299835669, "events": 148, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"}
{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0725614070033771, "elapsed_s": 18.94562807300099, "decode_tps": 108.46144064588867, "max_stream_gap_s": 0.031187818000034895, "events": 1189, "sha256": "e35a9f765a93cce044543990fe166bfafe2cb845b0e603c334dc449496e21491"}
{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0843369200010784, "elapsed_s": 18.989241095994657, "decode_tps": 108.27878210561818, "max_stream_gap_s": 0.03129963100218447, "events": 1191, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.07209744000283536, "elapsed_s": 18.980188051005825, "decode_tps": 108.26053471569521, "max_stream_gap_s": 0.03122235600312706, "events": 1191, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126130, "completion_tokens": 135, "completion_tokens_details": {"reasoning_tokens": 115}}, "ttft_s": 14.440546402998734, "elapsed_s": 15.590460605999397, "decode_tps": 116.53043300998588, "max_stream_gap_s": 0.032008403999498114, "events": 71, "sha256": "ee0579233e57d5accc2cf35a71509aa83fe7dcf3a2727f10f973dec75ff41759"}
{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126129, "completion_tokens": 134, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 0.856336849006766, "elapsed_s": 1.9747926430063671, "decode_tps": 118.91395325012499, "max_stream_gap_s": 0.03192400800617179, "events": 69, "sha256": "9400cbc20a09d71e03a4c31d5413dd314ee75a387fce3dc146f230a763b8faf3"}
{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126128, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 113}}, "ttft_s": 0.853535173999262, "elapsed_s": 1.9229757680004695, "decode_tps": 123.4290158241842, "max_stream_gap_s": 0.03247794899652945, "events": 67, "sha256": "48e028596401908f642e64f951e3f0c5c7b6314e6a46a87c047b4728d9438d99"}
{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111863, "completion_tokens": 193, "completion_tokens_details": {"reasoning_tokens": 148}}, "ttft_s": 12.791386062002857, "elapsed_s": 14.462968651001574, "decode_tps": 114.86121072546499, "max_stream_gap_s": 0.03197040300437948, "events": 103, "sha256": "410a369097c639859051822f809cd9adecfca2ac1f489544f9525a65f25ef94a"}
{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08371570699819131, "elapsed_s": 0.5917828360034036, "decode_tps": 123.99936229559475, "max_stream_gap_s": 0.03208907500084024, "events": 32, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"}
{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.6603063000002294, "elapsed_s": 0.9943764119961997, "decode_tps": 125.72211189160979, "max_stream_gap_s": 0.03203328799281735, "events": 21, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"}
{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.0632171030010795, "elapsed_s": 1.5722418850054964, "decode_tps": 123.76607628398992, "max_stream_gap_s": 0.01634497399936663, "events": 32, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"}
{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.6414810709975427, "elapsed_s": 2.150085207002121, "decode_tps": 123.86843822173104, "max_stream_gap_s": 0.016239169002801646, "events": 32, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"}
SUITE_COMPLETE
@@ -0,0 +1,339 @@
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] █ █ █▄ ▄█
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 1}, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 2], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:54:04 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
(APIServer pid=1) INFO 09-18 14:54:04 [model.py:2024] Using max model len 131072
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:54:07 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:54:07 [model.py:691] Resolved architecture: Qwen4ExpMTP
(APIServer pid=1) INFO 09-18 14:54:07 [model.py:2024] Using max model len 262144
(APIServer pid=1) INFO 09-18 14:54:07 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072
(APIServer pid=1) INFO 09-18 14:54:07 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
(APIServer pid=1) INFO 09-18 14:54:07 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(APIServer pid=1) INFO 09-18 14:54:07 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
(APIServer pid=1) INFO 09-18 14:54:07 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) WARNING 09-18 14:54:07 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(APIServer pid=1) INFO 09-18 14:54:07 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 14:54:20 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=1), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 2, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) INFO 09-18 14:54:21 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_d75d110cee7b4c8d8fc41c851dee4d8c backend=nccl
(EngineCore pid=112) INFO 09-18 14:54:21 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
(EngineCore pid=112) INFO 09-18 14:54:21 [gpu_worker.py:441] Using V2 Model Runner
(EngineCore pid=112) INFO 09-18 14:54:22 [model_runner.py:387] Loading model from scratch...
(EngineCore pid=112) INFO 09-18 14:54:22 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
(EngineCore pid=112) INFO 09-18 14:54:22 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
(EngineCore pid=112) INFO 09-18 14:54:22 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
(EngineCore pid=112) INFO 09-18 14:54:22 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
(EngineCore pid=112) INFO 09-18 14:54:24 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
(APIServer pid=1) INFO 09-18 14:54:26 [base.py:261] Multi-modal warmup completed in 12.108s
(APIServer pid=1) INFO 09-18 14:54:27 [base.py:261] Readonly multi-modal warmup completed in 1.295s
(EngineCore pid=112) INFO 09-18 14:54:58 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
(EngineCore pid=112) INFO 09-18 14:54:58 [flash_attn.py:1115] Using FlashAttention version 2
(EngineCore pid=112) WARNING 09-18 14:54:58 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
(EngineCore pid=112) INFO 09-18 14:54:58 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.31 GiB.
(EngineCore pid=112) INFO 09-18 14:54:58 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:17, 1.73s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:07<00:35, 3.98s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:13<00:38, 4.80s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:18<00:36, 5.19s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:24<00:32, 5.45s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:30<00:28, 5.61s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:36<00:22, 5.67s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:42<00:17, 5.74s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:43<00:08, 4.41s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:44<00:00, 2.51s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:44<00:00, 4.06s/it]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 14:55:43 [default_loader.py:430] Loading weights took 44.78 seconds
(EngineCore pid=112) INFO 09-18 14:55:43 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) INFO 09-18 14:55:43 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(EngineCore pid=112) INFO 09-18 14:55:43 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(EngineCore pid=112) WARNING 09-18 14:55:43 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(EngineCore pid=112) INFO 09-18 14:55:43 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(EngineCore pid=112) INFO 09-18 14:55:44 [fp8.py:433] Using DEEPGEMM Fp8 MoE backend out of potential backends: ['AITER', 'FLASHINFER_TRTLLM', 'FLASHINFER_CUTLASS', 'DEEPGEMM', 'TRITON', 'MARLIN', 'HUMMING', 'BATCHED_DEEPGEMM', 'BATCHED_TRITON', 'XPU', 'CPU', 'HPC'].
(EngineCore pid=112) INFO 09-18 14:55:44 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.17 GiB.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:00<00:01, 5.16it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:00<00:01, 4.04it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:01<00:01, 3.65it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:01<00:01, 3.44it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:01<00:01, 3.32it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:01<00:01, 3.26it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:02<00:00, 3.22it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:02<00:00, 3.72it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.91it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.70it/s]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 14:55:47 [default_loader.py:430] Loading weights took 2.98 seconds
(EngineCore pid=112) INFO 09-18 14:55:47 [deep_gemm.py:196] deep_gemm not found in site-packages, trying vendored vllm.third_party.deep_gemm
(EngineCore pid=112) INFO 09-18 14:55:47 [deep_gemm.py:223] DeepGEMM PDL enabled on vllm.third_party.deep_gemm.
(EngineCore pid=112) INFO 09-18 14:55:47 [deep_gemm.py:136] DeepGEMM E8M0 enabled on current platform.
(EngineCore pid=112) INFO 09-18 14:55:50 [fp8.py:733] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) WARNING 09-18 14:55:50 [speculator.py:235] Draft model Qwen4ExpMTP does not support external multimodal embeddings. Embeddings from the target model will not be passed to the drafter; using text-only draft inputs instead.
(EngineCore pid=112) INFO 09-18 14:55:51 [model_runner.py:419] Model loading took 76.36 GiB memory and 89.100896 seconds
(EngineCore pid=112) INFO 09-18 14:55:51 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
(EngineCore pid=112) INFO 09-18 14:55:51 [interface.py:918] Setting attention block size to 3152 tokens to ensure that attention page size is >= mamba page size.
(EngineCore pid=112) INFO 09-18 14:55:51 [utils.py:320] Using BLNHC KV cache layout.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=112) INFO 09-18 14:55:55 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
(EngineCore pid=112) WARNING 09-18 14:56:17 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) WARNING 09-18 14:56:17 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.81s/it]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.81s/it]
(EngineCore pid=112) INFO 09-18 14:56:19 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 1.95it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 1.95it/s]
(EngineCore pid=112) INFO 09-18 14:56:20 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.14 GiB
(EngineCore pid=112) INFO 09-18 14:56:21 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB
(EngineCore pid=112) INFO 09-18 14:56:21 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9850 is equivalent to --gpu-memory-utilization=0.9833 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9867. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
(EngineCore pid=112) WARNING 09-18 14:56:21 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) INFO 09-18 14:56:21 [kv_cache_utils.py:2404] GPU KV cache size: 159,669 tokens, Maximum concurrency for 131,072 tokens per request: 1.22x
(EngineCore pid=112) INFO 09-18 14:56:21 [kernel_warmup.py:171] JIT kernel warmup starting.
(EngineCore pid=112) INFO 09-18 14:56:21 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
(EngineCore pid=112) INFO 09-18 14:56:21 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
(EngineCore pid=112) INFO 09-18 14:56:21 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
(EngineCore pid=112) INFO 09-18 14:56:21 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
(EngineCore pid=112) INFO 09-18 14:56:22 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1), (2, 1)).
(EngineCore pid=112) INFO 09-18 14:56:27 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
(EngineCore pid=112) INFO 09-18 14:56:27 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.68it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.67it/s]
(EngineCore pid=112) INFO 09-18 14:56:36 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 120.04it/s]
(EngineCore pid=112) INFO 09-18 14:56:36 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.10 GiB
(EngineCore pid=112) INFO 09-18 14:56:36 [gpu_worker.py:824] CUDA graph pool memory: 0.1 GiB (actual), 0.14 GiB (estimated), difference: 0.04 GiB (37.7%).
(EngineCore pid=112) INFO 09-18 14:56:36 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.985, 81.8 GiB). Actual usage is 77.82 GiB for consumed memory (weights + non-torch), 1.2 GiB for peak activation, and 0.1 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=2718838744` (2.53 GiB) to fit into requested memory, or `--kv-cache-memory=3565505536` (3.32 GiB) to fully utilize gpu memory. Current kv cache memory in use is 2.78 GiB.
(EngineCore pid=112) INFO 09-18 14:56:37 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
(EngineCore pid=112) WARNING 09-18 14:56:38 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
(EngineCore pid=112) INFO 09-18 14:56:38 [core.py:380] init engine (profile, create kv cache, warmup model) took 47.02 s
(EngineCore pid=112) INFO 09-18 14:56:38 [kv_cache_utils.py:762] kv cache group sizes [3152, 3152, 3152, 3152, 8, 3152]
(EngineCore pid=112) INFO 09-18 14:56:38 [kv_cache_utils.py:763] kv lcm block sizes 3152
(EngineCore pid=112) INFO 09-18 14:56:38 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) INFO 09-18 14:56:38 [entry.py:132] Supported tasks: ['generate']
(APIServer pid=1) INFO 09-18 14:56:38 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
(APIServer pid=1) INFO 09-18 14:56:38 [parser_manager.py:34] "auto" tool choice has been enabled.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:56:38 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
(APIServer pid=1) WARNING 09-18 14:56:38 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
(APIServer pid=1) INFO 09-18 14:56:38 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:60] Available routes are:
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /openapi.json, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /docs, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /redoc, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /load, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /version, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /health, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /metrics, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /tokenize, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /detokenize, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/models, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /ping, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /ping, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /invocations, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/responses, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/completions, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/messages, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /generative_scoring, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
(APIServer pid=1) INFO 09-18 14:56:38 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
(APIServer pid=1) INFO: Started server process [1]
(APIServer pid=1) INFO: Waiting for application startup.
(APIServer pid=1) INFO: Application startup complete.
(APIServer pid=1) INFO: 172.21.0.1:38360 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38376 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(EngineCore pid=112) WARNING 09-18 14:56:41 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 14:56:41 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _count_expert_num_tokens. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 14:56:41 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _compute_local_logits_stats_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 14:56:41 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _rejection_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 14:56:41 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _resample_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(APIServer pid=1) INFO: 172.21.0.1:38386 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38400 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:41714 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38404 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38416 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52596 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52608 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52624 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52626 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52628 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52642 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52646 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:56:49 [loggers.py:323] Engine 000: Avg prompt throughput: 64.7 tokens/s, Avg generation throughput: 73.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%
(APIServer pid=1) INFO 09-18 14:56:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.99, Accepted throughput: 36.01 tokens/s, Drafted throughput: 36.56 tokens/s, Accepted: 396 tokens, Drafted: 402 tokens, Per-position acceptance rate: 0.985, Avg Draft acceptance rate: 98.5%
(APIServer pid=1) INFO: 172.21.0.1:52658 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52666 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52672 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52688 - "POST /v1/chat/completions HTTP/1.1" 200 OK
[rank0]:[W918 14:56:53.391378074 CUDACachingAllocator.cpp:3933] memory allocation failed with OOM on device 0 while trying to allocate 377487360 bytes (free: 83820544, total: 89173131264).
(APIServer pid=1) INFO: 172.21.0.1:53008 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53016 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:56:59 [loggers.py:323] Engine 000: Avg prompt throughput: 2778.3 tokens/s, Avg generation throughput: 59.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:56:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.90, Accepted throughput: 28.10 tokens/s, Drafted throughput: 31.20 tokens/s, Accepted: 281 tokens, Drafted: 312 tokens, Per-position acceptance rate: 0.901, Avg Draft acceptance rate: 90.1%
(APIServer pid=1) INFO: 172.21.0.1:53018 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:09 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 109.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.75, Accepted throughput: 46.60 tokens/s, Drafted throughput: 62.40 tokens/s, Accepted: 466 tokens, Drafted: 624 tokens, Per-position acceptance rate: 0.747, Avg Draft acceptance rate: 74.7%
(APIServer pid=1) INFO: 127.0.0.1:34068 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 108.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.72, Accepted throughput: 45.60 tokens/s, Drafted throughput: 62.90 tokens/s, Accepted: 456 tokens, Drafted: 629 tokens, Per-position acceptance rate: 0.725, Avg Draft acceptance rate: 72.5%
(APIServer pid=1) INFO: 172.21.0.1:44752 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:29 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 111.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.77, Accepted throughput: 48.40 tokens/s, Drafted throughput: 62.60 tokens/s, Accepted: 484 tokens, Drafted: 626 tokens, Per-position acceptance rate: 0.773, Avg Draft acceptance rate: 77.3%
(APIServer pid=1) INFO: 172.21.0.1:55170 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:39 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 107.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.72, Accepted throughput: 44.80 tokens/s, Drafted throughput: 62.40 tokens/s, Accepted: 448 tokens, Drafted: 624 tokens, Per-position acceptance rate: 0.718, Avg Draft acceptance rate: 71.8%
(APIServer pid=1) INFO: 127.0.0.1:39416 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 111.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.77, Accepted throughput: 48.30 tokens/s, Drafted throughput: 63.00 tokens/s, Accepted: 483 tokens, Drafted: 630 tokens, Per-position acceptance rate: 0.767, Avg Draft acceptance rate: 76.7%
(APIServer pid=1) INFO: 172.21.0.1:49960 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:49968 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:57:59 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 105.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:57:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.71, Accepted throughput: 43.90 tokens/s, Drafted throughput: 61.80 tokens/s, Accepted: 439 tokens, Drafted: 618 tokens, Per-position acceptance rate: 0.710, Avg Draft acceptance rate: 71.0%
(APIServer pid=1) INFO 09-18 14:58:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 113.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.80, Accepted throughput: 50.20 tokens/s, Drafted throughput: 63.10 tokens/s, Accepted: 502 tokens, Drafted: 631 tokens, Per-position acceptance rate: 0.796, Avg Draft acceptance rate: 79.6%
(APIServer pid=1) INFO: 127.0.0.1:41032 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35070 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:58:19 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 110.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.77, Accepted throughput: 48.20 tokens/s, Drafted throughput: 62.50 tokens/s, Accepted: 482 tokens, Drafted: 625 tokens, Per-position acceptance rate: 0.771, Avg Draft acceptance rate: 77.1%
(APIServer pid=1) INFO 09-18 14:58:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 113.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.80, Accepted throughput: 50.49 tokens/s, Drafted throughput: 63.09 tokens/s, Accepted: 505 tokens, Drafted: 631 tokens, Per-position acceptance rate: 0.800, Avg Draft acceptance rate: 80.0%
(APIServer pid=1) INFO: 172.21.0.1:36238 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:58:39 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 110.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.77, Accepted throughput: 48.00 tokens/s, Drafted throughput: 62.60 tokens/s, Accepted: 480 tokens, Drafted: 626 tokens, Per-position acceptance rate: 0.767, Avg Draft acceptance rate: 76.7%
(APIServer pid=1) INFO: 127.0.0.1:47540 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:58:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 113.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.81, Accepted throughput: 50.80 tokens/s, Drafted throughput: 63.09 tokens/s, Accepted: 508 tokens, Drafted: 631 tokens, Per-position acceptance rate: 0.805, Avg Draft acceptance rate: 80.5%
(APIServer pid=1) INFO: 172.21.0.1:46184 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:34866 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:58:59 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 109.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:58:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.75, Accepted throughput: 46.80 tokens/s, Drafted throughput: 62.10 tokens/s, Accepted: 468 tokens, Drafted: 621 tokens, Per-position acceptance rate: 0.754, Avg Draft acceptance rate: 75.4%
(APIServer pid=1) INFO 09-18 14:59:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 108.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.71, Accepted throughput: 45.00 tokens/s, Drafted throughput: 63.00 tokens/s, Accepted: 450 tokens, Drafted: 630 tokens, Per-position acceptance rate: 0.714, Avg Draft acceptance rate: 71.4%
(APIServer pid=1) INFO: 127.0.0.1:32954 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:54998 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:59:19 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 108.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.74, Accepted throughput: 46.20 tokens/s, Drafted throughput: 62.50 tokens/s, Accepted: 462 tokens, Drafted: 625 tokens, Per-position acceptance rate: 0.739, Avg Draft acceptance rate: 73.9%
(APIServer pid=1) INFO 09-18 14:59:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 107.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.71, Accepted throughput: 44.49 tokens/s, Drafted throughput: 62.99 tokens/s, Accepted: 445 tokens, Drafted: 630 tokens, Per-position acceptance rate: 0.706, Avg Draft acceptance rate: 70.6%
(APIServer pid=1) INFO: 172.21.0.1:48116 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:59:39 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 108.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.74, Accepted throughput: 46.10 tokens/s, Drafted throughput: 62.50 tokens/s, Accepted: 461 tokens, Drafted: 625 tokens, Per-position acceptance rate: 0.738, Avg Draft acceptance rate: 73.8%
(APIServer pid=1) INFO: 127.0.0.1:53944 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:59:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 107.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.70, Accepted throughput: 44.30 tokens/s, Drafted throughput: 63.00 tokens/s, Accepted: 443 tokens, Drafted: 630 tokens, Per-position acceptance rate: 0.703, Avg Draft acceptance rate: 70.3%
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
(APIServer pid=1) INFO: 172.21.0.1:43784 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43794 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43806 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43814 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43818 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40940 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40946 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40952 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40964 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40968 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40970 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40984 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40986 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:41002 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:41010 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:59:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 43.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 39.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:59:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.72, Accepted throughput: 18.10 tokens/s, Drafted throughput: 25.00 tokens/s, Accepted: 181 tokens, Drafted: 250 tokens, Per-position acceptance rate: 0.724, Avg Draft acceptance rate: 72.4%
(APIServer pid=1) INFO 09-18 15:00:09 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 0.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 74.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:00:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.00, Accepted throughput: 0.30 tokens/s, Drafted throughput: 0.30 tokens/s, Accepted: 3 tokens, Drafted: 3 tokens, Per-position acceptance rate: 1.000, Avg Draft acceptance rate: 100.0%
(APIServer pid=1) INFO: 172.21.0.1:52464 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52476 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:57076 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:52546 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:00:19 [loggers.py:323] Engine 000: Avg prompt throughput: 1243.7 tokens/s, Avg generation throughput: 39.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 42.4%, Prefix cache hit rate: 46.2%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:00:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.92, Accepted throughput: 18.90 tokens/s, Drafted throughput: 20.60 tokens/s, Accepted: 189 tokens, Drafted: 206 tokens, Per-position acceptance rate: 0.917, Avg Draft acceptance rate: 91.7%
(APIServer pid=1) INFO: 172.21.0.1:38136 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38144 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38150 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38158 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:00:29 [loggers.py:323] Engine 000: Avg prompt throughput: 11171.3 tokens/s, Avg generation throughput: 24.7 tokens/s, Running: 1 reqs, Waiting: 3 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.2%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:00:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.87, Accepted throughput: 11.40 tokens/s, Drafted throughput: 13.10 tokens/s, Accepted: 114 tokens, Drafted: 131 tokens, Per-position acceptance rate: 0.870, Avg Draft acceptance rate: 87.0%
(APIServer pid=1) INFO: 172.21.0.1:38172 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38188 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:34860 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:34870 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:00:39 [loggers.py:323] Engine 000: Avg prompt throughput: 58.8 tokens/s, Avg generation throughput: 114.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.1%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:00:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.95, Accepted throughput: 55.30 tokens/s, Drafted throughput: 58.40 tokens/s, Accepted: 553 tokens, Drafted: 584 tokens, Per-position acceptance rate: 0.947, Avg Draft acceptance rate: 94.7%
(APIServer pid=1) INFO: 127.0.0.1:44446 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:34880 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:00:49 [loggers.py:323] Engine 000: Avg prompt throughput: 9.2 tokens/s, Avg generation throughput: 117.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.1%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:00:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.89, Accepted throughput: 55.10 tokens/s, Drafted throughput: 62.10 tokens/s, Accepted: 551 tokens, Drafted: 621 tokens, Per-position acceptance rate: 0.887, Avg Draft acceptance rate: 88.7%
(APIServer pid=1) INFO: 172.21.0.1:34804 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:34816 - "POST /v1/chat/completions HTTP/1.1" 200 OK
@@ -0,0 +1,5 @@
{
"startup": true,
"records": 44,
"failures": []
}
@@ -0,0 +1,46 @@
name: qwen38-flash-rtx6000d
services:
vllm:
image: local/qwen38-flash-6000d:0bfc7a15
build: .
container_name: qwen38-flash-6000d
gpus: all
shm_size: 8g
mem_limit: 210g
memswap_limit: 210g
restart: "no"
ports:
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
environment:
HF_HUB_OFFLINE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_WORKER_MULTIPROC_METHOD: spawn
OMP_NUM_THREADS: "8"
TORCHINDUCTOR_COMPILE_THREADS: "2"
CUTE_DSL_ARCH: sm_120a
volumes:
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
- /data/flash-next/cache:/root/.cache
- ./secrets/api-key:/run/secrets/api-key:ro
entrypoint: ["/bin/bash", "-lc"]
command:
- >-
exec vllm serve /model
--served-model-name qwen3.8-flash-next
--host 0.0.0.0 --port 8000
--tensor-parallel-size 1 --dtype bfloat16
--engram-config '{"cpu_offload":true}'
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
--enable-chunked-prefill --enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":2}'
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune
--load-format safetensors
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
--api-key "$$(cat /run/secrets/api-key)"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30m
@@ -0,0 +1,26 @@
# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_total counter
vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 52554.0
# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_created gauge
vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528008e+09
# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_total counter
vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 105108.0
# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_created gauge
vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528184e+09
# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_total counter
vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 69785.0
# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_created gauge
vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528296e+09
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 40257.0
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 29528.0
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897410011528459e+09
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.78974100115285e+09
@@ -0,0 +1,7 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2711202999998932, "response": {"id": "chatcmpl-acf4252c63733754", "object": "chat.completion", "created": 1789743196, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.7527402949999669, "response": {"id": "chatcmpl-82388a5e0c331239", "object": "chat.completion", "created": 1789743198, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "schedule", "passed": true, "elapsed_s": 2.0548892629958573, "response": {"id": "chatcmpl-b59f8d39ded5a629", "object": "chat.completion", "created": 1789743199, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.3326087149980594, "response": {"id": "chatcmpl-8e1cd6b923cb571c", "object": "chat.completion", "created": 1789743201, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-848dfe0da82871ae", "object": "chat.completion", "created": 1789743203, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-acbd27655cbcc931", "object": "chat.completion", "created": 1789743210, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-aa08e51120ea5d43", "object": "chat.completion", "created": 1789743212, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-80911c02057f9d71", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
@@ -0,0 +1,8 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2711202999998932}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.7527402949999669}
{"name": "schedule", "passed": true, "elapsed_s": 2.0548892629958573}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.3326087149980594}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "tool-arguments", "passed": true}
QUALITY_COMPLETE
File diff suppressed because one or more lines are too long
@@ -0,0 +1,38 @@
{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08033495700510684, "elapsed_s": 0.5909661660043639, "decode_tps": 158.62720212253586, "max_stream_gap_s": 0.018763815001875628, "events": 29, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"}
{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08595889700518455, "elapsed_s": 0.3043062380020274, "decode_tps": 164.87491826392582, "max_stream_gap_s": 0.01860671999747865, "events": 13, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"}
{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08532555599958869, "elapsed_s": 0.5077504829969257, "decode_tps": 153.87349525519303, "max_stream_gap_s": 0.01870621599664446, "events": 24, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"}
{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08707080200110795, "elapsed_s": 0.32353630300349323, "decode_tps": 160.69997457944905, "max_stream_gap_s": 0.01859223899373319, "events": 14, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"}
{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08545495500584366, "elapsed_s": 0.5441159190013423, "decode_tps": 154.798436259984, "max_stream_gap_s": 0.018821514000592288, "events": 26, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"}
{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08643778399709845, "elapsed_s": 0.34146275900275214, "decode_tps": 160.7685678592501, "max_stream_gap_s": 0.018947848002426326, "events": 15, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"}
{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08607953199680196, "elapsed_s": 0.5258115449978504, "decode_tps": 159.18786426821532, "max_stream_gap_s": 0.018670357996597886, "events": 25, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"}
{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08593622500484344, "elapsed_s": 0.6902767670035246, "decode_tps": 153.8867468537349, "max_stream_gap_s": 0.018791513997712173, "events": 33, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"}
{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08543285400082823, "elapsed_s": 0.6162005010046414, "decode_tps": 158.2613418021625, "max_stream_gap_s": 0.018708337993302848, "events": 29, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"}
{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08579609300068114, "elapsed_s": 0.6164905580008053, "decode_tps": 156.39884241110482, "max_stream_gap_s": 0.018804239996825345, "events": 29, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"}
{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08569144899956882, "elapsed_s": 0.5975101969961543, "decode_tps": 160.21296664292387, "max_stream_gap_s": 0.018605278004542924, "events": 29, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"}
{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08546528800070519, "elapsed_s": 0.6353967890026979, "decode_tps": 161.8383377526822, "max_stream_gap_s": 0.018794105999404564, "events": 31, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"}
{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08568278700113297, "elapsed_s": 0.37689909800246824, "decode_tps": 144.2226908773919, "max_stream_gap_s": 0.018781819999276195, "events": 17, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"}
{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08570619000238366, "elapsed_s": 0.5379072569994605, "decode_tps": 137.10715105499912, "max_stream_gap_s": 0.018587483995361254, "events": 26, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"}
{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3276, "completion_tokens": 141, "completion_tokens_details": {"reasoning_tokens": 102}}, "ttft_s": 0.37922176600113744, "elapsed_s": 1.3786603870030376, "decode_tps": 140.07863720500933, "max_stream_gap_s": 0.01991734900366282, "events": 54, "sha256": "9e3c14a4e0973d378dc3fda85c529bfed85154aefd90d72c8313c78df80f6564"}
{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12213, "completion_tokens": 144, "completion_tokens_details": {"reasoning_tokens": 130}}, "ttft_s": 0.8094746360002318, "elapsed_s": 1.7980530589993577, "decode_tps": 144.65215573507055, "max_stream_gap_s": 0.018804600993462373, "events": 55, "sha256": "40374d1d01e9d30a2e10dfcf6ba861c98b6c27c52bed2fc369e82286625e2f93"}
{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12433, "completion_tokens": 77, "completion_tokens_details": {"reasoning_tokens": 66}}, "ttft_s": 1.4533180090002134, "elapsed_s": 1.9994508850068087, "decode_tps": 139.16027278145802, "max_stream_gap_s": 0.018682743997487705, "events": 31, "sha256": "6948cfaa73343b43893c8bbd5a7faf45ed78d55ac8ae50d67b8b890a8d1bf6c7"}
{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08620800900098402, "elapsed_s": 2.189102151998668, "decode_tps": 121.26145333996531, "max_stream_gap_s": 0.018632254999829456, "events": 117, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"}
{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07210104499972658, "elapsed_s": 16.617303178005386, "decode_tps": 123.72166768011161, "max_stream_gap_s": 0.019507625998812728, "events": 920, "sha256": "4b5a99c6a2d74362c14314126d243ab0ea75f0d203a637e5635b96bc429e3afb"}
{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07450222100305837, "elapsed_s": 16.60604691500339, "decode_tps": 123.82387961258709, "max_stream_gap_s": 0.018462731000909116, "events": 922, "sha256": "c6a65a378e367e11250c85739caf6c8d1093bd987b88899767d97d623c15bcb9"}
{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08693470500293188, "elapsed_s": 16.638716553999984, "decode_tps": 123.67248545654539, "max_stream_gap_s": 0.018704524001805112, "events": 923, "sha256": "c6aab7182e88cf796f5b16f8b786142eb9521c9123cb0a049524dde6362b299f"}
{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07400395700096851, "elapsed_s": 1.9887343450027402, "decode_tps": 133.17801900356326, "max_stream_gap_s": 0.0185166260052938, "events": 108, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"}
{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08585922900238074, "elapsed_s": 15.984570784006792, "decode_tps": 128.75257173627188, "max_stream_gap_s": 0.01882019199547358, "events": 889, "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"}
{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07313994099968113, "elapsed_s": 15.988134749000892, "decode_tps": 128.62083994968555, "max_stream_gap_s": 0.018578499999421183, "events": 890, "sha256": "8fc19ac2a6425782e63b5c26dd326f7124bce54fc957913963c9fc6ae4ee3341"}
{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08644359100435395, "elapsed_s": 15.966877792001469, "decode_tps": 128.90075762988087, "max_stream_gap_s": 0.018619577000208665, "events": 888, "sha256": "488769996df35d3bbedd2a91046c0a137677097523f73433b0919b61c68071b5"}
{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.07434646700130543, "elapsed_s": 2.126096015002986, "decode_tps": 124.28417505848094, "max_stream_gap_s": 0.01840589300263673, "events": 116, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"}
{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07352274699951522, "elapsed_s": 16.837296222998702, "decode_tps": 122.10854572395078, "max_stream_gap_s": 0.018720656000368763, "events": 935, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"}
{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07170502199733164, "elapsed_s": 16.85341221699491, "decode_tps": 121.97805480780796, "max_stream_gap_s": 0.01876352399995085, "events": 936, "sha256": "8a7cb2cde52eee53562fd48aca68d5267dceb315e35b12b86822b67066eacf58"}
{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07260531799693126, "elapsed_s": 16.818446777993813, "decode_tps": 122.23930370354654, "max_stream_gap_s": 0.01878616500471253, "events": 934, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"}
{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126111, "completion_tokens": 116, "completion_tokens_details": {"reasoning_tokens": 96}}, "ttft_s": 14.42599186499865, "elapsed_s": 15.14422588799789, "decode_tps": 160.11494348287312, "max_stream_gap_s": 0.01879984400147805, "events": 41, "sha256": "e054eb4577e203d59776c06555ff8dff56d50e49389606fa8a7344884f985155"}
{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7020562550023897, "elapsed_s": 1.4927438169979723, "decode_tps": 149.23720274818137, "max_stream_gap_s": 0.01888078500633128, "events": 45, "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"}
{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7015097360053915, "elapsed_s": 1.4736580300013884, "decode_tps": 152.82038556263618, "max_stream_gap_s": 0.018826871004421264, "events": 44, "sha256": "1275326ebe3452473a77c76ba9ec5b91f06f8896ffd6de203abf1f9129842a90"}
{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111902, "completion_tokens": 232, "completion_tokens_details": {"reasoning_tokens": 187}}, "ttft_s": 12.815331379999407, "elapsed_s": 14.372689276002347, "decode_tps": 148.32813998174495, "max_stream_gap_s": 0.01881139999750303, "events": 87, "sha256": "4b7c8fb1d0d992dda6a5a680074a3aa037b1c33fd732b7315720e712693c2335"}
{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08533696200174745, "elapsed_s": 0.4814725750038633, "decode_tps": 159.0364459346489, "max_stream_gap_s": 0.018437628001265693, "events": 22, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"}
{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5489882670008228, "elapsed_s": 0.8192126439971616, "decode_tps": 155.42639219617496, "max_stream_gap_s": 0.018371916994510684, "events": 15, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"}
{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8859058559974073, "elapsed_s": 1.2823282210010802, "decode_tps": 158.92140696808642, "max_stream_gap_s": 0.018409249001706485, "events": 22, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"}
{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3495687850008835, "elapsed_s": 1.743830876002903, "decode_tps": 159.79218250449873, "max_stream_gap_s": 0.018336048000492156, "events": 23, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"}
SUITE_COMPLETE
@@ -0,0 +1,730 @@
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] █ █ █▄ ▄█
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 2}, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:14:09 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
(APIServer pid=1) INFO 09-18 14:14:09 [model.py:2024] Using max model len 131072
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:14:13 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:14:13 [model.py:691] Resolved architecture: Qwen4ExpMTP
(APIServer pid=1) INFO 09-18 14:14:13 [model.py:2024] Using max model len 262144
(APIServer pid=1) WARNING 09-18 14:14:13 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate
(APIServer pid=1) INFO 09-18 14:14:13 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072
(APIServer pid=1) INFO 09-18 14:14:13 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
(APIServer pid=1) INFO 09-18 14:14:13 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(APIServer pid=1) INFO 09-18 14:14:13 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
(APIServer pid=1) INFO 09-18 14:14:13 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) WARNING 09-18 14:14:13 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(APIServer pid=1) INFO 09-18 14:14:13 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=113) INFO 09-18 14:14:25 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=2), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 3, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=113) INFO 09-18 14:14:26 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_ed11f4f25e124a16965c184de0adfebf backend=nccl
(EngineCore pid=113) INFO 09-18 14:14:26 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
(EngineCore pid=113) INFO 09-18 14:14:26 [gpu_worker.py:441] Using V2 Model Runner
(EngineCore pid=113) INFO 09-18 14:14:27 [model_runner.py:387] Loading model from scratch...
(EngineCore pid=113) INFO 09-18 14:14:27 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
(EngineCore pid=113) INFO 09-18 14:14:27 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
(EngineCore pid=113) INFO 09-18 14:14:27 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
(EngineCore pid=113) INFO 09-18 14:14:27 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
(EngineCore pid=113) INFO 09-18 14:14:29 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
(APIServer pid=1) INFO 09-18 14:14:31 [base.py:261] Multi-modal warmup completed in 12.320s
(APIServer pid=1) INFO 09-18 14:14:33 [base.py:261] Readonly multi-modal warmup completed in 1.302s
(EngineCore pid=113) INFO 09-18 14:15:03 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
(EngineCore pid=113) INFO 09-18 14:15:03 [flash_attn.py:1115] Using FlashAttention version 2
(EngineCore pid=113) WARNING 09-18 14:15:03 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
(EngineCore pid=113) INFO 09-18 14:15:03 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.31 GiB.
(EngineCore pid=113) INFO 09-18 14:15:03 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
(EngineCore pid=113)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:16, 1.63s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:06<00:33, 3.74s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:12<00:36, 4.52s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:17<00:34, 4.87s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:23<00:30, 5.11s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:28<00:26, 5.26s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:34<00:21, 5.36s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:40<00:16, 5.47s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:41<00:08, 4.20s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 2.38s/it]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 3.83s/it]
(EngineCore pid=113)
(EngineCore pid=113) INFO 09-18 14:15:45 [default_loader.py:430] Loading weights took 42.29 seconds
(EngineCore pid=113) INFO 09-18 14:15:45 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=113) INFO 09-18 14:15:46 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(EngineCore pid=113) INFO 09-18 14:15:46 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(EngineCore pid=113) WARNING 09-18 14:15:46 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(EngineCore pid=113) INFO 09-18 14:15:46 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(EngineCore pid=113) INFO 09-18 14:15:46 [fp8.py:433] Using DEEPGEMM Fp8 MoE backend out of potential backends: ['AITER', 'FLASHINFER_TRTLLM', 'FLASHINFER_CUTLASS', 'DEEPGEMM', 'TRITON', 'MARLIN', 'HUMMING', 'BATCHED_DEEPGEMM', 'BATCHED_TRITON', 'XPU', 'CPU', 'HPC'].
(EngineCore pid=113) INFO 09-18 14:15:47 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.17 GiB.
(EngineCore pid=113)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:00<00:01, 5.13it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:00<00:01, 4.03it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:01<00:01, 3.63it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:01<00:01, 3.41it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:01<00:01, 3.30it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:02<00:01, 3.24it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:02<00:00, 3.19it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:02<00:00, 3.69it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.89it/s]
(EngineCore pid=113)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.68it/s]
(EngineCore pid=113)
(EngineCore pid=113) INFO 09-18 14:15:50 [default_loader.py:430] Loading weights took 3.00 seconds
(EngineCore pid=113) INFO 09-18 14:15:50 [deep_gemm.py:196] deep_gemm not found in site-packages, trying vendored vllm.third_party.deep_gemm
(EngineCore pid=113) INFO 09-18 14:15:50 [deep_gemm.py:223] DeepGEMM PDL enabled on vllm.third_party.deep_gemm.
(EngineCore pid=113) INFO 09-18 14:15:50 [deep_gemm.py:136] DeepGEMM E8M0 enabled on current platform.
(EngineCore pid=113) INFO 09-18 14:15:52 [fp8.py:733] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=113) WARNING 09-18 14:15:52 [speculator.py:235] Draft model Qwen4ExpMTP does not support external multimodal embeddings. Embeddings from the target model will not be passed to the drafter; using text-only draft inputs instead.
(EngineCore pid=113) INFO 09-18 14:15:53 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.461433 seconds
(EngineCore pid=113) INFO 09-18 14:15:53 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
(EngineCore pid=113) INFO 09-18 14:15:53 [interface.py:918] Setting attention block size to 3184 tokens to ensure that attention page size is >= mamba page size.
(EngineCore pid=113) INFO 09-18 14:15:53 [interface.py:942] Padding mamba page size by 0.38% to ensure that mamba page size and attention page size are exactly equal.
(EngineCore pid=113) INFO 09-18 14:15:53 [utils.py:320] Using BLNHC KV cache layout.
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
(EngineCore pid=113) INFO 09-18 14:15:58 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
(EngineCore pid=113) WARNING 09-18 14:16:20 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=113) WARNING 09-18 14:16:20 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
(EngineCore pid=113) INFO 09-18 14:16:20 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps.
(EngineCore pid=113)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.83s/it]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.83s/it]
(EngineCore pid=113) INFO 09-18 14:16:22 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=113)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.17it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.17it/s]
(EngineCore pid=113)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.69it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.69it/s]
(EngineCore pid=113) INFO 09-18 14:16:23 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.14 GiB
(EngineCore pid=113) INFO 09-18 14:16:24 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB
(EngineCore pid=113) INFO 09-18 14:16:24 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9850 is equivalent to --gpu-memory-utilization=0.9833 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9867. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
(EngineCore pid=113) WARNING 09-18 14:16:24 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=113) INFO 09-18 14:16:24 [kv_cache_utils.py:2404] GPU KV cache size: 146,622 tokens, Maximum concurrency for 131,072 tokens per request: 1.12x
(EngineCore pid=113) INFO 09-18 14:16:24 [kernel_warmup.py:171] JIT kernel warmup starting.
(EngineCore pid=113) INFO 09-18 14:16:24 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
(EngineCore pid=113) INFO 09-18 14:16:24 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
(EngineCore pid=113) INFO 09-18 14:16:24 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
(EngineCore pid=113) INFO 09-18 14:16:24 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
(EngineCore pid=113) INFO 09-18 14:16:25 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1), (2, 1), (3, 1)).
(EngineCore pid=113) INFO 09-18 14:16:30 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
(EngineCore pid=113) INFO 09-18 14:16:30 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
(EngineCore pid=113)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.91it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.89it/s]
(EngineCore pid=113) INFO 09-18 14:16:39 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=113)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 118.66it/s]
(EngineCore pid=113)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 176.98it/s]
(EngineCore pid=113) INFO 09-18 14:16:39 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.13 GiB
(EngineCore pid=113) INFO 09-18 14:16:39 [gpu_worker.py:824] CUDA graph pool memory: 0.13 GiB (actual), 0.14 GiB (estimated), difference: 0.01 GiB (5.8%).
(EngineCore pid=113) INFO 09-18 14:16:39 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.985, 81.8 GiB). Actual usage is 77.85 GiB for consumed memory (weights + non-torch), 1.17 GiB for peak activation, and 0.13 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=2685284312` (2.5 GiB) to fit into requested memory, or `--kv-cache-memory=3531951104` (3.29 GiB) to fully utilize gpu memory. Current kv cache memory in use is 2.78 GiB.
(EngineCore pid=113) INFO 09-18 14:16:40 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
(EngineCore pid=113) WARNING 09-18 14:16:41 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
(EngineCore pid=113) INFO 09-18 14:16:41 [core.py:380] init engine (profile, create kv cache, warmup model) took 47.40 s
(EngineCore pid=113) INFO 09-18 14:16:41 [kv_cache_utils.py:762] kv cache group sizes [3184, 3184, 3184, 3184, 8, 3184]
(EngineCore pid=113) INFO 09-18 14:16:41 [kv_cache_utils.py:763] kv lcm block sizes 3184
(EngineCore pid=113) INFO 09-18 14:16:41 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) INFO 09-18 14:16:41 [entry.py:132] Supported tasks: ['generate']
(APIServer pid=1) INFO 09-18 14:16:41 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
(APIServer pid=1) INFO 09-18 14:16:41 [parser_manager.py:34] "auto" tool choice has been enabled.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 14:16:41 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
(APIServer pid=1) WARNING 09-18 14:16:41 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
(APIServer pid=1) INFO 09-18 14:16:41 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:60] Available routes are:
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /openapi.json, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /docs, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /redoc, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /load, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /version, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /health, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /metrics, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /tokenize, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /detokenize, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/models, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /ping, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /ping, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /invocations, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/responses, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/completions, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/messages, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /generative_scoring, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
(APIServer pid=1) INFO 09-18 14:16:41 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
(APIServer pid=1) INFO: Started server process [1]
(APIServer pid=1) INFO: Waiting for application startup.
(APIServer pid=1) INFO: Application startup complete.
(APIServer pid=1) INFO: 127.0.0.1:52136 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36256 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36262 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(EngineCore pid=113) WARNING 09-18 14:16:46 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=113) WARNING 09-18 14:16:46 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _count_expert_num_tokens. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=113) WARNING 09-18 14:16:46 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _compute_local_logits_stats_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=113) WARNING 09-18 14:16:46 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _rejection_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=113) WARNING 09-18 14:16:47 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _resample_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(APIServer pid=1) INFO: 172.21.0.1:36278 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36282 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36290 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36304 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36306 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36310 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36316 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36332 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36342 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36352 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:16:52 [loggers.py:323] Engine 000: Avg prompt throughput: 53.5 tokens/s, Avg generation throughput: 60.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
(APIServer pid=1) INFO 09-18 14:16:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.91, Accepted throughput: 39.90 tokens/s, Drafted throughput: 41.72 tokens/s, Accepted: 440 tokens, Drafted: 460 tokens, Per-position acceptance rate: 0.987, 0.926, Avg Draft acceptance rate: 95.7%
(APIServer pid=1) INFO: 172.21.0.1:36364 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36378 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36380 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33586 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33600 - "POST /v1/chat/completions HTTP/1.1" 200 OK
[rank0]:[W918 14:16:56.003832297 CUDACachingAllocator.cpp:3933] memory allocation failed with OOM on device 0 while trying to allocate 377487360 bytes (free: 64946176, total: 89173131264).
(APIServer pid=1) INFO: 172.21.0.1:33614 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33630 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:17:02 [loggers.py:323] Engine 000: Avg prompt throughput: 2790.6 tokens/s, Avg generation throughput: 72.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.72, Accepted throughput: 45.70 tokens/s, Drafted throughput: 53.00 tokens/s, Accepted: 457 tokens, Drafted: 530 tokens, Per-position acceptance rate: 0.925, 0.800, Avg Draft acceptance rate: 86.2%
(APIServer pid=1) INFO: 172.21.0.1:33642 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:17:12 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 124.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 69.90 tokens/s, Drafted throughput: 110.19 tokens/s, Accepted: 699 tokens, Drafted: 1102 tokens, Per-position acceptance rate: 0.737, 0.532, Avg Draft acceptance rate: 63.4%
(APIServer pid=1) INFO: 127.0.0.1:35366 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43852 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:17:22 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 118.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.14, Accepted throughput: 62.90 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 629 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.707, 0.433, Avg Draft acceptance rate: 57.0%
(APIServer pid=1) INFO 09-18 14:17:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 72.60 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 726 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.758, 0.543, Avg Draft acceptance rate: 65.1%
(APIServer pid=1) INFO: 172.21.0.1:44946 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:17:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 119.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 64.09 tokens/s, Drafted throughput: 110.38 tokens/s, Accepted: 641 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.707, 0.455, Avg Draft acceptance rate: 58.1%
(APIServer pid=1) INFO: 127.0.0.1:40726 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:17:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 123.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:17:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 68.10 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 681 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.727, 0.496, Avg Draft acceptance rate: 61.1%
(APIServer pid=1) INFO: 172.21.0.1:41494 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51114 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:18:02 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 124.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 69.80 tokens/s, Drafted throughput: 109.80 tokens/s, Accepted: 698 tokens, Drafted: 1098 tokens, Per-position acceptance rate: 0.754, 0.517, Avg Draft acceptance rate: 63.6%
(APIServer pid=1) INFO: 172.21.0.1:53922 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:18:12 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 129.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.50 tokens/s, Drafted throughput: 111.00 tokens/s, Accepted: 735 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.766, 0.559, Avg Draft acceptance rate: 66.2%
(APIServer pid=1) INFO: 127.0.0.1:57802 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:18:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.28, Accepted throughput: 71.50 tokens/s, Drafted throughput: 111.80 tokens/s, Accepted: 715 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.758, 0.521, Avg Draft acceptance rate: 64.0%
(APIServer pid=1) INFO: 172.21.0.1:36444 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:18:32 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 128.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.31, Accepted throughput: 72.70 tokens/s, Drafted throughput: 110.80 tokens/s, Accepted: 727 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.765, 0.547, Avg Draft acceptance rate: 65.6%
(APIServer pid=1) INFO 09-18 14:18:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 129.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.70 tokens/s, Drafted throughput: 111.79 tokens/s, Accepted: 737 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.764, 0.555, Avg Draft acceptance rate: 65.9%
(APIServer pid=1) INFO: 172.21.0.1:38818 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:34668 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40728 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:18:52 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 120.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:18:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 65.80 tokens/s, Drafted throughput: 109.99 tokens/s, Accepted: 658 tokens, Drafted: 1100 tokens, Per-position acceptance rate: 0.709, 0.487, Avg Draft acceptance rate: 59.8%
(APIServer pid=1) INFO 09-18 14:19:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 122.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 67.00 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 670 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.716, 0.487, Avg Draft acceptance rate: 60.1%
(APIServer pid=1) INFO: 172.21.0.1:50984 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:19:12 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 119.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 64.20 tokens/s, Drafted throughput: 110.80 tokens/s, Accepted: 642 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.702, 0.457, Avg Draft acceptance rate: 57.9%
(APIServer pid=1) INFO: 127.0.0.1:57380 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37540 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:19:22 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 122.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 67.59 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 676 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.714, 0.508, Avg Draft acceptance rate: 61.1%
(APIServer pid=1) INFO 09-18 14:19:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.17, Accepted throughput: 65.50 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 655 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.710, 0.464, Avg Draft acceptance rate: 58.7%
(APIServer pid=1) INFO: 172.21.0.1:43060 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:19:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.5 tokens/s, Avg generation throughput: 124.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.26, Accepted throughput: 69.60 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 696 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.741, 0.517, Avg Draft acceptance rate: 62.9%
(APIServer pid=1) INFO: 127.0.0.1:45568 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:19:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 137.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:19:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.47, Accepted throughput: 81.90 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 819 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.831, 0.642, Avg Draft acceptance rate: 73.7%
(APIServer pid=1) INFO 09-18 14:20:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 150.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.71, Accepted throughput: 95.00 tokens/s, Drafted throughput: 110.80 tokens/s, Accepted: 950 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.919, 0.796, Avg Draft acceptance rate: 85.7%
(APIServer pid=1) INFO 09-18 14:20:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 145.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.63, Accepted throughput: 90.20 tokens/s, Drafted throughput: 110.79 tokens/s, Accepted: 902 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.888, 0.740, Avg Draft acceptance rate: 81.4%
(APIServer pid=1) INFO: 127.0.0.1:41446 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:20:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 138.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.51, Accepted throughput: 83.29 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 833 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.837, 0.669, Avg Draft acceptance rate: 75.3%
(APIServer pid=1) INFO 09-18 14:20:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 134.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.43, Accepted throughput: 79.10 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 791 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.813, 0.620, Avg Draft acceptance rate: 71.6%
(APIServer pid=1) INFO 09-18 14:20:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 134.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.45, Accepted throughput: 79.80 tokens/s, Drafted throughput: 110.19 tokens/s, Accepted: 798 tokens, Drafted: 1102 tokens, Per-position acceptance rate: 0.815, 0.633, Avg Draft acceptance rate: 72.4%
(APIServer pid=1) INFO: 127.0.0.1:38588 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:20:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 131.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:20:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.43, Accepted throughput: 77.00 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 770 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.743, 0.683, Avg Draft acceptance rate: 71.3%
(APIServer pid=1) INFO 09-18 14:21:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 154.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:21:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.88, Accepted throughput: 100.70 tokens/s, Drafted throughput: 107.40 tokens/s, Accepted: 1007 tokens, Drafted: 1074 tokens, Per-position acceptance rate: 0.952, 0.924, Avg Draft acceptance rate: 93.8%
(APIServer pid=1) INFO 09-18 14:21:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 143.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 27.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:21:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.66, Accepted throughput: 89.40 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 894 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.833, 0.822, Avg Draft acceptance rate: 82.8%
(APIServer pid=1) INFO: 127.0.0.1:42148 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:21:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 151.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 27.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:21:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.80, Accepted throughput: 97.59 tokens/s, Drafted throughput: 108.19 tokens/s, Accepted: 976 tokens, Drafted: 1082 tokens, Per-position acceptance rate: 0.906, 0.898, Avg Draft acceptance rate: 90.2%
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
(APIServer pid=1) INFO: 172.21.0.1:48854 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48862 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48874 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48886 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48894 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48902 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48914 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:21:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 140.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 14:21:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 91.39 tokens/s, Drafted throughput: 98.39 tokens/s, Accepted: 914 tokens, Drafted: 984 tokens, Per-position acceptance rate: 0.929, 0.929, Avg Draft acceptance rate: 92.9%
(APIServer pid=1) INFO: 172.21.0.1:48926 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48928 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48934 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48948 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48950 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48964 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48968 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48976 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:21:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 67.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO: 127.0.0.1:49396 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38790 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38796 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38802 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:21:52 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.9 tokens/s, Avg generation throughput: 37.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 14:21:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.80, Accepted throughput: 11.95 tokens/s, Drafted throughput: 13.30 tokens/s, Accepted: 239 tokens, Drafted: 266 tokens, Per-position acceptance rate: 0.940, 0.857, Avg Draft acceptance rate: 89.8%
(APIServer pid=1) INFO 09-18 14:22:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 73.8%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO: 172.21.0.1:43834 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43840 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43848 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43858 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43874 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43878 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:43890 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:22:12 [loggers.py:323] Engine 000: Avg prompt throughput: 13594.6 tokens/s, Avg generation throughput: 70.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8%
(APIServer pid=1) INFO 09-18 14:22:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.77, Accepted throughput: 22.55 tokens/s, Drafted throughput: 25.50 tokens/s, Accepted: 451 tokens, Drafted: 510 tokens, Per-position acceptance rate: 0.929, 0.839, Avg Draft acceptance rate: 88.4%
(APIServer pid=1) INFO: 172.21.0.1:51984 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:58388 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:22:22 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.2 tokens/s, Avg generation throughput: 113.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8%
(APIServer pid=1) INFO 09-18 14:22:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 63.79 tokens/s, Drafted throughput: 99.19 tokens/s, Accepted: 638 tokens, Drafted: 992 tokens, Per-position acceptance rate: 0.754, 0.532, Avg Draft acceptance rate: 64.3%
(APIServer pid=1) INFO: 172.21.0.1:52418 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:22:32 [loggers.py:323] Engine 000: Avg prompt throughput: 8.6 tokens/s, Avg generation throughput: 139.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8%
(APIServer pid=1) INFO 09-18 14:22:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.53, Accepted throughput: 84.50 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 845 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.848, 0.680, Avg Draft acceptance rate: 76.4%
(APIServer pid=1) INFO 09-18 14:22:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 134.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8%
(APIServer pid=1) INFO 09-18 14:22:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 78.30 tokens/s, Drafted throughput: 112.00 tokens/s, Accepted: 783 tokens, Drafted: 1120 tokens, Per-position acceptance rate: 0.811, 0.588, Avg Draft acceptance rate: 69.9%
(APIServer pid=1) INFO: 127.0.0.1:56790 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53910 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:22:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 33.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 58.5%, Prefix cache hit rate: 44.1%, MM cache hit rate: 77.8%
(APIServer pid=1) INFO 09-18 14:22:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.42, Accepted throughput: 19.70 tokens/s, Drafted throughput: 27.80 tokens/s, Accepted: 197 tokens, Drafted: 278 tokens, Per-position acceptance rate: 0.799, 0.619, Avg Draft acceptance rate: 70.9%
(APIServer pid=1) INFO: 172.21.0.1:47694 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47700 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:23:02 [loggers.py:323] Engine 000: Avg prompt throughput: 12912.6 tokens/s, Avg generation throughput: 24.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 30.8%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.74, Accepted throughput: 15.50 tokens/s, Drafted throughput: 17.80 tokens/s, Accepted: 155 tokens, Drafted: 178 tokens, Per-position acceptance rate: 0.921, 0.820, Avg Draft acceptance rate: 87.1%
(APIServer pid=1) INFO: 172.21.0.1:47708 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:23:12 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.4 tokens/s, Avg generation throughput: 118.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.24, Accepted throughput: 65.70 tokens/s, Drafted throughput: 105.99 tokens/s, Accepted: 657 tokens, Drafted: 1060 tokens, Per-position acceptance rate: 0.743, 0.496, Avg Draft acceptance rate: 62.0%
(APIServer pid=1) INFO: 127.0.0.1:53052 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59822 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:23:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.4 tokens/s, Avg generation throughput: 120.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 65.40 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 654 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.716, 0.469, Avg Draft acceptance rate: 59.2%
(APIServer pid=1) INFO 09-18 14:23:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 66.00 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 660 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.704, 0.478, Avg Draft acceptance rate: 59.1%
(APIServer pid=1) INFO: 172.21.0.1:34388 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:23:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 61.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.2%, Prefix cache hit rate: 37.4%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.06, Accepted throughput: 31.80 tokens/s, Drafted throughput: 59.99 tokens/s, Accepted: 318 tokens, Drafted: 600 tokens, Per-position acceptance rate: 0.673, 0.387, Avg Draft acceptance rate: 53.0%
(APIServer pid=1) INFO: 127.0.0.1:50974 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:23:52 [loggers.py:323] Engine 000: Avg prompt throughput: 12598.6 tokens/s, Avg generation throughput: 2.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.5%, Prefix cache hit rate: 37.4%, MM cache hit rate: 81.8%
(APIServer pid=1) INFO 09-18 14:23:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.80, Accepted throughput: 1.80 tokens/s, Drafted throughput: 2.00 tokens/s, Accepted: 18 tokens, Drafted: 20 tokens, Per-position acceptance rate: 0.900, 0.900, Avg Draft acceptance rate: 90.0%
(APIServer pid=1) INFO: 172.21.0.1:50664 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38410 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38412 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:24:02 [loggers.py:323] Engine 000: Avg prompt throughput: 1529.0 tokens/s, Avg generation throughput: 105.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6%
(APIServer pid=1) INFO 09-18 14:24:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 60.20 tokens/s, Drafted throughput: 89.20 tokens/s, Accepted: 602 tokens, Drafted: 892 tokens, Per-position acceptance rate: 0.791, 0.558, Avg Draft acceptance rate: 67.5%
(APIServer pid=1) INFO 09-18 14:24:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6%
(APIServer pid=1) INFO 09-18 14:24:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.13, Accepted throughput: 62.90 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 629 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.689, 0.440, Avg Draft acceptance rate: 56.5%
(APIServer pid=1) INFO: 172.21.0.1:38508 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:37090 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:24:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 123.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6%
(APIServer pid=1) INFO 09-18 14:24:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 67.80 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 678 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.732, 0.494, Avg Draft acceptance rate: 61.3%
(APIServer pid=1) INFO: 172.21.0.1:55060 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:24:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 100.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 36.9%, Prefix cache hit rate: 32.5%, MM cache hit rate: 84.6%
(APIServer pid=1) INFO 09-18 14:24:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 57.60 tokens/s, Drafted throughput: 86.60 tokens/s, Accepted: 576 tokens, Drafted: 866 tokens, Per-position acceptance rate: 0.762, 0.568, Avg Draft acceptance rate: 66.5%
(APIServer pid=1) INFO 09-18 14:24:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 78.5%, Prefix cache hit rate: 32.5%, MM cache hit rate: 84.6%
(APIServer pid=1) INFO: 127.0.0.1:41864 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:44650 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:44654 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:44658 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:24:52 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.6 tokens/s, Avg generation throughput: 77.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7%
(APIServer pid=1) INFO 09-18 14:24:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.37, Accepted throughput: 22.20 tokens/s, Drafted throughput: 32.40 tokens/s, Accepted: 444 tokens, Drafted: 648 tokens, Per-position acceptance rate: 0.765, 0.605, Avg Draft acceptance rate: 68.5%
(APIServer pid=1) INFO 09-18 14:25:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7%
(APIServer pid=1) INFO 09-18 14:25:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 74.49 tokens/s, Drafted throughput: 111.79 tokens/s, Accepted: 745 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.767, 0.565, Avg Draft acceptance rate: 66.6%
(APIServer pid=1) INFO: 172.21.0.1:52324 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:25:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 124.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7%
(APIServer pid=1) INFO 09-18 14:25:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 69.30 tokens/s, Drafted throughput: 110.99 tokens/s, Accepted: 693 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.737, 0.512, Avg Draft acceptance rate: 62.4%
(APIServer pid=1) INFO: 127.0.0.1:54114 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47722 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:25:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 112.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.3%, Prefix cache hit rate: 28.7%, MM cache hit rate: 86.7%
(APIServer pid=1) INFO 09-18 14:25:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 63.90 tokens/s, Drafted throughput: 96.79 tokens/s, Accepted: 639 tokens, Drafted: 968 tokens, Per-position acceptance rate: 0.756, 0.564, Avg Draft acceptance rate: 66.0%
(APIServer pid=1) INFO 09-18 14:25:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 75.4%, Prefix cache hit rate: 28.7%, MM cache hit rate: 86.7%
(APIServer pid=1) INFO: 172.21.0.1:56320 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56330 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:25:42 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.9 tokens/s, Avg generation throughput: 67.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2%
(APIServer pid=1) INFO 09-18 14:25:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.45, Accepted throughput: 19.80 tokens/s, Drafted throughput: 27.30 tokens/s, Accepted: 396 tokens, Drafted: 546 tokens, Per-position acceptance rate: 0.806, 0.645, Avg Draft acceptance rate: 72.5%
(APIServer pid=1) INFO: 127.0.0.1:34304 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:25:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 126.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2%
(APIServer pid=1) INFO 09-18 14:25:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 70.90 tokens/s, Drafted throughput: 111.80 tokens/s, Accepted: 709 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.757, 0.512, Avg Draft acceptance rate: 63.4%
(APIServer pid=1) INFO: 172.21.0.1:33198 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:26:02 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 117.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2%
(APIServer pid=1) INFO 09-18 14:26:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.12, Accepted throughput: 62.09 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 621 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.681, 0.444, Avg Draft acceptance rate: 56.2%
(APIServer pid=1) INFO 09-18 14:26:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2%
(APIServer pid=1) INFO 09-18 14:26:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.34, Accepted throughput: 74.39 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 744 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.772, 0.564, Avg Draft acceptance rate: 66.8%
(APIServer pid=1) INFO: 172.21.0.1:52376 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:51278 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:26:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 69.2%, Prefix cache hit rate: 25.7%, MM cache hit rate: 88.2%
(APIServer pid=1) INFO 09-18 14:26:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.80, Accepted throughput: 0.40 tokens/s, Drafted throughput: 1.00 tokens/s, Accepted: 4 tokens, Drafted: 10 tokens, Per-position acceptance rate: 0.600, 0.200, Avg Draft acceptance rate: 40.0%
(APIServer pid=1) INFO: 172.21.0.1:56900 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56902 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56906 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:26:32 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.0 tokens/s, Avg generation throughput: 53.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5%
(APIServer pid=1) INFO 09-18 14:26:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.74, Accepted throughput: 33.90 tokens/s, Drafted throughput: 39.00 tokens/s, Accepted: 339 tokens, Drafted: 390 tokens, Per-position acceptance rate: 0.923, 0.815, Avg Draft acceptance rate: 86.9%
(APIServer pid=1) INFO 09-18 14:26:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 112.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5%
(APIServer pid=1) INFO 09-18 14:26:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.01, Accepted throughput: 56.60 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 566 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.658, 0.357, Avg Draft acceptance rate: 50.7%
(APIServer pid=1) INFO: 127.0.0.1:50340 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37702 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:26:52 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 117.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5%
(APIServer pid=1) INFO 09-18 14:26:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.12, Accepted throughput: 62.10 tokens/s, Drafted throughput: 110.80 tokens/s, Accepted: 621 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.704, 0.417, Avg Draft acceptance rate: 56.0%
(APIServer pid=1) INFO 09-18 14:27:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5%
(APIServer pid=1) INFO 09-18 14:27:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.17, Accepted throughput: 65.20 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 652 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.707, 0.463, Avg Draft acceptance rate: 58.5%
(APIServer pid=1) INFO: 172.21.0.1:41528 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:27:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 45.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 53.8%, Prefix cache hit rate: 23.2%, MM cache hit rate: 89.5%
(APIServer pid=1) INFO 09-18 14:27:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 25.40 tokens/s, Drafted throughput: 40.80 tokens/s, Accepted: 254 tokens, Drafted: 408 tokens, Per-position acceptance rate: 0.745, 0.500, Avg Draft acceptance rate: 62.3%
(APIServer pid=1) INFO: 127.0.0.1:34370 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:60904 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:27:22 [loggers.py:323] Engine 000: Avg prompt throughput: 12913.2 tokens/s, Avg generation throughput: 22.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 23.2%, MM cache hit rate: 90.0%
(APIServer pid=1) INFO 09-18 14:27:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.66, Accepted throughput: 14.10 tokens/s, Drafted throughput: 17.00 tokens/s, Accepted: 141 tokens, Drafted: 170 tokens, Per-position acceptance rate: 0.894, 0.765, Avg Draft acceptance rate: 82.9%
(APIServer pid=1) INFO: 172.21.0.1:60920 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33700 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:27:32 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.4 tokens/s, Avg generation throughput: 98.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5%
(APIServer pid=1) INFO 09-18 14:27:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.09, Accepted throughput: 51.20 tokens/s, Drafted throughput: 93.60 tokens/s, Accepted: 512 tokens, Drafted: 936 tokens, Per-position acceptance rate: 0.660, 0.434, Avg Draft acceptance rate: 54.7%
(APIServer pid=1) INFO: 172.21.0.1:44936 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:27:42 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 134.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5%
(APIServer pid=1) INFO 09-18 14:27:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.43, Accepted throughput: 79.30 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 793 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.817, 0.617, Avg Draft acceptance rate: 71.7%
(APIServer pid=1) INFO: 127.0.0.1:45276 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:27:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5%
(APIServer pid=1) INFO 09-18 14:27:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.11, Accepted throughput: 62.10 tokens/s, Drafted throughput: 111.80 tokens/s, Accepted: 621 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.699, 0.411, Avg Draft acceptance rate: 55.5%
(APIServer pid=1) INFO: 172.21.0.1:34630 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:28:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 43.1%, Prefix cache hit rate: 21.2%, MM cache hit rate: 90.5%
(APIServer pid=1) INFO 09-18 14:28:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 41.10 tokens/s, Drafted throughput: 70.60 tokens/s, Accepted: 411 tokens, Drafted: 706 tokens, Per-position acceptance rate: 0.711, 0.453, Avg Draft acceptance rate: 58.2%
(APIServer pid=1) INFO 09-18 14:28:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 84.6%, Prefix cache hit rate: 21.2%, MM cache hit rate: 90.5%
(APIServer pid=1) INFO: 172.21.0.1:47082 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37824 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:49298 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37832 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:28:22 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.0 tokens/s, Avg generation throughput: 98.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3%
(APIServer pid=1) INFO 09-18 14:28:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.41, Accepted throughput: 28.70 tokens/s, Drafted throughput: 40.60 tokens/s, Accepted: 574 tokens, Drafted: 812 tokens, Per-position acceptance rate: 0.793, 0.621, Avg Draft acceptance rate: 70.7%
(APIServer pid=1) INFO 09-18 14:28:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 124.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3%
(APIServer pid=1) INFO 09-18 14:28:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.24, Accepted throughput: 69.10 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 691 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.745, 0.498, Avg Draft acceptance rate: 62.1%
(APIServer pid=1) INFO: 172.21.0.1:53742 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:28:42 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 111.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3%
(APIServer pid=1) INFO 09-18 14:28:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.02, Accepted throughput: 56.50 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 565 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.650, 0.373, Avg Draft acceptance rate: 51.2%
(APIServer pid=1) INFO: 127.0.0.1:49254 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:44080 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:28:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 111.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 19.5%, MM cache hit rate: 91.3%
(APIServer pid=1) INFO 09-18 14:28:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 64.89 tokens/s, Drafted throughput: 92.59 tokens/s, Accepted: 649 tokens, Drafted: 926 tokens, Per-position acceptance rate: 0.803, 0.598, Avg Draft acceptance rate: 70.1%
(APIServer pid=1) INFO 09-18 14:29:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 76.9%, Prefix cache hit rate: 19.5%, MM cache hit rate: 91.3%
(APIServer pid=1) INFO: 172.21.0.1:53800 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53804 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53806 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:29:12 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.1 tokens/s, Avg generation throughput: 75.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0%
(APIServer pid=1) INFO 09-18 14:29:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.57, Accepted throughput: 23.05 tokens/s, Drafted throughput: 29.40 tokens/s, Accepted: 461 tokens, Drafted: 588 tokens, Per-position acceptance rate: 0.864, 0.704, Avg Draft acceptance rate: 78.4%
(APIServer pid=1) INFO: 127.0.0.1:51408 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:29:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0%
(APIServer pid=1) INFO 09-18 14:29:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.13, Accepted throughput: 62.89 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 629 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.703, 0.425, Avg Draft acceptance rate: 56.4%
(APIServer pid=1) INFO: 172.21.0.1:60596 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:29:32 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 124.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0%
(APIServer pid=1) INFO 09-18 14:29:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 68.89 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 689 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.732, 0.514, Avg Draft acceptance rate: 62.3%
(APIServer pid=1) INFO 09-18 14:29:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 119.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0%
(APIServer pid=1) INFO 09-18 14:29:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.20 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 642 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.716, 0.439, Avg Draft acceptance rate: 57.7%
(APIServer pid=1) INFO: 172.21.0.1:60372 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:41996 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:29:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 11.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 66.2%, Prefix cache hit rate: 18.1%, MM cache hit rate: 92.0%
(APIServer pid=1) INFO 09-18 14:29:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 6.40 tokens/s, Drafted throughput: 10.40 tokens/s, Accepted: 64 tokens, Drafted: 104 tokens, Per-position acceptance rate: 0.769, 0.462, Avg Draft acceptance rate: 61.5%
(APIServer pid=1) INFO: 172.21.0.1:59394 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59402 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59410 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:30:02 [loggers.py:323] Engine 000: Avg prompt throughput: 14127.4 tokens/s, Avg generation throughput: 41.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6%
(APIServer pid=1) INFO 09-18 14:30:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.76, Accepted throughput: 26.00 tokens/s, Drafted throughput: 29.60 tokens/s, Accepted: 260 tokens, Drafted: 296 tokens, Per-position acceptance rate: 0.912, 0.845, Avg Draft acceptance rate: 87.8%
(APIServer pid=1) INFO 09-18 14:30:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 117.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6%
(APIServer pid=1) INFO 09-18 14:30:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.10, Accepted throughput: 61.50 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 615 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.666, 0.438, Avg Draft acceptance rate: 55.2%
(APIServer pid=1) INFO: 127.0.0.1:43480 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48952 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:30:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 119.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6%
(APIServer pid=1) INFO 09-18 14:30:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 63.90 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 639 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.699, 0.458, Avg Draft acceptance rate: 57.9%
(APIServer pid=1) INFO 09-18 14:30:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6%
(APIServer pid=1) INFO 09-18 14:30:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 65.00 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 650 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.706, 0.459, Avg Draft acceptance rate: 58.2%
(APIServer pid=1) INFO: 172.21.0.1:50314 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:30:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 52.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 50.8%, Prefix cache hit rate: 16.9%, MM cache hit rate: 92.6%
(APIServer pid=1) INFO 09-18 14:30:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 28.10 tokens/s, Drafted throughput: 48.40 tokens/s, Accepted: 281 tokens, Drafted: 484 tokens, Per-position acceptance rate: 0.665, 0.496, Avg Draft acceptance rate: 58.1%
(APIServer pid=1) INFO: 127.0.0.1:50034 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:58630 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:30:52 [loggers.py:323] Engine 000: Avg prompt throughput: 12598.1 tokens/s, Avg generation throughput: 13.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.8%, MM cache hit rate: 92.9%
(APIServer pid=1) INFO 09-18 14:30:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.71, Accepted throughput: 8.40 tokens/s, Drafted throughput: 9.80 tokens/s, Accepted: 84 tokens, Drafted: 98 tokens, Per-position acceptance rate: 0.878, 0.837, Avg Draft acceptance rate: 85.7%
(APIServer pid=1) INFO: 172.21.0.1:58644 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:38342 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:31:02 [loggers.py:323] Engine 000: Avg prompt throughput: 1529.0 tokens/s, Avg generation throughput: 110.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1%
(APIServer pid=1) INFO 09-18 14:31:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 63.19 tokens/s, Drafted throughput: 93.59 tokens/s, Accepted: 632 tokens, Drafted: 936 tokens, Per-position acceptance rate: 0.778, 0.573, Avg Draft acceptance rate: 67.5%
(APIServer pid=1) INFO: 172.21.0.1:49568 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:31:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 126.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1%
(APIServer pid=1) INFO 09-18 14:31:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 71.59 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 716 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.766, 0.531, Avg Draft acceptance rate: 64.9%
(APIServer pid=1) INFO: 127.0.0.1:53428 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:31:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1%
(APIServer pid=1) INFO 09-18 14:31:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 65.00 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 650 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.703, 0.462, Avg Draft acceptance rate: 58.2%
(APIServer pid=1) INFO: 172.21.0.1:53214 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:31:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 78.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 40.0%, Prefix cache hit rate: 15.8%, MM cache hit rate: 93.1%
(APIServer pid=1) INFO 09-18 14:31:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.02, Accepted throughput: 39.70 tokens/s, Drafted throughput: 77.60 tokens/s, Accepted: 397 tokens, Drafted: 776 tokens, Per-position acceptance rate: 0.631, 0.392, Avg Draft acceptance rate: 51.2%
(APIServer pid=1) INFO 09-18 14:31:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.5%, Prefix cache hit rate: 15.8%, MM cache hit rate: 93.1%
(APIServer pid=1) INFO: 172.21.0.1:59288 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:56286 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59294 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59298 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:31:52 [loggers.py:323] Engine 000: Avg prompt throughput: 14129.1 tokens/s, Avg generation throughput: 88.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:31:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 25.75 tokens/s, Drafted throughput: 36.80 tokens/s, Accepted: 515 tokens, Drafted: 736 tokens, Per-position acceptance rate: 0.788, 0.611, Avg Draft acceptance rate: 70.0%
(APIServer pid=1) INFO 09-18 14:32:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.30 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 643 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.713, 0.442, Avg Draft acceptance rate: 57.7%
(APIServer pid=1) INFO: 172.21.0.1:40376 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:32:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 129.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 74.70 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 747 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.775, 0.578, Avg Draft acceptance rate: 67.7%
(APIServer pid=1) INFO: 127.0.0.1:51024 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:54062 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:32:22 [loggers.py:323] Engine 000: Avg prompt throughput: 25.2 tokens/s, Avg generation throughput: 127.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 73.19 tokens/s, Drafted throughput: 108.79 tokens/s, Accepted: 732 tokens, Drafted: 1088 tokens, Per-position acceptance rate: 0.774, 0.572, Avg Draft acceptance rate: 67.3%
(APIServer pid=1) INFO: 172.21.0.1:54076 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:54078 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53562 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:32:32 [loggers.py:323] Engine 000: Avg prompt throughput: 38.1 tokens/s, Avg generation throughput: 133.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.48, Accepted throughput: 79.80 tokens/s, Drafted throughput: 107.80 tokens/s, Accepted: 798 tokens, Drafted: 1078 tokens, Per-position acceptance rate: 0.831, 0.649, Avg Draft acceptance rate: 74.0%
(APIServer pid=1) INFO 09-18 14:32:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.10 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 731 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.750, 0.571, Avg Draft acceptance rate: 66.1%
(APIServer pid=1) INFO: 127.0.0.1:36210 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53576 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:32:52 [loggers.py:323] Engine 000: Avg prompt throughput: 13.4 tokens/s, Avg generation throughput: 131.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:32:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.42, Accepted throughput: 77.20 tokens/s, Drafted throughput: 109.00 tokens/s, Accepted: 772 tokens, Drafted: 1090 tokens, Per-position acceptance rate: 0.796, 0.620, Avg Draft acceptance rate: 70.8%
(APIServer pid=1) INFO: 172.21.0.1:38118 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47594 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47610 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47622 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:33:02 [loggers.py:323] Engine 000: Avg prompt throughput: 59.9 tokens/s, Avg generation throughput: 137.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:33:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.59, Accepted throughput: 84.40 tokens/s, Drafted throughput: 106.20 tokens/s, Accepted: 844 tokens, Drafted: 1062 tokens, Per-position acceptance rate: 0.868, 0.721, Avg Draft acceptance rate: 79.5%
(APIServer pid=1) INFO: 172.21.0.1:47632 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47648 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:45364 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:33:12 [loggers.py:323] Engine 000: Avg prompt throughput: 27.9 tokens/s, Avg generation throughput: 142.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:33:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.64, Accepted throughput: 88.50 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 885 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.891, 0.748, Avg Draft acceptance rate: 81.9%
(APIServer pid=1) INFO: 172.21.0.1:45372 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:45386 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:47374 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:33:22 [loggers.py:323] Engine 000: Avg prompt throughput: 32.5 tokens/s, Avg generation throughput: 28.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO 09-18 14:33:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 18.40 tokens/s, Drafted throughput: 19.80 tokens/s, Accepted: 184 tokens, Drafted: 198 tokens, Per-position acceptance rate: 0.970, 0.889, Avg Draft acceptance rate: 92.9%
(APIServer pid=1) INFO 09-18 14:33:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.5%
(APIServer pid=1) INFO: 127.0.0.1:51446 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:47488 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:50100 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:54884 - "GET /metrics HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:49228 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:45014 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:45300 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:46998 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:51132 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.16.100.3:62103 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.16.100.3:62104 - "GET /v1/models HTTP/1.1" 401 Unauthorized
(APIServer pid=1) INFO: 127.0.0.1:58480 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:46524 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:53330 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:55746 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:56396 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:55482 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:56702 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:36636 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:34060 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:44360 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:52262 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:60346 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:32864 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:47770 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:59988 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:47698 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:44138 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:52264 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:52764 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:54950 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:37836 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:50426 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:42744 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:36986 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:46082 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33154 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33156 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35076 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35092 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35102 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35110 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35124 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35134 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35136 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35150 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35160 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35174 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35188 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35192 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35200 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:35204 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:02 [loggers.py:323] Engine 000: Avg prompt throughput: 399.3 tokens/s, Avg generation throughput: 105.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.8%
(APIServer pid=1) INFO 09-18 14:50:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.84, Accepted throughput: 0.68 tokens/s, Drafted throughput: 0.74 tokens/s, Accepted: 679 tokens, Drafted: 740 tokens, Per-position acceptance rate: 0.957, 0.878, Avg Draft acceptance rate: 91.8%
(APIServer pid=1) INFO: 172.21.0.1:35214 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56506 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56510 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:56514 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:12 [loggers.py:323] Engine 000: Avg prompt throughput: 1821.0 tokens/s, Avg generation throughput: 96.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:50:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 55.10 tokens/s, Drafted throughput: 83.40 tokens/s, Accepted: 551 tokens, Drafted: 834 tokens, Per-position acceptance rate: 0.763, 0.559, Avg Draft acceptance rate: 66.1%
(APIServer pid=1) INFO: 127.0.0.1:55476 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:50:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 72.00 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 720 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.758, 0.532, Avg Draft acceptance rate: 64.5%
(APIServer pid=1) INFO: 172.21.0.1:37334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:32 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 122.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:50:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.21, Accepted throughput: 66.99 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 670 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.725, 0.489, Avg Draft acceptance rate: 60.7%
(APIServer pid=1) INFO: 172.21.0.1:37138 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 121.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:50:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 66.00 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 660 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.728, 0.467, Avg Draft acceptance rate: 59.8%
(APIServer pid=1) INFO: 127.0.0.1:53092 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:50:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 126.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:50:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 70.60 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 706 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.742, 0.523, Avg Draft acceptance rate: 63.3%
(APIServer pid=1) INFO: 172.21.0.1:39218 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:39220 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:51:02 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 121.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 66.50 tokens/s, Drafted throughput: 109.40 tokens/s, Accepted: 665 tokens, Drafted: 1094 tokens, Per-position acceptance rate: 0.729, 0.486, Avg Draft acceptance rate: 60.8%
(APIServer pid=1) INFO 09-18 14:51:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 71.20 tokens/s, Drafted throughput: 112.00 tokens/s, Accepted: 712 tokens, Drafted: 1120 tokens, Per-position acceptance rate: 0.750, 0.521, Avg Draft acceptance rate: 63.6%
(APIServer pid=1) INFO: 172.21.0.1:41374 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:37916 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:51:22 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 128.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.29 tokens/s, Drafted throughput: 110.79 tokens/s, Accepted: 733 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.769, 0.554, Avg Draft acceptance rate: 66.2%
(APIServer pid=1) INFO 09-18 14:51:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 72.70 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 727 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.758, 0.545, Avg Draft acceptance rate: 65.1%
(APIServer pid=1) INFO: 172.21.0.1:49184 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:51:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 127.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 72.30 tokens/s, Drafted throughput: 111.00 tokens/s, Accepted: 723 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.769, 0.533, Avg Draft acceptance rate: 65.1%
(APIServer pid=1) INFO: 172.21.0.1:39208 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:55292 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:39224 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:51:52 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 125.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:51:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 71.00 tokens/s, Drafted throughput: 109.99 tokens/s, Accepted: 710 tokens, Drafted: 1100 tokens, Per-position acceptance rate: 0.744, 0.547, Avg Draft acceptance rate: 64.5%
(APIServer pid=1) INFO 09-18 14:52:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:52:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.39 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 644 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.697, 0.457, Avg Draft acceptance rate: 57.7%
(APIServer pid=1) INFO: 172.21.0.1:53966 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:52:12 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 123.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:52:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 67.80 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 678 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.722, 0.505, Avg Draft acceptance rate: 61.3%
(APIServer pid=1) INFO: 127.0.0.1:53434 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:52:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:52:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 66.00 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 660 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.715, 0.470, Avg Draft acceptance rate: 59.2%
(APIServer pid=1) INFO: 172.21.0.1:58938 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:52:32 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 121.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:52:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.19, Accepted throughput: 65.60 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 656 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.712, 0.474, Avg Draft acceptance rate: 59.3%
(APIServer pid=1) INFO: 172.21.0.1:59878 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59890 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59894 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59900 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59908 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59912 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59914 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59930 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59932 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59944 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59950 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:52:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 105.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO 09-18 14:52:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 57.80 tokens/s, Drafted throughput: 96.20 tokens/s, Accepted: 578 tokens, Drafted: 962 tokens, Per-position acceptance rate: 0.711, 0.491, Avg Draft acceptance rate: 60.1%
(APIServer pid=1) INFO: 172.21.0.1:59952 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59958 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59968 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59980 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:55976 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:52:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 69.2%, Prefix cache hit rate: 14.9%, MM cache hit rate: 94.6%
(APIServer pid=1) INFO: 172.21.0.1:32818 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:32834 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:32836 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:53:02 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.1 tokens/s, Avg generation throughput: 35.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7%
(APIServer pid=1) INFO 09-18 14:53:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.79, Accepted throughput: 11.35 tokens/s, Drafted throughput: 12.70 tokens/s, Accepted: 227 tokens, Drafted: 254 tokens, Per-position acceptance rate: 0.913, 0.874, Avg Draft acceptance rate: 89.4%
(APIServer pid=1) INFO 09-18 14:53:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 76.9%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7%
(APIServer pid=1) INFO: 172.21.0.1:53668 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53676 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53688 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53696 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53700 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 127.0.0.1:44356 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53716 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53720 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:53:22 [loggers.py:323] Engine 000: Avg prompt throughput: 11222.0 tokens/s, Avg generation throughput: 125.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7%
(APIServer pid=1) INFO 09-18 14:53:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.77, Accepted throughput: 40.25 tokens/s, Drafted throughput: 45.40 tokens/s, Accepted: 805 tokens, Drafted: 908 tokens, Per-position acceptance rate: 0.943, 0.830, Avg Draft acceptance rate: 88.7%
(APIServer pid=1) INFO: 172.21.0.1:53732 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:53748 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 14:53:32 [loggers.py:323] Engine 000: Avg prompt throughput: 17.3 tokens/s, Avg generation throughput: 144.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7%
(APIServer pid=1) INFO 09-18 14:53:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.66, Accepted throughput: 90.39 tokens/s, Drafted throughput: 108.99 tokens/s, Accepted: 904 tokens, Drafted: 1090 tokens, Per-position acceptance rate: 0.899, 0.760, Avg Draft acceptance rate: 82.9%
(APIServer pid=1) INFO: 172.21.0.1:40154 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:40170 - "POST /v1/chat/completions HTTP/1.1" 200 OK
@@ -0,0 +1,5 @@
{
"startup": true,
"records": 44,
"failures": []
}
@@ -0,0 +1,46 @@
name: qwen38-flash-rtx6000d
services:
vllm:
image: local/qwen38-flash-6000d:0bfc7a15
build: .
container_name: qwen38-flash-6000d
gpus: all
shm_size: 8g
mem_limit: 210g
memswap_limit: 210g
restart: "no"
ports:
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
environment:
HF_HUB_OFFLINE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_WORKER_MULTIPROC_METHOD: spawn
OMP_NUM_THREADS: "8"
TORCHINDUCTOR_COMPILE_THREADS: "2"
CUTE_DSL_ARCH: sm_120a
volumes:
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
- /data/flash-next/cache:/root/.cache
- ./secrets/api-key:/run/secrets/api-key:ro
entrypoint: ["/bin/bash", "-lc"]
command:
- >-
exec vllm serve /model
--served-model-name qwen3.8-flash-next
--host 0.0.0.0 --port 8000
--tensor-parallel-size 1 --dtype bfloat16
--engram-config '{"cpu_offload":true}'
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
--enable-chunked-prefill --enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":3}'
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,4]}' --no-enable-flashinfer-autotune
--load-format safetensors
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
--api-key "$$(cat /run/secrets/api-key)"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30m
@@ -0,0 +1,28 @@
# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_total counter
vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 8872.0
# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts.
# TYPE vllm:spec_decode_num_drafts_created gauge
vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577659e+09
# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_total counter
vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 26616.0
# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens.
# TYPE vllm:spec_decode_num_draft_tokens_created gauge
vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897438345776758e+09
# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_total counter
vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 14825.0
# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens.
# TYPE vllm:spec_decode_num_accepted_tokens_created gauge
vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577687e+09
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 6682.0
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 4815.0
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="2"} 3328.0
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position.
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.789743834577702e+09
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.7897438345777066e+09
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="2"} 1.7897438345777097e+09
@@ -0,0 +1,7 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228, "response": {"id": "chatcmpl-bbb7d55702a41460", "object": "chat.completion", "created": 1789744068, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236, "response": {"id": "chatcmpl-8083ab34aa888310", "object": "chat.completion", "created": 1789744069, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427, "response": {"id": "chatcmpl-ae45cca384883fdb", "object": "chat.completion", "created": 1789744071, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661, "response": {"id": "chatcmpl-af2a24bebab98e35", "object": "chat.completion", "created": 1789744073, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-9c2f3c0c1afec9f2", "object": "chat.completion", "created": 1789744074, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-a61f6a3fb52eef25", "object": "chat.completion", "created": 1789744080, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-9b2fb34a4ff8498a", "object": "chat.completion", "created": 1789744082, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-b78447dfbb2c5061", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
@@ -0,0 +1,8 @@
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228}
{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236}
{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427}
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661}
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
{"name": "tool-arguments", "passed": true}
QUALITY_COMPLETE
File diff suppressed because one or more lines are too long
@@ -0,0 +1,38 @@
{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6857045710057719, "elapsed_s": 1.1168529220012715, "decode_tps": 187.87036947485734, "max_stream_gap_s": 0.021299945998180192, "events": 23, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"}
{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.09021790199767565, "elapsed_s": 0.30122662099893205, "decode_tps": 170.60906378842878, "max_stream_gap_s": 0.0215804309991654, "events": 10, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"}
{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0900614569982281, "elapsed_s": 0.43906909199722577, "decode_tps": 186.2423439538412, "max_stream_gap_s": 0.021113369999511633, "events": 18, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"}
{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.09134426800301298, "elapsed_s": 0.293823849002365, "decode_tps": 187.67324493881486, "max_stream_gap_s": 0.020979296998120844, "events": 11, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"}
{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08995563200005563, "elapsed_s": 0.4795566350003355, "decode_tps": 182.23772385911695, "max_stream_gap_s": 0.021156058995984495, "events": 20, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"}
{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08942628499789862, "elapsed_s": 0.3126287199993385, "decode_tps": 183.68975230819282, "max_stream_gap_s": 0.02103689500654582, "events": 12, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"}
{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08937776899983874, "elapsed_s": 0.4788391099937144, "decode_tps": 179.73542591253175, "max_stream_gap_s": 0.02123644899984356, "events": 20, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"}
{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.09002831100224284, "elapsed_s": 0.6215811549991486, "decode_tps": 174.95908647709422, "max_stream_gap_s": 0.021011450000514742, "events": 26, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"}
{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08917911599564832, "elapsed_s": 0.5392506789939944, "decode_tps": 186.6369859948443, "max_stream_gap_s": 0.021216852001089137, "events": 23, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"}
{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.09073446100228466, "elapsed_s": 0.539050845996826, "decode_tps": 185.13711025977918, "max_stream_gap_s": 0.020977412998036016, "events": 23, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"}
{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08987670400529169, "elapsed_s": 0.5182491330051562, "decode_tps": 191.42221685799936, "max_stream_gap_s": 0.02107875300134765, "events": 22, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"}
{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09002829199744156, "elapsed_s": 0.5601620039960835, "decode_tps": 189.3078452545796, "max_stream_gap_s": 0.021013982994190883, "events": 24, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"}
{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.1072728509971057, "elapsed_s": 0.40980977100116434, "decode_tps": 138.82603154496502, "max_stream_gap_s": 0.020985293995181564, "events": 16, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"}
{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.09010298399516614, "elapsed_s": 0.5101531999971485, "decode_tps": 147.60140011380784, "max_stream_gap_s": 0.020926092998706736, "events": 22, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"}
{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3285, "completion_tokens": 150, "completion_tokens_details": {"reasoning_tokens": 111}}, "ttft_s": 0.7640111340006115, "elapsed_s": 1.6888792590034427, "decode_tps": 161.10404929302098, "max_stream_gap_s": 0.026646995996998157, "events": 45, "sha256": "59acfe36a1bac3e2c16c79d5c3712068c6d77030f2161a2297c21a795d2a2f9c"}
{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12191, "completion_tokens": 122, "completion_tokens_details": {"reasoning_tokens": 108}}, "ttft_s": 2.5409359719997155, "elapsed_s": 3.243877524000709, "decode_tps": 172.1337992548049, "max_stream_gap_s": 0.03231711799890036, "events": 35, "sha256": "34f1ae68e14f16d3cbbeed12b616b538f07ebec97e9ad601ced414e453c29f65"}
{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.463653916005569, "elapsed_s": 1.8863289360015187, "decode_tps": 165.61186890266376, "max_stream_gap_s": 0.020653829000366386, "events": 22, "sha256": "c4d4ac93f59330deca1f3c5fe9842a60a2ac31170fdbc40f311385c571ed9024"}
{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08850599199649878, "elapsed_s": 2.1877699620017665, "decode_tps": 121.47114590803943, "max_stream_gap_s": 0.0206061229982879, "events": 106, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"}
{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08851322199916467, "elapsed_s": 16.576275200000964, "decode_tps": 124.152689900008, "max_stream_gap_s": 0.021029635005106684, "events": 824, "sha256": "9e8927825d4c68bc3bb71bc5d5dc66a8a35e5961a52ff04a44c5211082f2d4c9"}
{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.09001792799972463, "elapsed_s": 16.722888370000874, "decode_tps": 123.06955718424507, "max_stream_gap_s": 0.021017879000282846, "events": 831, "sha256": "649658539a14ad95cd3aba4a031955cf646f9c2cef5e606a2f3f25be99a3962b"}
{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07350860699807527, "elapsed_s": 16.641791551002825, "decode_tps": 123.54931448950835, "max_stream_gap_s": 0.021029645002272446, "events": 828, "sha256": "4af2092c8a16885a6236f0c3adc0227eb9dd8079ec421ba9c2eb5cc404b3440b"}
{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07293527999718208, "elapsed_s": 1.9857562760007568, "decode_tps": 133.31095828243588, "max_stream_gap_s": 0.020934283995302394, "events": 97, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"}
{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08870576600020286, "elapsed_s": 15.336975530000927, "decode_tps": 134.24473934955645, "max_stream_gap_s": 0.020986846000596415, "events": 765, "sha256": "d4b632309eb489d41b5803e52e0ae083ffe5c0f801ac927326c6037eb95c6474"}
{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08885356999962823, "elapsed_s": 15.44073144599679, "decode_tps": 133.33873657244945, "max_stream_gap_s": 0.02105036300054053, "events": 770, "sha256": "795e37bad007369fe0ce6ce658b4ebf659222130ea24f338f6f342b4ce6aff47"}
{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07399053900007857, "elapsed_s": 15.320730624000134, "decode_tps": 134.25820789152598, "max_stream_gap_s": 0.020978505002858583, "events": 765, "sha256": "ece641cb2f77db4659a64c75f36366c11b83095a29fc22cb320a1b306b6da1f5"}
{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.09028758799831849, "elapsed_s": 2.1541953229971114, "decode_tps": 123.55203465534235, "max_stream_gap_s": 0.02097705299820518, "events": 105, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"}
{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0739625900023384, "elapsed_s": 16.64171508599975, "decode_tps": 123.55327015505169, "max_stream_gap_s": 0.021100942998600658, "events": 828, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"}
{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07516271799977403, "elapsed_s": 16.646220769005595, "decode_tps": 123.52862404436223, "max_stream_gap_s": 0.021613906996208243, "events": 828, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07455741699959617, "elapsed_s": 16.66266934000305, "decode_tps": 123.40162699055197, "max_stream_gap_s": 0.021173682995140553, "events": 829, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126125, "completion_tokens": 130, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.516383726004278, "elapsed_s": 15.218214432999957, "decode_tps": 183.80500983237002, "max_stream_gap_s": 0.02093203100230312, "events": 36, "sha256": "9a2e65aa6f54431f5ea3bfa29bcdd76256bef03427fca286f07cc3232bfdc529"}
{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126119, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.510828671998752, "elapsed_s": 15.23376362700219, "decode_tps": 170.13978802479537, "max_stream_gap_s": 0.021043553999334108, "events": 37, "sha256": "1df90f428d94a645b711abef54df465c7fa0db783e3b3602ea5b4aa73f29aaab"}
{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 14.506387795998307, "elapsed_s": 15.205817506997846, "decode_tps": 168.7088754513573, "max_stream_gap_s": 0.020960481000656728, "events": 36, "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"}
{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 12.83541851100017, "elapsed_s": 13.968540412002767, "decode_tps": 167.67834054913806, "max_stream_gap_s": 0.021041211002739146, "events": 57, "sha256": "86bc2cd86b8219571bdb62f3c793782b1b6de944ea376483b0595f97b9bb55aa"}
{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08649174699530704, "elapsed_s": 0.44836226699408144, "decode_tps": 174.09541954457458, "max_stream_gap_s": 0.020703647001937497, "events": 19, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"}
{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.5161374739982421, "elapsed_s": 0.8586695900012273, "decode_tps": 183.9243593714609, "max_stream_gap_s": 0.021041070998762734, "events": 18, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"}
{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.9265989400009857, "elapsed_s": 1.1674528340008692, "decode_tps": 174.37957635852183, "max_stream_gap_s": 0.02082027800497599, "events": 13, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"}
{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.2341738940012874, "elapsed_s": 1.5963426840025932, "decode_tps": 173.95204042781506, "max_stream_gap_s": 0.02091294499405194, "events": 19, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"}
SUITE_COMPLETE
@@ -0,0 +1,345 @@
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █ █ █▄ ▄█
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347]
(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 3}, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:22 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
(APIServer pid=1) INFO 09-18 15:01:22 [model.py:2024] Using max model len 131072
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:25 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:01:25 [model.py:691] Resolved architecture: Qwen4ExpMTP
(APIServer pid=1) INFO 09-18 15:01:25 [model.py:2024] Using max model len 262144
(APIServer pid=1) WARNING 09-18 15:01:25 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate
(APIServer pid=1) INFO 09-18 15:01:25 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072
(APIServer pid=1) INFO 09-18 15:01:25 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
(APIServer pid=1) INFO 09-18 15:01:25 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) WARNING 09-18 15:01:25 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(APIServer pid=1) INFO 09-18 15:01:25 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:01:38 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=3), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 4, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_f7efa1f40cb14f80867628d23379f752 backend=nccl
(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
(EngineCore pid=112) INFO 09-18 15:01:39 [gpu_worker.py:441] Using V2 Model Runner
(EngineCore pid=112) INFO 09-18 15:01:40 [model_runner.py:387] Loading model from scratch...
(EngineCore pid=112) INFO 09-18 15:01:40 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
(EngineCore pid=112) INFO 09-18 15:01:40 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
(EngineCore pid=112) INFO 09-18 15:01:42 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
(APIServer pid=1) INFO 09-18 15:01:44 [base.py:261] Multi-modal warmup completed in 12.297s
(APIServer pid=1) INFO 09-18 15:01:45 [base.py:261] Readonly multi-modal warmup completed in 1.333s
(EngineCore pid=112) INFO 09-18 15:02:16 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
(EngineCore pid=112) INFO 09-18 15:02:16 [flash_attn.py:1115] Using FlashAttention version 2
(EngineCore pid=112) WARNING 09-18 15:02:16 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.27 GiB.
(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:16, 1.63s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:06<00:33, 3.75s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:12<00:36, 4.53s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:17<00:34, 4.89s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:23<00:30, 5.14s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:28<00:26, 5.29s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:34<00:21, 5.38s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:40<00:16, 5.48s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:41<00:08, 4.21s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 2.44s/it]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:42<00:00, 3.86s/it]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 15:02:59 [default_loader.py:430] Loading weights took 42.67 seconds
(EngineCore pid=112) INFO 09-18 15:02:59 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) INFO 09-18 15:02:59 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
(EngineCore pid=112) INFO 09-18 15:02:59 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(EngineCore pid=112) WARNING 09-18 15:02:59 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
(EngineCore pid=112) INFO 09-18 15:02:59 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
(EngineCore pid=112) INFO 09-18 15:03:00 [fp8.py:433] Using DEEPGEMM Fp8 MoE backend out of potential backends: ['AITER', 'FLASHINFER_TRTLLM', 'FLASHINFER_CUTLASS', 'DEEPGEMM', 'TRITON', 'MARLIN', 'HUMMING', 'BATCHED_DEEPGEMM', 'BATCHED_TRITON', 'XPU', 'CPU', 'HPC'].
(EngineCore pid=112) INFO 09-18 15:03:00 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.14 GiB.
(EngineCore pid=112)
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:00<00:01, 5.20it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:00<00:01, 4.09it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:01<00:01, 3.69it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:01<00:01, 3.47it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:01<00:01, 3.33it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:01<00:01, 3.26it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:02<00:00, 3.19it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:02<00:00, 3.69it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.89it/s]
(EngineCore pid=112)
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.70it/s]
(EngineCore pid=112)
(EngineCore pid=112) INFO 09-18 15:03:03 [default_loader.py:430] Loading weights took 2.98 seconds
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:196] deep_gemm not found in site-packages, trying vendored vllm.third_party.deep_gemm
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:223] DeepGEMM PDL enabled on vllm.third_party.deep_gemm.
(EngineCore pid=112) INFO 09-18 15:03:03 [deep_gemm.py:136] DeepGEMM E8M0 enabled on current platform.
(EngineCore pid=112) INFO 09-18 15:03:06 [fp8.py:733] Using MoEPrepareAndFinalizeNoDPEPModular
(EngineCore pid=112) WARNING 09-18 15:03:06 [speculator.py:235] Draft model Qwen4ExpMTP does not support external multimodal embeddings. Embeddings from the target model will not be passed to the drafter; using text-only draft inputs instead.
(EngineCore pid=112) INFO 09-18 15:03:06 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.716523 seconds
(EngineCore pid=112) INFO 09-18 15:03:06 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
(EngineCore pid=112) INFO 09-18 15:03:06 [interface.py:918] Setting attention block size to 3200 tokens to ensure that attention page size is >= mamba page size.
(EngineCore pid=112) INFO 09-18 15:03:06 [interface.py:942] Padding mamba page size by 0.25% to ensure that mamba page size and attention page size are exactly equal.
(EngineCore pid=112) INFO 09-18 15:03:06 [utils.py:320] Using BLNHC KV cache layout.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(EngineCore pid=112) INFO 09-18 15:03:11 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
(EngineCore pid=112) WARNING 09-18 15:03:33 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) WARNING 09-18 15:03:33 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
(EngineCore pid=112) INFO 09-18 15:03:33 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps.
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.85s/it]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.85s/it]
(EngineCore pid=112) INFO 09-18 15:03:35 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.16it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.16it/s]
(EngineCore pid=112)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.71it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.71it/s]
(EngineCore pid=112) INFO 09-18 15:03:36 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.15 GiB
(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB
(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9850 is equivalent to --gpu-memory-utilization=0.9832 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9868. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
(EngineCore pid=112) WARNING 09-18 15:03:37 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
(EngineCore pid=112) INFO 09-18 15:03:37 [kv_cache_utils.py:2404] GPU KV cache size: 137,414 tokens, Maximum concurrency for 131,072 tokens per request: 1.05x
(EngineCore pid=112) INFO 09-18 15:03:37 [kernel_warmup.py:171] JIT kernel warmup starting.
(EngineCore pid=112) INFO 09-18 15:03:37 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
(EngineCore pid=112) INFO 09-18 15:03:37 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
(EngineCore pid=112) INFO 09-18 15:03:38 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
(EngineCore pid=112) INFO 09-18 15:03:38 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
(EngineCore pid=112) INFO 09-18 15:03:38 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1), (2, 1), (3, 1), (4, 1)).
(EngineCore pid=112) INFO 09-18 15:03:44 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
(EngineCore pid=112) INFO 09-18 15:03:44 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
(EngineCore pid=112)
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.71it/s]
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.70it/s]
(EngineCore pid=112) INFO 09-18 15:03:52 [speculator.py:150] Capturing model for speculator...
(EngineCore pid=112)
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 117.82it/s]
(EngineCore pid=112)
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 175.37it/s]
(EngineCore pid=112) INFO 09-18 15:03:53 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.14 GiB
(EngineCore pid=112) INFO 09-18 15:03:53 [gpu_worker.py:824] CUDA graph pool memory: 0.14 GiB (actual), 0.15 GiB (estimated), difference: 0.01 GiB (5.6%).
(EngineCore pid=112) INFO 09-18 15:03:53 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.985, 81.8 GiB). Actual usage is 77.85 GiB for consumed memory (weights + non-torch), 1.17 GiB for peak activation, and 0.14 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=2676895704` (2.49 GiB) to fit into requested memory, or `--kv-cache-memory=3523562496` (3.28 GiB) to fully utilize gpu memory. Current kv cache memory in use is 2.78 GiB.
(EngineCore pid=112) INFO 09-18 15:03:54 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
(EngineCore pid=112) WARNING 09-18 15:03:54 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
(EngineCore pid=112) INFO 09-18 15:03:54 [core.py:380] init engine (profile, create kv cache, warmup model) took 47.63 s
(EngineCore pid=112) INFO 09-18 15:03:54 [kv_cache_utils.py:762] kv cache group sizes [3200, 3200, 3200, 3200, 8, 3200]
(EngineCore pid=112) INFO 09-18 15:03:54 [kv_cache_utils.py:763] kv lcm block sizes 3200
(EngineCore pid=112) INFO 09-18 15:03:54 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
(APIServer pid=1) INFO 09-18 15:03:54 [entry.py:132] Supported tasks: ['generate']
(APIServer pid=1) INFO 09-18 15:03:54 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
(APIServer pid=1) INFO 09-18 15:03:54 [parser_manager.py:34] "auto" tool choice has been enabled.
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
(APIServer pid=1) INFO 09-18 15:03:55 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
(APIServer pid=1) WARNING 09-18 15:03:55 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
(APIServer pid=1) INFO 09-18 15:03:55 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:60] Available routes are:
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /openapi.json, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /docs, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /redoc, Methods: HEAD, GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /load, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /version, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /health, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /metrics, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /tokenize, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /detokenize, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/models, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /ping, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /ping, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /invocations, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/completions, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/messages, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /generative_scoring, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
(APIServer pid=1) INFO 09-18 15:03:55 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
(APIServer pid=1) INFO: Started server process [1]
(APIServer pid=1) INFO: Waiting for application startup.
(APIServer pid=1) INFO: Application startup complete.
(APIServer pid=1) INFO: 127.0.0.1:37726 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42266 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42282 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _count_expert_num_tokens. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _compute_local_logits_stats_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _rejection_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(EngineCore pid=112) WARNING 09-18 15:03:59 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _resample_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
(APIServer pid=1) INFO: 172.21.0.1:42298 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42314 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42316 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42324 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42344 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42350 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42356 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:42358 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51442 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51456 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51468 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:05 [loggers.py:323] Engine 000: Avg prompt throughput: 70.7 tokens/s, Avg generation throughput: 78.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%
(APIServer pid=1) INFO 09-18 15:04:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.81, Accepted throughput: 57.96 tokens/s, Drafted throughput: 61.97 tokens/s, Accepted: 637 tokens, Drafted: 681 tokens, Per-position acceptance rate: 0.996, 0.934, 0.877, Avg Draft acceptance rate: 93.5%
(APIServer pid=1) INFO: 172.21.0.1:51480 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51482 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51486 - "POST /v1/chat/completions HTTP/1.1" 200 OK
[rank0]:[W918 15:04:08.952733466 CUDACachingAllocator.cpp:3933] memory allocation failed with OOM on device 0 while trying to allocate 377487360 bytes (free: 92209152, total: 89173131264).
(APIServer pid=1) INFO: 172.21.0.1:51502 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:51512 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50416 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:15 [loggers.py:323] Engine 000: Avg prompt throughput: 2779.5 tokens/s, Avg generation throughput: 72.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.95, Accepted throughput: 47.90 tokens/s, Drafted throughput: 73.80 tokens/s, Accepted: 479 tokens, Drafted: 738 tokens, Per-position acceptance rate: 0.821, 0.646, 0.480, Avg Draft acceptance rate: 64.9%
(APIServer pid=1) INFO 09-18 15:04:25 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 129.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.59, Accepted throughput: 79.59 tokens/s, Drafted throughput: 150.28 tokens/s, Accepted: 796 tokens, Drafted: 1503 tokens, Per-position acceptance rate: 0.743, 0.511, 0.335, Avg Draft acceptance rate: 53.0%
(APIServer pid=1) INFO: 127.0.0.1:48946 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:59844 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:35 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 116.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.36, Accepted throughput: 66.99 tokens/s, Drafted throughput: 147.89 tokens/s, Accepted: 670 tokens, Drafted: 1479 tokens, Per-position acceptance rate: 0.684, 0.422, 0.254, Avg Draft acceptance rate: 45.3%
(APIServer pid=1) INFO 09-18 15:04:45 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.60, Accepted throughput: 80.09 tokens/s, Drafted throughput: 149.99 tokens/s, Accepted: 801 tokens, Drafted: 1500 tokens, Per-position acceptance rate: 0.750, 0.506, 0.346, Avg Draft acceptance rate: 53.4%
(APIServer pid=1) INFO: 172.21.0.1:41026 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:04:55 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 121.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:04:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.44, Accepted throughput: 71.39 tokens/s, Drafted throughput: 148.79 tokens/s, Accepted: 714 tokens, Drafted: 1488 tokens, Per-position acceptance rate: 0.692, 0.456, 0.292, Avg Draft acceptance rate: 48.0%
(APIServer pid=1) INFO: 127.0.0.1:57870 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:33434 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:05 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 118.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 69.00 tokens/s, Drafted throughput: 148.19 tokens/s, Accepted: 690 tokens, Drafted: 1482 tokens, Per-position acceptance rate: 0.704, 0.433, 0.259, Avg Draft acceptance rate: 46.6%
(APIServer pid=1) INFO: 172.21.0.1:33450 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:15 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 132.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 82.20 tokens/s, Drafted throughput: 149.41 tokens/s, Accepted: 822 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.759, 0.534, 0.357, Avg Draft acceptance rate: 55.0%
(APIServer pid=1) INFO: 172.21.0.1:59598 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:25 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 131.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 81.60 tokens/s, Drafted throughput: 148.49 tokens/s, Accepted: 816 tokens, Drafted: 1485 tokens, Per-position acceptance rate: 0.731, 0.560, 0.358, Avg Draft acceptance rate: 54.9%
(APIServer pid=1) INFO: 127.0.0.1:37860 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:35 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 136.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.73, Accepted throughput: 86.80 tokens/s, Drafted throughput: 150.30 tokens/s, Accepted: 868 tokens, Drafted: 1503 tokens, Per-position acceptance rate: 0.774, 0.563, 0.395, Avg Draft acceptance rate: 57.8%
(APIServer pid=1) INFO: 172.21.0.1:57216 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:45 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 131.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.63, Accepted throughput: 81.10 tokens/s, Drafted throughput: 149.40 tokens/s, Accepted: 811 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.739, 0.528, 0.361, Avg Draft acceptance rate: 54.3%
(APIServer pid=1) INFO: 172.21.0.1:38916 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50652 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:05:55 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 130.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:05:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.65, Accepted throughput: 81.50 tokens/s, Drafted throughput: 147.89 tokens/s, Accepted: 815 tokens, Drafted: 1479 tokens, Per-position acceptance rate: 0.740, 0.550, 0.363, Avg Draft acceptance rate: 55.1%
(APIServer pid=1) INFO: 127.0.0.1:49510 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:05 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 119.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 70.00 tokens/s, Drafted throughput: 149.71 tokens/s, Accepted: 700 tokens, Drafted: 1497 tokens, Per-position acceptance rate: 0.697, 0.437, 0.269, Avg Draft acceptance rate: 46.8%
(APIServer pid=1) INFO: 172.21.0.1:34316 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:15 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 126.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:15 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.55, Accepted throughput: 77.00 tokens/s, Drafted throughput: 149.10 tokens/s, Accepted: 770 tokens, Drafted: 1491 tokens, Per-position acceptance rate: 0.714, 0.519, 0.316, Avg Draft acceptance rate: 51.6%
(APIServer pid=1) INFO 09-18 15:06:25 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 122.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.45, Accepted throughput: 72.29 tokens/s, Drafted throughput: 149.39 tokens/s, Accepted: 723 tokens, Drafted: 1494 tokens, Per-position acceptance rate: 0.709, 0.454, 0.289, Avg Draft acceptance rate: 48.4%
(APIServer pid=1) INFO: 127.0.0.1:54572 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:50616 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:35 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 123.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.49, Accepted throughput: 73.70 tokens/s, Drafted throughput: 148.49 tokens/s, Accepted: 737 tokens, Drafted: 1485 tokens, Per-position acceptance rate: 0.715, 0.477, 0.297, Avg Draft acceptance rate: 49.6%
(APIServer pid=1) INFO 09-18 15:06:45 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:06:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.49, Accepted throughput: 73.10 tokens/s, Drafted throughput: 147.29 tokens/s, Accepted: 731 tokens, Drafted: 1473 tokens, Per-position acceptance rate: 0.701, 0.491, 0.297, Avg Draft acceptance rate: 49.6%
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
(APIServer pid=1) INFO: 172.21.0.1:47280 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47290 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47300 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47314 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47322 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47324 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47340 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47356 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47372 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47378 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47382 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47384 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47398 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47414 - "POST /tokenize HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:47418 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:06:55 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 70.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO: 127.0.0.1:40192 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:41722 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:05 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 13.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 48.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:07:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.69, Accepted throughput: 4.70 tokens/s, Drafted throughput: 5.25 tokens/s, Accepted: 94 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.943, 0.943, 0.800, Avg Draft acceptance rate: 89.5%
(APIServer pid=1) INFO 09-18 15:07:15 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 90.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO: 172.21.0.1:45834 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:25 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.9 tokens/s, Avg generation throughput: 12.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 68.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
(APIServer pid=1) INFO 09-18 15:07:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.47, Accepted throughput: 4.45 tokens/s, Drafted throughput: 5.40 tokens/s, Accepted: 89 tokens, Drafted: 108 tokens, Per-position acceptance rate: 0.917, 0.833, 0.722, Avg Draft acceptance rate: 82.4%
(APIServer pid=1) INFO: 127.0.0.1:36836 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:36034 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:35 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.43, Accepted throughput: 8.50 tokens/s, Drafted throughput: 10.50 tokens/s, Accepted: 85 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.886, 0.857, 0.686, Avg Draft acceptance rate: 81.0%
(APIServer pid=1) INFO 09-18 15:07:45 [loggers.py:323] Engine 000: Avg prompt throughput: 11167.1 tokens/s, Avg generation throughput: 2.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.33, Accepted throughput: 1.40 tokens/s, Drafted throughput: 1.80 tokens/s, Accepted: 14 tokens, Drafted: 18 tokens, Per-position acceptance rate: 0.833, 0.833, 0.667, Avg Draft acceptance rate: 77.8%
(APIServer pid=1) INFO: 172.21.0.1:48276 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48278 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48294 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48308 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48318 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48332 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48334 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:48342 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO 09-18 15:07:55 [loggers.py:323] Engine 000: Avg prompt throughput: 64.3 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4%
(APIServer pid=1) INFO 09-18 15:07:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.42, Accepted throughput: 110.00 tokens/s, Drafted throughput: 136.50 tokens/s, Accepted: 1100 tokens, Drafted: 1365 tokens, Per-position acceptance rate: 0.923, 0.804, 0.690, Avg Draft acceptance rate: 80.6%
(APIServer pid=1) INFO: 127.0.0.1:49226 - "GET /health HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37666 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37670 - "POST /v1/chat/completions HTTP/1.1" 200 OK
(APIServer pid=1) INFO: 172.21.0.1:37676 - "POST /v1/chat/completions HTTP/1.1" 200 OK
@@ -0,0 +1,5 @@
{
"startup": true,
"records": 44,
"failures": []
}
+46
View File
@@ -0,0 +1,46 @@
name: qwen38-flash-rtx6000d
services:
vllm:
image: local/qwen38-flash-6000d:0bfc7a15
build: .
container_name: qwen38-flash-6000d
gpus: all
shm_size: 8g
mem_limit: 210g
memswap_limit: 210g
restart: unless-stopped
ports:
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
environment:
HF_HUB_OFFLINE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_WORKER_MULTIPROC_METHOD: spawn
OMP_NUM_THREADS: "8"
TORCHINDUCTOR_COMPILE_THREADS: "2"
CUTE_DSL_ARCH: sm_120a
volumes:
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
- /data/flash-next/cache:/root/.cache
- ./secrets/api-key:/run/secrets/api-key:ro
entrypoint: ["/bin/bash", "-lc"]
command:
- >-
exec vllm serve /model
--served-model-name qwen3.8-flash-next
--host 0.0.0.0 --port 8000
--tensor-parallel-size 1 --dtype bfloat16
--engram-config '{"cpu_offload":true}'
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
--enable-chunked-prefill --enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":2}'
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune
--load-format safetensors
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
--api-key "$$(cat /run/secrets/api-key)"
healthcheck:
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30m
+624
View File
@@ -0,0 +1,624 @@
{"time": 1789742993.8930721, "available_gib": 172.44721603393555, "oom_kill": 0, "gpu": "84797, 85651, 0, 34", "stop": false}
{"time": 1789742995.9187403, "available_gib": 172.4483184814453, "oom_kill": 0, "gpu": "84797, 85651, 100, 40", "stop": false}
{"time": 1789742997.9435408, "available_gib": 172.43352890014648, "oom_kill": 0, "gpu": "84797, 85651, 88, 41", "stop": false}
{"time": 1789742999.969222, "available_gib": 172.44828033447266, "oom_kill": 0, "gpu": "84797, 85651, 100, 42", "stop": false}
{"time": 1789743001.994626, "available_gib": 172.44638061523438, "oom_kill": 0, "gpu": "84797, 85651, 100, 43", "stop": false}
{"time": 1789743004.0204384, "available_gib": 172.44339752197266, "oom_kill": 0, "gpu": "84797, 85651, 100, 43", "stop": false}
{"time": 1789743006.0460222, "available_gib": 172.41333389282227, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false}
{"time": 1789743008.0727768, "available_gib": 172.41802978515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false}
{"time": 1789743010.0991132, "available_gib": 172.4132423400879, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false}
{"time": 1789743012.1248758, "available_gib": 172.42271041870117, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false}
{"time": 1789743014.15081, "available_gib": 172.42755508422852, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false}
{"time": 1789743016.175784, "available_gib": 172.4231948852539, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false}
{"time": 1789743018.2006702, "available_gib": 172.4263038635254, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false}
{"time": 1789743020.2257159, "available_gib": 172.4193344116211, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false}
{"time": 1789743022.2507591, "available_gib": 172.41914749145508, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false}
{"time": 1789743024.2756846, "available_gib": 172.4089813232422, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false}
{"time": 1789743026.3011591, "available_gib": 172.41101837158203, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false}
{"time": 1789743028.3262846, "available_gib": 172.41080474853516, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false}
{"time": 1789743030.3521566, "available_gib": 172.41079330444336, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false}
{"time": 1789743032.3783853, "available_gib": 172.41777420043945, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false}
{"time": 1789743034.403828, "available_gib": 172.42018127441406, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false}
{"time": 1789743036.4298394, "available_gib": 172.41797256469727, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false}
{"time": 1789743038.4559314, "available_gib": 172.42116928100586, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false}
{"time": 1789743040.4814112, "available_gib": 172.42036056518555, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743042.5068018, "available_gib": 172.41886138916016, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743044.5317273, "available_gib": 172.41489791870117, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743046.5577147, "available_gib": 172.41412353515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743048.5837293, "available_gib": 172.41419219970703, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743050.6099472, "available_gib": 172.40557861328125, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false}
{"time": 1789743052.635809, "available_gib": 172.40662002563477, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743054.6614828, "available_gib": 172.40903854370117, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743056.686891, "available_gib": 172.40744018554688, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743058.7122042, "available_gib": 172.4039192199707, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743060.737155, "available_gib": 172.4096450805664, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743062.762888, "available_gib": 172.41356658935547, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743064.789021, "available_gib": 172.41518020629883, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743066.8147395, "available_gib": 172.4062843322754, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false}
{"time": 1789743068.8408678, "available_gib": 172.40911483764648, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743070.8669722, "available_gib": 172.4068145751953, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743072.893002, "available_gib": 172.40983963012695, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743074.9192512, "available_gib": 172.40817642211914, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743076.945159, "available_gib": 172.40537643432617, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743078.9712658, "available_gib": 172.40187454223633, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743080.9970002, "available_gib": 172.39935302734375, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743083.0224788, "available_gib": 172.40352249145508, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743085.0485544, "available_gib": 172.4073257446289, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743087.0739734, "available_gib": 172.40684509277344, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false}
{"time": 1789743089.0995305, "available_gib": 172.40550994873047, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743091.125159, "available_gib": 172.4019660949707, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743093.151235, "available_gib": 172.40221786499023, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743095.1767147, "available_gib": 172.40421295166016, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743097.2028806, "available_gib": 172.4161262512207, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743099.2291465, "available_gib": 172.4316177368164, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743101.2551146, "available_gib": 172.43370056152344, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743103.2816598, "available_gib": 172.43728256225586, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743105.307239, "available_gib": 172.43943786621094, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743107.3330648, "available_gib": 172.44949340820312, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743109.359068, "available_gib": 172.44104385375977, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743111.385109, "available_gib": 172.43330764770508, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743113.4113765, "available_gib": 172.43706893920898, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743115.4370093, "available_gib": 172.44180297851562, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743117.4625585, "available_gib": 172.44410705566406, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743119.4885128, "available_gib": 172.43635177612305, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743121.5149198, "available_gib": 172.44395446777344, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false}
{"time": 1789743123.5413435, "available_gib": 172.44974899291992, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743125.5679505, "available_gib": 172.44622421264648, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743127.5935173, "available_gib": 172.44979095458984, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743129.6191273, "available_gib": 172.45713424682617, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743131.6455028, "available_gib": 172.44577407836914, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743133.672497, "available_gib": 172.45316314697266, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743135.697847, "available_gib": 172.45311737060547, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743137.7242105, "available_gib": 172.44537353515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743139.7507234, "available_gib": 172.4357032775879, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743141.7774155, "available_gib": 172.44175720214844, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743143.8043108, "available_gib": 172.43401336669922, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743145.8301265, "available_gib": 172.44019317626953, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743147.855514, "available_gib": 172.45041275024414, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743149.8819315, "available_gib": 172.4467430114746, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743151.9083953, "available_gib": 172.45286178588867, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743153.934035, "available_gib": 172.4552459716797, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743155.9600441, "available_gib": 172.44754028320312, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743157.987035, "available_gib": 172.4474754333496, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743160.012713, "available_gib": 172.45933532714844, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false}
{"time": 1789743162.0334756, "available_gib": 172.38756561279297, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false}
{"time": 1789743164.0600572, "available_gib": 172.37825393676758, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false}
{"time": 1789743166.085538, "available_gib": 172.386474609375, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false}
{"time": 1789743168.1112113, "available_gib": 172.39423751831055, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false}
{"time": 1789743170.135376, "available_gib": 172.39533615112305, "oom_kill": 0, "gpu": "84797, 85651, 100, 59", "stop": false}
{"time": 1789743172.1629245, "available_gib": 172.40753555297852, "oom_kill": 0, "gpu": "84797, 85651, 100, 59", "stop": false}
{"time": 1789743174.1864367, "available_gib": 172.4146385192871, "oom_kill": 0, "gpu": "84797, 85651, 100, 60", "stop": false}
{"time": 1789743176.2113686, "available_gib": 172.4199676513672, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false}
{"time": 1789743178.235966, "available_gib": 172.41577529907227, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false}
{"time": 1789743180.2613661, "available_gib": 172.42212295532227, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false}
{"time": 1789743182.2844982, "available_gib": 172.41865921020508, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false}
{"time": 1789743184.3087924, "available_gib": 172.41510391235352, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false}
{"time": 1789743186.3337767, "available_gib": 172.42054748535156, "oom_kill": 0, "gpu": "84797, 85651, 100, 62", "stop": false}
{"time": 1789743188.3582373, "available_gib": 172.41769409179688, "oom_kill": 0, "gpu": "84797, 85651, 100, 62", "stop": false}
{"time": 1789743190.3825104, "available_gib": 172.41735076904297, "oom_kill": 0, "gpu": "84797, 85651, 100, 63", "stop": false}
{"time": 1789743192.4065006, "available_gib": 172.42127227783203, "oom_kill": 0, "gpu": "84797, 85651, 100, 63", "stop": false}
{"time": 1789743194.4317603, "available_gib": 172.42076873779297, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false}
{"time": 1789743196.4574146, "available_gib": 172.41965103149414, "oom_kill": 0, "gpu": "84797, 85651, 99, 58", "stop": false}
{"time": 1789743198.4829125, "available_gib": 172.43689346313477, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false}
{"time": 1789743200.508612, "available_gib": 172.41334915161133, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false}
{"time": 1789743202.5339994, "available_gib": 172.41661071777344, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false}
{"time": 1789743204.55841, "available_gib": 172.39996337890625, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false}
{"time": 1789743206.58313, "available_gib": 172.39534378051758, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false}
{"time": 1789743208.6087117, "available_gib": 172.3869171142578, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false}
{"time": 1789743210.6334622, "available_gib": 172.39297485351562, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false}
{"time": 1789743212.6587076, "available_gib": 172.39804077148438, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false}
{"time": 1789743214.679407, "available_gib": 174.36180877685547, "oom_kill": 0, "gpu": "84797, 85651, 0, 51", "stop": false}
{"time": 1789743216.6994534, "available_gib": 175.1852035522461, "oom_kill": 0, "gpu": "84797, 85651, 0, 50", "stop": false}
{"time": 1789743218.7198522, "available_gib": 175.1703224182129, "oom_kill": 0, "gpu": "84797, 85651, 0, 49", "stop": false}
{"time": 1789743220.740253, "available_gib": 175.17662811279297, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false}
{"time": 1789743222.7602718, "available_gib": 175.17460250854492, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false}
{"time": 1789743229.6084123, "available_gib": 190.91661071777344, "oom_kill": 0, "gpu": "3, 85651, 0, 45", "stop": false}
{"time": 1789743231.630355, "available_gib": 242.44229888916016, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false}
{"time": 1789743233.6530468, "available_gib": 242.35516738891602, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false}
{"time": 1789743235.6799445, "available_gib": 242.30181884765625, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false}
{"time": 1789743237.7048993, "available_gib": 242.2751808166504, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
{"time": 1789743239.7298052, "available_gib": 242.2560272216797, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
{"time": 1789743241.7549934, "available_gib": 242.20417404174805, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
{"time": 1789743243.779685, "available_gib": 242.14367294311523, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false}
{"time": 1789743245.8037167, "available_gib": 242.09040069580078, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false}
{"time": 1789743247.8466423, "available_gib": 241.7714080810547, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789743249.8889477, "available_gib": 241.40981674194336, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789743251.9310393, "available_gib": 241.1487274169922, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789743253.9736116, "available_gib": 240.97250366210938, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789743256.0156634, "available_gib": 240.2674903869629, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789743258.0573394, "available_gib": 240.00533294677734, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
{"time": 1789743260.1017191, "available_gib": 239.7233428955078, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
{"time": 1789743262.1220598, "available_gib": 239.19343948364258, "oom_kill": 0, "gpu": "683, 85651, 2, 37", "stop": false}
{"time": 1789743264.1423042, "available_gib": 238.81260681152344, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789743266.1623287, "available_gib": 234.10787963867188, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789743268.1824176, "available_gib": 229.910400390625, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743270.2026012, "available_gib": 225.91469192504883, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743272.2226171, "available_gib": 221.90094757080078, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743274.2424407, "available_gib": 217.90217971801758, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743276.2622116, "available_gib": 213.95528411865234, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743278.2818873, "available_gib": 210.04045486450195, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743280.302443, "available_gib": 206.0911102294922, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743282.3223915, "available_gib": 202.18833541870117, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743284.342848, "available_gib": 198.23841857910156, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743286.3631046, "available_gib": 194.30224227905273, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743288.3831742, "available_gib": 190.38965225219727, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743290.4034154, "available_gib": 186.47941970825195, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743292.4239209, "available_gib": 182.56446838378906, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743294.4439209, "available_gib": 178.62585067749023, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743296.4635558, "available_gib": 174.67407989501953, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743298.484325, "available_gib": 173.4862289428711, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false}
{"time": 1789743300.5052922, "available_gib": 173.47106552124023, "oom_kill": 0, "gpu": "76995, 85651, 7, 36", "stop": false}
{"time": 1789743302.5263877, "available_gib": 173.44770050048828, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743304.5477655, "available_gib": 173.44190216064453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743306.572195, "available_gib": 173.42721557617188, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743308.5933337, "available_gib": 173.46649551391602, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743310.614528, "available_gib": 173.50944900512695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743312.6357765, "available_gib": 173.54594039916992, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743314.6571312, "available_gib": 173.57780838012695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743316.6779072, "available_gib": 173.67039108276367, "oom_kill": 0, "gpu": "76995, 85651, 8, 36", "stop": false}
{"time": 1789743318.6988745, "available_gib": 173.6916618347168, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false}
{"time": 1789743320.7197983, "available_gib": 173.72816467285156, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false}
{"time": 1789743322.7410035, "available_gib": 173.74269485473633, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743324.7621593, "available_gib": 173.79139709472656, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false}
{"time": 1789743326.7830343, "available_gib": 173.8392562866211, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743328.8042507, "available_gib": 173.84291458129883, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743330.8253624, "available_gib": 173.9206657409668, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743332.846458, "available_gib": 173.9108123779297, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743334.8676717, "available_gib": 173.95557403564453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743336.8889298, "available_gib": 173.98338317871094, "oom_kill": 0, "gpu": "76995, 85651, 6, 35", "stop": false}
{"time": 1789743338.9100575, "available_gib": 174.0068702697754, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false}
{"time": 1789743340.9305177, "available_gib": 174.01351165771484, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false}
{"time": 1789743342.9574733, "available_gib": 174.00801849365234, "oom_kill": 0, "gpu": "76995, 85651, 0, 35", "stop": false}
{"time": 1789743344.9781544, "available_gib": 173.8778190612793, "oom_kill": 0, "gpu": "82531, 85651, 0, 35", "stop": false}
{"time": 1789743346.9989328, "available_gib": 173.89619827270508, "oom_kill": 0, "gpu": "82531, 85651, 50, 36", "stop": false}
{"time": 1789743349.0198088, "available_gib": 173.94148635864258, "oom_kill": 0, "gpu": "82559, 85651, 0, 35", "stop": false}
{"time": 1789743351.040807, "available_gib": 173.9337158203125, "oom_kill": 0, "gpu": "79121, 85651, 0, 35", "stop": false}
{"time": 1789743353.0616393, "available_gib": 173.59209823608398, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false}
{"time": 1789743355.0827384, "available_gib": 173.4678955078125, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false}
{"time": 1789743357.104013, "available_gib": 172.85984802246094, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743359.1243138, "available_gib": 173.2513542175293, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743361.1446815, "available_gib": 173.23069763183594, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743363.1656852, "available_gib": 173.2288589477539, "oom_kill": 0, "gpu": "82317, 85651, 100, 43", "stop": false}
{"time": 1789743365.1863644, "available_gib": 173.2270965576172, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false}
{"time": 1789743367.206995, "available_gib": 173.21285247802734, "oom_kill": 0, "gpu": "82317, 85651, 64, 42", "stop": false}
{"time": 1789743369.227728, "available_gib": 173.15103149414062, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743371.2485144, "available_gib": 173.17145538330078, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743373.2686968, "available_gib": 173.10255432128906, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743375.289341, "available_gib": 173.1105842590332, "oom_kill": 0, "gpu": "79683, 85651, 63, 42", "stop": false}
{"time": 1789743377.310061, "available_gib": 173.2584686279297, "oom_kill": 0, "gpu": "80157, 85651, 0, 37", "stop": false}
{"time": 1789743379.330606, "available_gib": 173.22555923461914, "oom_kill": 0, "gpu": "80239, 85651, 0, 36", "stop": false}
{"time": 1789743381.3509958, "available_gib": 173.3180694580078, "oom_kill": 0, "gpu": "83047, 85651, 1, 36", "stop": false}
{"time": 1789743383.370858, "available_gib": 173.2816619873047, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false}
{"time": 1789743385.3907626, "available_gib": 173.23712921142578, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false}
{"time": 1789743387.4104686, "available_gib": 173.25224685668945, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false}
{"time": 1789743389.4309165, "available_gib": 173.26059341430664, "oom_kill": 0, "gpu": "83049, 85651, 1, 36", "stop": false}
{"time": 1789743391.450981, "available_gib": 173.16680526733398, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false}
{"time": 1789743393.4715426, "available_gib": 173.06145477294922, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false}
{"time": 1789743395.492277, "available_gib": 173.10271835327148, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false}
{"time": 1789743397.5132124, "available_gib": 173.11201095581055, "oom_kill": 0, "gpu": "83157, 85651, 0, 36", "stop": false}
{"time": 1789743399.5333133, "available_gib": 173.06784439086914, "oom_kill": 0, "gpu": "83157, 85651, 0, 36", "stop": false}
{"time": 1789743401.5540614, "available_gib": 173.05779266357422, "oom_kill": 0, "gpu": "83163, 85651, 0, 36", "stop": false}
{"time": 1789743403.579573, "available_gib": 173.06655502319336, "oom_kill": 0, "gpu": "83163, 85651, 84, 41", "stop": false}
{"time": 1789743405.6054838, "available_gib": 173.08317184448242, "oom_kill": 0, "gpu": "83163, 85651, 100, 42", "stop": false}
{"time": 1789743407.6315184, "available_gib": 173.0826072692871, "oom_kill": 0, "gpu": "83163, 85651, 100, 43", "stop": false}
{"time": 1789743409.6563766, "available_gib": 173.06651306152344, "oom_kill": 0, "gpu": "83165, 85651, 77, 43", "stop": false}
{"time": 1789743411.6810968, "available_gib": 172.9534683227539, "oom_kill": 0, "gpu": "83835, 85651, 100, 44", "stop": false}
{"time": 1789743413.705223, "available_gib": 172.33104705810547, "oom_kill": 0, "gpu": "84753, 85651, 100, 48", "stop": false}
{"time": 1789743415.7302306, "available_gib": 172.4871826171875, "oom_kill": 0, "gpu": "84755, 85651, 100, 44", "stop": false}
{"time": 1789743417.75524, "available_gib": 172.52772903442383, "oom_kill": 0, "gpu": "84755, 85651, 100, 46", "stop": false}
{"time": 1789743419.7805443, "available_gib": 172.54979705810547, "oom_kill": 0, "gpu": "84759, 85651, 100, 45", "stop": false}
{"time": 1789743421.806648, "available_gib": 172.57553100585938, "oom_kill": 0, "gpu": "84759, 85651, 100, 45", "stop": false}
{"time": 1789743423.8317566, "available_gib": 172.58489227294922, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false}
{"time": 1789743425.8571992, "available_gib": 172.58294677734375, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false}
{"time": 1789743427.8820453, "available_gib": 172.57914352416992, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false}
{"time": 1789743429.9071174, "available_gib": 172.58489227294922, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false}
{"time": 1789743431.9314394, "available_gib": 172.58298873901367, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false}
{"time": 1789743433.9564836, "available_gib": 172.58153915405273, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false}
{"time": 1789743435.9823027, "available_gib": 172.58718490600586, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false}
{"time": 1789743438.0072143, "available_gib": 172.58710861206055, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false}
{"time": 1789743440.0316548, "available_gib": 172.58600234985352, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743442.0570004, "available_gib": 172.58583068847656, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743444.081906, "available_gib": 172.5858039855957, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743446.1073487, "available_gib": 172.58935546875, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743448.1323597, "available_gib": 172.58035278320312, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743450.15754, "available_gib": 172.58575820922852, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false}
{"time": 1789743452.1833196, "available_gib": 172.5850830078125, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743454.208151, "available_gib": 172.5833282470703, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743456.2327328, "available_gib": 172.58715438842773, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743458.2573335, "available_gib": 172.58446502685547, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743460.2828667, "available_gib": 172.5850715637207, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743462.3078372, "available_gib": 172.5800018310547, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743464.33269, "available_gib": 172.58169174194336, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false}
{"time": 1789743466.3571842, "available_gib": 172.57829666137695, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743468.3811376, "available_gib": 172.57852172851562, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743470.4063337, "available_gib": 172.59275436401367, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743472.431251, "available_gib": 172.59864044189453, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743474.4564416, "available_gib": 172.60233306884766, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743476.481559, "available_gib": 172.59457778930664, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743478.5064373, "available_gib": 172.6103286743164, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743480.5307262, "available_gib": 172.62775039672852, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743482.5562549, "available_gib": 172.61786270141602, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743484.5813863, "available_gib": 172.62933731079102, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743486.6072564, "available_gib": 172.62554931640625, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743488.63282, "available_gib": 172.62833404541016, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false}
{"time": 1789743490.6583722, "available_gib": 172.62057876586914, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743492.684071, "available_gib": 172.62283325195312, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743494.7092838, "available_gib": 172.6273536682129, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743496.7351062, "available_gib": 172.62580108642578, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743498.7601368, "available_gib": 172.62276077270508, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743500.78492, "available_gib": 172.62115859985352, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743502.8107402, "available_gib": 172.6251106262207, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743504.8364482, "available_gib": 172.6214942932129, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743506.8617234, "available_gib": 172.6231918334961, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743508.8863335, "available_gib": 172.62788772583008, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743510.9119735, "available_gib": 172.62989044189453, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743512.9368422, "available_gib": 172.63046264648438, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743514.9624221, "available_gib": 172.62270736694336, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743516.987541, "available_gib": 172.62264251708984, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743519.0120552, "available_gib": 172.61489868164062, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743521.0366497, "available_gib": 172.61508560180664, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743523.0619605, "available_gib": 172.61526107788086, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743525.0872154, "available_gib": 172.61988067626953, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743527.1127303, "available_gib": 172.62136459350586, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false}
{"time": 1789743529.137909, "available_gib": 172.61513137817383, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743531.1632333, "available_gib": 172.62389373779297, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743533.188662, "available_gib": 172.6221923828125, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743535.2138846, "available_gib": 172.62632751464844, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743537.2387412, "available_gib": 172.6260643005371, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743539.2632637, "available_gib": 172.62511825561523, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743541.2876751, "available_gib": 172.62701797485352, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743543.3130155, "available_gib": 172.62727737426758, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743545.3385234, "available_gib": 172.63215255737305, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743547.3637378, "available_gib": 172.6363410949707, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743549.3886216, "available_gib": 172.6325225830078, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743551.4133384, "available_gib": 172.6255989074707, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743553.438609, "available_gib": 172.62480926513672, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743555.4636686, "available_gib": 172.62656784057617, "oom_kill": 0, "gpu": "84759, 85651, 86, 52", "stop": false}
{"time": 1789743557.4889739, "available_gib": 172.62308883666992, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743559.5141568, "available_gib": 172.6314353942871, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743561.539715, "available_gib": 172.62345504760742, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743563.565002, "available_gib": 172.63013458251953, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743565.5897946, "available_gib": 172.61325073242188, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743567.614607, "available_gib": 172.60382843017578, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743569.6399143, "available_gib": 172.61295318603516, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743571.665013, "available_gib": 172.6142234802246, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743573.691099, "available_gib": 172.61298370361328, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743575.716717, "available_gib": 172.61080169677734, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743577.741419, "available_gib": 172.60554885864258, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743579.7667089, "available_gib": 172.60740661621094, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743581.7916934, "available_gib": 172.6053695678711, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743583.8170035, "available_gib": 172.60088729858398, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743585.8422914, "available_gib": 172.6147918701172, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743587.8678956, "available_gib": 172.61311721801758, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743589.893016, "available_gib": 172.61607360839844, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743591.9179227, "available_gib": 172.6152458190918, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false}
{"time": 1789743593.9394135, "available_gib": 172.49266815185547, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false}
{"time": 1789743595.96263, "available_gib": 172.47529983520508, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false}
{"time": 1789743597.9865532, "available_gib": 172.47649383544922, "oom_kill": 0, "gpu": "84759, 85651, 100, 58", "stop": false}
{"time": 1789743600.011351, "available_gib": 172.47513961791992, "oom_kill": 0, "gpu": "84759, 85651, 100, 58", "stop": false}
{"time": 1789743602.0350573, "available_gib": 172.47442626953125, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false}
{"time": 1789743604.0597992, "available_gib": 172.47433471679688, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false}
{"time": 1789743606.0856547, "available_gib": 172.47597885131836, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false}
{"time": 1789743608.1109443, "available_gib": 172.4771270751953, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false}
{"time": 1789743610.136922, "available_gib": 172.47300720214844, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743612.1622267, "available_gib": 172.46633911132812, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743614.1874115, "available_gib": 172.46381378173828, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743616.210489, "available_gib": 172.47092056274414, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false}
{"time": 1789743618.2337203, "available_gib": 172.46821975708008, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false}
{"time": 1789743620.2577622, "available_gib": 172.47286987304688, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false}
{"time": 1789743622.2841165, "available_gib": 172.46925735473633, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false}
{"time": 1789743624.3088126, "available_gib": 172.4719467163086, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false}
{"time": 1789743626.3321114, "available_gib": 172.4673614501953, "oom_kill": 0, "gpu": "84759, 85651, 100, 63", "stop": false}
{"time": 1789743628.3577669, "available_gib": 172.45921325683594, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false}
{"time": 1789743630.381975, "available_gib": 172.45735931396484, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false}
{"time": 1789743632.4074128, "available_gib": 172.47180557250977, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false}
{"time": 1789743634.433758, "available_gib": 172.4716796875, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false}
{"time": 1789743636.4596326, "available_gib": 172.47660446166992, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false}
{"time": 1789743638.4853396, "available_gib": 172.4737548828125, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false}
{"time": 1789743640.5112894, "available_gib": 172.4694061279297, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false}
{"time": 1789743642.5373013, "available_gib": 172.4714698791504, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743644.5635104, "available_gib": 172.4744110107422, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743646.5894952, "available_gib": 172.47317504882812, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743648.6155689, "available_gib": 172.47101211547852, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743650.6409676, "available_gib": 172.46884155273438, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false}
{"time": 1789743652.6616406, "available_gib": 174.40550231933594, "oom_kill": 0, "gpu": "84759, 85651, 0, 50", "stop": false}
{"time": 1789743654.6818483, "available_gib": 174.40292358398438, "oom_kill": 0, "gpu": "84759, 85651, 0, 50", "stop": false}
{"time": 1789743656.7018807, "available_gib": 175.23258590698242, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false}
{"time": 1789743658.7224524, "available_gib": 175.22650909423828, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false}
{"time": 1789743660.7429912, "available_gib": 175.2299690246582, "oom_kill": 0, "gpu": "84759, 85651, 0, 47", "stop": false}
{"time": 1789743667.684546, "available_gib": 189.53302764892578, "oom_kill": 0, "gpu": "3, 85651, 0, 45", "stop": false}
{"time": 1789743669.7074604, "available_gib": 242.46288299560547, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false}
{"time": 1789743671.7295043, "available_gib": 242.38314819335938, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false}
{"time": 1789743673.7539766, "available_gib": 242.30734252929688, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false}
{"time": 1789743675.77852, "available_gib": 242.25893783569336, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
{"time": 1789743677.8033016, "available_gib": 242.2384796142578, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
{"time": 1789743679.8281374, "available_gib": 242.18122482299805, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
{"time": 1789743681.8531134, "available_gib": 242.11988830566406, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false}
{"time": 1789743683.877929, "available_gib": 242.06714248657227, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false}
{"time": 1789743685.9199724, "available_gib": 241.74395751953125, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789743687.9635887, "available_gib": 241.41803741455078, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789743690.005523, "available_gib": 241.17269897460938, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789743692.048056, "available_gib": 240.93939590454102, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789743694.0926096, "available_gib": 240.32037353515625, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
{"time": 1789743696.134711, "available_gib": 240.09738159179688, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
{"time": 1789743698.1778443, "available_gib": 239.82658004760742, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
{"time": 1789743700.1996496, "available_gib": 239.30536651611328, "oom_kill": 0, "gpu": "683, 85651, 2, 37", "stop": false}
{"time": 1789743702.2195978, "available_gib": 239.08550262451172, "oom_kill": 0, "gpu": "2915, 85651, 0, 37", "stop": false}
{"time": 1789743704.239852, "available_gib": 234.2517204284668, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743706.2606213, "available_gib": 230.40094375610352, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743708.280376, "available_gib": 226.4360122680664, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743710.3012598, "available_gib": 222.46880340576172, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743712.32121, "available_gib": 218.4713897705078, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743714.341048, "available_gib": 214.47138214111328, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743716.360451, "available_gib": 210.4367904663086, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743718.3802285, "available_gib": 206.42337799072266, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743720.4000227, "available_gib": 202.46843719482422, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743722.4194143, "available_gib": 198.42879104614258, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789743724.4388602, "available_gib": 194.38553619384766, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743726.4584868, "available_gib": 190.3592414855957, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743728.4785507, "available_gib": 186.39047622680664, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743730.4990554, "available_gib": 182.43556594848633, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743732.5190928, "available_gib": 178.44296646118164, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743734.5388536, "available_gib": 174.4428482055664, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false}
{"time": 1789743736.5602393, "available_gib": 173.5626335144043, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false}
{"time": 1789743738.581365, "available_gib": 173.58563232421875, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743740.602393, "available_gib": 173.59395599365234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743742.623244, "available_gib": 173.592041015625, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743744.644461, "available_gib": 173.58086395263672, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743746.6653383, "available_gib": 173.5805892944336, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743748.6863484, "available_gib": 173.57102584838867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743750.7072456, "available_gib": 173.6135025024414, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743752.7283082, "available_gib": 173.61641693115234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743754.7494485, "available_gib": 173.60739517211914, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743756.770568, "available_gib": 173.62178421020508, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743758.791231, "available_gib": 173.6512794494629, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743760.8119733, "available_gib": 173.6450080871582, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743762.8327234, "available_gib": 173.65488052368164, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743764.8535624, "available_gib": 173.63638305664062, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false}
{"time": 1789743766.8746958, "available_gib": 173.65426635742188, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743768.89607, "available_gib": 173.6699333190918, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743770.9170897, "available_gib": 173.67607879638672, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789743772.9382615, "available_gib": 173.70412826538086, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false}
{"time": 1789743774.959427, "available_gib": 173.7211685180664, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false}
{"time": 1789743776.9805934, "available_gib": 173.74411010742188, "oom_kill": 0, "gpu": "76995, 85651, 37, 35", "stop": false}
{"time": 1789743779.0047975, "available_gib": 173.68999481201172, "oom_kill": 0, "gpu": "76995, 85651, 0, 35", "stop": false}
{"time": 1789743781.0252128, "available_gib": 173.55987930297852, "oom_kill": 0, "gpu": "82531, 85651, 0, 35", "stop": false}
{"time": 1789743783.0469394, "available_gib": 173.57495498657227, "oom_kill": 0, "gpu": "82531, 85651, 10, 35", "stop": false}
{"time": 1789743785.0683353, "available_gib": 173.6102638244629, "oom_kill": 0, "gpu": "82559, 85651, 0, 35", "stop": false}
{"time": 1789743787.0894022, "available_gib": 173.57339477539062, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false}
{"time": 1789743789.109543, "available_gib": 173.24087524414062, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false}
{"time": 1789743791.13055, "available_gib": 173.11621475219727, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false}
{"time": 1789743793.152408, "available_gib": 172.55051803588867, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743795.1728363, "available_gib": 172.77595138549805, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743797.1929188, "available_gib": 172.77360916137695, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789743799.214182, "available_gib": 172.7921257019043, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false}
{"time": 1789743801.2358875, "available_gib": 172.80052947998047, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false}
{"time": 1789743803.2571647, "available_gib": 172.80226516723633, "oom_kill": 0, "gpu": "82317, 85651, 0, 39", "stop": false}
{"time": 1789743805.2780788, "available_gib": 172.79526901245117, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743807.299253, "available_gib": 172.80913162231445, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743809.3191426, "available_gib": 172.6723861694336, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789743811.3417072, "available_gib": 173.06355667114258, "oom_kill": 0, "gpu": "80519, 85651, 2, 37", "stop": false}
{"time": 1789743813.3622966, "available_gib": 173.13029861450195, "oom_kill": 0, "gpu": "80189, 85651, 0, 37", "stop": false}
{"time": 1789743815.3829036, "available_gib": 173.09708404541016, "oom_kill": 0, "gpu": "80241, 85651, 0, 36", "stop": false}
{"time": 1789743817.4042423, "available_gib": 173.31471633911133, "oom_kill": 0, "gpu": "80197, 85651, 0, 36", "stop": false}
{"time": 1789743819.4245021, "available_gib": 173.30271530151367, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false}
{"time": 1789743821.4452205, "available_gib": 173.21561431884766, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false}
{"time": 1789743823.4653258, "available_gib": 173.16995239257812, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false}
{"time": 1789743825.4860725, "available_gib": 173.21529006958008, "oom_kill": 0, "gpu": "83005, 85651, 0, 36", "stop": false}
{"time": 1789743827.5069244, "available_gib": 173.12132263183594, "oom_kill": 0, "gpu": "83011, 85651, 2, 36", "stop": false}
{"time": 1789743829.5271885, "available_gib": 172.99286651611328, "oom_kill": 0, "gpu": "83011, 85651, 0, 36", "stop": false}
{"time": 1789743831.549038, "available_gib": 173.04050064086914, "oom_kill": 0, "gpu": "83011, 85651, 5, 36", "stop": false}
{"time": 1789743833.570074, "available_gib": 173.05452728271484, "oom_kill": 0, "gpu": "83149, 85651, 0, 36", "stop": false}
{"time": 1789743835.5897286, "available_gib": 173.02277755737305, "oom_kill": 0, "gpu": "83149, 85651, 0, 36", "stop": false}
{"time": 1789743837.6093662, "available_gib": 173.02072143554688, "oom_kill": 0, "gpu": "83149, 85651, 0, 35", "stop": false}
{"time": 1789743839.6298525, "available_gib": 173.01683044433594, "oom_kill": 0, "gpu": "83155, 85651, 0, 36", "stop": false}
{"time": 1789743841.6556835, "available_gib": 173.02434158325195, "oom_kill": 0, "gpu": "83155, 85651, 100, 42", "stop": false}
{"time": 1789743843.681355, "available_gib": 173.0242156982422, "oom_kill": 0, "gpu": "83155, 85651, 100, 43", "stop": false}
{"time": 1789743845.7065227, "available_gib": 173.01176071166992, "oom_kill": 0, "gpu": "83155, 85651, 100, 44", "stop": false}
{"time": 1789743847.7316642, "available_gib": 172.9279899597168, "oom_kill": 0, "gpu": "83827, 85651, 100, 43", "stop": false}
{"time": 1789743849.7550159, "available_gib": 172.27711868286133, "oom_kill": 0, "gpu": "84693, 85651, 100, 48", "stop": false}
{"time": 1789743851.779772, "available_gib": 172.35914993286133, "oom_kill": 0, "gpu": "84693, 85651, 100, 48", "stop": false}
{"time": 1789743853.8058956, "available_gib": 172.42580795288086, "oom_kill": 0, "gpu": "84697, 85651, 100, 45", "stop": false}
{"time": 1789743855.83155, "available_gib": 172.41658782958984, "oom_kill": 0, "gpu": "84697, 85651, 100, 45", "stop": false}
{"time": 1789743857.8563097, "available_gib": 172.4184684753418, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false}
{"time": 1789743859.8813326, "available_gib": 172.41818618774414, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false}
{"time": 1789743861.9067163, "available_gib": 172.41766738891602, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false}
{"time": 1789743863.932184, "available_gib": 172.41712951660156, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false}
{"time": 1789743865.956786, "available_gib": 172.42663192749023, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false}
{"time": 1789743867.982782, "available_gib": 172.42996215820312, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false}
{"time": 1789743870.0088744, "available_gib": 172.4624481201172, "oom_kill": 0, "gpu": "84697, 85651, 100, 48", "stop": false}
{"time": 1789743872.0358696, "available_gib": 172.49190139770508, "oom_kill": 0, "gpu": "84697, 85651, 80, 48", "stop": false}
{"time": 1789743874.0616338, "available_gib": 172.50891876220703, "oom_kill": 0, "gpu": "84697, 85651, 100, 48", "stop": false}
{"time": 1789743876.0869436, "available_gib": 172.508544921875, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false}
{"time": 1789743878.1125872, "available_gib": 172.5115089416504, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false}
{"time": 1789743880.1391616, "available_gib": 172.51123046875, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false}
{"time": 1789743882.1654341, "available_gib": 172.51721954345703, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false}
{"time": 1789743884.1917255, "available_gib": 172.52043914794922, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false}
{"time": 1789743886.2185214, "available_gib": 172.52492904663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false}
{"time": 1789743888.2450676, "available_gib": 172.5250587463379, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false}
{"time": 1789743890.2708652, "available_gib": 172.5223503112793, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false}
{"time": 1789743892.2970574, "available_gib": 172.5244483947754, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false}
{"time": 1789743894.323345, "available_gib": 172.52897262573242, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false}
{"time": 1789743896.3477945, "available_gib": 172.5200309753418, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743898.3742619, "available_gib": 172.52397537231445, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743900.4001124, "available_gib": 172.53362274169922, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743902.4257197, "available_gib": 172.53076934814453, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743904.4522634, "available_gib": 172.53030395507812, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743906.4786067, "available_gib": 172.5284767150879, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743908.5038824, "available_gib": 172.53790283203125, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false}
{"time": 1789743910.5290341, "available_gib": 172.55227279663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743912.5544894, "available_gib": 172.5442771911621, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743914.580608, "available_gib": 172.5365219116211, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743916.6071196, "available_gib": 172.54441452026367, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743918.6325831, "available_gib": 172.5501251220703, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743920.6588407, "available_gib": 172.54729080200195, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743922.6839206, "available_gib": 172.56165313720703, "oom_kill": 0, "gpu": "84697, 85651, 94, 52", "stop": false}
{"time": 1789743924.710274, "available_gib": 172.56248474121094, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743926.7364788, "available_gib": 172.56631088256836, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743928.7624378, "available_gib": 172.57713317871094, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743930.7882645, "available_gib": 172.57723236083984, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false}
{"time": 1789743932.8145838, "available_gib": 172.57740783691406, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743934.8404663, "available_gib": 172.57373809814453, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743936.866788, "available_gib": 172.57367324829102, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743938.8923123, "available_gib": 172.57363891601562, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743940.9182467, "available_gib": 172.57357788085938, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743942.943654, "available_gib": 172.57180404663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743944.9701402, "available_gib": 172.5783462524414, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743946.996388, "available_gib": 172.5782928466797, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743949.0224254, "available_gib": 172.58384323120117, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743951.0482817, "available_gib": 172.586669921875, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743953.0737877, "available_gib": 172.59230041503906, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743955.1002982, "available_gib": 172.59587478637695, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743957.125969, "available_gib": 172.58476638793945, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743959.152024, "available_gib": 172.58797073364258, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743961.1781206, "available_gib": 172.5951156616211, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743963.2033892, "available_gib": 172.58736038208008, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false}
{"time": 1789743965.2290244, "available_gib": 172.58730697631836, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743967.2558095, "available_gib": 172.58724212646484, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743969.2818677, "available_gib": 172.58523559570312, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743971.30743, "available_gib": 172.58755493164062, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743973.3339229, "available_gib": 172.58671188354492, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743975.3602042, "available_gib": 172.59050369262695, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743977.3864458, "available_gib": 172.58787536621094, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743979.412776, "available_gib": 172.59032821655273, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743981.4392207, "available_gib": 172.58793258666992, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743983.4650486, "available_gib": 172.59032821655273, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743985.4914, "available_gib": 172.58974838256836, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743987.5175338, "available_gib": 172.57958984375, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743989.5431314, "available_gib": 172.57130432128906, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743991.5687299, "available_gib": 172.5556182861328, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743993.5943153, "available_gib": 172.5477066040039, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743995.6200294, "available_gib": 172.55400466918945, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743997.6459754, "available_gib": 172.5567626953125, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789743999.6716835, "available_gib": 172.5649642944336, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789744001.6973622, "available_gib": 172.5669174194336, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789744003.7233665, "available_gib": 172.56511688232422, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false}
{"time": 1789744005.7436235, "available_gib": 172.42804336547852, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false}
{"time": 1789744007.767596, "available_gib": 172.40644073486328, "oom_kill": 0, "gpu": "84697, 85651, 100, 58", "stop": false}
{"time": 1789744009.7923508, "available_gib": 172.39516830444336, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744011.8161004, "available_gib": 172.38743209838867, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744013.839925, "available_gib": 172.37675857543945, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false}
{"time": 1789744015.8648205, "available_gib": 172.38694763183594, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false}
{"time": 1789744017.8896184, "available_gib": 172.39286422729492, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false}
{"time": 1789744019.913305, "available_gib": 172.3947982788086, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false}
{"time": 1789744021.9392803, "available_gib": 172.4147834777832, "oom_kill": 0, "gpu": "84697, 85651, 100, 57", "stop": false}
{"time": 1789744023.9630058, "available_gib": 172.41676330566406, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false}
{"time": 1789744025.9881828, "available_gib": 172.40471267700195, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false}
{"time": 1789744028.012432, "available_gib": 172.42510604858398, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false}
{"time": 1789744030.036954, "available_gib": 172.42960357666016, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false}
{"time": 1789744032.061324, "available_gib": 172.42523956298828, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false}
{"time": 1789744034.0878584, "available_gib": 172.42697143554688, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false}
{"time": 1789744036.1117172, "available_gib": 172.42796325683594, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false}
{"time": 1789744038.136802, "available_gib": 172.43342208862305, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744040.160596, "available_gib": 172.43236541748047, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744042.1858406, "available_gib": 172.44089889526367, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744044.210708, "available_gib": 172.44932556152344, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744046.2357752, "available_gib": 172.44157791137695, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744048.2619452, "available_gib": 172.4429702758789, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false}
{"time": 1789744050.2856293, "available_gib": 172.44015502929688, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744052.3109236, "available_gib": 172.44293594360352, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false}
{"time": 1789744054.3350937, "available_gib": 172.43926620483398, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744056.3591633, "available_gib": 172.44363403320312, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744058.3845017, "available_gib": 172.43565368652344, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744060.409055, "available_gib": 172.41972732543945, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744062.4337656, "available_gib": 172.42286682128906, "oom_kill": 0, "gpu": "84697, 85651, 100, 66", "stop": false}
{"time": 1789744064.460525, "available_gib": 172.43537521362305, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false}
{"time": 1789744066.4872258, "available_gib": 172.42713928222656, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false}
{"time": 1789744068.512237, "available_gib": 172.4395408630371, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false}
{"time": 1789744070.538391, "available_gib": 172.43132400512695, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false}
{"time": 1789744072.5637136, "available_gib": 172.43741989135742, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false}
{"time": 1789744074.590121, "available_gib": 172.4353141784668, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false}
{"time": 1789744076.6162558, "available_gib": 172.43359756469727, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744078.6413624, "available_gib": 172.43523025512695, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744080.6638365, "available_gib": 172.4354362487793, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744082.689977, "available_gib": 172.4274673461914, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false}
{"time": 1789744084.7105415, "available_gib": 174.46597290039062, "oom_kill": 0, "gpu": "84697, 85651, 0, 53", "stop": false}
{"time": 1789744086.7306116, "available_gib": 174.4582176208496, "oom_kill": 0, "gpu": "84697, 85651, 0, 52", "stop": false}
{"time": 1789744088.7536929, "available_gib": 175.251953125, "oom_kill": 0, "gpu": "84697, 85651, 0, 51", "stop": false}
{"time": 1789744090.7746408, "available_gib": 175.2590446472168, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false}
{"time": 1789744092.7950041, "available_gib": 175.26241302490234, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false}
{"time": 1789744099.8720243, "available_gib": 189.0756950378418, "oom_kill": 0, "gpu": "3, 85651, 0, 47", "stop": false}
{"time": 1789744101.8936296, "available_gib": 242.46984481811523, "oom_kill": 0, "gpu": "0, 85651, 0, 46", "stop": false}
{"time": 1789744103.9150841, "available_gib": 242.3806266784668, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false}
{"time": 1789744105.9400728, "available_gib": 242.3127899169922, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false}
{"time": 1789744107.9647262, "available_gib": 242.27752685546875, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false}
{"time": 1789744109.989206, "available_gib": 242.25381088256836, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false}
{"time": 1789744112.0140321, "available_gib": 242.22451782226562, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
{"time": 1789744114.0383747, "available_gib": 242.15388107299805, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
{"time": 1789744116.0627222, "available_gib": 242.1028938293457, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
{"time": 1789744118.1058602, "available_gib": 241.75963973999023, "oom_kill": 0, "gpu": "3, 85651, 0, 41", "stop": false}
{"time": 1789744120.149718, "available_gib": 241.42746353149414, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false}
{"time": 1789744122.1952055, "available_gib": 241.17255783081055, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false}
{"time": 1789744124.2375867, "available_gib": 240.98880767822266, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789744126.2794218, "available_gib": 240.34492874145508, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
{"time": 1789744128.3214467, "available_gib": 240.07582092285156, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789744130.3638144, "available_gib": 239.7653465270996, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789744132.4073355, "available_gib": 239.3140525817871, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
{"time": 1789744134.4280436, "available_gib": 239.03064346313477, "oom_kill": 0, "gpu": "2915, 85651, 0, 39", "stop": false}
{"time": 1789744136.447735, "available_gib": 234.87797164916992, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744138.4675038, "available_gib": 231.21246337890625, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744140.487026, "available_gib": 227.27257919311523, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744142.5069096, "available_gib": 223.3726921081543, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744144.5267863, "available_gib": 219.43412399291992, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744146.5464373, "available_gib": 215.4873390197754, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744148.566585, "available_gib": 211.5369987487793, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744150.5869632, "available_gib": 207.59728240966797, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
{"time": 1789744152.60729, "available_gib": 203.638671875, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744154.6276596, "available_gib": 199.62559509277344, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744156.6482484, "available_gib": 195.57702255249023, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744158.668383, "available_gib": 191.5196304321289, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744160.6888082, "available_gib": 187.50440216064453, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744162.708834, "available_gib": 183.54587936401367, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744164.7295425, "available_gib": 179.56959915161133, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744166.7498033, "available_gib": 175.61302947998047, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
{"time": 1789744168.7754214, "available_gib": 173.72442626953125, "oom_kill": 0, "gpu": "4413, 85651, 8, 37", "stop": false}
{"time": 1789744170.7964997, "available_gib": 173.6231575012207, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
{"time": 1789744172.8178098, "available_gib": 173.62652969360352, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
{"time": 1789744174.8396492, "available_gib": 173.62175369262695, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
{"time": 1789744176.8605325, "available_gib": 173.6126823425293, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
{"time": 1789744178.8817682, "available_gib": 173.62749099731445, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744180.9026644, "available_gib": 173.63713836669922, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false}
{"time": 1789744182.9237735, "available_gib": 173.64041900634766, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744184.944897, "available_gib": 173.63914108276367, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744186.9659956, "available_gib": 173.65515899658203, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744188.9869823, "available_gib": 173.68014907836914, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744191.0077991, "available_gib": 173.66809844970703, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744193.0285199, "available_gib": 173.6796112060547, "oom_kill": 0, "gpu": "76995, 85651, 2, 36", "stop": false}
{"time": 1789744195.049452, "available_gib": 173.67911529541016, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744197.0703564, "available_gib": 173.70183181762695, "oom_kill": 0, "gpu": "76995, 85651, 8, 36", "stop": false}
{"time": 1789744199.0914025, "available_gib": 173.7275505065918, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744201.1120594, "available_gib": 173.74087524414062, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744203.1332436, "available_gib": 173.74426651000977, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744205.1543298, "available_gib": 173.76029586791992, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744207.1753323, "available_gib": 173.78207397460938, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744209.1961696, "available_gib": 173.77024459838867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744211.2166562, "available_gib": 173.79063415527344, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
{"time": 1789744213.2376473, "available_gib": 173.60406875610352, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false}
{"time": 1789744215.2581506, "available_gib": 173.57421112060547, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false}
{"time": 1789744217.2796466, "available_gib": 173.60267639160156, "oom_kill": 0, "gpu": "82559, 85651, 0, 36", "stop": false}
{"time": 1789744219.301794, "available_gib": 173.5945167541504, "oom_kill": 0, "gpu": "82613, 85651, 0, 36", "stop": false}
{"time": 1789744221.3223813, "available_gib": 173.4375114440918, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false}
{"time": 1789744223.3421645, "available_gib": 173.10824966430664, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false}
{"time": 1789744225.3638763, "available_gib": 172.49812698364258, "oom_kill": 0, "gpu": "81095, 85651, 0, 36", "stop": false}
{"time": 1789744227.3846838, "available_gib": 172.73027420043945, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789744229.4048204, "available_gib": 172.87688064575195, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
{"time": 1789744231.4258337, "available_gib": 172.88410186767578, "oom_kill": 0, "gpu": "82317, 85651, 100, 43", "stop": false}
{"time": 1789744233.448499, "available_gib": 173.24000930786133, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false}
{"time": 1789744235.4696748, "available_gib": 173.2450294494629, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false}
{"time": 1789744237.490613, "available_gib": 173.23291778564453, "oom_kill": 0, "gpu": "82319, 85651, 0, 38", "stop": false}
{"time": 1789744239.511178, "available_gib": 173.19620895385742, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789744241.5316548, "available_gib": 173.20362854003906, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
{"time": 1789744243.5523272, "available_gib": 173.18954849243164, "oom_kill": 0, "gpu": "79465, 85651, 11, 37", "stop": false}
{"time": 1789744245.5724537, "available_gib": 172.98081588745117, "oom_kill": 0, "gpu": "80553, 85651, 0, 37", "stop": false}
{"time": 1789744247.5929437, "available_gib": 173.15055084228516, "oom_kill": 0, "gpu": "80239, 85651, 0, 37", "stop": false}
{"time": 1789744249.6134512, "available_gib": 173.34084701538086, "oom_kill": 0, "gpu": "80335, 85651, 0, 37", "stop": false}
{"time": 1789744251.6334639, "available_gib": 173.3362808227539, "oom_kill": 0, "gpu": "83033, 85651, 0, 37", "stop": false}
{"time": 1789744253.6532722, "available_gib": 173.2673683166504, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
{"time": 1789744255.6729264, "available_gib": 173.26757431030273, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
{"time": 1789744257.6932175, "available_gib": 173.25699615478516, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
{"time": 1789744259.7134147, "available_gib": 173.12919235229492, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false}
{"time": 1789744261.733558, "available_gib": 173.114990234375, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false}
{"time": 1789744263.7532177, "available_gib": 173.02079010009766, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false}
{"time": 1789744265.773419, "available_gib": 173.05761337280273, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
{"time": 1789744267.793305, "available_gib": 173.0266342163086, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
{"time": 1789744269.8131666, "available_gib": 173.04352188110352, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
+1
View File
@@ -0,0 +1 @@
{"healthy": true, "time": 1789744271.2613404}