Compare MTP steps and prefill batches; retain qualified MTP 2 baseline
This commit is contained in:
@@ -0,0 +1,46 @@
|
||||
name: qwen38-flash-rtx6000d
|
||||
services:
|
||||
vllm:
|
||||
image: local/qwen38-flash-6000d:0bfc7a15
|
||||
build: .
|
||||
container_name: qwen38-flash-6000d
|
||||
gpus: all
|
||||
shm_size: 8g
|
||||
mem_limit: 210g
|
||||
memswap_limit: 210g
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
|
||||
environment:
|
||||
HF_HUB_OFFLINE: "1"
|
||||
VLLM_USE_V2_MODEL_RUNNER: "1"
|
||||
VLLM_WORKER_MULTIPROC_METHOD: spawn
|
||||
OMP_NUM_THREADS: "8"
|
||||
TORCHINDUCTOR_COMPILE_THREADS: "2"
|
||||
CUTE_DSL_ARCH: sm_120a
|
||||
volumes:
|
||||
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
|
||||
- /data/flash-next/cache:/root/.cache
|
||||
- ./secrets/api-key:/run/secrets/api-key:ro
|
||||
entrypoint: ["/bin/bash", "-lc"]
|
||||
command:
|
||||
- >-
|
||||
exec vllm serve /model
|
||||
--served-model-name qwen3.8-flash-next
|
||||
--host 0.0.0.0 --port 8000
|
||||
--tensor-parallel-size 1 --dtype bfloat16
|
||||
--engram-config '{"cpu_offload":true}'
|
||||
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
|
||||
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 1024
|
||||
--enable-chunked-prefill --enable-prefix-caching
|
||||
--speculative-config '{"method":"mtp","num_speculative_tokens":2}'
|
||||
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune
|
||||
--load-format safetensors
|
||||
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
|
||||
--api-key "$$(cat /run/secrets/api-key)"
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30m
|
||||
@@ -0,0 +1,26 @@
|
||||
# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts.
|
||||
# TYPE vllm:spec_decode_num_drafts_total counter
|
||||
vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 10192.0
|
||||
# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts.
|
||||
# TYPE vllm:spec_decode_num_drafts_created gauge
|
||||
vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758813e+09
|
||||
# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens.
|
||||
# TYPE vllm:spec_decode_num_draft_tokens_total counter
|
||||
vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 20384.0
|
||||
# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens.
|
||||
# TYPE vllm:spec_decode_num_draft_tokens_created gauge
|
||||
vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758831e+09
|
||||
# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens.
|
||||
# TYPE vllm:spec_decode_num_accepted_tokens_total counter
|
||||
vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 13500.0
|
||||
# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens.
|
||||
# TYPE vllm:spec_decode_num_accepted_tokens_created gauge
|
||||
vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758843e+09
|
||||
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position.
|
||||
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter
|
||||
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 7818.0
|
||||
vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 5682.0
|
||||
# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position.
|
||||
# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge
|
||||
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897444587588596e+09
|
||||
vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.7897444587588646e+09
|
||||
@@ -0,0 +1,7 @@
|
||||
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2744407369973487, "response": {"id": "chatcmpl-ab334d6076aab97f", "object": "chat.completion", "created": 1789744677, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "deduplicate", "passed": true, "elapsed_s": 1.7563880499947118, "response": {"id": "chatcmpl-83976fdccc7a7950", "object": "chat.completion", "created": 1789744678, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "schedule", "passed": true, "elapsed_s": 2.0583331399975577, "response": {"id": "chatcmpl-b19c133f07424b43", "object": "chat.completion", "created": 1789744680, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.335008790003485, "response": {"id": "chatcmpl-94c59814be10c281", "object": "chat.completion", "created": 1789744682, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-98f22c30a07f3320", "object": "chat.completion", "created": 1789744683, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-aa0947832062704c", "object": "chat.completion", "created": 1789744690, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-a4458e991f7ae8ed", "object": "chat.completion", "created": 1789744692, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-8473d9a80e8964e8", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}}
|
||||
@@ -0,0 +1,8 @@
|
||||
{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2744407369973487}
|
||||
{"name": "deduplicate", "passed": true, "elapsed_s": 1.7563880499947118}
|
||||
{"name": "schedule", "passed": true, "elapsed_s": 2.0583331399975577}
|
||||
{"name": "chinese-policy", "passed": true, "elapsed_s": 1.335008790003485}
|
||||
{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
|
||||
{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""}
|
||||
{"name": "tool-arguments", "passed": true}
|
||||
QUALITY_COMPLETE
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,38 @@
|
||||
{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6775439599950914, "elapsed_s": 1.1741499959971406, "decode_tps": 163.10715965535658, "max_stream_gap_s": 0.01910071699967375, "events": 29, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"}
|
||||
{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08832487399922684, "elapsed_s": 0.30722334199526813, "decode_tps": 164.45980791720777, "max_stream_gap_s": 0.018707998999161646, "events": 13, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"}
|
||||
{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0868781070021214, "elapsed_s": 0.5104481859962107, "decode_tps": 153.4574872577509, "max_stream_gap_s": 0.019009973002539482, "events": 24, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"}
|
||||
{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08706215999700362, "elapsed_s": 0.3244306199994753, "decode_tps": 160.08866552702204, "max_stream_gap_s": 0.0187255120035843, "events": 14, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"}
|
||||
{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08671738699922571, "elapsed_s": 0.5463336720058578, "decode_tps": 154.47668482629047, "max_stream_gap_s": 0.018846639002731536, "events": 26, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"}
|
||||
{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08789612600230612, "elapsed_s": 0.34364471300068544, "decode_tps": 160.31369119650236, "max_stream_gap_s": 0.018786797001666855, "events": 15, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"}
|
||||
{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08695229000295512, "elapsed_s": 0.5275881220004521, "decode_tps": 158.8613428977733, "max_stream_gap_s": 0.018750836999970488, "events": 25, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"}
|
||||
{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08669126099994173, "elapsed_s": 0.6922984420016292, "decode_tps": 153.5648897791733, "max_stream_gap_s": 0.01870115900237579, "events": 33, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"}
|
||||
{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08650394399592187, "elapsed_s": 0.6182316369959153, "decode_tps": 157.97559748312946, "max_stream_gap_s": 0.018719275001785718, "events": 29, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"}
|
||||
{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08662291799555533, "elapsed_s": 0.6184161519995541, "decode_tps": 156.07569764487806, "max_stream_gap_s": 0.0187153379956726, "events": 29, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"}
|
||||
{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08671497400064254, "elapsed_s": 0.5994503900001291, "decode_tps": 159.92653801796698, "max_stream_gap_s": 0.01877488000172889, "events": 29, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"}
|
||||
{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09096695000334876, "elapsed_s": 0.6402136610049638, "decode_tps": 162.0401146102417, "max_stream_gap_s": 0.018958202002977487, "events": 31, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"}
|
||||
{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08711282000149367, "elapsed_s": 0.3791177230014, "decode_tps": 143.8332013213267, "max_stream_gap_s": 0.01870136900106445, "events": 17, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"}
|
||||
{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08702069400169421, "elapsed_s": 0.5400721170008183, "decode_tps": 136.84980744474976, "max_stream_gap_s": 0.018700218002777547, "events": 26, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"}
|
||||
{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3259, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8501223290004418, "elapsed_s": 1.6364744279999286, "decode_tps": 156.4184799105881, "max_stream_gap_s": 0.018985629998496734, "events": 44, "sha256": "b98f4b1d895e3f0407d8a57eb03b0f63758ea19b579c4e088432b7866a2d4690"}
|
||||
{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12202, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 119}}, "ttft_s": 2.827090976003092, "elapsed_s": 3.731427850005275, "decode_tps": 145.96330614699818, "max_stream_gap_s": 0.028689479004242457, "events": 50, "sha256": "d83e0557fd930a4bb9a1993da480eb91dac833e571d201ed45f1cd6dcdada19d"}
|
||||
{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12440, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 73}}, "ttft_s": 1.7833786369956215, "elapsed_s": 2.3456459599983646, "decode_tps": 147.61661687317203, "max_stream_gap_s": 0.018566765000286978, "events": 32, "sha256": "b4bb08e18e5f49a28d5345524402e08c2a16f3c78b1833c26c875cca3f3fc609"}
|
||||
{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08747289400344016, "elapsed_s": 2.1725823170054355, "decode_tps": 122.29574006378465, "max_stream_gap_s": 0.018369523997534998, "events": 117, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"}
|
||||
{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07303110099746846, "elapsed_s": 16.646995861003234, "decode_tps": 123.50695983978235, "max_stream_gap_s": 0.018655836996913422, "events": 922, "sha256": "e053aa6dfa5732848b372c8cb52c9ba10c97c457a2440c10916d60f425686130"}
|
||||
{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08565058299427619, "elapsed_s": 16.716109230997972, "decode_tps": 123.08740506358313, "max_stream_gap_s": 0.018705554997723084, "events": 926, "sha256": "9e66ae8aada53d4b44ab4d175580b8c3dfc5775eb380db30f457db621fd2e2a1"}
|
||||
{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07342425100068795, "elapsed_s": 16.633013375998416, "decode_tps": 123.61417813863065, "max_stream_gap_s": 0.018731199998001102, "events": 922, "sha256": "33d28bd2b5e88242eb4cb19327d4fe8de4422634d578e9cab27019b562c17538"}
|
||||
{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07248010500188684, "elapsed_s": 1.990966349003429, "decode_tps": 132.91729393280707, "max_stream_gap_s": 0.01848529199924087, "events": 108, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"}
|
||||
{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08546512899920344, "elapsed_s": 15.991916218001279, "decode_tps": 128.68992514711985, "max_stream_gap_s": 0.01876928300043801, "events": 888, "sha256": "e334bcd262d61ff4bcb2054046c80cecc4f1d5f3cb1038a8ba88491374656262"}
|
||||
{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08559878199594095, "elapsed_s": 16.024810650997097, "decode_tps": 128.42542133347507, "max_stream_gap_s": 0.018655616004252806, "events": 890, "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"}
|
||||
{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07282382699486334, "elapsed_s": 15.980729622999206, "decode_tps": 128.67815702769335, "max_stream_gap_s": 0.01888122699892847, "events": 888, "sha256": "4d9e0b72967ce710e8b72877a43dad452f89f5b17d29711eec2d359e922e535e"}
|
||||
{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08633529300277587, "elapsed_s": 2.140694021996751, "decode_tps": 124.12632535938559, "max_stream_gap_s": 0.018333694999455474, "events": 116, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"}
|
||||
{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07246749800106045, "elapsed_s": 16.79015202500159, "decode_tps": 122.44518651455081, "max_stream_gap_s": 0.01858688200445613, "events": 931, "sha256": "7030090191298f50ad3861e9a54ab069c220645f2d9ae3d3adc1518c326737f9"}
|
||||
{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07343941200088011, "elapsed_s": 16.83907349400397, "decode_tps": 122.09499443849444, "max_stream_gap_s": 0.01887494800030254, "events": 934, "sha256": "c6787dd161d44b1ce0d6102e509688d98221200201f15b37bbe28ad0abe65926"}
|
||||
{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07259925900143571, "elapsed_s": 16.8790069530005, "decode_tps": 121.79878277800597, "max_stream_gap_s": 0.01869672300381353, "events": 936, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"}
|
||||
{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 17.993200959004753, "elapsed_s": 18.83731006600283, "decode_tps": 155.193207742869, "max_stream_gap_s": 0.018897349000326358, "events": 48, "sha256": "e7fe3af72b6276b64a57b8dbc84e9f7f1af850672cddc9cc0c52b712188d3da6"}
|
||||
{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 0.8536870109965093, "elapsed_s": 1.7167949249997037, "decode_tps": 151.77708126021793, "max_stream_gap_s": 0.01887543899647426, "events": 49, "sha256": "179eba41d263a1f8ac9539955273bb72bab4be7c5f08db5d1dd5ce4a54183138"}
|
||||
{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126122, "completion_tokens": 127, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8531978610044462, "elapsed_s": 1.698211060000176, "decode_tps": 149.11009691889643, "max_stream_gap_s": 0.018885962999775074, "events": 48, "sha256": "c218b2baa413014b261435911b363b44341c05e35463b7652f562b35127fb79c"}
|
||||
{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 15.955126308996114, "elapsed_s": 17.181451591997757, "decode_tps": 154.93442289222173, "max_stream_gap_s": 0.018894102999183815, "events": 68, "sha256": "a6ca22b1812683511b811b4709da81363362811fb9f4b26449f2dee67e705cfd"}
|
||||
{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.0856839689949993, "elapsed_s": 0.48237196899572155, "decode_tps": 158.81498810119112, "max_stream_gap_s": 0.01842227699671639, "events": 22, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"}
|
||||
{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5501441320011509, "elapsed_s": 0.8208450640013325, "decode_tps": 155.15277206349558, "max_stream_gap_s": 0.018418501000269316, "events": 15, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"}
|
||||
{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8888816709950333, "elapsed_s": 1.2858870880008908, "decode_tps": 158.68801104814773, "max_stream_gap_s": 0.018424560003040824, "events": 22, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"}
|
||||
{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3535734030010644, "elapsed_s": 1.748511461999442, "decode_tps": 159.51868543583137, "max_stream_gap_s": 0.018308389000594616, "events": 23, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"}
|
||||
SUITE_COMPLETE
|
||||
@@ -0,0 +1,343 @@
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347]
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] █ █ █▄ ▄█
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] █▄█▀ █ █ █ █ model /model
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347]
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 1024, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 2}, 'compilation_config': {'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration
|
||||
(APIServer pid=1) INFO 09-18 15:11:46 [model.py:2024] Using max model len 131072
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [model.py:691] Resolved architecture: Qwen4ExpMTP
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [model.py:2024] Using max model len 262144
|
||||
(APIServer pid=1) WARNING 09-18 15:11:50 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [scheduler.py:295] Chunked prefill is enabled with max_num_batched_tokens=1024.
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out.
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
|
||||
(APIServer pid=1) WARNING 09-18 15:11:50 [vllm.py:2152] max_num_scheduled_tokens is set to 1024 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
|
||||
(APIServer pid=1) INFO 09-18 15:11:50 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) INFO 09-18 15:12:02 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=2), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': <CompilationMode.NONE: 0>, 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [1024], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': <CUDAGraphMode.FULL: 2>, 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 3, 'dynamic_shapes_config': {'type': <DynamicShapesType.BACKED: 'backed'>, 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None)
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(EngineCore pid=113) INFO 09-18 15:12:03 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_7e0858f85a864c69a8f6d1aab70eda0b backend=nccl
|
||||
(EngineCore pid=113) INFO 09-18 15:12:03 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A
|
||||
(EngineCore pid=113) INFO 09-18 15:12:03 [gpu_worker.py:441] Using V2 Model Runner
|
||||
(EngineCore pid=113) INFO 09-18 15:12:04 [model_runner.py:387] Loading model from scratch...
|
||||
(EngineCore pid=113) INFO 09-18 15:12:04 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention
|
||||
(EngineCore pid=113) INFO 09-18 15:12:04 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention.
|
||||
(EngineCore pid=113) INFO 09-18 15:12:04 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128).
|
||||
(EngineCore pid=113) INFO 09-18 15:12:04 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda
|
||||
(EngineCore pid=113) INFO 09-18 15:12:06 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION'].
|
||||
(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning.
|
||||
(APIServer pid=1) INFO 09-18 15:12:09 [base.py:261] Multi-modal warmup completed in 12.331s
|
||||
(APIServer pid=1) INFO 09-18 15:12:10 [base.py:261] Readonly multi-modal warmup completed in 1.293s
|
||||
(EngineCore pid=113) INFO 09-18 15:12:40 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True
|
||||
(EngineCore pid=113) INFO 09-18 15:12:40 [flash_attn.py:1115] Using FlashAttention version 2
|
||||
(EngineCore pid=113) WARNING 09-18 15:12:40 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect
|
||||
(EngineCore pid=113) INFO 09-18 15:12:41 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.35 GiB.
|
||||
(EngineCore pid=113) INFO 09-18 15:12:41 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch.
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 9% Completed | 1/11 [00:01<00:16, 1.65s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:06<00:33, 3.76s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:12<00:36, 4.53s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:17<00:34, 4.87s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:23<00:30, 5.12s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:28<00:26, 5.28s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:34<00:21, 5.35s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:39<00:16, 5.42s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:41<00:08, 4.16s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:41<00:00, 2.35s/it]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:41<00:00, 3.82s/it]
|
||||
(EngineCore pid=113)
|
||||
(EngineCore pid=113) INFO 09-18 15:13:23 [default_loader.py:430] Loading weights took 42.13 seconds
|
||||
(EngineCore pid=113) INFO 09-18 15:13:23 [nvfp4.py:611] Using MoEPrepareAndFinalizeNoDPEPModular
|
||||
(EngineCore pid=113) INFO 09-18 15:13:23 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)
|
||||
(EngineCore pid=113) INFO 09-18 15:13:23 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
|
||||
(EngineCore pid=113) WARNING 09-18 15:13:23 [vllm.py:2152] max_num_scheduled_tokens is set to 1024 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:23 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant
|
||||
(EngineCore pid=113) INFO 09-18 15:13:24 [fp8.py:433] Using DEEPGEMM Fp8 MoE backend out of potential backends: ['AITER', 'FLASHINFER_TRTLLM', 'FLASHINFER_CUTLASS', 'DEEPGEMM', 'TRITON', 'MARLIN', 'HUMMING', 'BATCHED_DEEPGEMM', 'BATCHED_TRITON', 'XPU', 'CPU', 'HPC'].
|
||||
(EngineCore pid=113) INFO 09-18 15:13:24 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.22 GiB.
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00<?, ?it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 18% Completed | 2/11 [00:00<00:01, 5.16it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 27% Completed | 3/11 [00:00<00:01, 4.05it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 36% Completed | 4/11 [00:01<00:01, 3.65it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 45% Completed | 5/11 [00:01<00:01, 3.42it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 55% Completed | 6/11 [00:01<00:01, 3.29it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 64% Completed | 7/11 [00:02<00:01, 3.23it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 73% Completed | 8/11 [00:02<00:00, 3.18it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 82% Completed | 9/11 [00:02<00:00, 3.67it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.87it/s]
|
||||
(EngineCore pid=113)
|
||||
Loading safetensors checkpoint shards: 100% Completed | 11/11 [00:02<00:00, 3.68it/s]
|
||||
(EngineCore pid=113)
|
||||
(EngineCore pid=113) INFO 09-18 15:13:27 [default_loader.py:430] Loading weights took 3.00 seconds
|
||||
(EngineCore pid=113) INFO 09-18 15:13:27 [deep_gemm.py:196] deep_gemm not found in site-packages, trying vendored vllm.third_party.deep_gemm
|
||||
(EngineCore pid=113) INFO 09-18 15:13:27 [deep_gemm.py:223] DeepGEMM PDL enabled on vllm.third_party.deep_gemm.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:27 [deep_gemm.py:136] DeepGEMM E8M0 enabled on current platform.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:30 [fp8.py:733] Using MoEPrepareAndFinalizeNoDPEPModular
|
||||
(EngineCore pid=113) WARNING 09-18 15:13:30 [speculator.py:235] Draft model Qwen4ExpMTP does not support external multimodal embeddings. Embeddings from the target model will not be passed to the drafter; using text-only draft inputs instead.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:31 [model_runner.py:419] Model loading took 76.23 GiB memory and 86.731724 seconds
|
||||
(EngineCore pid=113) INFO 09-18 15:13:31 [topk_topp_sampler.py:78] Using FlashInfer for top-p & top-k sampling.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:31 [interface.py:918] Setting attention block size to 3184 tokens to ensure that attention page size is >= mamba page size.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:31 [interface.py:942] Padding mamba page size by 0.38% to ensure that mamba page size and attention page size are exactly equal.
|
||||
(EngineCore pid=113) INFO 09-18 15:13:31 [utils.py:320] Using BLNHC KV cache layout.
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`.
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'}
|
||||
(EngineCore pid=113) INFO 09-18 15:13:35 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size.
|
||||
(EngineCore pid=113) WARNING 09-18 15:13:57 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
|
||||
(EngineCore pid=113) WARNING 09-18 15:13:57 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY
|
||||
(EngineCore pid=113) INFO 09-18 15:13:57 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps.
|
||||
(EngineCore pid=113)
|
||||
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.84s/it]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:01<00:00, 1.84s/it]
|
||||
(EngineCore pid=113) INFO 09-18 15:14:00 [speculator.py:150] Capturing model for speculator...
|
||||
(EngineCore pid=113)
|
||||
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.17it/s]
|
||||
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 2.17it/s]
|
||||
(EngineCore pid=113)
|
||||
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.71it/s]
|
||||
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 5.70it/s]
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [model_runner.py:1057] Graph capturing finished in 3 secs, took 0.14 GiB
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [gpu_worker.py:641] Available KV cache memory: 2.96 GiB
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [gpu_worker.py:656] CUDA graph memory profiling is enabled (default since v0.21.0). The current --gpu-memory-utilization=0.9850 is equivalent to --gpu-memory-utilization=0.9833 without CUDA graph memory profiling. To maintain the same effective KV cache size as before, increase --gpu-memory-utilization to 0.9867. To disable, set VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:01 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [kv_cache_utils.py:2404] GPU KV cache size: 155,509 tokens, Maximum concurrency for 131,072 tokens per request: 1.19x
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [kernel_warmup.py:171] JIT kernel warmup starting.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:01 [kernel_warmup.py:184] JIT kernel warmup finished in 0.00s.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:02 [qwen_vl_triton_warmup.py:57] Warmed position embedding and vision rotary kernels on grids=[(1, 16, 16), (1, 16, 2), (1, 2, 16), (1, 2, 2)].
|
||||
(EngineCore pid=113) INFO 09-18 15:14:02 [qwen_vl_triton_warmup.py:98] Warmed M-RoPE Triton kernels.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:02 [mamba_triton_warmup.py:42] Warmed Mamba batch_memcpy_kernel.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:03 [qwen4_exp_qsa_warmup.py:71] Warmed up Qwen4Exp QSA decode kernels: ((1, 1), (2, 1), (3, 1)).
|
||||
(EngineCore pid=113) INFO 09-18 15:14:08 [qwen4_exp_qsa_warmup.py:85] Warmed up Qwen4Exp QSA sparse attention kernels: ((32, 2, 1), (32, 4, 4), (64, 1, 2), (64, 4, 4), (64, 8, 4), (64, 33, 8), (128, 4, 4), (128, 8, 4)).
|
||||
(EngineCore pid=113) INFO 09-18 15:14:08 [kernel_warmup.py:254] Skipping FlashInfer autotune because it is disabled.
|
||||
(EngineCore pid=113)
|
||||
Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.70it/s]
|
||||
Capturing CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 9.68it/s]
|
||||
(EngineCore pid=113) INFO 09-18 15:14:16 [speculator.py:150] Capturing model for speculator...
|
||||
(EngineCore pid=113)
|
||||
Capturing prefill CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing prefill CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 121.35it/s]
|
||||
(EngineCore pid=113)
|
||||
Capturing decode CUDA graphs (FULL): 0%| | 0/1 [00:00<?, ?it/s]
|
||||
Capturing decode CUDA graphs (FULL): 100%|██████████| 1/1 [00:00<00:00, 178.63it/s]
|
||||
(EngineCore pid=113) INFO 09-18 15:14:17 [model_runner.py:1057] Graph capturing finished in 1 secs, took 0.13 GiB
|
||||
(EngineCore pid=113) INFO 09-18 15:14:17 [gpu_worker.py:824] CUDA graph pool memory: 0.13 GiB (actual), 0.14 GiB (estimated), difference: 0.01 GiB (4.3%).
|
||||
(EngineCore pid=113) INFO 09-18 15:14:17 [gpu_worker.py:887] Free memory on device (82.59/83.05 GiB) on startup. Desired GPU memory utilization is (0.985, 81.8 GiB). Actual usage is 77.58 GiB for consumed memory (weights + non-torch), 1.27 GiB for peak activation, and 0.13 GiB for CUDAGraph memory. Replace gpu_memory_utilization config with `--kv-cache-memory=2878222296` (2.68 GiB) to fit into requested memory, or `--kv-cache-memory=3724889088` (3.47 GiB) to fully utilize gpu memory. Current kv cache memory in use is 2.96 GiB.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:18 [jit_monitor.py:84] Kernel JIT monitor activated; monitored JIT compilations during inference will use mode=warn.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:18 [torch_utils.py:274] OMP_NUM_THREADS=8 is set; leaving Torch threads at 8 for serving. Multi-threaded torch CPU ops during serving can degrade performance through spin-wait contention and cgroup CPU-quota throttling.
|
||||
(EngineCore pid=113) INFO 09-18 15:14:18 [core.py:380] init engine (profile, create kv cache, warmup model) took 47.56 s
|
||||
(EngineCore pid=113) INFO 09-18 15:14:18 [kv_cache_utils.py:762] kv cache group sizes [3184, 3184, 3184, 3184, 8, 3184]
|
||||
(EngineCore pid=113) INFO 09-18 15:14:18 [kv_cache_utils.py:763] kv lcm block sizes 3184
|
||||
(EngineCore pid=113) INFO 09-18 15:14:18 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native'])
|
||||
(APIServer pid=1) INFO 09-18 15:14:18 [entry.py:132] Supported tasks: ['generate']
|
||||
(APIServer pid=1) INFO 09-18 15:14:18 [factories.py:76] Scale-out endpoints are disabled. Set --enable-scale-out to enable them.
|
||||
(APIServer pid=1) INFO 09-18 15:14:18 [parser_manager.py:34] "auto" tool choice has been enabled.
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'}
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [hf.py:642] Detected the chat template content format to be 'openai'. You can set `--chat-template-content-format` to override this.
|
||||
(APIServer pid=1) WARNING 09-18 15:14:19 [model.py:1769] Default vLLM sampling parameters have been overridden by the model's `generation_config.json`: `{'temperature': 1.0, 'top_k': 20, 'top_p': 0.95}`. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [entry.py:136] Starting vLLM server on http://0.0.0.0:8000
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:60] Available routes are:
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /openapi.json, Methods: HEAD, GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /docs, Methods: HEAD, GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /redoc, Methods: HEAD, GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /load, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /version, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /health, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /metrics, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /tokenize, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /detokenize, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/models, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /ping, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /ping, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /invocations, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/chat/completions, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/chat/completions/batch, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/responses, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/responses/{response_id}, Methods: GET
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/responses/{response_id}/cancel, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/completions, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/messages, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /v1/messages/count_tokens, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /generative_scoring, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /scale_elastic_ep, Methods: POST
|
||||
(APIServer pid=1) INFO 09-18 15:14:19 [launcher.py:69] Route: /is_scaling_elastic_ep, Methods: POST
|
||||
(APIServer pid=1) INFO: Started server process [1]
|
||||
(APIServer pid=1) INFO: Waiting for application startup.
|
||||
(APIServer pid=1) INFO: Application startup complete.
|
||||
(APIServer pid=1) INFO: 127.0.0.1:43058 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:49896 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:49906 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:23 [jit_monitor.py:140] Triton kernel JIT compilation during inference: layer_norm_fwd_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:24 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _count_expert_num_tokens. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:24 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _compute_local_logits_stats_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:24 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _rejection_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(EngineCore pid=113) WARNING 09-18 15:14:24 [jit_monitor.py:140] Triton kernel JIT compilation during inference: _resample_kernel. This causes a latency spike; consider extending warmup to cover this shape/config.
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43098 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43100 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43114 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43122 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43136 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43142 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43154 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43168 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43174 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43186 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:14:29 [loggers.py:323] Engine 000: Avg prompt throughput: 59.1 tokens/s, Avg generation throughput: 63.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%
|
||||
(APIServer pid=1) INFO 09-18 15:14:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.91, Accepted throughput: 41.35 tokens/s, Drafted throughput: 43.26 tokens/s, Accepted: 455 tokens, Drafted: 476 tokens, Per-position acceptance rate: 0.987, 0.924, Avg Draft acceptance rate: 95.6%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43194 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43196 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43202 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43218 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:43220 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
[rank0]:[W918 15:14:34.279891387 CUDACachingAllocator.cpp:3933] memory allocation failed with OOM on device 0 while trying to allocate 377487360 bytes (free: 209649664, total: 89173131264).
|
||||
(APIServer pid=1) INFO: 172.21.0.1:49914 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:49920 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:14:39 [loggers.py:323] Engine 000: Avg prompt throughput: 2784.2 tokens/s, Avg generation throughput: 64.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:14:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.78, Accepted throughput: 40.90 tokens/s, Drafted throughput: 45.99 tokens/s, Accepted: 409 tokens, Drafted: 460 tokens, Per-position acceptance rate: 0.935, 0.843, Avg Draft acceptance rate: 88.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:49926 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:14:49 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 125.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:14:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 70.19 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 702 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.743, 0.529, Avg Draft acceptance rate: 63.6%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:53100 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35742 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:14:59 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 119.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:14:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.17, Accepted throughput: 64.19 tokens/s, Drafted throughput: 109.99 tokens/s, Accepted: 642 tokens, Drafted: 1100 tokens, Per-position acceptance rate: 0.720, 0.447, Avg Draft acceptance rate: 58.4%
|
||||
(APIServer pid=1) INFO 09-18 15:15:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.28, Accepted throughput: 71.49 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 715 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.751, 0.530, Avg Draft acceptance rate: 64.1%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:51870 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:15:19 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 118.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.14, Accepted throughput: 63.00 tokens/s, Drafted throughput: 110.20 tokens/s, Accepted: 630 tokens, Drafted: 1102 tokens, Per-position acceptance rate: 0.695, 0.448, Avg Draft acceptance rate: 57.2%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:58158 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:15:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 125.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 69.60 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 696 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.740, 0.510, Avg Draft acceptance rate: 62.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33192 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33200 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:15:39 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 124.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.26, Accepted throughput: 69.20 tokens/s, Drafted throughput: 109.59 tokens/s, Accepted: 692 tokens, Drafted: 1096 tokens, Per-position acceptance rate: 0.752, 0.511, Avg Draft acceptance rate: 63.1%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47080 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:15:49 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 128.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 73.30 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 733 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.768, 0.560, Avg Draft acceptance rate: 66.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:33908 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:15:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:15:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 71.90 tokens/s, Drafted throughput: 111.79 tokens/s, Accepted: 719 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.760, 0.526, Avg Draft acceptance rate: 64.3%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:47368 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:16:09 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 127.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 71.89 tokens/s, Drafted throughput: 110.58 tokens/s, Accepted: 719 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.756, 0.544, Avg Draft acceptance rate: 65.0%
|
||||
(APIServer pid=1) INFO 09-18 15:16:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 129.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 74.00 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 740 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.772, 0.554, Avg Draft acceptance rate: 66.3%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:34654 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35668 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35672 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:16:29 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 121.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.21, Accepted throughput: 66.70 tokens/s, Drafted throughput: 109.80 tokens/s, Accepted: 667 tokens, Drafted: 1098 tokens, Per-position acceptance rate: 0.719, 0.495, Avg Draft acceptance rate: 60.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:39 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.19, Accepted throughput: 66.10 tokens/s, Drafted throughput: 111.00 tokens/s, Accepted: 661 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.712, 0.479, Avg Draft acceptance rate: 59.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:35474 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:16:49 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 120.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 65.30 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 653 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.712, 0.468, Avg Draft acceptance rate: 59.0%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:42942 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:36048 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:16:59 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 123.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:16:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 67.90 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 679 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.725, 0.503, Avg Draft acceptance rate: 61.4%
|
||||
(APIServer pid=1) INFO 09-18 15:17:09 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:17:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 65.70 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 657 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.714, 0.468, Avg Draft acceptance rate: 59.1%
|
||||
(APIServer pid=1) [transformers] Token indices sequence length is longer than the specified maximum sequence length for this model (268298 > 262144). Running this sequence through the model will result in indexing errors
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39466 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39476 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39480 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39492 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39500 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39510 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39520 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39524 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39532 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39534 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39542 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39558 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39574 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39590 - "POST /tokenize HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:39598 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:17:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 54.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 31.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO 09-18 15:17:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 29.90 tokens/s, Drafted throughput: 49.20 tokens/s, Accepted: 299 tokens, Drafted: 492 tokens, Per-position acceptance rate: 0.711, 0.504, Avg Draft acceptance rate: 60.8%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:52170 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:17:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 65.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60232 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60240 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:60254 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:17:39 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.9 tokens/s, Avg generation throughput: 39.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.6%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO 09-18 15:17:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.75, Accepted throughput: 12.45 tokens/s, Drafted throughput: 14.20 tokens/s, Accepted: 249 tokens, Drafted: 284 tokens, Per-position acceptance rate: 0.894, 0.859, Avg Draft acceptance rate: 87.7%
|
||||
(APIServer pid=1) INFO 09-18 15:17:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 58.0%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO: 127.0.0.1:40916 - "GET /health HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52448 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52464 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52478 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52486 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52492 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:17:59 [loggers.py:323] Engine 000: Avg prompt throughput: 11206.2 tokens/s, Avg generation throughput: 82.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO 09-18 15:17:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 26.95 tokens/s, Drafted throughput: 29.00 tokens/s, Accepted: 539 tokens, Drafted: 580 tokens, Per-position acceptance rate: 0.969, 0.890, Avg Draft acceptance rate: 92.9%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52504 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52508 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52522 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO 09-18 15:18:09 [loggers.py:323] Engine 000: Avg prompt throughput: 25.5 tokens/s, Avg generation throughput: 141.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4%
|
||||
(APIServer pid=1) INFO 09-18 15:18:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.61, Accepted throughput: 86.90 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 869 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.881, 0.728, Avg Draft acceptance rate: 80.5%
|
||||
(APIServer pid=1) INFO: 172.21.0.1:52536 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33904 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
(APIServer pid=1) INFO: 172.21.0.1:33916 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"startup": true,
|
||||
"records": 44,
|
||||
"failures": []
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
name: qwen38-flash-rtx6000d
|
||||
services:
|
||||
vllm:
|
||||
image: local/qwen38-flash-6000d:0bfc7a15
|
||||
build: .
|
||||
container_name: qwen38-flash-6000d
|
||||
gpus: all
|
||||
shm_size: 8g
|
||||
mem_limit: 210g
|
||||
memswap_limit: 210g
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000"
|
||||
environment:
|
||||
HF_HUB_OFFLINE: "1"
|
||||
VLLM_USE_V2_MODEL_RUNNER: "1"
|
||||
VLLM_WORKER_MULTIPROC_METHOD: spawn
|
||||
OMP_NUM_THREADS: "8"
|
||||
TORCHINDUCTOR_COMPILE_THREADS: "2"
|
||||
CUTE_DSL_ARCH: sm_120a
|
||||
volumes:
|
||||
- /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro
|
||||
- /data/flash-next/cache:/root/.cache
|
||||
- ./secrets/api-key:/run/secrets/api-key:ro
|
||||
entrypoint: ["/bin/bash", "-lc"]
|
||||
command:
|
||||
- >-
|
||||
exec vllm serve /model
|
||||
--served-model-name qwen3.8-flash-next
|
||||
--host 0.0.0.0 --port 8000
|
||||
--tensor-parallel-size 1 --dtype bfloat16
|
||||
--engram-config '{"cpu_offload":true}'
|
||||
--kv-cache-dtype fp8 --gpu-memory-utilization 0.985
|
||||
--max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048
|
||||
--enable-chunked-prefill --enable-prefix-caching
|
||||
--speculative-config '{"method":"mtp","num_speculative_tokens":2}'
|
||||
--compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune
|
||||
--load-format safetensors
|
||||
--reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice
|
||||
--api-key "$$(cat /run/secrets/api-key)"
|
||||
healthcheck:
|
||||
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30m
|
||||
@@ -0,0 +1,292 @@
|
||||
{"time": 1789744276.214623, "available_gib": 173.0637550354004, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744278.2351742, "available_gib": 174.80208587646484, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744280.2557561, "available_gib": 175.59978103637695, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744282.276541, "available_gib": 175.6155548095703, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744284.2998052, "available_gib": 175.61663055419922, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744292.4291055, "available_gib": 179.1224250793457, "oom_kill": 0, "gpu": "3, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744294.4508595, "available_gib": 242.44279098510742, "oom_kill": 0, "gpu": "0, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744296.47304, "available_gib": 242.34941482543945, "oom_kill": 0, "gpu": "0, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744298.4976192, "available_gib": 242.27065658569336, "oom_kill": 0, "gpu": "0, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744300.5220234, "available_gib": 242.24016952514648, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744302.5465665, "available_gib": 242.22364044189453, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744304.5709565, "available_gib": 242.16735076904297, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744306.5958922, "available_gib": 242.09473419189453, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744308.6201835, "available_gib": 242.06150817871094, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744310.6645873, "available_gib": 241.44736862182617, "oom_kill": 0, "gpu": "3, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744312.707104, "available_gib": 241.14602661132812, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744314.75055, "available_gib": 240.90560913085938, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744316.792225, "available_gib": 240.69082641601562, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744318.8365598, "available_gib": 240.03157424926758, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744320.878574, "available_gib": 239.80778884887695, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744322.9206219, "available_gib": 239.54640579223633, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false}
|
||||
{"time": 1789744324.940512, "available_gib": 239.1410789489746, "oom_kill": 0, "gpu": "2893, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744326.9599116, "available_gib": 238.2663345336914, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744328.9792445, "available_gib": 233.60436630249023, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744330.999156, "available_gib": 229.81796646118164, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744333.018626, "available_gib": 225.89186096191406, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744335.0391977, "available_gib": 221.95586776733398, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744337.058962, "available_gib": 218.04838180541992, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744339.0786686, "available_gib": 214.0251693725586, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744341.0983963, "available_gib": 210.03896713256836, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false}
|
||||
{"time": 1789744343.1179674, "available_gib": 206.0509033203125, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744345.137733, "available_gib": 202.11498641967773, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744347.1575859, "available_gib": 198.17535018920898, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744349.1773338, "available_gib": 194.24428176879883, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744351.1974428, "available_gib": 190.24589157104492, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744353.2174225, "available_gib": 186.31936645507812, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744355.237774, "available_gib": 182.4132080078125, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744357.258104, "available_gib": 178.48447036743164, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744359.27828, "available_gib": 174.53839492797852, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744361.299235, "available_gib": 173.70015335083008, "oom_kill": 0, "gpu": "76825, 85651, 6, 34", "stop": false}
|
||||
{"time": 1789744363.320299, "available_gib": 173.70723342895508, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744365.341664, "available_gib": 173.72999572753906, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744367.3625295, "available_gib": 173.7398796081543, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744369.3839087, "available_gib": 173.75790405273438, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744371.4052994, "available_gib": 173.76782608032227, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744373.426473, "available_gib": 173.79326248168945, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744375.447874, "available_gib": 173.82366561889648, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744377.469133, "available_gib": 173.84239959716797, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744379.4954257, "available_gib": 173.85829544067383, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744381.5174913, "available_gib": 173.86966705322266, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744383.53876, "available_gib": 173.87560272216797, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744385.5598798, "available_gib": 173.89007186889648, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744387.5811183, "available_gib": 173.89695358276367, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744389.6018076, "available_gib": 173.90465545654297, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744391.622792, "available_gib": 173.89840698242188, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744393.6431725, "available_gib": 173.93378448486328, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744395.6640935, "available_gib": 173.99294662475586, "oom_kill": 0, "gpu": "76825, 85651, 1, 34", "stop": false}
|
||||
{"time": 1789744397.6851332, "available_gib": 174.01663208007812, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744399.7062628, "available_gib": 174.02749633789062, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false}
|
||||
{"time": 1789744401.7283592, "available_gib": 174.0538558959961, "oom_kill": 0, "gpu": "76825, 85651, 4, 35", "stop": false}
|
||||
{"time": 1789744403.748965, "available_gib": 174.08990859985352, "oom_kill": 0, "gpu": "78573, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744405.7693496, "available_gib": 173.97901916503906, "oom_kill": 0, "gpu": "82343, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744407.790143, "available_gib": 174.0068016052246, "oom_kill": 0, "gpu": "82385, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744409.8103971, "available_gib": 174.01287460327148, "oom_kill": 0, "gpu": "82385, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744411.8309193, "available_gib": 173.89633178710938, "oom_kill": 0, "gpu": "78947, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744413.8503227, "available_gib": 173.6109504699707, "oom_kill": 0, "gpu": "78947, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744415.8708775, "available_gib": 172.81961822509766, "oom_kill": 0, "gpu": "79469, 85651, 0, 34", "stop": false}
|
||||
{"time": 1789744417.8913932, "available_gib": 172.99155044555664, "oom_kill": 0, "gpu": "80911, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744419.9111977, "available_gib": 172.97130966186523, "oom_kill": 0, "gpu": "80911, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744421.9319136, "available_gib": 172.9793930053711, "oom_kill": 0, "gpu": "82133, 85651, 100, 41", "stop": false}
|
||||
{"time": 1789744423.9523408, "available_gib": 172.97999572753906, "oom_kill": 0, "gpu": "82133, 85651, 100, 43", "stop": false}
|
||||
{"time": 1789744425.9735954, "available_gib": 173.01754760742188, "oom_kill": 0, "gpu": "82133, 85651, 100, 44", "stop": false}
|
||||
{"time": 1789744427.9937978, "available_gib": 173.03707122802734, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744430.014089, "available_gib": 173.0387420654297, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744432.034095, "available_gib": 173.0599822998047, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744434.053616, "available_gib": 172.9596939086914, "oom_kill": 0, "gpu": "82135, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744436.073761, "available_gib": 172.88088989257812, "oom_kill": 0, "gpu": "80167, 85651, 1, 36", "stop": false}
|
||||
{"time": 1789744438.0942142, "available_gib": 172.89511108398438, "oom_kill": 0, "gpu": "79907, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744440.1141858, "available_gib": 173.14619064331055, "oom_kill": 0, "gpu": "79957, 85651, 12, 36", "stop": false}
|
||||
{"time": 1789744442.134541, "available_gib": 173.14928817749023, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744444.1541903, "available_gib": 173.16854095458984, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744446.1740553, "available_gib": 173.13195419311523, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744448.1940587, "available_gib": 173.15098571777344, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744450.2143607, "available_gib": 173.18714904785156, "oom_kill": 0, "gpu": "82923, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744452.2385378, "available_gib": 173.07194900512695, "oom_kill": 0, "gpu": "82929, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744454.2581725, "available_gib": 172.97765350341797, "oom_kill": 0, "gpu": "82929, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744456.278477, "available_gib": 172.9991798400879, "oom_kill": 0, "gpu": "82929, 85651, 7, 36", "stop": false}
|
||||
{"time": 1789744458.2987852, "available_gib": 173.00883865356445, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744460.3185778, "available_gib": 172.9944725036621, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744462.3388991, "available_gib": 173.0210838317871, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744464.3591044, "available_gib": 173.00377655029297, "oom_kill": 0, "gpu": "83069, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744466.3856246, "available_gib": 173.01154708862305, "oom_kill": 0, "gpu": "83069, 85651, 76, 41", "stop": false}
|
||||
{"time": 1789744468.4109452, "available_gib": 173.03825759887695, "oom_kill": 0, "gpu": "83069, 85651, 76, 42", "stop": false}
|
||||
{"time": 1789744470.4362907, "available_gib": 173.05249786376953, "oom_kill": 0, "gpu": "83069, 85651, 100, 43", "stop": false}
|
||||
{"time": 1789744472.4610434, "available_gib": 172.96158599853516, "oom_kill": 0, "gpu": "83523, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744474.4873188, "available_gib": 172.29580307006836, "oom_kill": 0, "gpu": "84647, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744476.5117893, "available_gib": 172.28779220581055, "oom_kill": 0, "gpu": "84649, 85651, 100, 45", "stop": false}
|
||||
{"time": 1789744478.5356812, "available_gib": 172.33455276489258, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744480.5609157, "available_gib": 172.34068298339844, "oom_kill": 0, "gpu": "84649, 85651, 100, 45", "stop": false}
|
||||
{"time": 1789744482.5864017, "available_gib": 172.34762573242188, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false}
|
||||
{"time": 1789744484.6110964, "available_gib": 172.34587860107422, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false}
|
||||
{"time": 1789744486.6363378, "available_gib": 172.3453598022461, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false}
|
||||
{"time": 1789744488.661498, "available_gib": 172.39890670776367, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744490.6863167, "available_gib": 172.45629119873047, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744492.7115188, "available_gib": 172.47060012817383, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744494.7373624, "available_gib": 172.5518569946289, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false}
|
||||
{"time": 1789744496.7631228, "available_gib": 172.57133865356445, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false}
|
||||
{"time": 1789744498.788501, "available_gib": 172.58835983276367, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false}
|
||||
{"time": 1789744500.8139327, "available_gib": 172.58929061889648, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false}
|
||||
{"time": 1789744502.839819, "available_gib": 172.5894432067871, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false}
|
||||
{"time": 1789744504.8652601, "available_gib": 172.59299087524414, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744506.8909621, "available_gib": 172.59333419799805, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744508.9166646, "available_gib": 172.59065628051758, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744510.9422443, "available_gib": 172.59705352783203, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744512.9670806, "available_gib": 172.5942840576172, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false}
|
||||
{"time": 1789744514.9919732, "available_gib": 172.5932502746582, "oom_kill": 0, "gpu": "84649, 85651, 76, 49", "stop": false}
|
||||
{"time": 1789744517.0184538, "available_gib": 172.57636260986328, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744519.0439193, "available_gib": 172.5904884338379, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744521.0691228, "available_gib": 172.59058380126953, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744523.0951657, "available_gib": 172.58983993530273, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744525.1207883, "available_gib": 172.5893211364746, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744527.1458921, "available_gib": 172.5901107788086, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false}
|
||||
{"time": 1789744529.1724672, "available_gib": 172.5957794189453, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744531.1980112, "available_gib": 172.59324264526367, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744533.2228138, "available_gib": 172.59070587158203, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744535.2482636, "available_gib": 172.6019172668457, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744537.2732003, "available_gib": 172.5978012084961, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744539.2988782, "available_gib": 172.60141372680664, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744541.3248637, "available_gib": 172.60461044311523, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744543.3506434, "available_gib": 172.60347747802734, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744545.3766954, "available_gib": 172.60011291503906, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744547.4024527, "available_gib": 172.60373306274414, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false}
|
||||
{"time": 1789744549.427506, "available_gib": 172.6028594970703, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744551.4528599, "available_gib": 172.59807586669922, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744553.4781616, "available_gib": 172.58591842651367, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744555.503374, "available_gib": 172.59807586669922, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744557.5295079, "available_gib": 172.60498809814453, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744559.5547729, "available_gib": 172.58954238891602, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744561.5798154, "available_gib": 172.57410049438477, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744563.6050577, "available_gib": 172.57672882080078, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744565.6282964, "available_gib": 172.58179092407227, "oom_kill": 0, "gpu": "84649, 85651, 91, 51", "stop": false}
|
||||
{"time": 1789744567.6539595, "available_gib": 172.58893585205078, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744569.6795833, "available_gib": 172.5806999206543, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744571.7047844, "available_gib": 172.57934188842773, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744573.7309473, "available_gib": 172.58593368530273, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744575.7565713, "available_gib": 172.58402633666992, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744577.7820156, "available_gib": 172.59289169311523, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false}
|
||||
{"time": 1789744579.806945, "available_gib": 172.5892791748047, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744581.8329036, "available_gib": 172.58263397216797, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744583.8582916, "available_gib": 172.59558486938477, "oom_kill": 0, "gpu": "84649, 85651, 75, 53", "stop": false}
|
||||
{"time": 1789744585.8839097, "available_gib": 172.59697341918945, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744587.9094357, "available_gib": 172.5935935974121, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744589.9352567, "available_gib": 172.59527969360352, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744591.9604733, "available_gib": 172.60036849975586, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744593.986449, "available_gib": 172.6029052734375, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744596.012588, "available_gib": 172.60062789916992, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744598.0393686, "available_gib": 172.60357284545898, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744600.0652373, "available_gib": 172.6021270751953, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744602.0909824, "available_gib": 172.59929656982422, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744604.1165397, "available_gib": 172.59819412231445, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744606.1419337, "available_gib": 172.59766387939453, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744608.1679437, "available_gib": 172.59556198120117, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744610.1933706, "available_gib": 172.59738159179688, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744612.2190385, "available_gib": 172.59409713745117, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744614.2444997, "available_gib": 172.59764862060547, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744616.2701576, "available_gib": 172.6019172668457, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744618.2950556, "available_gib": 172.60046005249023, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744620.320425, "available_gib": 172.59806442260742, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744622.3454478, "available_gib": 172.5938377380371, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744624.3713562, "available_gib": 172.59962844848633, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744626.396724, "available_gib": 172.59697723388672, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744628.421861, "available_gib": 172.59457397460938, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744630.4473107, "available_gib": 172.59413146972656, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744632.4730318, "available_gib": 172.59252166748047, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false}
|
||||
{"time": 1789744634.4932597, "available_gib": 172.46091079711914, "oom_kill": 0, "gpu": "84649, 85651, 0, 50", "stop": false}
|
||||
{"time": 1789744636.5176382, "available_gib": 172.46760177612305, "oom_kill": 0, "gpu": "84649, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744638.5421684, "available_gib": 172.45936584472656, "oom_kill": 0, "gpu": "84649, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744640.5671206, "available_gib": 172.4629135131836, "oom_kill": 0, "gpu": "84649, 85651, 100, 59", "stop": false}
|
||||
{"time": 1789744642.5907373, "available_gib": 172.46268463134766, "oom_kill": 0, "gpu": "84649, 85651, 100, 58", "stop": false}
|
||||
{"time": 1789744644.6149285, "available_gib": 172.47219467163086, "oom_kill": 0, "gpu": "84649, 85651, 100, 60", "stop": false}
|
||||
{"time": 1789744646.6392212, "available_gib": 172.47919845581055, "oom_kill": 0, "gpu": "84649, 85651, 100, 60", "stop": false}
|
||||
{"time": 1789744648.6654615, "available_gib": 172.48125457763672, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false}
|
||||
{"time": 1789744650.6908221, "available_gib": 172.47984313964844, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false}
|
||||
{"time": 1789744652.7153213, "available_gib": 172.4798126220703, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false}
|
||||
{"time": 1789744654.7413292, "available_gib": 172.46713256835938, "oom_kill": 0, "gpu": "84649, 85651, 37, 54", "stop": false}
|
||||
{"time": 1789744656.7650921, "available_gib": 172.46598052978516, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false}
|
||||
{"time": 1789744658.788567, "available_gib": 172.46988677978516, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false}
|
||||
{"time": 1789744660.8137739, "available_gib": 172.46673965454102, "oom_kill": 0, "gpu": "84649, 85651, 100, 62", "stop": false}
|
||||
{"time": 1789744662.8383768, "available_gib": 172.47161865234375, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false}
|
||||
{"time": 1789744664.8626714, "available_gib": 172.469970703125, "oom_kill": 0, "gpu": "84649, 85651, 100, 62", "stop": false}
|
||||
{"time": 1789744666.8868442, "available_gib": 172.4693832397461, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false}
|
||||
{"time": 1789744668.9095283, "available_gib": 172.47140884399414, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false}
|
||||
{"time": 1789744670.9342623, "available_gib": 172.47360229492188, "oom_kill": 0, "gpu": "84649, 85651, 100, 64", "stop": false}
|
||||
{"time": 1789744672.9589663, "available_gib": 172.47840118408203, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false}
|
||||
{"time": 1789744674.9843948, "available_gib": 172.4886360168457, "oom_kill": 0, "gpu": "84669, 85651, 100, 59", "stop": false}
|
||||
{"time": 1789744677.0047867, "available_gib": 172.51124954223633, "oom_kill": 0, "gpu": "84669, 85651, 84, 58", "stop": false}
|
||||
{"time": 1789744679.029943, "available_gib": 172.49595260620117, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false}
|
||||
{"time": 1789744681.055466, "available_gib": 172.49994659423828, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false}
|
||||
{"time": 1789744683.0806797, "available_gib": 172.4964599609375, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false}
|
||||
{"time": 1789744685.1061206, "available_gib": 172.49471282958984, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false}
|
||||
{"time": 1789744687.131367, "available_gib": 172.50520706176758, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744689.1567147, "available_gib": 172.5035514831543, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744691.182143, "available_gib": 172.50299072265625, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744693.2079492, "available_gib": 172.50631713867188, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false}
|
||||
{"time": 1789744695.2286391, "available_gib": 174.55757522583008, "oom_kill": 0, "gpu": "84669, 85651, 0, 52", "stop": false}
|
||||
{"time": 1789744697.2490587, "available_gib": 175.3351173400879, "oom_kill": 0, "gpu": "84669, 85651, 0, 51", "stop": false}
|
||||
{"time": 1789744699.2692952, "available_gib": 175.33306503295898, "oom_kill": 0, "gpu": "84669, 85651, 0, 50", "stop": false}
|
||||
{"time": 1789744701.2899477, "available_gib": 175.34216690063477, "oom_kill": 0, "gpu": "84669, 85651, 0, 49", "stop": false}
|
||||
{"time": 1789744703.310319, "available_gib": 175.33515167236328, "oom_kill": 0, "gpu": "84669, 85651, 0, 48", "stop": false}
|
||||
{"time": 1789744709.8561594, "available_gib": 193.49578094482422, "oom_kill": 0, "gpu": "3, 85651, 0, 46", "stop": false}
|
||||
{"time": 1789744711.8781233, "available_gib": 242.45547485351562, "oom_kill": 0, "gpu": "0, 85651, 0, 46", "stop": false}
|
||||
{"time": 1789744713.8998814, "available_gib": 242.3747444152832, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false}
|
||||
{"time": 1789744715.924673, "available_gib": 242.27092742919922, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false}
|
||||
{"time": 1789744717.9494975, "available_gib": 242.2499008178711, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false}
|
||||
{"time": 1789744719.9739065, "available_gib": 242.2431297302246, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false}
|
||||
{"time": 1789744721.9986234, "available_gib": 242.1970977783203, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
|
||||
{"time": 1789744724.023132, "available_gib": 242.1283721923828, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false}
|
||||
{"time": 1789744726.047791, "available_gib": 242.09485626220703, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false}
|
||||
{"time": 1789744728.0922565, "available_gib": 241.76881408691406, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false}
|
||||
{"time": 1789744730.1342695, "available_gib": 241.43723678588867, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
|
||||
{"time": 1789744732.1755586, "available_gib": 241.15095138549805, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false}
|
||||
{"time": 1789744734.2230153, "available_gib": 241.0072364807129, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744736.2652614, "available_gib": 240.35486221313477, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744738.3089554, "available_gib": 240.07362365722656, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744740.350472, "available_gib": 239.79244232177734, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744742.3701403, "available_gib": 239.2092056274414, "oom_kill": 0, "gpu": "703, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744744.3898382, "available_gib": 238.29636001586914, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744746.409347, "available_gib": 233.53387832641602, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744748.4292598, "available_gib": 229.6477928161621, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744750.4486876, "available_gib": 225.60779190063477, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744752.4681375, "available_gib": 221.5484733581543, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744754.4875712, "available_gib": 217.4948959350586, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744756.5070164, "available_gib": 213.44181060791016, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744758.526479, "available_gib": 209.36876678466797, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744760.5463486, "available_gib": 205.36697387695312, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744762.5667367, "available_gib": 201.33798217773438, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744764.5860617, "available_gib": 197.3143424987793, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744766.6053874, "available_gib": 193.25495529174805, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744768.6250842, "available_gib": 189.2132225036621, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744770.6450837, "available_gib": 185.17929458618164, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744772.665085, "available_gib": 181.16746139526367, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744774.684941, "available_gib": 177.13880157470703, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744777.100537, "available_gib": 173.67180252075195, "oom_kill": 0, "gpu": "4293, 85651, 1, 36", "stop": false}
|
||||
{"time": 1789744779.1218758, "available_gib": 173.65283584594727, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
|
||||
{"time": 1789744781.1431904, "available_gib": 173.6810760498047, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false}
|
||||
{"time": 1789744783.1644783, "available_gib": 173.68351364135742, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false}
|
||||
{"time": 1789744785.185819, "available_gib": 173.69385528564453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744787.207108, "available_gib": 173.71892547607422, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744789.2283504, "available_gib": 173.72776412963867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744791.2495914, "available_gib": 173.7425079345703, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744793.2705739, "available_gib": 173.75009155273438, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744795.2918007, "available_gib": 173.75555419921875, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744797.312934, "available_gib": 173.78668975830078, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744799.3340576, "available_gib": 173.80676651000977, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false}
|
||||
{"time": 1789744801.355535, "available_gib": 173.82130432128906, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744803.3765295, "available_gib": 173.82918548583984, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744805.3975353, "available_gib": 173.84343338012695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744807.4185128, "available_gib": 173.85370635986328, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744809.4393783, "available_gib": 173.8660545349121, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744811.4602966, "available_gib": 173.88482284545898, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744813.481391, "available_gib": 173.90010833740234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744815.502314, "available_gib": 173.91947174072266, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false}
|
||||
{"time": 1789744817.523715, "available_gib": 173.92353057861328, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false}
|
||||
{"time": 1789744819.549531, "available_gib": 173.93406677246094, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744821.5700142, "available_gib": 173.76786422729492, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744823.5898106, "available_gib": 173.7619857788086, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744825.6101544, "available_gib": 173.74892807006836, "oom_kill": 0, "gpu": "82559, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744827.6301765, "available_gib": 173.73380661010742, "oom_kill": 0, "gpu": "79121, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744829.6503675, "available_gib": 173.36173629760742, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744831.6701827, "available_gib": 173.22573471069336, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744833.6909227, "available_gib": 172.6994743347168, "oom_kill": 0, "gpu": "81095, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744835.7112474, "available_gib": 172.84538650512695, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744837.7309992, "available_gib": 172.86717224121094, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false}
|
||||
{"time": 1789744839.753742, "available_gib": 172.88789749145508, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false}
|
||||
{"time": 1789744841.774714, "available_gib": 172.90548706054688, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false}
|
||||
{"time": 1789744843.795579, "available_gib": 172.92531967163086, "oom_kill": 0, "gpu": "82317, 85651, 90, 45", "stop": false}
|
||||
{"time": 1789744845.8162441, "available_gib": 172.8922233581543, "oom_kill": 0, "gpu": "82319, 85651, 0, 38", "stop": false}
|
||||
{"time": 1789744847.8363395, "available_gib": 172.93823623657227, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744849.8560581, "available_gib": 172.86439895629883, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744851.878012, "available_gib": 172.8914680480957, "oom_kill": 0, "gpu": "79683, 85651, 24, 40", "stop": false}
|
||||
{"time": 1789744853.8984156, "available_gib": 172.936767578125, "oom_kill": 0, "gpu": "80189, 85651, 1, 37", "stop": false}
|
||||
{"time": 1789744855.918923, "available_gib": 172.94670486450195, "oom_kill": 0, "gpu": "80239, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744857.9392416, "available_gib": 173.1967430114746, "oom_kill": 0, "gpu": "80335, 85651, 0, 37", "stop": false}
|
||||
{"time": 1789744859.9595659, "available_gib": 173.17830276489258, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744861.9793572, "available_gib": 173.14974975585938, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744863.999041, "available_gib": 173.08945846557617, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744866.0192883, "available_gib": 173.08757400512695, "oom_kill": 0, "gpu": "83035, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744868.0397038, "available_gib": 173.0578498840332, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744870.0599365, "available_gib": 172.92647552490234, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744872.0805118, "available_gib": 172.9449348449707, "oom_kill": 0, "gpu": "83041, 85651, 3, 36", "stop": false}
|
||||
{"time": 1789744874.1012588, "available_gib": 173.0093879699707, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
|
||||
{"time": 1789744876.121015, "available_gib": 172.98486709594727, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false}
|
||||
@@ -0,0 +1 @@
|
||||
{"healthy": true, "time": 1789744876.2350338}
|
||||
Reference in New Issue
Block a user