diff --git a/README.md b/README.md index 5d60438..aa1a100 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,8 @@ ## 仓库导航 +- [参数筛选结果](docs/mtp-tuning.md):已比较 MTP 1/2/3 和 prefill batch 1024,继续保留 MTP 2+2048。 + - `compose.yaml`、`Dockerfile`:已验收部署配置与固定镜像。 - `patches/`:最小补丁、精确源码保护及许可证。 - [MTP 完整验证](docs/mtp-qualification.md):长解码提升约 58%–67%,98 个负载请求及两组各 7 个质量任务通过;包含缓存延迟取舍。 diff --git a/audit/tuning/batch-1024/mtp2-b1024/compose.yaml b/audit/tuning/batch-1024/mtp2-b1024/compose.yaml new file mode 100644 index 0000000..9454c39 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/compose.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: "no" + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 1024 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":2}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/batch-1024/mtp2-b1024/mtp-metrics.txt b/audit/tuning/batch-1024/mtp2-b1024/mtp-metrics.txt new file mode 100644 index 0000000..57695f6 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/mtp-metrics.txt @@ -0,0 +1,26 @@ +# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_total counter +vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 10192.0 +# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_created gauge +vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758813e+09 +# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_total counter +vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 20384.0 +# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_created gauge +vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758831e+09 +# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_total counter +vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 13500.0 +# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_created gauge +vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789744458758843e+09 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 7818.0 +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 5682.0 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897444587588596e+09 +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.7897444587588646e+09 \ No newline at end of file diff --git a/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl b/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl new file mode 100644 index 0000000..03b6d50 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl @@ -0,0 +1,7 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2744407369973487, "response": {"id": "chatcmpl-ab334d6076aab97f", "object": "chat.completion", "created": 1789744677, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.7563880499947118, "response": {"id": "chatcmpl-83976fdccc7a7950", "object": "chat.completion", "created": 1789744678, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "schedule", "passed": true, "elapsed_s": 2.0583331399975577, "response": {"id": "chatcmpl-b19c133f07424b43", "object": "chat.completion", "created": 1789744680, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.335008790003485, "response": {"id": "chatcmpl-94c59814be10c281", "object": "chat.completion", "created": 1789744682, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-98f22c30a07f3320", "object": "chat.completion", "created": 1789744683, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-aa0947832062704c", "object": "chat.completion", "created": 1789744690, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-a4458e991f7ae8ed", "object": "chat.completion", "created": 1789744692, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-8473d9a80e8964e8", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-8f7f7f72", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} diff --git a/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl.log b/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl.log new file mode 100644 index 0000000..385b256 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/quality-medium.jsonl.log @@ -0,0 +1,8 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2744407369973487} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.7563880499947118} +{"name": "schedule", "passed": true, "elapsed_s": 2.0583331399975577} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.335008790003485} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "tool-arguments", "passed": true} +QUALITY_COMPLETE diff --git a/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl b/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl new file mode 100644 index 0000000..821c3a9 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl @@ -0,0 +1,37 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6775439599950914, "elapsed_s": 1.1741499959971406, "decode_tps": 163.10715965535658, "max_stream_gap_s": 0.01910071699967375, "events": 29, "content": "\n\n3973", "reasoning": "I need to compute 137 * 29 + 0.\n\n137 * 29 = 137 * 30 - 137 = 4110 - 137 = 3973\n\nAdding 0 doesn't change anything, so the answer is 3973.\n", "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08832487399922684, "elapsed_s": 0.30722334199526813, "decode_tps": 164.45980791720777, "max_stream_gap_s": 0.018707998999161646, "events": 13, "content": "\n\n5761", "reasoning": "160 * 36 = 5760. 5760 + 1 = 5761.\n", "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0868781070021214, "elapsed_s": 0.5104481859962107, "decode_tps": 153.4574872577509, "max_stream_gap_s": 0.019009973002539482, "events": 24, "content": "\n\n7871", "reasoning": "183 * 43 = 183 * 40 + 183 * 3 = 7320 + 549 = 7869. Then 7869 + 2 = 7871.\n", "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08706215999700362, "elapsed_s": 0.3244306199994753, "decode_tps": 160.08866552702204, "max_stream_gap_s": 0.0187255120035843, "events": 14, "content": "\n\n10303", "reasoning": "206 * 50 = 10300\n10300 + 3 = 10303\n", "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08671738699922571, "elapsed_s": 0.5463336720058578, "decode_tps": 154.47668482629047, "max_stream_gap_s": 0.018846639002731536, "events": 26, "content": "\n\n13057", "reasoning": "229 * 57 = 229 * 50 + 229 * 7 = 11450 + 1603 = 13053. Then 13053 + 4 = 13057.\n", "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08789612600230612, "elapsed_s": 0.34364471300068544, "decode_tps": 160.31369119650236, "max_stream_gap_s": 0.018786797001666855, "events": 15, "content": "\n\n16133", "reasoning": "252 * 64 = 16128. Then 16128 + 5 = 16133.\n", "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08695229000295512, "elapsed_s": 0.5275881220004521, "decode_tps": 158.8613428977733, "max_stream_gap_s": 0.018750836999970488, "events": 25, "content": "\n\n19531", "reasoning": "275 * 71 = 275 * 70 + 275 * 1 = 19250 + 275 = 19525. Then 19525 + 6 = 19531.\n", "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08669126099994173, "elapsed_s": 0.6922984420016292, "decode_tps": 153.5648897791733, "max_stream_gap_s": 0.01870115900237579, "events": 33, "content": "\n\n23251", "reasoning": "298 * 78 = 298 * 78. Let me compute this step by step.\n\n298 * 78 = 298 * 80 - 298 * 2 = 23840 - 596 = 23244\n\n23244 + 7 = 23251\n", "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08650394399592187, "elapsed_s": 0.6182316369959153, "decode_tps": 157.97559748312946, "max_stream_gap_s": 0.018719275001785718, "events": 29, "content": "\n\n27293", "reasoning": "Let me compute 321 * 85 + 8.\n\n321 * 85 = 321 * 80 + 321 * 5 = 25680 + 1605 = 27285\n\n27285 + 8 = 27293\n", "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08662291799555533, "elapsed_s": 0.6184161519995541, "decode_tps": 156.07569764487806, "max_stream_gap_s": 0.0187153379956726, "events": 29, "content": "\n\n31657", "reasoning": "Let me compute 344 * 92 + 9.\n\n344 * 92 = 344 * 90 + 344 * 2 = 30960 + 688 = 31648\n\n31648 + 9 = 31657\n", "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08671497400064254, "elapsed_s": 0.5994503900001291, "decode_tps": 159.92653801796698, "max_stream_gap_s": 0.01877488000172889, "events": 29, "content": "\n\n36343", "reasoning": "Let me compute 367 * 99 + 10.\n\n367 * 99 = 367 * (100 - 1) = 36700 - 367 = 36333\n\n36333 + 10 = 36343\n", "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09096695000334876, "elapsed_s": 0.6402136610049638, "decode_tps": 162.0401146102417, "max_stream_gap_s": 0.018958202002977487, "events": 31, "content": "\n\n41351", "reasoning": "Let me compute 390 * 106 + 11.\n\n390 * 106 = 390 * 100 + 390 * 6 = 39000 + 2340 = 41340\n\n41340 + 11 = 41351\n", "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08711282000149367, "elapsed_s": 0.3791177230014, "decode_tps": 143.8332013213267, "max_stream_gap_s": 0.01870136900106445, "events": 17, "content": "\n\n{\"name\":\"设备A\",\"count\":7,\"enabled\":true}", "reasoning": "The user wants me to return exactly a specific JSON object with no markdown formatting. Let me provide exactly what they asked for.\n", "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08702069400169421, "elapsed_s": 0.5400721170008183, "decode_tps": 136.84980744474976, "max_stream_gap_s": 0.018700218002777547, "events": 26, "content": "\n\nAQUA-58321", "reasoning": "The user is asking me to recall the code they previously told me to remember. They said \"Remember the code AQUA-58321.\" and now they're asking what it was, with the instruction to reply only with the code.\n", "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3259, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8501223290004418, "elapsed_s": 1.6364744279999286, "decode_tps": 156.4184799105881, "max_stream_gap_s": 0.018985629998496734, "events": 44, "content": "\n\nSPARK-82741, 1368", "reasoning": "The user wants the verification code and the sum of three amounts from the invoice image. Reading the image: Verification: SPARK-82741. Amount A: 125, Amount B: 478, Amount C: 765. Sum = 125 + 478 = 603; 603 + 765 = 1368. I need to provide a clean, self-contained response with the code and the sum.\n", "sha256": "b98f4b1d895e3f0407d8a57eb03b0f63758ea19b579c4e088432b7866a2d4690"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12202, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 119}}, "ttft_s": 2.827090976003092, "elapsed_s": 3.731427850005275, "decode_tps": 145.96330614699818, "max_stream_gap_s": 0.028689479004242457, "events": 50, "content": "\n\nRed, Green, Blue, Yellow, Purple, White", "reasoning": "The user wants the six background colors in chronological order, replying only with English color names. Looking at the frames: 0-19s RED, 20-39s GREEN, 40-59s BLUE, 60-79s YELLOW, 80-99s PURPLE, 100-119s WHITE. That's six colors. I need to provide just the names, comma-separated or listed. The instruction says \"Reply only with the English color names.\" So I'll give a simple list.\n", "sha256": "d83e0557fd930a4bb9a1993da480eb91dac833e571d201ed45f1cd6dcdada19d"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12440, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 73}}, "ttft_s": 1.7833786369956215, "elapsed_s": 2.3456459599983646, "decode_tps": 147.61661687317203, "max_stream_gap_s": 0.018566765000286978, "events": 32, "content": "\n\nSPARK-82741", "reasoning": "The user wants the verification code from the invoice. All four copies show the same line: \"Verification: SPARK-82741\". I need to extract just the code value. The instruction says reply only with the code. So the answer is SPARK-82741. No need for elaborate reasoning beyond confirming consistency across copies.\n", "sha256": "b4bb08e18e5f49a28d5345524402e08c2a16f3c78b1833c26c875cca3f3fc609"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08747289400344016, "elapsed_s": 2.1725823170054355, "decode_tps": 122.29574006378465, "max_stream_gap_s": 0.018369523997534998, "events": 117, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling,", "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07303110099746846, "elapsed_s": 16.646995861003234, "decode_tps": 123.50695983978235, "max_stream_gap_s": 0.018655836996913422, "events": 922, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and stop? Or error row.\n REPLACE: if exists repo.update else insert; also seen? For within file, replace previous? Could update.\n- If no duplicate_keys: no duplicate handling.\n\nNeed repository transaction? Could have Repository with begin/commit/rollback? Maybe optional. For production, use transaction. But", "sha256": "e053aa6dfa5732848b372c8cb52c9ba10c97c457a2440c10916d60f425686130"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08565058299427619, "elapsed_s": 16.716109230997972, "decode_tps": 123.08740506358313, "max_stream_gap_s": 0.018705554997723084, "events": 926, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY, continue? or fail. Could add error and skip.\n REPLACE: if exists repo.update? same as update? Maybe REPLACE means overwrite regardless. Could implement: if exists update else insert.\n- Add key to seen_in_file after insert/update? For duplicates within file: if strategy SKIP, second duplicate skipped. If UPDATE", "sha256": "9e66ae8aada53d4b44ab4d175580b8c3dfc5775eb380db30f457db621fd2e2a1"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07342425100068795, "elapsed_s": 16.633013375998416, "decode_tps": 123.61417813863065, "max_stream_gap_s": 0.018731199998001102, "events": 922, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and maybe stop? If on_error fail? Could treat as error.\n REPLACE: if exists repo.update else insert; also seen? For within file, replace previous? Could update.\n- If no duplicate_keys: no duplicate detection.\n\nNeed repository transaction? Could wrap in try/except and rollback? InMemory no.", "sha256": "33d28bd2b5e88242eb4cb19327d4fe8de4422634d578e9cab27019b562c17538"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07248010500188684, "elapsed_s": 1.990966349003429, "decode_tps": 132.91729393280707, "max_stream_gap_s": 0.01848529199924087, "events": 108, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08546512899920344, "elapsed_s": 15.991916218001279, "decode_tps": 128.68992514711985, "max_stream_gap_s": 0.01876928300043801, "events": 888, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动拒绝(按配置) |\n| 会签中部分人拒绝 | 按规则处理:任一拒绝即拒绝,或全部通过才通过 |\n| 或签中多人同时处理 | 以第一个有效操作为准,后续操作提示“已处理” |\n| 审批", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "e334bcd262d61ff4bcb2054046c80cecc4f1d5f3cb1038a8ba88491374656262"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08559878199594095, "elapsed_s": 16.024810650997097, "decode_tps": 128.42542133347507, "max_stream_gap_s": 0.018655616004252806, "events": 890, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止 |\n\n---\n\n## 4.2.2 需求项:审批发起\n\n### 功能说明", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07282382699486334, "elapsed_s": 15.980729622999206, "decode_tps": 128.67815702769335, "max_stream_gap_s": 0.01888122699892847, "events": 888, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过/拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 并发审批同一单据 | 加锁处理,后操作者提示“单据已被处理” |\n\n---\n\n## 4.2.2 需求项:", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "4d9e0b72967ce710e8b72877a43dad452f89f5b17d29711eec2d359e922e535e"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08633529300277587, "elapsed_s": 2.140694021996751, "decode_tps": 124.12632535938559, "max_stream_gap_s": 0.018333694999455474, "events": 116, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsett", "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07246749800106045, "elapsed_s": 16.79015202500159, "decode_tps": 122.44518651455081, "max_stream_gap_s": 0.01858688200445613, "events": 931, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a already settled.\n\nNeed maybe include step-by-step table for counterexample:\nVertices s,a,b. Edges: s→a 1, s→b 2, b→a -2.\nInitial d: s=0,a=1,b=∞? Wait if initialize by relaxing source? Dijkstra often initial", "sha256": "7030090191298f50ad3861e9a54ab069c220645f2d9ae3d3adc1518c326737f9"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07343941200088011, "elapsed_s": 16.83907349400397, "decode_tps": 122.09499443849444, "max_stream_gap_s": 0.01887494800030254, "events": 934, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled. Good.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=∞? Actually if initialize all", "sha256": "c6787dd161d44b1ce0d6102e509688d98221200201f15b37bbe28ad0abe65926"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07259925900143571, "elapsed_s": 16.8790069530005, "decode_tps": 121.79878277800597, "max_stream_gap_s": 0.01869672300381353, "events": 936, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax.\n", "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 17.993200959004753, "elapsed_s": 18.83731006600283, "decode_tps": 155.193207742869, "max_stream_gap_s": 0.018897349000326358, "events": 48, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes that are embedded in the data.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (between Item 002499 and Item 002501)\n\nThe user wants both codes only.\n", "sha256": "e7fe3af72b6276b64a57b8dbc84e9f7f1af850672cddc9cc0c52b712188d3da6"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 0.8536870109965093, "elapsed_s": 1.7167949249997037, "decode_tps": 151.77708126021793, "max_stream_gap_s": 0.01887543899647426, "events": 49, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n\nThe user wants both codes only.\n", "sha256": "179eba41d263a1f8ac9539955273bb72bab4be7c5f08db5d1dd5ce4a54183138"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126122, "completion_tokens": 127, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8531978610044462, "elapsed_s": 1.698211060000176, "decode_tps": 149.11009691889643, "max_stream_gap_s": 0.018885962999775074, "events": 48, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data. Let me find them in the text.\n\nLooking through the data:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n", "sha256": "c218b2baa413014b261435911b363b44341c05e35463b7652f562b35127fb79c"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 15.955126308996114, "elapsed_s": 17.181451591997757, "decode_tps": 154.93442289222173, "max_stream_gap_s": 0.018894102999183815, "events": 68, "content": "\n\n- **Northern access code:** NORTH-49273\n- **Southern access code:** SOUTH-18362\n- **Invoice verification code:** SPARK-82741", "reasoning": "The user wants three pieces of information: the northern access code, the southern access code, and the invoice verification code. These are embedded in the text and the image.\n\nScanning the text: \"The northern access code is NORTH-49273.\" appears after Item 000019. \"The southern access code is SOUTH-18362.\" appears after Item 002499. The image shows \"Verification: SPARK-82741\".\n\nSo the answer is straightforward extraction. No computation needed. I'll present all three clearly. The reasoning is in English matching the user's English request.\n\nI need a self-contained final response.\n", "sha256": "a6ca22b1812683511b811b4709da81363362811fb9f4b26449f2dee67e705cfd"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.0856839689949993, "elapsed_s": 0.48237196899572155, "decode_tps": 158.81498810119112, "max_stream_gap_s": 0.01842227699671639, "events": 22, "content": "\n\n731", "reasoning": "The user wants me to compute 43 * 17 and reply with only the integer.\n\n43 * 17 = 43 * 10 + 43 * 7 = 430 + 301 = 731\n", "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5501441320011509, "elapsed_s": 0.8208450640013325, "decode_tps": 155.15277206349558, "max_stream_gap_s": 0.018418501000269316, "events": 15, "content": "\n\n748", "reasoning": "44 * 17 = 44 * 10 + 44 * 7 = 440 + 308 = 748\n", "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8888816709950333, "elapsed_s": 1.2858870880008908, "decode_tps": 158.68801104814773, "max_stream_gap_s": 0.018424560003040824, "events": 22, "content": "\n\n765", "reasoning": "The user wants me to compute 45 * 17 and reply with only the integer.\n\n45 * 17 = 45 * 10 + 45 * 7 = 450 + 315 = 765\n", "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3535734030010644, "elapsed_s": 1.748511461999442, "decode_tps": 159.51868543583137, "max_stream_gap_s": 0.018308389000594616, "events": 23, "content": "\n\n782", "reasoning": "The user wants me to compute 46 * 17 and reply with just the integer.\n\n46 * 17 = 46 * 10 + 46 * 7 = 460 + 322 = 782\n", "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} diff --git a/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl.log b/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl.log new file mode 100644 index 0000000..0e852a9 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/screen.jsonl.log @@ -0,0 +1,38 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6775439599950914, "elapsed_s": 1.1741499959971406, "decode_tps": 163.10715965535658, "max_stream_gap_s": 0.01910071699967375, "events": 29, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08832487399922684, "elapsed_s": 0.30722334199526813, "decode_tps": 164.45980791720777, "max_stream_gap_s": 0.018707998999161646, "events": 13, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0868781070021214, "elapsed_s": 0.5104481859962107, "decode_tps": 153.4574872577509, "max_stream_gap_s": 0.019009973002539482, "events": 24, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08706215999700362, "elapsed_s": 0.3244306199994753, "decode_tps": 160.08866552702204, "max_stream_gap_s": 0.0187255120035843, "events": 14, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08671738699922571, "elapsed_s": 0.5463336720058578, "decode_tps": 154.47668482629047, "max_stream_gap_s": 0.018846639002731536, "events": 26, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08789612600230612, "elapsed_s": 0.34364471300068544, "decode_tps": 160.31369119650236, "max_stream_gap_s": 0.018786797001666855, "events": 15, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08695229000295512, "elapsed_s": 0.5275881220004521, "decode_tps": 158.8613428977733, "max_stream_gap_s": 0.018750836999970488, "events": 25, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08669126099994173, "elapsed_s": 0.6922984420016292, "decode_tps": 153.5648897791733, "max_stream_gap_s": 0.01870115900237579, "events": 33, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08650394399592187, "elapsed_s": 0.6182316369959153, "decode_tps": 157.97559748312946, "max_stream_gap_s": 0.018719275001785718, "events": 29, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08662291799555533, "elapsed_s": 0.6184161519995541, "decode_tps": 156.07569764487806, "max_stream_gap_s": 0.0187153379956726, "events": 29, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08671497400064254, "elapsed_s": 0.5994503900001291, "decode_tps": 159.92653801796698, "max_stream_gap_s": 0.01877488000172889, "events": 29, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09096695000334876, "elapsed_s": 0.6402136610049638, "decode_tps": 162.0401146102417, "max_stream_gap_s": 0.018958202002977487, "events": 31, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08711282000149367, "elapsed_s": 0.3791177230014, "decode_tps": 143.8332013213267, "max_stream_gap_s": 0.01870136900106445, "events": 17, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08702069400169421, "elapsed_s": 0.5400721170008183, "decode_tps": 136.84980744474976, "max_stream_gap_s": 0.018700218002777547, "events": 26, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3259, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8501223290004418, "elapsed_s": 1.6364744279999286, "decode_tps": 156.4184799105881, "max_stream_gap_s": 0.018985629998496734, "events": 44, "sha256": "b98f4b1d895e3f0407d8a57eb03b0f63758ea19b579c4e088432b7866a2d4690"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12202, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 119}}, "ttft_s": 2.827090976003092, "elapsed_s": 3.731427850005275, "decode_tps": 145.96330614699818, "max_stream_gap_s": 0.028689479004242457, "events": 50, "sha256": "d83e0557fd930a4bb9a1993da480eb91dac833e571d201ed45f1cd6dcdada19d"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12440, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 73}}, "ttft_s": 1.7833786369956215, "elapsed_s": 2.3456459599983646, "decode_tps": 147.61661687317203, "max_stream_gap_s": 0.018566765000286978, "events": 32, "sha256": "b4bb08e18e5f49a28d5345524402e08c2a16f3c78b1833c26c875cca3f3fc609"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08747289400344016, "elapsed_s": 2.1725823170054355, "decode_tps": 122.29574006378465, "max_stream_gap_s": 0.018369523997534998, "events": 117, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07303110099746846, "elapsed_s": 16.646995861003234, "decode_tps": 123.50695983978235, "max_stream_gap_s": 0.018655836996913422, "events": 922, "sha256": "e053aa6dfa5732848b372c8cb52c9ba10c97c457a2440c10916d60f425686130"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08565058299427619, "elapsed_s": 16.716109230997972, "decode_tps": 123.08740506358313, "max_stream_gap_s": 0.018705554997723084, "events": 926, "sha256": "9e66ae8aada53d4b44ab4d175580b8c3dfc5775eb380db30f457db621fd2e2a1"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07342425100068795, "elapsed_s": 16.633013375998416, "decode_tps": 123.61417813863065, "max_stream_gap_s": 0.018731199998001102, "events": 922, "sha256": "33d28bd2b5e88242eb4cb19327d4fe8de4422634d578e9cab27019b562c17538"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07248010500188684, "elapsed_s": 1.990966349003429, "decode_tps": 132.91729393280707, "max_stream_gap_s": 0.01848529199924087, "events": 108, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08546512899920344, "elapsed_s": 15.991916218001279, "decode_tps": 128.68992514711985, "max_stream_gap_s": 0.01876928300043801, "events": 888, "sha256": "e334bcd262d61ff4bcb2054046c80cecc4f1d5f3cb1038a8ba88491374656262"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08559878199594095, "elapsed_s": 16.024810650997097, "decode_tps": 128.42542133347507, "max_stream_gap_s": 0.018655616004252806, "events": 890, "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07282382699486334, "elapsed_s": 15.980729622999206, "decode_tps": 128.67815702769335, "max_stream_gap_s": 0.01888122699892847, "events": 888, "sha256": "4d9e0b72967ce710e8b72877a43dad452f89f5b17d29711eec2d359e922e535e"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08633529300277587, "elapsed_s": 2.140694021996751, "decode_tps": 124.12632535938559, "max_stream_gap_s": 0.018333694999455474, "events": 116, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07246749800106045, "elapsed_s": 16.79015202500159, "decode_tps": 122.44518651455081, "max_stream_gap_s": 0.01858688200445613, "events": 931, "sha256": "7030090191298f50ad3861e9a54ab069c220645f2d9ae3d3adc1518c326737f9"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07343941200088011, "elapsed_s": 16.83907349400397, "decode_tps": 122.09499443849444, "max_stream_gap_s": 0.01887494800030254, "events": 934, "sha256": "c6787dd161d44b1ce0d6102e509688d98221200201f15b37bbe28ad0abe65926"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07259925900143571, "elapsed_s": 16.8790069530005, "decode_tps": 121.79878277800597, "max_stream_gap_s": 0.01869672300381353, "events": 936, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 17.993200959004753, "elapsed_s": 18.83731006600283, "decode_tps": 155.193207742869, "max_stream_gap_s": 0.018897349000326358, "events": 48, "sha256": "e7fe3af72b6276b64a57b8dbc84e9f7f1af850672cddc9cc0c52b712188d3da6"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126127, "completion_tokens": 132, "completion_tokens_details": {"reasoning_tokens": 112}}, "ttft_s": 0.8536870109965093, "elapsed_s": 1.7167949249997037, "decode_tps": 151.77708126021793, "max_stream_gap_s": 0.01887543899647426, "events": 49, "sha256": "179eba41d263a1f8ac9539955273bb72bab4be7c5f08db5d1dd5ce4a54183138"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126122, "completion_tokens": 127, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.8531978610044462, "elapsed_s": 1.698211060000176, "decode_tps": 149.11009691889643, "max_stream_gap_s": 0.018885962999775074, "events": 48, "sha256": "c218b2baa413014b261435911b363b44341c05e35463b7652f562b35127fb79c"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 15.955126308996114, "elapsed_s": 17.181451591997757, "decode_tps": 154.93442289222173, "max_stream_gap_s": 0.018894102999183815, "events": 68, "sha256": "a6ca22b1812683511b811b4709da81363362811fb9f4b26449f2dee67e705cfd"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.0856839689949993, "elapsed_s": 0.48237196899572155, "decode_tps": 158.81498810119112, "max_stream_gap_s": 0.01842227699671639, "events": 22, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5501441320011509, "elapsed_s": 0.8208450640013325, "decode_tps": 155.15277206349558, "max_stream_gap_s": 0.018418501000269316, "events": 15, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8888816709950333, "elapsed_s": 1.2858870880008908, "decode_tps": 158.68801104814773, "max_stream_gap_s": 0.018424560003040824, "events": 22, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3535734030010644, "elapsed_s": 1.748511461999442, "decode_tps": 159.51868543583137, "max_stream_gap_s": 0.018308389000594616, "events": 23, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} +SUITE_COMPLETE diff --git a/audit/tuning/batch-1024/mtp2-b1024/server.log b/audit/tuning/batch-1024/mtp2-b1024/server.log new file mode 100644 index 0000000..d940e7a --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/server.log @@ -0,0 +1,343 @@ +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] █ █ █▄ ▄█ +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] █▄█▀ █ █ █ █ model /model +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 15:11:46 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 1024, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 2}, 'compilation_config': {'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:11:46 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration +(APIServer pid=1) INFO 09-18 15:11:46 [model.py:2024] Using max model len 131072 +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:11:50 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:11:50 [model.py:691] Resolved architecture: Qwen4ExpMTP +(APIServer pid=1) INFO 09-18 15:11:50 [model.py:2024] Using max model len 262144 +(APIServer pid=1) WARNING 09-18 15:11:50 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate +(APIServer pid=1) INFO 09-18 15:11:50 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072 +(APIServer pid=1) INFO 09-18 15:11:50 [scheduler.py:295] Chunked prefill is enabled with max_num_batched_tokens=1024. +(APIServer pid=1) INFO 09-18 15:11:50 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled +(APIServer pid=1) INFO 09-18 15:11:50 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False) +(APIServer pid=1) INFO 09-18 15:11:50 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out. +(APIServer pid=1) INFO 09-18 15:11:50 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']) +(APIServer pid=1) WARNING 09-18 15:11:50 [vllm.py:2152] max_num_scheduled_tokens is set to 1024 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens. +(APIServer pid=1) INFO 09-18 15:11:50 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=113) INFO 09-18 15:12:02 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=2), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [1024], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 3, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None) +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=113) INFO 09-18 15:12:03 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_7e0858f85a864c69a8f6d1aab70eda0b backend=nccl +(EngineCore pid=113) INFO 09-18 15:12:03 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=113) INFO 09-18 15:12:03 [gpu_worker.py:441] Using V2 Model Runner +(EngineCore pid=113) INFO 09-18 15:12:04 [model_runner.py:387] Loading model from scratch... +(EngineCore pid=113) INFO 09-18 15:12:04 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention +(EngineCore pid=113) INFO 09-18 15:12:04 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention. +(EngineCore pid=113) INFO 09-18 15:12:04 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128). +(EngineCore pid=113) INFO 09-18 15:12:04 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda +(EngineCore pid=113) INFO 09-18 15:12:06 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION']. +(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. +(APIServer pid=1) INFO 09-18 15:12:09 [base.py:261] Multi-modal warmup completed in 12.331s +(APIServer pid=1) INFO 09-18 15:12:10 [base.py:261] Readonly multi-modal warmup completed in 1.293s +(EngineCore pid=113) INFO 09-18 15:12:40 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True +(EngineCore pid=113) INFO 09-18 15:12:40 [flash_attn.py:1115] Using FlashAttention version 2 +(EngineCore pid=113) WARNING 09-18 15:12:40 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect +(EngineCore pid=113) INFO 09-18 15:12:41 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.35 GiB. +(EngineCore pid=113) INFO 09-18 15:12:41 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch. +(EngineCore pid=113) +Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00= mamba page size. +(EngineCore pid=113) INFO 09-18 15:13:31 [interface.py:942] Padding mamba page size by 0.38% to ensure that mamba page size and attention page size are exactly equal. +(EngineCore pid=113) INFO 09-18 15:13:31 [utils.py:320] Using BLNHC KV cache layout. +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) INFO 09-18 15:13:35 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size. +(EngineCore pid=113) WARNING 09-18 15:13:57 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's. +(EngineCore pid=113) WARNING 09-18 15:13:57 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY +(EngineCore pid=113) INFO 09-18 15:13:57 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps. +(EngineCore pid=113) +Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00 262144). Running this sequence through the model will result in indexing errors +(APIServer pid=1) INFO: 172.21.0.1:39466 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39476 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39480 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39492 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39500 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39510 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39520 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39524 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39532 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39534 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39542 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39558 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39574 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39590 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39598 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:17:19 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 54.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 31.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 15:17:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 29.90 tokens/s, Drafted throughput: 49.20 tokens/s, Accepted: 299 tokens, Drafted: 492 tokens, Per-position acceptance rate: 0.711, 0.504, Avg Draft acceptance rate: 60.8% +(APIServer pid=1) INFO: 127.0.0.1:52170 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:17:29 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 65.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO: 172.21.0.1:60232 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:60240 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:60254 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:17:39 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.9 tokens/s, Avg generation throughput: 39.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.6%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:17:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.75, Accepted throughput: 12.45 tokens/s, Drafted throughput: 14.20 tokens/s, Accepted: 249 tokens, Drafted: 284 tokens, Per-position acceptance rate: 0.894, 0.859, Avg Draft acceptance rate: 87.7% +(APIServer pid=1) INFO 09-18 15:17:49 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 58.0%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO: 127.0.0.1:40916 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52448 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52464 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52478 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52486 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52492 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:17:59 [loggers.py:323] Engine 000: Avg prompt throughput: 11206.2 tokens/s, Avg generation throughput: 82.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:17:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 26.95 tokens/s, Drafted throughput: 29.00 tokens/s, Accepted: 539 tokens, Drafted: 580 tokens, Per-position acceptance rate: 0.969, 0.890, Avg Draft acceptance rate: 92.9% +(APIServer pid=1) INFO: 172.21.0.1:52504 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52508 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52522 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:18:09 [loggers.py:323] Engine 000: Avg prompt throughput: 25.5 tokens/s, Avg generation throughput: 141.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:18:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.61, Accepted throughput: 86.90 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 869 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.881, 0.728, Avg Draft acceptance rate: 80.5% +(APIServer pid=1) INFO: 172.21.0.1:52536 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:33904 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:33916 - "POST /v1/chat/completions HTTP/1.1" 200 OK diff --git a/audit/tuning/batch-1024/mtp2-b1024/status.json b/audit/tuning/batch-1024/mtp2-b1024/status.json new file mode 100644 index 0000000..256e394 --- /dev/null +++ b/audit/tuning/batch-1024/mtp2-b1024/status.json @@ -0,0 +1,5 @@ +{ + "startup": true, + "records": 44, + "failures": [] +} \ No newline at end of file diff --git a/audit/tuning/batch-1024/original.yaml b/audit/tuning/batch-1024/original.yaml new file mode 100644 index 0000000..693de66 --- /dev/null +++ b/audit/tuning/batch-1024/original.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: unless-stopped + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":2}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/batch-1024/resources.jsonl b/audit/tuning/batch-1024/resources.jsonl new file mode 100644 index 0000000..6d52fcf --- /dev/null +++ b/audit/tuning/batch-1024/resources.jsonl @@ -0,0 +1,292 @@ +{"time": 1789744276.214623, "available_gib": 173.0637550354004, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} +{"time": 1789744278.2351742, "available_gib": 174.80208587646484, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} +{"time": 1789744280.2557561, "available_gib": 175.59978103637695, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false} +{"time": 1789744282.276541, "available_gib": 175.6155548095703, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false} +{"time": 1789744284.2998052, "available_gib": 175.61663055419922, "oom_kill": 0, "gpu": "83175, 85651, 0, 35", "stop": false} +{"time": 1789744292.4291055, "available_gib": 179.1224250793457, "oom_kill": 0, "gpu": "3, 85651, 0, 35", "stop": false} +{"time": 1789744294.4508595, "available_gib": 242.44279098510742, "oom_kill": 0, "gpu": "0, 85651, 0, 35", "stop": false} +{"time": 1789744296.47304, "available_gib": 242.34941482543945, "oom_kill": 0, "gpu": "0, 85651, 0, 34", "stop": false} +{"time": 1789744298.4976192, "available_gib": 242.27065658569336, "oom_kill": 0, "gpu": "0, 85651, 0, 34", "stop": false} +{"time": 1789744300.5220234, "available_gib": 242.24016952514648, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false} +{"time": 1789744302.5465665, "available_gib": 242.22364044189453, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false} +{"time": 1789744304.5709565, "available_gib": 242.16735076904297, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false} +{"time": 1789744306.5958922, "available_gib": 242.09473419189453, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false} +{"time": 1789744308.6201835, "available_gib": 242.06150817871094, "oom_kill": 0, "gpu": "0, 85651, 0, 33", "stop": false} +{"time": 1789744310.6645873, "available_gib": 241.44736862182617, "oom_kill": 0, "gpu": "3, 85651, 0, 33", "stop": false} +{"time": 1789744312.707104, "available_gib": 241.14602661132812, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744314.75055, "available_gib": 240.90560913085938, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744316.792225, "available_gib": 240.69082641601562, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744318.8365598, "available_gib": 240.03157424926758, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744320.878574, "available_gib": 239.80778884887695, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744322.9206219, "available_gib": 239.54640579223633, "oom_kill": 0, "gpu": "3, 85651, 0, 32", "stop": false} +{"time": 1789744324.940512, "available_gib": 239.1410789489746, "oom_kill": 0, "gpu": "2893, 85651, 0, 33", "stop": false} +{"time": 1789744326.9599116, "available_gib": 238.2663345336914, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744328.9792445, "available_gib": 233.60436630249023, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744330.999156, "available_gib": 229.81796646118164, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744333.018626, "available_gib": 225.89186096191406, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744335.0391977, "available_gib": 221.95586776733398, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744337.058962, "available_gib": 218.04838180541992, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744339.0786686, "available_gib": 214.0251693725586, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744341.0983963, "available_gib": 210.03896713256836, "oom_kill": 0, "gpu": "4265, 85651, 0, 33", "stop": false} +{"time": 1789744343.1179674, "available_gib": 206.0509033203125, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744345.137733, "available_gib": 202.11498641967773, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744347.1575859, "available_gib": 198.17535018920898, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744349.1773338, "available_gib": 194.24428176879883, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744351.1974428, "available_gib": 190.24589157104492, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744353.2174225, "available_gib": 186.31936645507812, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744355.237774, "available_gib": 182.4132080078125, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744357.258104, "available_gib": 178.48447036743164, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744359.27828, "available_gib": 174.53839492797852, "oom_kill": 0, "gpu": "4265, 85651, 0, 34", "stop": false} +{"time": 1789744361.299235, "available_gib": 173.70015335083008, "oom_kill": 0, "gpu": "76825, 85651, 6, 34", "stop": false} +{"time": 1789744363.320299, "available_gib": 173.70723342895508, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744365.341664, "available_gib": 173.72999572753906, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744367.3625295, "available_gib": 173.7398796081543, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744369.3839087, "available_gib": 173.75790405273438, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744371.4052994, "available_gib": 173.76782608032227, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744373.426473, "available_gib": 173.79326248168945, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744375.447874, "available_gib": 173.82366561889648, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744377.469133, "available_gib": 173.84239959716797, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744379.4954257, "available_gib": 173.85829544067383, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744381.5174913, "available_gib": 173.86966705322266, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744383.53876, "available_gib": 173.87560272216797, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744385.5598798, "available_gib": 173.89007186889648, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744387.5811183, "available_gib": 173.89695358276367, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744389.6018076, "available_gib": 173.90465545654297, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744391.622792, "available_gib": 173.89840698242188, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744393.6431725, "available_gib": 173.93378448486328, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744395.6640935, "available_gib": 173.99294662475586, "oom_kill": 0, "gpu": "76825, 85651, 1, 34", "stop": false} +{"time": 1789744397.6851332, "available_gib": 174.01663208007812, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744399.7062628, "available_gib": 174.02749633789062, "oom_kill": 0, "gpu": "76825, 85651, 4, 34", "stop": false} +{"time": 1789744401.7283592, "available_gib": 174.0538558959961, "oom_kill": 0, "gpu": "76825, 85651, 4, 35", "stop": false} +{"time": 1789744403.748965, "available_gib": 174.08990859985352, "oom_kill": 0, "gpu": "78573, 85651, 0, 36", "stop": false} +{"time": 1789744405.7693496, "available_gib": 173.97901916503906, "oom_kill": 0, "gpu": "82343, 85651, 0, 35", "stop": false} +{"time": 1789744407.790143, "available_gib": 174.0068016052246, "oom_kill": 0, "gpu": "82385, 85651, 0, 35", "stop": false} +{"time": 1789744409.8103971, "available_gib": 174.01287460327148, "oom_kill": 0, "gpu": "82385, 85651, 0, 35", "stop": false} +{"time": 1789744411.8309193, "available_gib": 173.89633178710938, "oom_kill": 0, "gpu": "78947, 85651, 0, 35", "stop": false} +{"time": 1789744413.8503227, "available_gib": 173.6109504699707, "oom_kill": 0, "gpu": "78947, 85651, 0, 35", "stop": false} +{"time": 1789744415.8708775, "available_gib": 172.81961822509766, "oom_kill": 0, "gpu": "79469, 85651, 0, 34", "stop": false} +{"time": 1789744417.8913932, "available_gib": 172.99155044555664, "oom_kill": 0, "gpu": "80911, 85651, 0, 35", "stop": false} +{"time": 1789744419.9111977, "available_gib": 172.97130966186523, "oom_kill": 0, "gpu": "80911, 85651, 0, 35", "stop": false} +{"time": 1789744421.9319136, "available_gib": 172.9793930053711, "oom_kill": 0, "gpu": "82133, 85651, 100, 41", "stop": false} +{"time": 1789744423.9523408, "available_gib": 172.97999572753906, "oom_kill": 0, "gpu": "82133, 85651, 100, 43", "stop": false} +{"time": 1789744425.9735954, "available_gib": 173.01754760742188, "oom_kill": 0, "gpu": "82133, 85651, 100, 44", "stop": false} +{"time": 1789744427.9937978, "available_gib": 173.03707122802734, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false} +{"time": 1789744430.014089, "available_gib": 173.0387420654297, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false} +{"time": 1789744432.034095, "available_gib": 173.0599822998047, "oom_kill": 0, "gpu": "82135, 85651, 0, 37", "stop": false} +{"time": 1789744434.053616, "available_gib": 172.9596939086914, "oom_kill": 0, "gpu": "82135, 85651, 0, 36", "stop": false} +{"time": 1789744436.073761, "available_gib": 172.88088989257812, "oom_kill": 0, "gpu": "80167, 85651, 1, 36", "stop": false} +{"time": 1789744438.0942142, "available_gib": 172.89511108398438, "oom_kill": 0, "gpu": "79907, 85651, 0, 36", "stop": false} +{"time": 1789744440.1141858, "available_gib": 173.14619064331055, "oom_kill": 0, "gpu": "79957, 85651, 12, 36", "stop": false} +{"time": 1789744442.134541, "available_gib": 173.14928817749023, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false} +{"time": 1789744444.1541903, "available_gib": 173.16854095458984, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false} +{"time": 1789744446.1740553, "available_gib": 173.13195419311523, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false} +{"time": 1789744448.1940587, "available_gib": 173.15098571777344, "oom_kill": 0, "gpu": "82921, 85651, 0, 36", "stop": false} +{"time": 1789744450.2143607, "available_gib": 173.18714904785156, "oom_kill": 0, "gpu": "82923, 85651, 0, 36", "stop": false} +{"time": 1789744452.2385378, "available_gib": 173.07194900512695, "oom_kill": 0, "gpu": "82929, 85651, 0, 36", "stop": false} +{"time": 1789744454.2581725, "available_gib": 172.97765350341797, "oom_kill": 0, "gpu": "82929, 85651, 0, 35", "stop": false} +{"time": 1789744456.278477, "available_gib": 172.9991798400879, "oom_kill": 0, "gpu": "82929, 85651, 7, 36", "stop": false} +{"time": 1789744458.2987852, "available_gib": 173.00883865356445, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false} +{"time": 1789744460.3185778, "available_gib": 172.9944725036621, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false} +{"time": 1789744462.3388991, "available_gib": 173.0210838317871, "oom_kill": 0, "gpu": "83063, 85651, 0, 35", "stop": false} +{"time": 1789744464.3591044, "available_gib": 173.00377655029297, "oom_kill": 0, "gpu": "83069, 85651, 0, 36", "stop": false} +{"time": 1789744466.3856246, "available_gib": 173.01154708862305, "oom_kill": 0, "gpu": "83069, 85651, 76, 41", "stop": false} +{"time": 1789744468.4109452, "available_gib": 173.03825759887695, "oom_kill": 0, "gpu": "83069, 85651, 76, 42", "stop": false} +{"time": 1789744470.4362907, "available_gib": 173.05249786376953, "oom_kill": 0, "gpu": "83069, 85651, 100, 43", "stop": false} +{"time": 1789744472.4610434, "available_gib": 172.96158599853516, "oom_kill": 0, "gpu": "83523, 85651, 100, 47", "stop": false} +{"time": 1789744474.4873188, "available_gib": 172.29580307006836, "oom_kill": 0, "gpu": "84647, 85651, 100, 47", "stop": false} +{"time": 1789744476.5117893, "available_gib": 172.28779220581055, "oom_kill": 0, "gpu": "84649, 85651, 100, 45", "stop": false} +{"time": 1789744478.5356812, "available_gib": 172.33455276489258, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744480.5609157, "available_gib": 172.34068298339844, "oom_kill": 0, "gpu": "84649, 85651, 100, 45", "stop": false} +{"time": 1789744482.5864017, "available_gib": 172.34762573242188, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false} +{"time": 1789744484.6110964, "available_gib": 172.34587860107422, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false} +{"time": 1789744486.6363378, "available_gib": 172.3453598022461, "oom_kill": 0, "gpu": "84649, 85651, 100, 46", "stop": false} +{"time": 1789744488.661498, "available_gib": 172.39890670776367, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false} +{"time": 1789744490.6863167, "available_gib": 172.45629119873047, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false} +{"time": 1789744492.7115188, "available_gib": 172.47060012817383, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false} +{"time": 1789744494.7373624, "available_gib": 172.5518569946289, "oom_kill": 0, "gpu": "84649, 85651, 100, 47", "stop": false} +{"time": 1789744496.7631228, "available_gib": 172.57133865356445, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false} +{"time": 1789744498.788501, "available_gib": 172.58835983276367, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false} +{"time": 1789744500.8139327, "available_gib": 172.58929061889648, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false} +{"time": 1789744502.839819, "available_gib": 172.5894432067871, "oom_kill": 0, "gpu": "84649, 85651, 100, 48", "stop": false} +{"time": 1789744504.8652601, "available_gib": 172.59299087524414, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744506.8909621, "available_gib": 172.59333419799805, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744508.9166646, "available_gib": 172.59065628051758, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744510.9422443, "available_gib": 172.59705352783203, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744512.9670806, "available_gib": 172.5942840576172, "oom_kill": 0, "gpu": "84649, 85651, 100, 49", "stop": false} +{"time": 1789744514.9919732, "available_gib": 172.5932502746582, "oom_kill": 0, "gpu": "84649, 85651, 76, 49", "stop": false} +{"time": 1789744517.0184538, "available_gib": 172.57636260986328, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744519.0439193, "available_gib": 172.5904884338379, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744521.0691228, "available_gib": 172.59058380126953, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744523.0951657, "available_gib": 172.58983993530273, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744525.1207883, "available_gib": 172.5893211364746, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744527.1458921, "available_gib": 172.5901107788086, "oom_kill": 0, "gpu": "84649, 85651, 100, 50", "stop": false} +{"time": 1789744529.1724672, "available_gib": 172.5957794189453, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744531.1980112, "available_gib": 172.59324264526367, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744533.2228138, "available_gib": 172.59070587158203, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744535.2482636, "available_gib": 172.6019172668457, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744537.2732003, "available_gib": 172.5978012084961, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744539.2988782, "available_gib": 172.60141372680664, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744541.3248637, "available_gib": 172.60461044311523, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744543.3506434, "available_gib": 172.60347747802734, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744545.3766954, "available_gib": 172.60011291503906, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744547.4024527, "available_gib": 172.60373306274414, "oom_kill": 0, "gpu": "84649, 85651, 100, 51", "stop": false} +{"time": 1789744549.427506, "available_gib": 172.6028594970703, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744551.4528599, "available_gib": 172.59807586669922, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744553.4781616, "available_gib": 172.58591842651367, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744555.503374, "available_gib": 172.59807586669922, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744557.5295079, "available_gib": 172.60498809814453, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744559.5547729, "available_gib": 172.58954238891602, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744561.5798154, "available_gib": 172.57410049438477, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744563.6050577, "available_gib": 172.57672882080078, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744565.6282964, "available_gib": 172.58179092407227, "oom_kill": 0, "gpu": "84649, 85651, 91, 51", "stop": false} +{"time": 1789744567.6539595, "available_gib": 172.58893585205078, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744569.6795833, "available_gib": 172.5806999206543, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744571.7047844, "available_gib": 172.57934188842773, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744573.7309473, "available_gib": 172.58593368530273, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744575.7565713, "available_gib": 172.58402633666992, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744577.7820156, "available_gib": 172.59289169311523, "oom_kill": 0, "gpu": "84649, 85651, 100, 52", "stop": false} +{"time": 1789744579.806945, "available_gib": 172.5892791748047, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744581.8329036, "available_gib": 172.58263397216797, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744583.8582916, "available_gib": 172.59558486938477, "oom_kill": 0, "gpu": "84649, 85651, 75, 53", "stop": false} +{"time": 1789744585.8839097, "available_gib": 172.59697341918945, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744587.9094357, "available_gib": 172.5935935974121, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744589.9352567, "available_gib": 172.59527969360352, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744591.9604733, "available_gib": 172.60036849975586, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744593.986449, "available_gib": 172.6029052734375, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744596.012588, "available_gib": 172.60062789916992, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744598.0393686, "available_gib": 172.60357284545898, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744600.0652373, "available_gib": 172.6021270751953, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744602.0909824, "available_gib": 172.59929656982422, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744604.1165397, "available_gib": 172.59819412231445, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744606.1419337, "available_gib": 172.59766387939453, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744608.1679437, "available_gib": 172.59556198120117, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744610.1933706, "available_gib": 172.59738159179688, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744612.2190385, "available_gib": 172.59409713745117, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744614.2444997, "available_gib": 172.59764862060547, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744616.2701576, "available_gib": 172.6019172668457, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744618.2950556, "available_gib": 172.60046005249023, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744620.320425, "available_gib": 172.59806442260742, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744622.3454478, "available_gib": 172.5938377380371, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744624.3713562, "available_gib": 172.59962844848633, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744626.396724, "available_gib": 172.59697723388672, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744628.421861, "available_gib": 172.59457397460938, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744630.4473107, "available_gib": 172.59413146972656, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744632.4730318, "available_gib": 172.59252166748047, "oom_kill": 0, "gpu": "84649, 85651, 100, 53", "stop": false} +{"time": 1789744634.4932597, "available_gib": 172.46091079711914, "oom_kill": 0, "gpu": "84649, 85651, 0, 50", "stop": false} +{"time": 1789744636.5176382, "available_gib": 172.46760177612305, "oom_kill": 0, "gpu": "84649, 85651, 100, 57", "stop": false} +{"time": 1789744638.5421684, "available_gib": 172.45936584472656, "oom_kill": 0, "gpu": "84649, 85651, 100, 57", "stop": false} +{"time": 1789744640.5671206, "available_gib": 172.4629135131836, "oom_kill": 0, "gpu": "84649, 85651, 100, 59", "stop": false} +{"time": 1789744642.5907373, "available_gib": 172.46268463134766, "oom_kill": 0, "gpu": "84649, 85651, 100, 58", "stop": false} +{"time": 1789744644.6149285, "available_gib": 172.47219467163086, "oom_kill": 0, "gpu": "84649, 85651, 100, 60", "stop": false} +{"time": 1789744646.6392212, "available_gib": 172.47919845581055, "oom_kill": 0, "gpu": "84649, 85651, 100, 60", "stop": false} +{"time": 1789744648.6654615, "available_gib": 172.48125457763672, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false} +{"time": 1789744650.6908221, "available_gib": 172.47984313964844, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false} +{"time": 1789744652.7153213, "available_gib": 172.4798126220703, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false} +{"time": 1789744654.7413292, "available_gib": 172.46713256835938, "oom_kill": 0, "gpu": "84649, 85651, 37, 54", "stop": false} +{"time": 1789744656.7650921, "available_gib": 172.46598052978516, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false} +{"time": 1789744658.788567, "available_gib": 172.46988677978516, "oom_kill": 0, "gpu": "84649, 85651, 100, 61", "stop": false} +{"time": 1789744660.8137739, "available_gib": 172.46673965454102, "oom_kill": 0, "gpu": "84649, 85651, 100, 62", "stop": false} +{"time": 1789744662.8383768, "available_gib": 172.47161865234375, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false} +{"time": 1789744664.8626714, "available_gib": 172.469970703125, "oom_kill": 0, "gpu": "84649, 85651, 100, 62", "stop": false} +{"time": 1789744666.8868442, "available_gib": 172.4693832397461, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false} +{"time": 1789744668.9095283, "available_gib": 172.47140884399414, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false} +{"time": 1789744670.9342623, "available_gib": 172.47360229492188, "oom_kill": 0, "gpu": "84649, 85651, 100, 64", "stop": false} +{"time": 1789744672.9589663, "available_gib": 172.47840118408203, "oom_kill": 0, "gpu": "84649, 85651, 100, 63", "stop": false} +{"time": 1789744674.9843948, "available_gib": 172.4886360168457, "oom_kill": 0, "gpu": "84669, 85651, 100, 59", "stop": false} +{"time": 1789744677.0047867, "available_gib": 172.51124954223633, "oom_kill": 0, "gpu": "84669, 85651, 84, 58", "stop": false} +{"time": 1789744679.029943, "available_gib": 172.49595260620117, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false} +{"time": 1789744681.055466, "available_gib": 172.49994659423828, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false} +{"time": 1789744683.0806797, "available_gib": 172.4964599609375, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false} +{"time": 1789744685.1061206, "available_gib": 172.49471282958984, "oom_kill": 0, "gpu": "84669, 85651, 100, 58", "stop": false} +{"time": 1789744687.131367, "available_gib": 172.50520706176758, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false} +{"time": 1789744689.1567147, "available_gib": 172.5035514831543, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false} +{"time": 1789744691.182143, "available_gib": 172.50299072265625, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false} +{"time": 1789744693.2079492, "available_gib": 172.50631713867188, "oom_kill": 0, "gpu": "84669, 85651, 100, 57", "stop": false} +{"time": 1789744695.2286391, "available_gib": 174.55757522583008, "oom_kill": 0, "gpu": "84669, 85651, 0, 52", "stop": false} +{"time": 1789744697.2490587, "available_gib": 175.3351173400879, "oom_kill": 0, "gpu": "84669, 85651, 0, 51", "stop": false} +{"time": 1789744699.2692952, "available_gib": 175.33306503295898, "oom_kill": 0, "gpu": "84669, 85651, 0, 50", "stop": false} +{"time": 1789744701.2899477, "available_gib": 175.34216690063477, "oom_kill": 0, "gpu": "84669, 85651, 0, 49", "stop": false} +{"time": 1789744703.310319, "available_gib": 175.33515167236328, "oom_kill": 0, "gpu": "84669, 85651, 0, 48", "stop": false} +{"time": 1789744709.8561594, "available_gib": 193.49578094482422, "oom_kill": 0, "gpu": "3, 85651, 0, 46", "stop": false} +{"time": 1789744711.8781233, "available_gib": 242.45547485351562, "oom_kill": 0, "gpu": "0, 85651, 0, 46", "stop": false} +{"time": 1789744713.8998814, "available_gib": 242.3747444152832, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false} +{"time": 1789744715.924673, "available_gib": 242.27092742919922, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false} +{"time": 1789744717.9494975, "available_gib": 242.2499008178711, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false} +{"time": 1789744719.9739065, "available_gib": 242.2431297302246, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789744721.9986234, "available_gib": 242.1970977783203, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789744724.023132, "available_gib": 242.1283721923828, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789744726.047791, "available_gib": 242.09485626220703, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false} +{"time": 1789744728.0922565, "available_gib": 241.76881408691406, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false} +{"time": 1789744730.1342695, "available_gib": 241.43723678588867, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789744732.1755586, "available_gib": 241.15095138549805, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789744734.2230153, "available_gib": 241.0072364807129, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744736.2652614, "available_gib": 240.35486221313477, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744738.3089554, "available_gib": 240.07362365722656, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744740.350472, "available_gib": 239.79244232177734, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789744742.3701403, "available_gib": 239.2092056274414, "oom_kill": 0, "gpu": "703, 85651, 0, 38", "stop": false} +{"time": 1789744744.3898382, "available_gib": 238.29636001586914, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744746.409347, "available_gib": 233.53387832641602, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744748.4292598, "available_gib": 229.6477928161621, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744750.4486876, "available_gib": 225.60779190063477, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744752.4681375, "available_gib": 221.5484733581543, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744754.4875712, "available_gib": 217.4948959350586, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744756.5070164, "available_gib": 213.44181060791016, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744758.526479, "available_gib": 209.36876678466797, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744760.5463486, "available_gib": 205.36697387695312, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744762.5667367, "available_gib": 201.33798217773438, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744764.5860617, "available_gib": 197.3143424987793, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744766.6053874, "available_gib": 193.25495529174805, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744768.6250842, "available_gib": 189.2132225036621, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744770.6450837, "available_gib": 185.17929458618164, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744772.665085, "available_gib": 181.16746139526367, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744774.684941, "available_gib": 177.13880157470703, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744777.100537, "available_gib": 173.67180252075195, "oom_kill": 0, "gpu": "4293, 85651, 1, 36", "stop": false} +{"time": 1789744779.1218758, "available_gib": 173.65283584594727, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744781.1431904, "available_gib": 173.6810760498047, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744783.1644783, "available_gib": 173.68351364135742, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false} +{"time": 1789744785.185819, "available_gib": 173.69385528564453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744787.207108, "available_gib": 173.71892547607422, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744789.2283504, "available_gib": 173.72776412963867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744791.2495914, "available_gib": 173.7425079345703, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744793.2705739, "available_gib": 173.75009155273438, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744795.2918007, "available_gib": 173.75555419921875, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744797.312934, "available_gib": 173.78668975830078, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744799.3340576, "available_gib": 173.80676651000977, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false} +{"time": 1789744801.355535, "available_gib": 173.82130432128906, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744803.3765295, "available_gib": 173.82918548583984, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744805.3975353, "available_gib": 173.84343338012695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744807.4185128, "available_gib": 173.85370635986328, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744809.4393783, "available_gib": 173.8660545349121, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744811.4602966, "available_gib": 173.88482284545898, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744813.481391, "available_gib": 173.90010833740234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744815.502314, "available_gib": 173.91947174072266, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744817.523715, "available_gib": 173.92353057861328, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false} +{"time": 1789744819.549531, "available_gib": 173.93406677246094, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false} +{"time": 1789744821.5700142, "available_gib": 173.76786422729492, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false} +{"time": 1789744823.5898106, "available_gib": 173.7619857788086, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false} +{"time": 1789744825.6101544, "available_gib": 173.74892807006836, "oom_kill": 0, "gpu": "82559, 85651, 0, 36", "stop": false} +{"time": 1789744827.6301765, "available_gib": 173.73380661010742, "oom_kill": 0, "gpu": "79121, 85651, 0, 36", "stop": false} +{"time": 1789744829.6503675, "available_gib": 173.36173629760742, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false} +{"time": 1789744831.6701827, "available_gib": 173.22573471069336, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false} +{"time": 1789744833.6909227, "available_gib": 172.6994743347168, "oom_kill": 0, "gpu": "81095, 85651, 0, 36", "stop": false} +{"time": 1789744835.7112474, "available_gib": 172.84538650512695, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789744837.7309992, "available_gib": 172.86717224121094, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789744839.753742, "available_gib": 172.88789749145508, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false} +{"time": 1789744841.774714, "available_gib": 172.90548706054688, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false} +{"time": 1789744843.795579, "available_gib": 172.92531967163086, "oom_kill": 0, "gpu": "82317, 85651, 90, 45", "stop": false} +{"time": 1789744845.8162441, "available_gib": 172.8922233581543, "oom_kill": 0, "gpu": "82319, 85651, 0, 38", "stop": false} +{"time": 1789744847.8363395, "available_gib": 172.93823623657227, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789744849.8560581, "available_gib": 172.86439895629883, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789744851.878012, "available_gib": 172.8914680480957, "oom_kill": 0, "gpu": "79683, 85651, 24, 40", "stop": false} +{"time": 1789744853.8984156, "available_gib": 172.936767578125, "oom_kill": 0, "gpu": "80189, 85651, 1, 37", "stop": false} +{"time": 1789744855.918923, "available_gib": 172.94670486450195, "oom_kill": 0, "gpu": "80239, 85651, 0, 37", "stop": false} +{"time": 1789744857.9392416, "available_gib": 173.1967430114746, "oom_kill": 0, "gpu": "80335, 85651, 0, 37", "stop": false} +{"time": 1789744859.9595659, "available_gib": 173.17830276489258, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744861.9793572, "available_gib": 173.14974975585938, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744863.999041, "available_gib": 173.08945846557617, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744866.0192883, "available_gib": 173.08757400512695, "oom_kill": 0, "gpu": "83035, 85651, 0, 36", "stop": false} +{"time": 1789744868.0397038, "available_gib": 173.0578498840332, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false} +{"time": 1789744870.0599365, "available_gib": 172.92647552490234, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false} +{"time": 1789744872.0805118, "available_gib": 172.9449348449707, "oom_kill": 0, "gpu": "83041, 85651, 3, 36", "stop": false} +{"time": 1789744874.1012588, "available_gib": 173.0093879699707, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} +{"time": 1789744876.121015, "available_gib": 172.98486709594727, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} diff --git a/audit/tuning/batch-1024/restored.json b/audit/tuning/batch-1024/restored.json new file mode 100644 index 0000000..0053d64 --- /dev/null +++ b/audit/tuning/batch-1024/restored.json @@ -0,0 +1 @@ +{"healthy": true, "time": 1789744876.2350338} \ No newline at end of file diff --git a/audit/tuning/final-state.json b/audit/tuning/final-state.json new file mode 100644 index 0000000..5959099 --- /dev/null +++ b/audit/tuning/final-state.json @@ -0,0 +1,7 @@ +{ + "time": 1789744894.941237, + "health": "healthy", + "restart": "unless-stopped", + "oom_killed": false, + "compose_matches_original": true +} \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp1-b2048/compose.yaml b/audit/tuning/steps-123/mtp1-b2048/compose.yaml new file mode 100644 index 0000000..566dfe5 --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/compose.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: "no" + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":1}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,2]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/steps-123/mtp1-b2048/mtp-metrics.txt b/audit/tuning/steps-123/mtp1-b2048/mtp-metrics.txt new file mode 100644 index 0000000..7a18534 --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/mtp-metrics.txt @@ -0,0 +1,24 @@ +# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_total counter +vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 13297.0 +# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_created gauge +vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897433982360141e+09 +# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_total counter +vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 13297.0 +# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_created gauge +vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897433982360315e+09 +# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_total counter +vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 10374.0 +# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_created gauge +vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743398236043e+09 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 10374.0 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897433982360573e+09 \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl b/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl new file mode 100644 index 0000000..c469b9c --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl @@ -0,0 +1,7 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.6113444969960256, "response": {"id": "chatcmpl-b16e6f7979109685", "object": "chat.completion", "created": 1789743630, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "deduplicate", "passed": true, "elapsed_s": 2.2101684480003314, "response": {"id": "chatcmpl-b7a2d1e2b74737cb", "object": "chat.completion", "created": 1789743632, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "schedule", "passed": true, "elapsed_s": 2.493778939002368, "response": {"id": "chatcmpl-94ebf857ca1ee68c", "object": "chat.completion", "created": 1789743634, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.6490989489975618, "response": {"id": "chatcmpl-9c57abdc32a627c6", "object": "chat.completion", "created": 1789743637, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-b3de4e6b44566b44", "object": "chat.completion", "created": 1789743638, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-b6b995fb183886f7", "object": "chat.completion", "created": 1789743647, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-a63e6f2654b5a582", "object": "chat.completion", "created": 1789743650, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-8666af43eb56d793", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9cae8d63", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} diff --git a/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl.log b/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl.log new file mode 100644 index 0000000..bbdd406 --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/quality-medium.jsonl.log @@ -0,0 +1,8 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.6113444969960256} +{"name": "deduplicate", "passed": true, "elapsed_s": 2.2101684480003314} +{"name": "schedule", "passed": true, "elapsed_s": 2.493778939002368} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.6490989489975618} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "tool-arguments", "passed": true} +QUALITY_COMPLETE diff --git a/audit/tuning/steps-123/mtp1-b2048/screen.jsonl b/audit/tuning/steps-123/mtp1-b2048/screen.jsonl new file mode 100644 index 0000000..eecf9ad --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/screen.jsonl @@ -0,0 +1,37 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6788992599977064, "elapsed_s": 1.330652390999603, "decode_tps": 124.28018546759276, "max_stream_gap_s": 0.016711308002413716, "events": 42, "content": "\n\n3973", "reasoning": "I need to compute 137 * 29 + 0.\n\n137 * 29 = 137 * 30 - 137 = 4110 - 137 = 3973\n\nAdding 0 doesn't change anything, so the answer is 3973.\n", "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08442727399960859, "elapsed_s": 0.38878584299527574, "decode_tps": 118.28153916873126, "max_stream_gap_s": 0.017301366999163292, "events": 19, "content": "\n\n5761", "reasoning": "160 * 36 = 5760. 5760 + 1 = 5761.\n", "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08499549300177023, "elapsed_s": 0.6222111509996466, "decode_tps": 120.99423952430095, "max_stream_gap_s": 0.03257976900204085, "events": 32, "content": "\n\n7871", "reasoning": "183 * 43 = 183 * 40 + 183 * 3 = 7320 + 549 = 7869. Then 7869 + 2 = 7871.\n", "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08491974999924423, "elapsed_s": 0.39251107800373575, "decode_tps": 123.54054402809794, "max_stream_gap_s": 0.03264573799970094, "events": 19, "content": "\n\n10303", "reasoning": "206 * 50 = 10300\n10300 + 3 = 10303\n", "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.085520082997391, "elapsed_s": 0.6701685270018061, "decode_tps": 121.4405010876311, "max_stream_gap_s": 0.032602148996375035, "events": 36, "content": "\n\n13057", "reasoning": "229 * 57 = 229 * 50 + 229 * 7 = 11450 + 1603 = 13053. Then 13053 + 4 = 13057.\n", "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08477247699920554, "elapsed_s": 0.4252711029985221, "decode_tps": 120.41164594914487, "max_stream_gap_s": 0.032577335005044006, "events": 21, "content": "\n\n16133", "reasoning": "252 * 64 = 16128. Then 16128 + 5 = 16133.\n", "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08490634099871386, "elapsed_s": 0.6701965460015344, "decode_tps": 119.59878945806493, "max_stream_gap_s": 0.03245343499293085, "events": 36, "content": "\n\n19531", "reasoning": "275 * 71 = 275 * 70 + 275 * 1 = 19250 + 275 = 19525. Then 19525 + 6 = 19531.\n", "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.0855920120011433, "elapsed_s": 0.8476354470039951, "decode_tps": 122.04028763747826, "max_stream_gap_s": 0.03259085400350159, "events": 47, "content": "\n\n23251", "reasoning": "298 * 78 = 298 * 78. Let me compute this step by step.\n\n298 * 78 = 298 * 80 - 298 * 2 = 23840 - 596 = 23244\n\n23244 + 7 = 23251\n", "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08551327399618458, "elapsed_s": 0.7655381610020413, "decode_tps": 123.52489093428804, "max_stream_gap_s": 0.03254275799554307, "events": 42, "content": "\n\n27293", "reasoning": "Let me compute 321 * 85 + 8.\n\n321 * 85 = 321 * 80 + 321 * 5 = 25680 + 1605 = 27285\n\n27285 + 8 = 27293\n", "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08536207699944498, "elapsed_s": 0.7670097699956386, "decode_tps": 121.7637803997724, "max_stream_gap_s": 0.016597622001427226, "events": 42, "content": "\n\n31657", "reasoning": "Let me compute 344 * 92 + 9.\n\n344 * 92 = 344 * 90 + 344 * 2 = 30960 + 688 = 31648\n\n31648 + 9 = 31657\n", "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08568318800098496, "elapsed_s": 0.7488038730007247, "decode_tps": 123.65773207637488, "max_stream_gap_s": 0.03235569100070279, "events": 41, "content": "\n\n36343", "reasoning": "Let me compute 367 * 99 + 10.\n\n367 * 99 = 367 * (100 - 1) = 36700 - 367 = 36333\n\n36333 + 10 = 36343\n", "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08726995399774751, "elapsed_s": 0.8155018589968677, "decode_tps": 122.21381594110125, "max_stream_gap_s": 0.016604981996351853, "events": 45, "content": "\n\n41351", "reasoning": "Let me compute 390 * 106 + 11.\n\n390 * 106 = 390 * 100 + 390 * 6 = 39000 + 2340 = 41340\n\n41340 + 11 = 41351\n", "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.10169535600289237, "elapsed_s": 0.4396858489999431, "decode_tps": 124.26385022719111, "max_stream_gap_s": 0.016457880003144965, "events": 22, "content": "\n\n{\"name\":\"设备A\",\"count\":7,\"enabled\":true}", "reasoning": "The user wants me to return exactly a specific JSON object with no markdown formatting. Let me provide exactly what they asked for.\n", "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08512602300470462, "elapsed_s": 0.6315631280012894, "decode_tps": 113.46228034860022, "max_stream_gap_s": 0.0164907750004204, "events": 34, "content": "\n\nAQUA-58321", "reasoning": "The user is asking me to recall the code they previously told me to remember. They said \"Remember the code AQUA-58321.\" and now they're asking what it was, with the instruction to reply only with the code.\n", "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3263, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.7587772789993323, "elapsed_s": 1.8307137399970088, "decode_tps": 118.4771715683578, "max_stream_gap_s": 0.017939804994966835, "events": 67, "content": "\n\nCode: SPARK-82741\nSum: 1368", "reasoning": "The user wants the verification code and the sum of three amounts from the invoice image. Reading the image: Verification: SPARK-82741. Amount A: 125, Amount B: 478, Amount C: 765. Sum = 125 + 478 = 603; 603 + 765 = 1368. I need to provide a concise, self-contained answer with the code and the sum.\n", "sha256": "8921e4d6739375a20a9acdb345ebda84acb52f498071a986927bd0a4e61a3779"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12197, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 2.490573155002494, "elapsed_s": 3.5669434599985834, "decode_tps": 117.98913386082441, "max_stream_gap_s": 0.03252299100131495, "events": 66, "content": "\n\nRed, Green, Blue, Yellow, Purple, White", "reasoning": "The user wants the six background colors in chronological order, based on the video frames. Let me trace through the timestamps: 0-19s shows RED, 20-39s shows GREEN, 40-59s shows BLUE, 60-79s shows YELLOW, 80-99s shows PURPLE, 100-119s shows WHITE. That's six colors in order. The instruction says reply only with the English color names. I'll provide them as a simple list.\n", "sha256": "3c4546e143b0d70a31d19afcd37f4462f9f4bc75d2ed102649b295b696577524"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.4522779820035794, "elapsed_s": 2.0256833830062533, "decode_tps": 122.07767816207503, "max_stream_gap_s": 0.01650294200226199, "events": 37, "content": "\n\nSPARK-82741", "reasoning": "The user wants the verification code from the invoice. All four copies show the same text: \"Verification: SPARK-82741\". The instruction says to reply only with the code. So the answer is SPARK-82741. No need for elaboration.\n", "sha256": "81142145348f20b3b8803d7c6d9426ccc5641f864e2bd78da3e927825dc764b3"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08386934799636947, "elapsed_s": 2.4533141759966384, "decode_tps": 107.62014670550964, "max_stream_gap_s": 0.016294626002490986, "events": 150, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling,", "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0830643839944969, "elapsed_s": 18.754658626996388, "decode_tps": 109.6317739856207, "max_stream_gap_s": 0.016825594000692945, "events": 1176, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and continue? Or raise. Let's add error and continue? Strategy FAIL means fail import on duplicate. Could add error and if on_error? Better: if duplicate_strategy == FAIL: add RowError code DUPLICATE_KEY, then if on_error FAIL raise? But if on_error SKIP_ROW, continue. Hmm.\n REPLACE:", "sha256": "4906c37a2e701fd6633345dd53b182e28095d6b5da93f478eac8d410d6af322a"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07255172300210688, "elapsed_s": 18.743193971000437, "decode_tps": 109.63736398620448, "max_stream_gap_s": 0.01642427300248528, "events": 1176, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error code DUPLICATE_KEY, continue? or fail. We'll add error and count duplicate_rows. If on_error FAIL? duplicate is error.\n REPLACE: if exists repo.update else insert; also seen.\n- If no duplicate_keys: no duplicate handling.\n\nNeed transaction? Could wrap repository in transaction? Provide optional transaction callback? Maybe not", "sha256": "bfcbf92932af2de82cc8324af42ba2d393695f14eb500ea32fbcad6f781dcfaf"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07399964099749923, "elapsed_s": 18.77744392999739, "decode_tps": 109.44508232656955, "max_stream_gap_s": 0.01670600000215927, "events": 1178, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and stop? Or error row.\n REPLACE: if exists repo.update else insert. Similar to UPDATE? Maybe UPDATE only if exists, REPLACE always upsert. Need differentiate: UPDATE: if duplicate in file? maybe update in-memory? For repository, if exists update. If duplicate within file and strategy SKIP, skip. If UPDATE", "sha256": "685a932841c3272e4da5b4f12e188bdb7fd6b50b4b564bc426e3e139793f068c"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08481309299531858, "elapsed_s": 2.3875851429984323, "decode_tps": 110.73610173427943, "max_stream_gap_s": 0.03143068100325763, "events": 145, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08440493299713125, "elapsed_s": 18.277994324002066, "decode_tps": 112.5121577720147, "max_stream_gap_s": 0.03122073299891781, "events": 1147, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 并发审批冲突 | 同一节点仅允许一次有效审批,后提交者提示“已处理” |\n\n---\n\n## 4.2.2 需求项:", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "5e28251162bb5f8117e86f4f2cb747928f5bb5287e6cec1443d2d64deefb8fb3"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07261836499674246, "elapsed_s": 18.21824221299903, "decode_tps": 112.80956869528413, "max_stream_gap_s": 0.03128572199784685, "events": 1144, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交 |\n| 会签中部分人拒绝 | 整体拒绝,流程终止 |\n| 或签中一人通过 | 节点通过,其余人待办自动关闭 |\n| 审批意见为空 | 拒绝时必填意见,通过时可选 |\n| 审批并发冲突 | ", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "314d5c03936def50095096c59aab4bfc5743e49e546edd449ec7db880db067df"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07247121300315484, "elapsed_s": 18.250864649002324, "decode_tps": 112.60621062069599, "max_stream_gap_s": 0.03126876999886008, "events": 1146, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过/拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止 |\n\n---\n\n## 4.2.2 需求项:审批发起\n\n### ", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "18736c57814a1186850c8f28964837b22549363a2ab0e0e00f2332ebe1aeaa88"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.0831944410019787, "elapsed_s": 2.406979079998564, "decode_tps": 109.7347816663895, "max_stream_gap_s": 0.01621198299835669, "events": 148, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsett", "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0725614070033771, "elapsed_s": 18.94562807300099, "decode_tps": 108.46144064588867, "max_stream_gap_s": 0.031187818000034895, "events": 1189, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a already settled.\n\nNeed maybe include step-by-step table for counterexample:\nVertices s,a,b. Edges: s→a:1, s→b:2, b→a:-2.\nInitial d: s=0,a=1,b=∞? Wait if initialize by relaxing source? Standard Dijkstra:", "sha256": "e35a9f765a93cce044543990fe166bfafe2cb845b0e603c334dc449496e21491"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0843369200010784, "elapsed_s": 18.989241095994657, "decode_tps": 108.27878210561818, "max_stream_gap_s": 0.03129963100218447, "events": 1191, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax.\n", "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.07209744000283536, "elapsed_s": 18.980188051005825, "decode_tps": 108.26053471569521, "max_stream_gap_s": 0.03122235600312706, "events": 1191, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax.\n", "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126130, "completion_tokens": 135, "completion_tokens_details": {"reasoning_tokens": 115}}, "ttft_s": 14.440546402998734, "elapsed_s": 15.590460605999397, "decode_tps": 116.53043300998588, "max_stream_gap_s": 0.032008403999498114, "events": 71, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data they've provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n\nThe user wants only these two codes.\n", "sha256": "ee0579233e57d5accc2cf35a71509aa83fe7dcf3a2727f10f973dec75ff41759"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126129, "completion_tokens": 134, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 0.856336849006766, "elapsed_s": 1.9747926430063671, "decode_tps": 118.91395325012499, "max_stream_gap_s": 0.03192400800617179, "events": 69, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data they've provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n\nThe user wants both codes only.\n", "sha256": "9400cbc20a09d71e03a4c31d5413dd314ee75a387fce3dc146f230a763b8faf3"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126128, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 113}}, "ttft_s": 0.853535173999262, "elapsed_s": 1.9229757680004695, "decode_tps": 123.4290158241842, "max_stream_gap_s": 0.03247794899652945, "events": 67, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n\nThe user wants only these two codes.\n", "sha256": "48e028596401908f642e64f951e3f0c5c7b6314e6a46a87c047b4728d9438d99"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111863, "completion_tokens": 193, "completion_tokens_details": {"reasoning_tokens": 148}}, "ttft_s": 12.791386062002857, "elapsed_s": 14.462968651001574, "decode_tps": 114.86121072546499, "max_stream_gap_s": 0.03197040300437948, "events": 103, "content": "\n\n- **Northern access code:** NORTH-49273\n- **Southern access code:** SOUTH-18362\n- **Invoice verification code:** SPARK-82741", "reasoning": "The user wants three pieces of information: the northern access code, the southern access code, and the invoice verification code. These are scattered across the text and the image.\n\nScanning the text: \"The northern access code is NORTH-49273.\" appears after Item 000019. \"The southern access code is SOUTH-18362.\" appears after Item 002499.\n\nFrom the image: \"Verification: SPARK-82741\".\n\nSo the answer is straightforward extraction. No computation needed. I'll present all three clearly. The amounts A/B/C are distractors not requested.\n\nI'll write a concise self-contained response.\n", "sha256": "410a369097c639859051822f809cd9adecfca2ac1f489544f9525a65f25ef94a"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08371570699819131, "elapsed_s": 0.5917828360034036, "decode_tps": 123.99936229559475, "max_stream_gap_s": 0.03208907500084024, "events": 32, "content": "\n\n731", "reasoning": "The user wants me to compute 43 * 17 and reply with only the integer.\n\n43 * 17 = 43 * 10 + 43 * 7 = 430 + 301 = 731\n", "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.6603063000002294, "elapsed_s": 0.9943764119961997, "decode_tps": 125.72211189160979, "max_stream_gap_s": 0.03203328799281735, "events": 21, "content": "\n\n748", "reasoning": "44 * 17 = 44 * 10 + 44 * 7 = 440 + 308 = 748\n", "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.0632171030010795, "elapsed_s": 1.5722418850054964, "decode_tps": 123.76607628398992, "max_stream_gap_s": 0.01634497399936663, "events": 32, "content": "\n\n765", "reasoning": "The user wants me to compute 45 * 17 and reply with only the integer.\n\n45 * 17 = 45 * 10 + 45 * 7 = 450 + 315 = 765\n", "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.6414810709975427, "elapsed_s": 2.150085207002121, "decode_tps": 123.86843822173104, "max_stream_gap_s": 0.016239169002801646, "events": 32, "content": "\n\n782", "reasoning": "The user wants me to compute 46 * 17 and reply with just the integer.\n\n46 * 17 = 46 * 10 + 46 * 7 = 460 + 322 = 782\n", "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} diff --git a/audit/tuning/steps-123/mtp1-b2048/screen.jsonl.log b/audit/tuning/steps-123/mtp1-b2048/screen.jsonl.log new file mode 100644 index 0000000..34d5fb2 --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/screen.jsonl.log @@ -0,0 +1,38 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6788992599977064, "elapsed_s": 1.330652390999603, "decode_tps": 124.28018546759276, "max_stream_gap_s": 0.016711308002413716, "events": 42, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08442727399960859, "elapsed_s": 0.38878584299527574, "decode_tps": 118.28153916873126, "max_stream_gap_s": 0.017301366999163292, "events": 19, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08499549300177023, "elapsed_s": 0.6222111509996466, "decode_tps": 120.99423952430095, "max_stream_gap_s": 0.03257976900204085, "events": 32, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08491974999924423, "elapsed_s": 0.39251107800373575, "decode_tps": 123.54054402809794, "max_stream_gap_s": 0.03264573799970094, "events": 19, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.085520082997391, "elapsed_s": 0.6701685270018061, "decode_tps": 121.4405010876311, "max_stream_gap_s": 0.032602148996375035, "events": 36, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08477247699920554, "elapsed_s": 0.4252711029985221, "decode_tps": 120.41164594914487, "max_stream_gap_s": 0.032577335005044006, "events": 21, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08490634099871386, "elapsed_s": 0.6701965460015344, "decode_tps": 119.59878945806493, "max_stream_gap_s": 0.03245343499293085, "events": 36, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.0855920120011433, "elapsed_s": 0.8476354470039951, "decode_tps": 122.04028763747826, "max_stream_gap_s": 0.03259085400350159, "events": 47, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08551327399618458, "elapsed_s": 0.7655381610020413, "decode_tps": 123.52489093428804, "max_stream_gap_s": 0.03254275799554307, "events": 42, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08536207699944498, "elapsed_s": 0.7670097699956386, "decode_tps": 121.7637803997724, "max_stream_gap_s": 0.016597622001427226, "events": 42, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08568318800098496, "elapsed_s": 0.7488038730007247, "decode_tps": 123.65773207637488, "max_stream_gap_s": 0.03235569100070279, "events": 41, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08726995399774751, "elapsed_s": 0.8155018589968677, "decode_tps": 122.21381594110125, "max_stream_gap_s": 0.016604981996351853, "events": 45, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.10169535600289237, "elapsed_s": 0.4396858489999431, "decode_tps": 124.26385022719111, "max_stream_gap_s": 0.016457880003144965, "events": 22, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08512602300470462, "elapsed_s": 0.6315631280012894, "decode_tps": 113.46228034860022, "max_stream_gap_s": 0.0164907750004204, "events": 34, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3263, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 107}}, "ttft_s": 0.7587772789993323, "elapsed_s": 1.8307137399970088, "decode_tps": 118.4771715683578, "max_stream_gap_s": 0.017939804994966835, "events": 67, "sha256": "8921e4d6739375a20a9acdb345ebda84acb52f498071a986927bd0a4e61a3779"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12197, "completion_tokens": 128, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 2.490573155002494, "elapsed_s": 3.5669434599985834, "decode_tps": 117.98913386082441, "max_stream_gap_s": 0.03252299100131495, "events": 66, "sha256": "3c4546e143b0d70a31d19afcd37f4462f9f4bc75d2ed102649b295b696577524"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.4522779820035794, "elapsed_s": 2.0256833830062533, "decode_tps": 122.07767816207503, "max_stream_gap_s": 0.01650294200226199, "events": 37, "sha256": "81142145348f20b3b8803d7c6d9426ccc5641f864e2bd78da3e927825dc764b3"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08386934799636947, "elapsed_s": 2.4533141759966384, "decode_tps": 107.62014670550964, "max_stream_gap_s": 0.016294626002490986, "events": 150, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0830643839944969, "elapsed_s": 18.754658626996388, "decode_tps": 109.6317739856207, "max_stream_gap_s": 0.016825594000692945, "events": 1176, "sha256": "4906c37a2e701fd6633345dd53b182e28095d6b5da93f478eac8d410d6af322a"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07255172300210688, "elapsed_s": 18.743193971000437, "decode_tps": 109.63736398620448, "max_stream_gap_s": 0.01642427300248528, "events": 1176, "sha256": "bfcbf92932af2de82cc8324af42ba2d393695f14eb500ea32fbcad6f781dcfaf"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07399964099749923, "elapsed_s": 18.77744392999739, "decode_tps": 109.44508232656955, "max_stream_gap_s": 0.01670600000215927, "events": 1178, "sha256": "685a932841c3272e4da5b4f12e188bdb7fd6b50b4b564bc426e3e139793f068c"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08481309299531858, "elapsed_s": 2.3875851429984323, "decode_tps": 110.73610173427943, "max_stream_gap_s": 0.03143068100325763, "events": 145, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08440493299713125, "elapsed_s": 18.277994324002066, "decode_tps": 112.5121577720147, "max_stream_gap_s": 0.03122073299891781, "events": 1147, "sha256": "5e28251162bb5f8117e86f4f2cb747928f5bb5287e6cec1443d2d64deefb8fb3"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07261836499674246, "elapsed_s": 18.21824221299903, "decode_tps": 112.80956869528413, "max_stream_gap_s": 0.03128572199784685, "events": 1144, "sha256": "314d5c03936def50095096c59aab4bfc5743e49e546edd449ec7db880db067df"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07247121300315484, "elapsed_s": 18.250864649002324, "decode_tps": 112.60621062069599, "max_stream_gap_s": 0.03126876999886008, "events": 1146, "sha256": "18736c57814a1186850c8f28964837b22549363a2ab0e0e00f2332ebe1aeaa88"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.0831944410019787, "elapsed_s": 2.406979079998564, "decode_tps": 109.7347816663895, "max_stream_gap_s": 0.01621198299835669, "events": 148, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0725614070033771, "elapsed_s": 18.94562807300099, "decode_tps": 108.46144064588867, "max_stream_gap_s": 0.031187818000034895, "events": 1189, "sha256": "e35a9f765a93cce044543990fe166bfafe2cb845b0e603c334dc449496e21491"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.0843369200010784, "elapsed_s": 18.989241095994657, "decode_tps": 108.27878210561818, "max_stream_gap_s": 0.03129963100218447, "events": 1191, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2047}}, "ttft_s": 0.07209744000283536, "elapsed_s": 18.980188051005825, "decode_tps": 108.26053471569521, "max_stream_gap_s": 0.03122235600312706, "events": 1191, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126130, "completion_tokens": 135, "completion_tokens_details": {"reasoning_tokens": 115}}, "ttft_s": 14.440546402998734, "elapsed_s": 15.590460605999397, "decode_tps": 116.53043300998588, "max_stream_gap_s": 0.032008403999498114, "events": 71, "sha256": "ee0579233e57d5accc2cf35a71509aa83fe7dcf3a2727f10f973dec75ff41759"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126129, "completion_tokens": 134, "completion_tokens_details": {"reasoning_tokens": 114}}, "ttft_s": 0.856336849006766, "elapsed_s": 1.9747926430063671, "decode_tps": 118.91395325012499, "max_stream_gap_s": 0.03192400800617179, "events": 69, "sha256": "9400cbc20a09d71e03a4c31d5413dd314ee75a387fce3dc146f230a763b8faf3"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126128, "completion_tokens": 133, "completion_tokens_details": {"reasoning_tokens": 113}}, "ttft_s": 0.853535173999262, "elapsed_s": 1.9229757680004695, "decode_tps": 123.4290158241842, "max_stream_gap_s": 0.03247794899652945, "events": 67, "sha256": "48e028596401908f642e64f951e3f0c5c7b6314e6a46a87c047b4728d9438d99"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111863, "completion_tokens": 193, "completion_tokens_details": {"reasoning_tokens": 148}}, "ttft_s": 12.791386062002857, "elapsed_s": 14.462968651001574, "decode_tps": 114.86121072546499, "max_stream_gap_s": 0.03197040300437948, "events": 103, "sha256": "410a369097c639859051822f809cd9adecfca2ac1f489544f9525a65f25ef94a"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08371570699819131, "elapsed_s": 0.5917828360034036, "decode_tps": 123.99936229559475, "max_stream_gap_s": 0.03208907500084024, "events": 32, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.6603063000002294, "elapsed_s": 0.9943764119961997, "decode_tps": 125.72211189160979, "max_stream_gap_s": 0.03203328799281735, "events": 21, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.0632171030010795, "elapsed_s": 1.5722418850054964, "decode_tps": 123.76607628398992, "max_stream_gap_s": 0.01634497399936663, "events": 32, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.6414810709975427, "elapsed_s": 2.150085207002121, "decode_tps": 123.86843822173104, "max_stream_gap_s": 0.016239169002801646, "events": 32, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} +SUITE_COMPLETE diff --git a/audit/tuning/steps-123/mtp1-b2048/server.log b/audit/tuning/steps-123/mtp1-b2048/server.log new file mode 100644 index 0000000..3619acb --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/server.log @@ -0,0 +1,339 @@ +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] █ █ █▄ ▄█ +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] █▄█▀ █ █ █ █ model /model +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 14:54:04 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 1}, 'compilation_config': {'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 2], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:54:04 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration +(APIServer pid=1) INFO 09-18 14:54:04 [model.py:2024] Using max model len 131072 +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:54:07 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:54:07 [model.py:691] Resolved architecture: Qwen4ExpMTP +(APIServer pid=1) INFO 09-18 14:54:07 [model.py:2024] Using max model len 262144 +(APIServer pid=1) INFO 09-18 14:54:07 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072 +(APIServer pid=1) INFO 09-18 14:54:07 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled +(APIServer pid=1) INFO 09-18 14:54:07 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False) +(APIServer pid=1) INFO 09-18 14:54:07 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out. +(APIServer pid=1) INFO 09-18 14:54:07 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']) +(APIServer pid=1) WARNING 09-18 14:54:07 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens. +(APIServer pid=1) INFO 09-18 14:54:07 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) INFO 09-18 14:54:20 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=1), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 2, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None) +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) INFO 09-18 14:54:21 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_d75d110cee7b4c8d8fc41c851dee4d8c backend=nccl +(EngineCore pid=112) INFO 09-18 14:54:21 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=112) INFO 09-18 14:54:21 [gpu_worker.py:441] Using V2 Model Runner +(EngineCore pid=112) INFO 09-18 14:54:22 [model_runner.py:387] Loading model from scratch... +(EngineCore pid=112) INFO 09-18 14:54:22 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention +(EngineCore pid=112) INFO 09-18 14:54:22 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention. +(EngineCore pid=112) INFO 09-18 14:54:22 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128). +(EngineCore pid=112) INFO 09-18 14:54:22 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda +(EngineCore pid=112) INFO 09-18 14:54:24 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION']. +(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. +(APIServer pid=1) INFO 09-18 14:54:26 [base.py:261] Multi-modal warmup completed in 12.108s +(APIServer pid=1) INFO 09-18 14:54:27 [base.py:261] Readonly multi-modal warmup completed in 1.295s +(EngineCore pid=112) INFO 09-18 14:54:58 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True +(EngineCore pid=112) INFO 09-18 14:54:58 [flash_attn.py:1115] Using FlashAttention version 2 +(EngineCore pid=112) WARNING 09-18 14:54:58 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect +(EngineCore pid=112) INFO 09-18 14:54:58 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.31 GiB. +(EngineCore pid=112) INFO 09-18 14:54:58 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch. +(EngineCore pid=112) +Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00= mamba page size. +(EngineCore pid=112) INFO 09-18 14:55:51 [utils.py:320] Using BLNHC KV cache layout. +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=112) INFO 09-18 14:55:55 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size. +(EngineCore pid=112) WARNING 09-18 14:56:17 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's. +(EngineCore pid=112) WARNING 09-18 14:56:17 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY +(EngineCore pid=112) +Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00 262144). Running this sequence through the model will result in indexing errors +(APIServer pid=1) INFO: 172.21.0.1:43784 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43794 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43806 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43814 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43818 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40940 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40946 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40952 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40964 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40968 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40970 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40984 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40986 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:41002 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:41010 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:59:59 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 43.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 39.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 14:59:59 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.72, Accepted throughput: 18.10 tokens/s, Drafted throughput: 25.00 tokens/s, Accepted: 181 tokens, Drafted: 250 tokens, Per-position acceptance rate: 0.724, Avg Draft acceptance rate: 72.4% +(APIServer pid=1) INFO 09-18 15:00:09 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 0.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 74.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 15:00:09 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.00, Accepted throughput: 0.30 tokens/s, Drafted throughput: 0.30 tokens/s, Accepted: 3 tokens, Drafted: 3 tokens, Per-position acceptance rate: 1.000, Avg Draft acceptance rate: 100.0% +(APIServer pid=1) INFO: 172.21.0.1:52464 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52476 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:57076 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:52546 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:00:19 [loggers.py:323] Engine 000: Avg prompt throughput: 1243.7 tokens/s, Avg generation throughput: 39.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 42.4%, Prefix cache hit rate: 46.2%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:00:19 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.92, Accepted throughput: 18.90 tokens/s, Drafted throughput: 20.60 tokens/s, Accepted: 189 tokens, Drafted: 206 tokens, Per-position acceptance rate: 0.917, Avg Draft acceptance rate: 91.7% +(APIServer pid=1) INFO: 172.21.0.1:38136 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38144 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38150 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38158 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:00:29 [loggers.py:323] Engine 000: Avg prompt throughput: 11171.3 tokens/s, Avg generation throughput: 24.7 tokens/s, Running: 1 reqs, Waiting: 3 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.2%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:00:29 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.87, Accepted throughput: 11.40 tokens/s, Drafted throughput: 13.10 tokens/s, Accepted: 114 tokens, Drafted: 131 tokens, Per-position acceptance rate: 0.870, Avg Draft acceptance rate: 87.0% +(APIServer pid=1) INFO: 172.21.0.1:38172 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38188 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:34860 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:34870 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:00:39 [loggers.py:323] Engine 000: Avg prompt throughput: 58.8 tokens/s, Avg generation throughput: 114.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.1%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:00:39 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.95, Accepted throughput: 55.30 tokens/s, Drafted throughput: 58.40 tokens/s, Accepted: 553 tokens, Drafted: 584 tokens, Per-position acceptance rate: 0.947, Avg Draft acceptance rate: 94.7% +(APIServer pid=1) INFO: 127.0.0.1:44446 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:34880 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:00:49 [loggers.py:323] Engine 000: Avg prompt throughput: 9.2 tokens/s, Avg generation throughput: 117.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 46.1%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:00:49 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.89, Accepted throughput: 55.10 tokens/s, Drafted throughput: 62.10 tokens/s, Accepted: 551 tokens, Drafted: 621 tokens, Per-position acceptance rate: 0.887, Avg Draft acceptance rate: 88.7% +(APIServer pid=1) INFO: 172.21.0.1:34804 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:34816 - "POST /v1/chat/completions HTTP/1.1" 200 OK diff --git a/audit/tuning/steps-123/mtp1-b2048/status.json b/audit/tuning/steps-123/mtp1-b2048/status.json new file mode 100644 index 0000000..256e394 --- /dev/null +++ b/audit/tuning/steps-123/mtp1-b2048/status.json @@ -0,0 +1,5 @@ +{ + "startup": true, + "records": 44, + "failures": [] +} \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp2-b2048/compose.yaml b/audit/tuning/steps-123/mtp2-b2048/compose.yaml new file mode 100644 index 0000000..67f19dc --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/compose.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: "no" + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":2}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/steps-123/mtp2-b2048/mtp-metrics.txt b/audit/tuning/steps-123/mtp2-b2048/mtp-metrics.txt new file mode 100644 index 0000000..cbb71e9 --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/mtp-metrics.txt @@ -0,0 +1,26 @@ +# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_total counter +vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 52554.0 +# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_created gauge +vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528008e+09 +# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_total counter +vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 105108.0 +# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_created gauge +vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528184e+09 +# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_total counter +vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 69785.0 +# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_created gauge +vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897410011528296e+09 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 40257.0 +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 29528.0 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.7897410011528459e+09 +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.78974100115285e+09 \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl b/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl new file mode 100644 index 0000000..786364c --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl @@ -0,0 +1,7 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2711202999998932, "response": {"id": "chatcmpl-acf4252c63733754", "object": "chat.completion", "created": 1789743196, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.7527402949999669, "response": {"id": "chatcmpl-82388a5e0c331239", "object": "chat.completion", "created": 1789743198, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "schedule", "passed": true, "elapsed_s": 2.0548892629958573, "response": {"id": "chatcmpl-b59f8d39ded5a629", "object": "chat.completion", "created": 1789743199, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.3326087149980594, "response": {"id": "chatcmpl-8e1cd6b923cb571c", "object": "chat.completion", "created": 1789743201, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-848dfe0da82871ae", "object": "chat.completion", "created": 1789743203, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-acbd27655cbcc931", "object": "chat.completion", "created": 1789743210, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-aa08e51120ea5d43", "object": "chat.completion", "created": 1789743212, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-80911c02057f9d71", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9e1c6e32", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} diff --git a/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl.log b/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl.log new file mode 100644 index 0000000..2d6a3fb --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/quality-medium.jsonl.log @@ -0,0 +1,8 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.2711202999998932} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.7527402949999669} +{"name": "schedule", "passed": true, "elapsed_s": 2.0548892629958573} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.3326087149980594} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "tool-arguments", "passed": true} +QUALITY_COMPLETE diff --git a/audit/tuning/steps-123/mtp2-b2048/screen.jsonl b/audit/tuning/steps-123/mtp2-b2048/screen.jsonl new file mode 100644 index 0000000..2f2c1e8 --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/screen.jsonl @@ -0,0 +1,37 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08033495700510684, "elapsed_s": 0.5909661660043639, "decode_tps": 158.62720212253586, "max_stream_gap_s": 0.018763815001875628, "events": 29, "content": "\n\n3973", "reasoning": "I need to compute 137 * 29 + 0.\n\n137 * 29 = 137 * 30 - 137 = 4110 - 137 = 3973\n\nAdding 0 doesn't change anything, so the answer is 3973.\n", "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08595889700518455, "elapsed_s": 0.3043062380020274, "decode_tps": 164.87491826392582, "max_stream_gap_s": 0.01860671999747865, "events": 13, "content": "\n\n5761", "reasoning": "160 * 36 = 5760. 5760 + 1 = 5761.\n", "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08532555599958869, "elapsed_s": 0.5077504829969257, "decode_tps": 153.87349525519303, "max_stream_gap_s": 0.01870621599664446, "events": 24, "content": "\n\n7871", "reasoning": "183 * 43 = 183 * 40 + 183 * 3 = 7320 + 549 = 7869. Then 7869 + 2 = 7871.\n", "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08707080200110795, "elapsed_s": 0.32353630300349323, "decode_tps": 160.69997457944905, "max_stream_gap_s": 0.01859223899373319, "events": 14, "content": "\n\n10303", "reasoning": "206 * 50 = 10300\n10300 + 3 = 10303\n", "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08545495500584366, "elapsed_s": 0.5441159190013423, "decode_tps": 154.798436259984, "max_stream_gap_s": 0.018821514000592288, "events": 26, "content": "\n\n13057", "reasoning": "229 * 57 = 229 * 50 + 229 * 7 = 11450 + 1603 = 13053. Then 13053 + 4 = 13057.\n", "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08643778399709845, "elapsed_s": 0.34146275900275214, "decode_tps": 160.7685678592501, "max_stream_gap_s": 0.018947848002426326, "events": 15, "content": "\n\n16133", "reasoning": "252 * 64 = 16128. Then 16128 + 5 = 16133.\n", "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08607953199680196, "elapsed_s": 0.5258115449978504, "decode_tps": 159.18786426821532, "max_stream_gap_s": 0.018670357996597886, "events": 25, "content": "\n\n19531", "reasoning": "275 * 71 = 275 * 70 + 275 * 1 = 19250 + 275 = 19525. Then 19525 + 6 = 19531.\n", "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08593622500484344, "elapsed_s": 0.6902767670035246, "decode_tps": 153.8867468537349, "max_stream_gap_s": 0.018791513997712173, "events": 33, "content": "\n\n23251", "reasoning": "298 * 78 = 298 * 78. Let me compute this step by step.\n\n298 * 78 = 298 * 80 - 298 * 2 = 23840 - 596 = 23244\n\n23244 + 7 = 23251\n", "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08543285400082823, "elapsed_s": 0.6162005010046414, "decode_tps": 158.2613418021625, "max_stream_gap_s": 0.018708337993302848, "events": 29, "content": "\n\n27293", "reasoning": "Let me compute 321 * 85 + 8.\n\n321 * 85 = 321 * 80 + 321 * 5 = 25680 + 1605 = 27285\n\n27285 + 8 = 27293\n", "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08579609300068114, "elapsed_s": 0.6164905580008053, "decode_tps": 156.39884241110482, "max_stream_gap_s": 0.018804239996825345, "events": 29, "content": "\n\n31657", "reasoning": "Let me compute 344 * 92 + 9.\n\n344 * 92 = 344 * 90 + 344 * 2 = 30960 + 688 = 31648\n\n31648 + 9 = 31657\n", "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08569144899956882, "elapsed_s": 0.5975101969961543, "decode_tps": 160.21296664292387, "max_stream_gap_s": 0.018605278004542924, "events": 29, "content": "\n\n36343", "reasoning": "Let me compute 367 * 99 + 10.\n\n367 * 99 = 367 * (100 - 1) = 36700 - 367 = 36333\n\n36333 + 10 = 36343\n", "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08546528800070519, "elapsed_s": 0.6353967890026979, "decode_tps": 161.8383377526822, "max_stream_gap_s": 0.018794105999404564, "events": 31, "content": "\n\n41351", "reasoning": "Let me compute 390 * 106 + 11.\n\n390 * 106 = 390 * 100 + 390 * 6 = 39000 + 2340 = 41340\n\n41340 + 11 = 41351\n", "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08568278700113297, "elapsed_s": 0.37689909800246824, "decode_tps": 144.2226908773919, "max_stream_gap_s": 0.018781819999276195, "events": 17, "content": "\n\n{\"name\":\"设备A\",\"count\":7,\"enabled\":true}", "reasoning": "The user wants me to return exactly a specific JSON object with no markdown formatting. Let me provide exactly what they asked for.\n", "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08570619000238366, "elapsed_s": 0.5379072569994605, "decode_tps": 137.10715105499912, "max_stream_gap_s": 0.018587483995361254, "events": 26, "content": "\n\nAQUA-58321", "reasoning": "The user is asking me to recall the code they previously told me to remember. They said \"Remember the code AQUA-58321.\" and now they're asking what it was, with the instruction to reply only with the code.\n", "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3276, "completion_tokens": 141, "completion_tokens_details": {"reasoning_tokens": 102}}, "ttft_s": 0.37922176600113744, "elapsed_s": 1.3786603870030376, "decode_tps": 140.07863720500933, "max_stream_gap_s": 0.01991734900366282, "events": 54, "content": "\n\nVerification code: SPARK-82741\nSum of amounts: 1368 (125 + 478 + 765)", "reasoning": "The user wants the verification code and the sum of three amounts. From the image: code is SPARK-82741; amounts are 125, 478, 765. Sum: 125+478=603; 603+765=1368. I need to provide a clean, self-contained answer. Reasoning is in English matching the user. Final response: code and sum.\n", "sha256": "9e3c14a4e0973d378dc3fda85c529bfed85154aefd90d72c8313c78df80f6564"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12213, "completion_tokens": 144, "completion_tokens_details": {"reasoning_tokens": 130}}, "ttft_s": 0.8094746360002318, "elapsed_s": 1.7980530589993577, "decode_tps": 144.65215573507055, "max_stream_gap_s": 0.018804600993462373, "events": 55, "content": "\n\nRed, Green, Blue, Yellow, Purple, White", "reasoning": "The user wants the six background colors in chronological order, replying only with English color names. Looking at the frames: 0-19s red, 20-39s green, 40-59s blue, 60-79s yellow, 80-99s purple, 100-119s white. So the sequence is Red, Green, Blue, Yellow, Purple, White. I need to provide just the names, comma-separated or listed. The instruction says \"Reply only with the English color names.\" I'll give a simple comma-separated list.\n", "sha256": "40374d1d01e9d30a2e10dfcf6ba861c98b6c27c52bed2fc369e82286625e2f93"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12433, "completion_tokens": 77, "completion_tokens_details": {"reasoning_tokens": 66}}, "ttft_s": 1.4533180090002134, "elapsed_s": 1.9994508850068087, "decode_tps": 139.16027278145802, "max_stream_gap_s": 0.018682743997487705, "events": 31, "content": "\n\nSPARK-82741", "reasoning": "The user wants the verification code from the invoice. All four copies show the same line: \"Verification: SPARK-82741\". I need to extract just the code value, not the label. The code is SPARK-82741. Reply only with the code, no extra text.\n", "sha256": "6948cfaa73343b43893c8bbd5a7faf45ed78d55ac8ae50d67b8b890a8d1bf6c7"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08620800900098402, "elapsed_s": 2.189102151998668, "decode_tps": 121.26145333996531, "max_stream_gap_s": 0.018632254999829456, "events": 117, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling,", "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07210104499972658, "elapsed_s": 16.617303178005386, "decode_tps": 123.72166768011161, "max_stream_gap_s": 0.019507625998812728, "events": 920, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and maybe stop? If on_error fail? Could treat as error.\n REPLACE: if exists repo.update else insert. Similar to UPDATE? Maybe UPDATE only if exists, REPLACE same. Could define UPDATE = update existing, insert new; REPLACE = delete then insert? Not needed. Maybe strategies: SKIP, UPDATE, FAIL, ERROR", "sha256": "4b5a99c6a2d74362c14314126d243ab0ea75f0d203a637e5635b96bc429e3afb"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07450222100305837, "elapsed_s": 16.60604691500339, "decode_tps": 123.82387961258709, "max_stream_gap_s": 0.018462731000909116, "events": 922, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and stop? Or error row.\n REPLACE: if exists repo.update else insert; but within file? If seen, update? Could treat as update.\n- If no duplicate_keys: no duplicate handling.\n\nNeed repository transaction? Could have repository.begin/commit/rollback optional. Use contextlib? Maybe define Repository protocol with", "sha256": "c6a65a378e367e11250c85739caf6c8d1093bd987b88899767d97d623c15bcb9"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08693470500293188, "elapsed_s": 16.638716553999984, "decode_tps": 123.67248545654539, "max_stream_gap_s": 0.018704524001805112, "events": 923, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY, continue? or fail. We'll add error and count duplicate_rows. If on_error FAIL? duplicate is error.\n REPLACE: if exists repo.update else insert; also seen? For within file, if duplicate, update? Could treat as update.\n- If no duplicate_keys: no duplicate handling.\n\nNeed repository transaction", "sha256": "c6aab7182e88cf796f5b16f8b786142eb9521c9123cb0a049524dde6362b299f"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07400395700096851, "elapsed_s": 1.9887343450027402, "decode_tps": 133.17801900356326, "max_stream_gap_s": 0.0185166260052938, "events": 108, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08585922900238074, "elapsed_s": 15.984570784006792, "decode_tps": 128.75257173627188, "max_stream_gap_s": 0.01882019199547358, "events": 889, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止 |\n\n---\n\n## 4.2.2 需求项:审批发起\n\n### 功能说明", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07313994099968113, "elapsed_s": 15.988134749000892, "decode_tps": 128.62083994968555, "max_stream_gap_s": 0.018578499999421183, "events": 890, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止、转交 |\n\n---\n\n## 4.2.2 需求项:审批提交与路由", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "8fc19ac2a6425782e63b5c26dd326f7124bce54fc957913963c9fc6ae4ee3341"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08644359100435395, "elapsed_s": 15.966877792001469, "decode_tps": 128.90075762988087, "max_stream_gap_s": 0.018619577000208665, "events": 888, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止 |\n\n---\n\n## 4.2.2 需求项:审批发起\n\n### 功能说明", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "488769996df35d3bbedd2a91046c0a137677097523f73433b0919b61c68071b5"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.07434646700130543, "elapsed_s": 2.126096015002986, "decode_tps": 124.28417505848094, "max_stream_gap_s": 0.01840589300263673, "events": 116, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsett", "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07352274699951522, "elapsed_s": 16.837296222998702, "decode_tps": 122.10854572395078, "max_stream_gap_s": 0.018720656000368763, "events": 935, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled. Good.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax", "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07170502199733164, "elapsed_s": 16.85341221699491, "decode_tps": 121.97805480780796, "max_stream_gap_s": 0.01876352399995085, "events": 936, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax: d", "sha256": "8a7cb2cde52eee53562fd48aca68d5267dceb315e35b12b86822b67066eacf58"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07260531799693126, "elapsed_s": 16.818446777993813, "decode_tps": 122.23930370354654, "max_stream_gap_s": 0.01878616500471253, "events": 934, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled. Good.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax", "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126111, "completion_tokens": 116, "completion_tokens_details": {"reasoning_tokens": 96}}, "ttft_s": 14.42599186499865, "elapsed_s": 15.14422588799789, "decode_tps": 160.11494348287312, "max_stream_gap_s": 0.01879984400147805, "events": 41, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide the northern and southern access codes that are embedded in the data. Let me find them:\n\n- The northern access code is NORTH-49273 (found between Item 000019 and Item 000021)\n- The southern access code is SOUTH-18362 (found between Item 002499 and Item 002501)\n", "sha256": "e054eb4577e203d59776c06555ff8dff56d50e49389606fa8a7344884f985155"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7020562550023897, "elapsed_s": 1.4927438169979723, "decode_tps": 149.23720274818137, "max_stream_gap_s": 0.01888078500633128, "events": 45, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data they've provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (appears after Item 000019)\n- \"The southern access code is SOUTH-18362.\" (appears after Item 002499)\n\nThe user wants only these two codes.\n", "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7015097360053915, "elapsed_s": 1.4736580300013884, "decode_tps": 152.82038556263618, "max_stream_gap_s": 0.018826871004421264, "events": 44, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes that are embedded in the data.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (appears after Item 000019)\n- \"The southern access code is SOUTH-18362.\" (appears after Item 002499)\n\nThe user wants only these two codes.\n", "sha256": "1275326ebe3452473a77c76ba9ec5b91f06f8896ffd6de203abf1f9129842a90"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111902, "completion_tokens": 232, "completion_tokens_details": {"reasoning_tokens": 187}}, "ttft_s": 12.815331379999407, "elapsed_s": 14.372689276002347, "decode_tps": 148.32813998174495, "max_stream_gap_s": 0.01881139999750303, "events": 87, "content": "\n\n- **Northern access code:** NORTH-49273\n- **Southern access code:** SOUTH-18362\n- **Invoice verification code:** SPARK-82741", "reasoning": "The user wants three pieces of information: the northern access code, the southern access code, and the invoice verification code. These are embedded in the text and the image.\n\nScanning the text: \"The northern access code is NORTH-49273.\" appears after Item 000019. \"The southern access code is SOUTH-18362.\" appears after Item 002499.\n\nFrom the image: \"Verification: SPARK-82741\".\n\nSo the answer is straightforward extraction. No computation needed. I'll present all three clearly. The reasoning is in English matching the user's English request.\n\nI need to make sure I read the codes correctly: NORTH-49273, SOUTH-18362, SPARK-82741. Yes.\n\nI'll provide a concise complete response.\n", "sha256": "4b7c8fb1d0d992dda6a5a680074a3aa037b1c33fd732b7315720e712693c2335"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08533696200174745, "elapsed_s": 0.4814725750038633, "decode_tps": 159.0364459346489, "max_stream_gap_s": 0.018437628001265693, "events": 22, "content": "\n\n731", "reasoning": "The user wants me to compute 43 * 17 and reply with only the integer.\n\n43 * 17 = 43 * 10 + 43 * 7 = 430 + 301 = 731\n", "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5489882670008228, "elapsed_s": 0.8192126439971616, "decode_tps": 155.42639219617496, "max_stream_gap_s": 0.018371916994510684, "events": 15, "content": "\n\n748", "reasoning": "44 * 17 = 44 * 10 + 44 * 7 = 440 + 308 = 748\n", "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8859058559974073, "elapsed_s": 1.2823282210010802, "decode_tps": 158.92140696808642, "max_stream_gap_s": 0.018409249001706485, "events": 22, "content": "\n\n765", "reasoning": "The user wants me to compute 45 * 17 and reply with only the integer.\n\n45 * 17 = 45 * 10 + 45 * 7 = 450 + 315 = 765\n", "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3495687850008835, "elapsed_s": 1.743830876002903, "decode_tps": 159.79218250449873, "max_stream_gap_s": 0.018336048000492156, "events": 23, "content": "\n\n782", "reasoning": "The user wants me to compute 46 * 17 and reply with just the integer.\n\n46 * 17 = 46 * 10 + 46 * 7 = 460 + 322 = 782\n", "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} diff --git a/audit/tuning/steps-123/mtp2-b2048/screen.jsonl.log b/audit/tuning/steps-123/mtp2-b2048/screen.jsonl.log new file mode 100644 index 0000000..27ee8c5 --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/screen.jsonl.log @@ -0,0 +1,38 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08033495700510684, "elapsed_s": 0.5909661660043639, "decode_tps": 158.62720212253586, "max_stream_gap_s": 0.018763815001875628, "events": 29, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.08595889700518455, "elapsed_s": 0.3043062380020274, "decode_tps": 164.87491826392582, "max_stream_gap_s": 0.01860671999747865, "events": 13, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.08532555599958869, "elapsed_s": 0.5077504829969257, "decode_tps": 153.87349525519303, "max_stream_gap_s": 0.01870621599664446, "events": 24, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.08707080200110795, "elapsed_s": 0.32353630300349323, "decode_tps": 160.69997457944905, "max_stream_gap_s": 0.01859223899373319, "events": 14, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08545495500584366, "elapsed_s": 0.5441159190013423, "decode_tps": 154.798436259984, "max_stream_gap_s": 0.018821514000592288, "events": 26, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08643778399709845, "elapsed_s": 0.34146275900275214, "decode_tps": 160.7685678592501, "max_stream_gap_s": 0.018947848002426326, "events": 15, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08607953199680196, "elapsed_s": 0.5258115449978504, "decode_tps": 159.18786426821532, "max_stream_gap_s": 0.018670357996597886, "events": 25, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.08593622500484344, "elapsed_s": 0.6902767670035246, "decode_tps": 153.8867468537349, "max_stream_gap_s": 0.018791513997712173, "events": 33, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08543285400082823, "elapsed_s": 0.6162005010046414, "decode_tps": 158.2613418021625, "max_stream_gap_s": 0.018708337993302848, "events": 29, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.08579609300068114, "elapsed_s": 0.6164905580008053, "decode_tps": 156.39884241110482, "max_stream_gap_s": 0.018804239996825345, "events": 29, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08569144899956882, "elapsed_s": 0.5975101969961543, "decode_tps": 160.21296664292387, "max_stream_gap_s": 0.018605278004542924, "events": 29, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.08546528800070519, "elapsed_s": 0.6353967890026979, "decode_tps": 161.8383377526822, "max_stream_gap_s": 0.018794105999404564, "events": 31, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.08568278700113297, "elapsed_s": 0.37689909800246824, "decode_tps": 144.2226908773919, "max_stream_gap_s": 0.018781819999276195, "events": 17, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.08570619000238366, "elapsed_s": 0.5379072569994605, "decode_tps": 137.10715105499912, "max_stream_gap_s": 0.018587483995361254, "events": 26, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3276, "completion_tokens": 141, "completion_tokens_details": {"reasoning_tokens": 102}}, "ttft_s": 0.37922176600113744, "elapsed_s": 1.3786603870030376, "decode_tps": 140.07863720500933, "max_stream_gap_s": 0.01991734900366282, "events": 54, "sha256": "9e3c14a4e0973d378dc3fda85c529bfed85154aefd90d72c8313c78df80f6564"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12213, "completion_tokens": 144, "completion_tokens_details": {"reasoning_tokens": 130}}, "ttft_s": 0.8094746360002318, "elapsed_s": 1.7980530589993577, "decode_tps": 144.65215573507055, "max_stream_gap_s": 0.018804600993462373, "events": 55, "sha256": "40374d1d01e9d30a2e10dfcf6ba861c98b6c27c52bed2fc369e82286625e2f93"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12433, "completion_tokens": 77, "completion_tokens_details": {"reasoning_tokens": 66}}, "ttft_s": 1.4533180090002134, "elapsed_s": 1.9994508850068087, "decode_tps": 139.16027278145802, "max_stream_gap_s": 0.018682743997487705, "events": 31, "sha256": "6948cfaa73343b43893c8bbd5a7faf45ed78d55ac8ae50d67b8b890a8d1bf6c7"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08620800900098402, "elapsed_s": 2.189102151998668, "decode_tps": 121.26145333996531, "max_stream_gap_s": 0.018632254999829456, "events": 117, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07210104499972658, "elapsed_s": 16.617303178005386, "decode_tps": 123.72166768011161, "max_stream_gap_s": 0.019507625998812728, "events": 920, "sha256": "4b5a99c6a2d74362c14314126d243ab0ea75f0d203a637e5635b96bc429e3afb"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07450222100305837, "elapsed_s": 16.60604691500339, "decode_tps": 123.82387961258709, "max_stream_gap_s": 0.018462731000909116, "events": 922, "sha256": "c6a65a378e367e11250c85739caf6c8d1093bd987b88899767d97d623c15bcb9"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08693470500293188, "elapsed_s": 16.638716553999984, "decode_tps": 123.67248545654539, "max_stream_gap_s": 0.018704524001805112, "events": 923, "sha256": "c6aab7182e88cf796f5b16f8b786142eb9521c9123cb0a049524dde6362b299f"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07400395700096851, "elapsed_s": 1.9887343450027402, "decode_tps": 133.17801900356326, "max_stream_gap_s": 0.0185166260052938, "events": 108, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08585922900238074, "elapsed_s": 15.984570784006792, "decode_tps": 128.75257173627188, "max_stream_gap_s": 0.01882019199547358, "events": 889, "sha256": "d4c974412272e4560fbce42c1ad1cf265991e6545fd8518f4c0823f8c56bc7ea"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07313994099968113, "elapsed_s": 15.988134749000892, "decode_tps": 128.62083994968555, "max_stream_gap_s": 0.018578499999421183, "events": 890, "sha256": "8fc19ac2a6425782e63b5c26dd326f7124bce54fc957913963c9fc6ae4ee3341"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08644359100435395, "elapsed_s": 15.966877792001469, "decode_tps": 128.90075762988087, "max_stream_gap_s": 0.018619577000208665, "events": 888, "sha256": "488769996df35d3bbedd2a91046c0a137677097523f73433b0919b61c68071b5"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.07434646700130543, "elapsed_s": 2.126096015002986, "decode_tps": 124.28417505848094, "max_stream_gap_s": 0.01840589300263673, "events": 116, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07352274699951522, "elapsed_s": 16.837296222998702, "decode_tps": 122.10854572395078, "max_stream_gap_s": 0.018720656000368763, "events": 935, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07170502199733164, "elapsed_s": 16.85341221699491, "decode_tps": 121.97805480780796, "max_stream_gap_s": 0.01876352399995085, "events": 936, "sha256": "8a7cb2cde52eee53562fd48aca68d5267dceb315e35b12b86822b67066eacf58"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07260531799693126, "elapsed_s": 16.818446777993813, "decode_tps": 122.23930370354654, "max_stream_gap_s": 0.01878616500471253, "events": 934, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126111, "completion_tokens": 116, "completion_tokens_details": {"reasoning_tokens": 96}}, "ttft_s": 14.42599186499865, "elapsed_s": 15.14422588799789, "decode_tps": 160.11494348287312, "max_stream_gap_s": 0.01879984400147805, "events": 41, "sha256": "e054eb4577e203d59776c06555ff8dff56d50e49389606fa8a7344884f985155"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7020562550023897, "elapsed_s": 1.4927438169979723, "decode_tps": 149.23720274818137, "max_stream_gap_s": 0.01888078500633128, "events": 45, "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 0.7015097360053915, "elapsed_s": 1.4736580300013884, "decode_tps": 152.82038556263618, "max_stream_gap_s": 0.018826871004421264, "events": 44, "sha256": "1275326ebe3452473a77c76ba9ec5b91f06f8896ffd6de203abf1f9129842a90"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111902, "completion_tokens": 232, "completion_tokens_details": {"reasoning_tokens": 187}}, "ttft_s": 12.815331379999407, "elapsed_s": 14.372689276002347, "decode_tps": 148.32813998174495, "max_stream_gap_s": 0.01881139999750303, "events": 87, "sha256": "4b7c8fb1d0d992dda6a5a680074a3aa037b1c33fd732b7315720e712693c2335"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08533696200174745, "elapsed_s": 0.4814725750038633, "decode_tps": 159.0364459346489, "max_stream_gap_s": 0.018437628001265693, "events": 22, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.5489882670008228, "elapsed_s": 0.8192126439971616, "decode_tps": 155.42639219617496, "max_stream_gap_s": 0.018371916994510684, "events": 15, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.8859058559974073, "elapsed_s": 1.2823282210010802, "decode_tps": 158.92140696808642, "max_stream_gap_s": 0.018409249001706485, "events": 22, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.3495687850008835, "elapsed_s": 1.743830876002903, "decode_tps": 159.79218250449873, "max_stream_gap_s": 0.018336048000492156, "events": 23, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} +SUITE_COMPLETE diff --git a/audit/tuning/steps-123/mtp2-b2048/server.log b/audit/tuning/steps-123/mtp2-b2048/server.log new file mode 100644 index 0000000..e01412a --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/server.log @@ -0,0 +1,730 @@ +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] █ █ █▄ ▄█ +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] █▄█▀ █ █ █ █ model /model +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 14:14:09 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 2}, 'compilation_config': {'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:14:09 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration +(APIServer pid=1) INFO 09-18 14:14:09 [model.py:2024] Using max model len 131072 +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:14:13 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 14:14:13 [model.py:691] Resolved architecture: Qwen4ExpMTP +(APIServer pid=1) INFO 09-18 14:14:13 [model.py:2024] Using max model len 262144 +(APIServer pid=1) WARNING 09-18 14:14:13 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate +(APIServer pid=1) INFO 09-18 14:14:13 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072 +(APIServer pid=1) INFO 09-18 14:14:13 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled +(APIServer pid=1) INFO 09-18 14:14:13 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False) +(APIServer pid=1) INFO 09-18 14:14:13 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out. +(APIServer pid=1) INFO 09-18 14:14:13 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']) +(APIServer pid=1) WARNING 09-18 14:14:13 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens. +(APIServer pid=1) INFO 09-18 14:14:13 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=113) INFO 09-18 14:14:25 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=2), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 3, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None) +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=113) INFO 09-18 14:14:26 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_ed11f4f25e124a16965c184de0adfebf backend=nccl +(EngineCore pid=113) INFO 09-18 14:14:26 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=113) INFO 09-18 14:14:26 [gpu_worker.py:441] Using V2 Model Runner +(EngineCore pid=113) INFO 09-18 14:14:27 [model_runner.py:387] Loading model from scratch... +(EngineCore pid=113) INFO 09-18 14:14:27 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention +(EngineCore pid=113) INFO 09-18 14:14:27 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention. +(EngineCore pid=113) INFO 09-18 14:14:27 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128). +(EngineCore pid=113) INFO 09-18 14:14:27 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda +(EngineCore pid=113) INFO 09-18 14:14:29 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION']. +(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. +(APIServer pid=1) INFO 09-18 14:14:31 [base.py:261] Multi-modal warmup completed in 12.320s +(APIServer pid=1) INFO 09-18 14:14:33 [base.py:261] Readonly multi-modal warmup completed in 1.302s +(EngineCore pid=113) INFO 09-18 14:15:03 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True +(EngineCore pid=113) INFO 09-18 14:15:03 [flash_attn.py:1115] Using FlashAttention version 2 +(EngineCore pid=113) WARNING 09-18 14:15:03 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect +(EngineCore pid=113) INFO 09-18 14:15:03 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.31 GiB. +(EngineCore pid=113) INFO 09-18 14:15:03 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch. +(EngineCore pid=113) +Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00= mamba page size. +(EngineCore pid=113) INFO 09-18 14:15:53 [interface.py:942] Padding mamba page size by 0.38% to ensure that mamba page size and attention page size are exactly equal. +(EngineCore pid=113) INFO 09-18 14:15:53 [utils.py:320] Using BLNHC KV cache layout. +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} +(EngineCore pid=113) INFO 09-18 14:15:58 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size. +(EngineCore pid=113) WARNING 09-18 14:16:20 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's. +(EngineCore pid=113) WARNING 09-18 14:16:20 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY +(EngineCore pid=113) INFO 09-18 14:16:20 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps. +(EngineCore pid=113) +Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00 262144). Running this sequence through the model will result in indexing errors +(APIServer pid=1) INFO: 172.21.0.1:48854 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48862 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48874 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48886 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48894 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48902 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48914 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:21:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 140.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 14:21:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 91.39 tokens/s, Drafted throughput: 98.39 tokens/s, Accepted: 914 tokens, Drafted: 984 tokens, Per-position acceptance rate: 0.929, 0.929, Avg Draft acceptance rate: 92.9% +(APIServer pid=1) INFO: 172.21.0.1:48926 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48928 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48934 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48948 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48950 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48964 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48968 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48976 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:21:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 67.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO: 127.0.0.1:49396 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38790 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38796 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38802 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:21:52 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.9 tokens/s, Avg generation throughput: 37.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.3%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 14:21:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.80, Accepted throughput: 11.95 tokens/s, Drafted throughput: 13.30 tokens/s, Accepted: 239 tokens, Drafted: 266 tokens, Per-position acceptance rate: 0.940, 0.857, Avg Draft acceptance rate: 89.8% +(APIServer pid=1) INFO 09-18 14:22:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 73.8%, Prefix cache hit rate: 46.6%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO: 172.21.0.1:43834 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43840 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43848 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43858 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43874 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43878 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:43890 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:22:12 [loggers.py:323] Engine 000: Avg prompt throughput: 13594.6 tokens/s, Avg generation throughput: 70.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8% +(APIServer pid=1) INFO 09-18 14:22:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.77, Accepted throughput: 22.55 tokens/s, Drafted throughput: 25.50 tokens/s, Accepted: 451 tokens, Drafted: 510 tokens, Per-position acceptance rate: 0.929, 0.839, Avg Draft acceptance rate: 88.4% +(APIServer pid=1) INFO: 172.21.0.1:51984 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:58388 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:22:22 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.2 tokens/s, Avg generation throughput: 113.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8% +(APIServer pid=1) INFO 09-18 14:22:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 63.79 tokens/s, Drafted throughput: 99.19 tokens/s, Accepted: 638 tokens, Drafted: 992 tokens, Per-position acceptance rate: 0.754, 0.532, Avg Draft acceptance rate: 64.3% +(APIServer pid=1) INFO: 172.21.0.1:52418 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:22:32 [loggers.py:323] Engine 000: Avg prompt throughput: 8.6 tokens/s, Avg generation throughput: 139.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8% +(APIServer pid=1) INFO 09-18 14:22:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.53, Accepted throughput: 84.50 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 845 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.848, 0.680, Avg Draft acceptance rate: 76.4% +(APIServer pid=1) INFO 09-18 14:22:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 134.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 52.5%, MM cache hit rate: 77.8% +(APIServer pid=1) INFO 09-18 14:22:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 78.30 tokens/s, Drafted throughput: 112.00 tokens/s, Accepted: 783 tokens, Drafted: 1120 tokens, Per-position acceptance rate: 0.811, 0.588, Avg Draft acceptance rate: 69.9% +(APIServer pid=1) INFO: 127.0.0.1:56790 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53910 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:22:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 33.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 58.5%, Prefix cache hit rate: 44.1%, MM cache hit rate: 77.8% +(APIServer pid=1) INFO 09-18 14:22:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.42, Accepted throughput: 19.70 tokens/s, Drafted throughput: 27.80 tokens/s, Accepted: 197 tokens, Drafted: 278 tokens, Per-position acceptance rate: 0.799, 0.619, Avg Draft acceptance rate: 70.9% +(APIServer pid=1) INFO: 172.21.0.1:47694 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47700 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:23:02 [loggers.py:323] Engine 000: Avg prompt throughput: 12912.6 tokens/s, Avg generation throughput: 24.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 30.8%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.74, Accepted throughput: 15.50 tokens/s, Drafted throughput: 17.80 tokens/s, Accepted: 155 tokens, Drafted: 178 tokens, Per-position acceptance rate: 0.921, 0.820, Avg Draft acceptance rate: 87.1% +(APIServer pid=1) INFO: 172.21.0.1:47708 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:23:12 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.4 tokens/s, Avg generation throughput: 118.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.24, Accepted throughput: 65.70 tokens/s, Drafted throughput: 105.99 tokens/s, Accepted: 657 tokens, Drafted: 1060 tokens, Per-position acceptance rate: 0.743, 0.496, Avg Draft acceptance rate: 62.0% +(APIServer pid=1) INFO: 127.0.0.1:53052 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59822 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:23:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.4 tokens/s, Avg generation throughput: 120.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 65.40 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 654 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.716, 0.469, Avg Draft acceptance rate: 59.2% +(APIServer pid=1) INFO 09-18 14:23:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 43.3%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 66.00 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 660 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.704, 0.478, Avg Draft acceptance rate: 59.1% +(APIServer pid=1) INFO: 172.21.0.1:34388 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:23:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 61.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.2%, Prefix cache hit rate: 37.4%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.06, Accepted throughput: 31.80 tokens/s, Drafted throughput: 59.99 tokens/s, Accepted: 318 tokens, Drafted: 600 tokens, Per-position acceptance rate: 0.673, 0.387, Avg Draft acceptance rate: 53.0% +(APIServer pid=1) INFO: 127.0.0.1:50974 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:23:52 [loggers.py:323] Engine 000: Avg prompt throughput: 12598.6 tokens/s, Avg generation throughput: 2.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.5%, Prefix cache hit rate: 37.4%, MM cache hit rate: 81.8% +(APIServer pid=1) INFO 09-18 14:23:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.80, Accepted throughput: 1.80 tokens/s, Drafted throughput: 2.00 tokens/s, Accepted: 18 tokens, Drafted: 20 tokens, Per-position acceptance rate: 0.900, 0.900, Avg Draft acceptance rate: 90.0% +(APIServer pid=1) INFO: 172.21.0.1:50664 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38410 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38412 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:24:02 [loggers.py:323] Engine 000: Avg prompt throughput: 1529.0 tokens/s, Avg generation throughput: 105.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6% +(APIServer pid=1) INFO 09-18 14:24:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 60.20 tokens/s, Drafted throughput: 89.20 tokens/s, Accepted: 602 tokens, Drafted: 892 tokens, Per-position acceptance rate: 0.791, 0.558, Avg Draft acceptance rate: 67.5% +(APIServer pid=1) INFO 09-18 14:24:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6% +(APIServer pid=1) INFO 09-18 14:24:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.13, Accepted throughput: 62.90 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 629 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.689, 0.440, Avg Draft acceptance rate: 56.5% +(APIServer pid=1) INFO: 172.21.0.1:38508 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:37090 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:24:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 123.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 36.8%, MM cache hit rate: 84.6% +(APIServer pid=1) INFO 09-18 14:24:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 67.80 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 678 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.732, 0.494, Avg Draft acceptance rate: 61.3% +(APIServer pid=1) INFO: 172.21.0.1:55060 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:24:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 100.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 36.9%, Prefix cache hit rate: 32.5%, MM cache hit rate: 84.6% +(APIServer pid=1) INFO 09-18 14:24:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 57.60 tokens/s, Drafted throughput: 86.60 tokens/s, Accepted: 576 tokens, Drafted: 866 tokens, Per-position acceptance rate: 0.762, 0.568, Avg Draft acceptance rate: 66.5% +(APIServer pid=1) INFO 09-18 14:24:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 78.5%, Prefix cache hit rate: 32.5%, MM cache hit rate: 84.6% +(APIServer pid=1) INFO: 127.0.0.1:41864 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:44650 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:44654 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:44658 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:24:52 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.6 tokens/s, Avg generation throughput: 77.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7% +(APIServer pid=1) INFO 09-18 14:24:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.37, Accepted throughput: 22.20 tokens/s, Drafted throughput: 32.40 tokens/s, Accepted: 444 tokens, Drafted: 648 tokens, Per-position acceptance rate: 0.765, 0.605, Avg Draft acceptance rate: 68.5% +(APIServer pid=1) INFO 09-18 14:25:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7% +(APIServer pid=1) INFO 09-18 14:25:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.33, Accepted throughput: 74.49 tokens/s, Drafted throughput: 111.79 tokens/s, Accepted: 745 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.767, 0.565, Avg Draft acceptance rate: 66.6% +(APIServer pid=1) INFO: 172.21.0.1:52324 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:25:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 124.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 32.0%, MM cache hit rate: 86.7% +(APIServer pid=1) INFO 09-18 14:25:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 69.30 tokens/s, Drafted throughput: 110.99 tokens/s, Accepted: 693 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.737, 0.512, Avg Draft acceptance rate: 62.4% +(APIServer pid=1) INFO: 127.0.0.1:54114 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47722 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:25:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 112.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.3%, Prefix cache hit rate: 28.7%, MM cache hit rate: 86.7% +(APIServer pid=1) INFO 09-18 14:25:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 63.90 tokens/s, Drafted throughput: 96.79 tokens/s, Accepted: 639 tokens, Drafted: 968 tokens, Per-position acceptance rate: 0.756, 0.564, Avg Draft acceptance rate: 66.0% +(APIServer pid=1) INFO 09-18 14:25:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 75.4%, Prefix cache hit rate: 28.7%, MM cache hit rate: 86.7% +(APIServer pid=1) INFO: 172.21.0.1:56320 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56330 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56334 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:25:42 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.9 tokens/s, Avg generation throughput: 67.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2% +(APIServer pid=1) INFO 09-18 14:25:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.45, Accepted throughput: 19.80 tokens/s, Drafted throughput: 27.30 tokens/s, Accepted: 396 tokens, Drafted: 546 tokens, Per-position acceptance rate: 0.806, 0.645, Avg Draft acceptance rate: 72.5% +(APIServer pid=1) INFO: 127.0.0.1:34304 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:25:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 126.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2% +(APIServer pid=1) INFO 09-18 14:25:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 70.90 tokens/s, Drafted throughput: 111.80 tokens/s, Accepted: 709 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.757, 0.512, Avg Draft acceptance rate: 63.4% +(APIServer pid=1) INFO: 172.21.0.1:33198 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:26:02 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 117.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2% +(APIServer pid=1) INFO 09-18 14:26:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.12, Accepted throughput: 62.09 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 621 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.681, 0.444, Avg Draft acceptance rate: 56.2% +(APIServer pid=1) INFO 09-18 14:26:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 28.3%, MM cache hit rate: 88.2% +(APIServer pid=1) INFO 09-18 14:26:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.34, Accepted throughput: 74.39 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 744 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.772, 0.564, Avg Draft acceptance rate: 66.8% +(APIServer pid=1) INFO: 172.21.0.1:52376 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:51278 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:26:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 69.2%, Prefix cache hit rate: 25.7%, MM cache hit rate: 88.2% +(APIServer pid=1) INFO 09-18 14:26:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 1.80, Accepted throughput: 0.40 tokens/s, Drafted throughput: 1.00 tokens/s, Accepted: 4 tokens, Drafted: 10 tokens, Per-position acceptance rate: 0.600, 0.200, Avg Draft acceptance rate: 40.0% +(APIServer pid=1) INFO: 172.21.0.1:56900 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56902 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56906 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:26:32 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.0 tokens/s, Avg generation throughput: 53.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5% +(APIServer pid=1) INFO 09-18 14:26:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.74, Accepted throughput: 33.90 tokens/s, Drafted throughput: 39.00 tokens/s, Accepted: 339 tokens, Drafted: 390 tokens, Per-position acceptance rate: 0.923, 0.815, Avg Draft acceptance rate: 86.9% +(APIServer pid=1) INFO 09-18 14:26:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 112.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5% +(APIServer pid=1) INFO 09-18 14:26:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.01, Accepted throughput: 56.60 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 566 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.658, 0.357, Avg Draft acceptance rate: 50.7% +(APIServer pid=1) INFO: 127.0.0.1:50340 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37702 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:26:52 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 117.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5% +(APIServer pid=1) INFO 09-18 14:26:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.12, Accepted throughput: 62.10 tokens/s, Drafted throughput: 110.80 tokens/s, Accepted: 621 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.704, 0.417, Avg Draft acceptance rate: 56.0% +(APIServer pid=1) INFO 09-18 14:27:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 25.4%, MM cache hit rate: 89.5% +(APIServer pid=1) INFO 09-18 14:27:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.17, Accepted throughput: 65.20 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 652 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.707, 0.463, Avg Draft acceptance rate: 58.5% +(APIServer pid=1) INFO: 172.21.0.1:41528 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:27:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 45.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 53.8%, Prefix cache hit rate: 23.2%, MM cache hit rate: 89.5% +(APIServer pid=1) INFO 09-18 14:27:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 25.40 tokens/s, Drafted throughput: 40.80 tokens/s, Accepted: 254 tokens, Drafted: 408 tokens, Per-position acceptance rate: 0.745, 0.500, Avg Draft acceptance rate: 62.3% +(APIServer pid=1) INFO: 127.0.0.1:34370 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:60904 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:27:22 [loggers.py:323] Engine 000: Avg prompt throughput: 12913.2 tokens/s, Avg generation throughput: 22.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 23.2%, MM cache hit rate: 90.0% +(APIServer pid=1) INFO 09-18 14:27:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.66, Accepted throughput: 14.10 tokens/s, Drafted throughput: 17.00 tokens/s, Accepted: 141 tokens, Drafted: 170 tokens, Per-position acceptance rate: 0.894, 0.765, Avg Draft acceptance rate: 82.9% +(APIServer pid=1) INFO: 172.21.0.1:60920 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:33700 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:27:32 [loggers.py:323] Engine 000: Avg prompt throughput: 1215.4 tokens/s, Avg generation throughput: 98.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5% +(APIServer pid=1) INFO 09-18 14:27:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.09, Accepted throughput: 51.20 tokens/s, Drafted throughput: 93.60 tokens/s, Accepted: 512 tokens, Drafted: 936 tokens, Per-position acceptance rate: 0.660, 0.434, Avg Draft acceptance rate: 54.7% +(APIServer pid=1) INFO: 172.21.0.1:44936 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:27:42 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 134.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5% +(APIServer pid=1) INFO 09-18 14:27:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.43, Accepted throughput: 79.30 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 793 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.817, 0.617, Avg Draft acceptance rate: 71.7% +(APIServer pid=1) INFO: 127.0.0.1:45276 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:27:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 23.0%, MM cache hit rate: 90.5% +(APIServer pid=1) INFO 09-18 14:27:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.11, Accepted throughput: 62.10 tokens/s, Drafted throughput: 111.80 tokens/s, Accepted: 621 tokens, Drafted: 1118 tokens, Per-position acceptance rate: 0.699, 0.411, Avg Draft acceptance rate: 55.5% +(APIServer pid=1) INFO: 172.21.0.1:34630 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:28:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 76.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 43.1%, Prefix cache hit rate: 21.2%, MM cache hit rate: 90.5% +(APIServer pid=1) INFO 09-18 14:28:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 41.10 tokens/s, Drafted throughput: 70.60 tokens/s, Accepted: 411 tokens, Drafted: 706 tokens, Per-position acceptance rate: 0.711, 0.453, Avg Draft acceptance rate: 58.2% +(APIServer pid=1) INFO 09-18 14:28:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 84.6%, Prefix cache hit rate: 21.2%, MM cache hit rate: 90.5% +(APIServer pid=1) INFO: 172.21.0.1:47082 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37824 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:49298 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37832 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:28:22 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.0 tokens/s, Avg generation throughput: 98.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3% +(APIServer pid=1) INFO 09-18 14:28:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.41, Accepted throughput: 28.70 tokens/s, Drafted throughput: 40.60 tokens/s, Accepted: 574 tokens, Drafted: 812 tokens, Per-position acceptance rate: 0.793, 0.621, Avg Draft acceptance rate: 70.7% +(APIServer pid=1) INFO 09-18 14:28:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 124.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3% +(APIServer pid=1) INFO 09-18 14:28:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.24, Accepted throughput: 69.10 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 691 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.745, 0.498, Avg Draft acceptance rate: 62.1% +(APIServer pid=1) INFO: 172.21.0.1:53742 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:28:42 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 111.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.0%, MM cache hit rate: 91.3% +(APIServer pid=1) INFO 09-18 14:28:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.02, Accepted throughput: 56.50 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 565 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.650, 0.373, Avg Draft acceptance rate: 51.2% +(APIServer pid=1) INFO: 127.0.0.1:49254 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:44080 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:28:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 111.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 19.5%, MM cache hit rate: 91.3% +(APIServer pid=1) INFO 09-18 14:28:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 64.89 tokens/s, Drafted throughput: 92.59 tokens/s, Accepted: 649 tokens, Drafted: 926 tokens, Per-position acceptance rate: 0.803, 0.598, Avg Draft acceptance rate: 70.1% +(APIServer pid=1) INFO 09-18 14:29:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 76.9%, Prefix cache hit rate: 19.5%, MM cache hit rate: 91.3% +(APIServer pid=1) INFO: 172.21.0.1:53800 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53804 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53806 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:29:12 [loggers.py:323] Engine 000: Avg prompt throughput: 14128.1 tokens/s, Avg generation throughput: 75.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0% +(APIServer pid=1) INFO 09-18 14:29:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.57, Accepted throughput: 23.05 tokens/s, Drafted throughput: 29.40 tokens/s, Accepted: 461 tokens, Drafted: 588 tokens, Per-position acceptance rate: 0.864, 0.704, Avg Draft acceptance rate: 78.4% +(APIServer pid=1) INFO: 127.0.0.1:51408 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:29:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 118.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0% +(APIServer pid=1) INFO 09-18 14:29:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.13, Accepted throughput: 62.89 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 629 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.703, 0.425, Avg Draft acceptance rate: 56.4% +(APIServer pid=1) INFO: 172.21.0.1:60596 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:29:32 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 124.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0% +(APIServer pid=1) INFO 09-18 14:29:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.25, Accepted throughput: 68.89 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 689 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.732, 0.514, Avg Draft acceptance rate: 62.3% +(APIServer pid=1) INFO 09-18 14:29:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 119.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 19.4%, MM cache hit rate: 92.0% +(APIServer pid=1) INFO 09-18 14:29:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.20 tokens/s, Drafted throughput: 111.20 tokens/s, Accepted: 642 tokens, Drafted: 1112 tokens, Per-position acceptance rate: 0.716, 0.439, Avg Draft acceptance rate: 57.7% +(APIServer pid=1) INFO: 172.21.0.1:60372 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:41996 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:29:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 11.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 66.2%, Prefix cache hit rate: 18.1%, MM cache hit rate: 92.0% +(APIServer pid=1) INFO 09-18 14:29:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 6.40 tokens/s, Drafted throughput: 10.40 tokens/s, Accepted: 64 tokens, Drafted: 104 tokens, Per-position acceptance rate: 0.769, 0.462, Avg Draft acceptance rate: 61.5% +(APIServer pid=1) INFO: 172.21.0.1:59394 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59402 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59410 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:30:02 [loggers.py:323] Engine 000: Avg prompt throughput: 14127.4 tokens/s, Avg generation throughput: 41.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6% +(APIServer pid=1) INFO 09-18 14:30:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.76, Accepted throughput: 26.00 tokens/s, Drafted throughput: 29.60 tokens/s, Accepted: 260 tokens, Drafted: 296 tokens, Per-position acceptance rate: 0.912, 0.845, Avg Draft acceptance rate: 87.8% +(APIServer pid=1) INFO 09-18 14:30:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 117.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6% +(APIServer pid=1) INFO 09-18 14:30:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.10, Accepted throughput: 61.50 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 615 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.666, 0.438, Avg Draft acceptance rate: 55.2% +(APIServer pid=1) INFO: 127.0.0.1:43480 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48952 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:30:22 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 119.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6% +(APIServer pid=1) INFO 09-18 14:30:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 63.90 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 639 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.699, 0.458, Avg Draft acceptance rate: 57.9% +(APIServer pid=1) INFO 09-18 14:30:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%, MM cache hit rate: 92.6% +(APIServer pid=1) INFO 09-18 14:30:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 65.00 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 650 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.706, 0.459, Avg Draft acceptance rate: 58.2% +(APIServer pid=1) INFO: 172.21.0.1:50314 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:30:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 52.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 50.8%, Prefix cache hit rate: 16.9%, MM cache hit rate: 92.6% +(APIServer pid=1) INFO 09-18 14:30:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 28.10 tokens/s, Drafted throughput: 48.40 tokens/s, Accepted: 281 tokens, Drafted: 484 tokens, Per-position acceptance rate: 0.665, 0.496, Avg Draft acceptance rate: 58.1% +(APIServer pid=1) INFO: 127.0.0.1:50034 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:58630 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:30:52 [loggers.py:323] Engine 000: Avg prompt throughput: 12598.1 tokens/s, Avg generation throughput: 13.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.8%, MM cache hit rate: 92.9% +(APIServer pid=1) INFO 09-18 14:30:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.71, Accepted throughput: 8.40 tokens/s, Drafted throughput: 9.80 tokens/s, Accepted: 84 tokens, Drafted: 98 tokens, Per-position acceptance rate: 0.878, 0.837, Avg Draft acceptance rate: 85.7% +(APIServer pid=1) INFO: 172.21.0.1:58644 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:38342 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:31:02 [loggers.py:323] Engine 000: Avg prompt throughput: 1529.0 tokens/s, Avg generation throughput: 110.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1% +(APIServer pid=1) INFO 09-18 14:31:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 63.19 tokens/s, Drafted throughput: 93.59 tokens/s, Accepted: 632 tokens, Drafted: 936 tokens, Per-position acceptance rate: 0.778, 0.573, Avg Draft acceptance rate: 67.5% +(APIServer pid=1) INFO: 172.21.0.1:49568 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:31:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.5 tokens/s, Avg generation throughput: 126.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1% +(APIServer pid=1) INFO 09-18 14:31:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 71.59 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 716 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.766, 0.531, Avg Draft acceptance rate: 64.9% +(APIServer pid=1) INFO: 127.0.0.1:53428 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:31:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 16.7%, MM cache hit rate: 93.1% +(APIServer pid=1) INFO 09-18 14:31:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.16, Accepted throughput: 65.00 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 650 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.703, 0.462, Avg Draft acceptance rate: 58.2% +(APIServer pid=1) INFO: 172.21.0.1:53214 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:31:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 78.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 40.0%, Prefix cache hit rate: 15.8%, MM cache hit rate: 93.1% +(APIServer pid=1) INFO 09-18 14:31:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.02, Accepted throughput: 39.70 tokens/s, Drafted throughput: 77.60 tokens/s, Accepted: 397 tokens, Drafted: 776 tokens, Per-position acceptance rate: 0.631, 0.392, Avg Draft acceptance rate: 51.2% +(APIServer pid=1) INFO 09-18 14:31:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.5%, Prefix cache hit rate: 15.8%, MM cache hit rate: 93.1% +(APIServer pid=1) INFO: 172.21.0.1:59288 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:56286 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59294 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59298 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:31:52 [loggers.py:323] Engine 000: Avg prompt throughput: 14129.1 tokens/s, Avg generation throughput: 88.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:31:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.40, Accepted throughput: 25.75 tokens/s, Drafted throughput: 36.80 tokens/s, Accepted: 515 tokens, Drafted: 736 tokens, Per-position acceptance rate: 0.788, 0.611, Avg Draft acceptance rate: 70.0% +(APIServer pid=1) INFO 09-18 14:32:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.30 tokens/s, Drafted throughput: 111.40 tokens/s, Accepted: 643 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.713, 0.442, Avg Draft acceptance rate: 57.7% +(APIServer pid=1) INFO: 172.21.0.1:40376 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:32:12 [loggers.py:323] Engine 000: Avg prompt throughput: 8.7 tokens/s, Avg generation throughput: 129.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 74.70 tokens/s, Drafted throughput: 110.40 tokens/s, Accepted: 747 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.775, 0.578, Avg Draft acceptance rate: 67.7% +(APIServer pid=1) INFO: 127.0.0.1:51024 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:54062 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:32:22 [loggers.py:323] Engine 000: Avg prompt throughput: 25.2 tokens/s, Avg generation throughput: 127.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.35, Accepted throughput: 73.19 tokens/s, Drafted throughput: 108.79 tokens/s, Accepted: 732 tokens, Drafted: 1088 tokens, Per-position acceptance rate: 0.774, 0.572, Avg Draft acceptance rate: 67.3% +(APIServer pid=1) INFO: 172.21.0.1:54076 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:54078 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53562 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:32:32 [loggers.py:323] Engine 000: Avg prompt throughput: 38.1 tokens/s, Avg generation throughput: 133.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.48, Accepted throughput: 79.80 tokens/s, Drafted throughput: 107.80 tokens/s, Accepted: 798 tokens, Drafted: 1078 tokens, Per-position acceptance rate: 0.831, 0.649, Avg Draft acceptance rate: 74.0% +(APIServer pid=1) INFO 09-18 14:32:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.10 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 731 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.750, 0.571, Avg Draft acceptance rate: 66.1% +(APIServer pid=1) INFO: 127.0.0.1:36210 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53576 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:32:52 [loggers.py:323] Engine 000: Avg prompt throughput: 13.4 tokens/s, Avg generation throughput: 131.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:32:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.42, Accepted throughput: 77.20 tokens/s, Drafted throughput: 109.00 tokens/s, Accepted: 772 tokens, Drafted: 1090 tokens, Per-position acceptance rate: 0.796, 0.620, Avg Draft acceptance rate: 70.8% +(APIServer pid=1) INFO: 172.21.0.1:38118 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47594 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47610 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47622 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:33:02 [loggers.py:323] Engine 000: Avg prompt throughput: 59.9 tokens/s, Avg generation throughput: 137.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:33:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.59, Accepted throughput: 84.40 tokens/s, Drafted throughput: 106.20 tokens/s, Accepted: 844 tokens, Drafted: 1062 tokens, Per-position acceptance rate: 0.868, 0.721, Avg Draft acceptance rate: 79.5% +(APIServer pid=1) INFO: 172.21.0.1:47632 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47648 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:45364 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:33:12 [loggers.py:323] Engine 000: Avg prompt throughput: 27.9 tokens/s, Avg generation throughput: 142.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:33:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.64, Accepted throughput: 88.50 tokens/s, Drafted throughput: 108.00 tokens/s, Accepted: 885 tokens, Drafted: 1080 tokens, Per-position acceptance rate: 0.891, 0.748, Avg Draft acceptance rate: 81.9% +(APIServer pid=1) INFO: 172.21.0.1:45372 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:45386 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:47374 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:33:22 [loggers.py:323] Engine 000: Avg prompt throughput: 32.5 tokens/s, Avg generation throughput: 28.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO 09-18 14:33:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.86, Accepted throughput: 18.40 tokens/s, Drafted throughput: 19.80 tokens/s, Accepted: 184 tokens, Drafted: 198 tokens, Per-position acceptance rate: 0.970, 0.889, Avg Draft acceptance rate: 92.9% +(APIServer pid=1) INFO 09-18 14:33:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.5% +(APIServer pid=1) INFO: 127.0.0.1:51446 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:47488 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:50100 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:54884 - "GET /metrics HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:49228 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:45014 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:45300 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:46998 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:51132 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.16.100.3:62103 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.16.100.3:62104 - "GET /v1/models HTTP/1.1" 401 Unauthorized +(APIServer pid=1) INFO: 127.0.0.1:58480 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:46524 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:53330 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:55746 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:56396 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:55482 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:56702 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:36636 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:34060 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:44360 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:52262 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:60346 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:32864 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:47770 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:59988 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:47698 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:44138 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:52264 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:52764 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:54950 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:37836 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:50426 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:42744 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:36986 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:46082 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:33154 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:33156 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35076 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35092 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35102 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35110 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35124 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35134 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35136 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35150 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35160 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35174 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35188 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35192 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35200 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:35204 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:02 [loggers.py:323] Engine 000: Avg prompt throughput: 399.3 tokens/s, Avg generation throughput: 105.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 15.6%, MM cache hit rate: 93.8% +(APIServer pid=1) INFO 09-18 14:50:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.84, Accepted throughput: 0.68 tokens/s, Drafted throughput: 0.74 tokens/s, Accepted: 679 tokens, Drafted: 740 tokens, Per-position acceptance rate: 0.957, 0.878, Avg Draft acceptance rate: 91.8% +(APIServer pid=1) INFO: 172.21.0.1:35214 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56506 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56510 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:56514 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:12 [loggers.py:323] Engine 000: Avg prompt throughput: 1821.0 tokens/s, Avg generation throughput: 96.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:50:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 55.10 tokens/s, Drafted throughput: 83.40 tokens/s, Accepted: 551 tokens, Drafted: 834 tokens, Per-position acceptance rate: 0.763, 0.559, Avg Draft acceptance rate: 66.1% +(APIServer pid=1) INFO: 127.0.0.1:55476 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:50:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 72.00 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 720 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.758, 0.532, Avg Draft acceptance rate: 64.5% +(APIServer pid=1) INFO: 172.21.0.1:37334 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:32 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 122.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:50:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.21, Accepted throughput: 66.99 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 670 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.725, 0.489, Avg Draft acceptance rate: 60.7% +(APIServer pid=1) INFO: 172.21.0.1:37138 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 121.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:50:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 66.00 tokens/s, Drafted throughput: 110.39 tokens/s, Accepted: 660 tokens, Drafted: 1104 tokens, Per-position acceptance rate: 0.728, 0.467, Avg Draft acceptance rate: 59.8% +(APIServer pid=1) INFO: 127.0.0.1:53092 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:50:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 126.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:50:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 70.60 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 706 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.742, 0.523, Avg Draft acceptance rate: 63.3% +(APIServer pid=1) INFO: 172.21.0.1:39218 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39220 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:51:02 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 121.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.22, Accepted throughput: 66.50 tokens/s, Drafted throughput: 109.40 tokens/s, Accepted: 665 tokens, Drafted: 1094 tokens, Per-position acceptance rate: 0.729, 0.486, Avg Draft acceptance rate: 60.8% +(APIServer pid=1) INFO 09-18 14:51:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 127.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.27, Accepted throughput: 71.20 tokens/s, Drafted throughput: 112.00 tokens/s, Accepted: 712 tokens, Drafted: 1120 tokens, Per-position acceptance rate: 0.750, 0.521, Avg Draft acceptance rate: 63.6% +(APIServer pid=1) INFO: 172.21.0.1:41374 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:37916 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:51:22 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 128.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.32, Accepted throughput: 73.29 tokens/s, Drafted throughput: 110.79 tokens/s, Accepted: 733 tokens, Drafted: 1108 tokens, Per-position acceptance rate: 0.769, 0.554, Avg Draft acceptance rate: 66.2% +(APIServer pid=1) INFO 09-18 14:51:32 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 72.70 tokens/s, Drafted throughput: 111.60 tokens/s, Accepted: 727 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.758, 0.545, Avg Draft acceptance rate: 65.1% +(APIServer pid=1) INFO: 172.21.0.1:49184 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:51:42 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 127.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.30, Accepted throughput: 72.30 tokens/s, Drafted throughput: 111.00 tokens/s, Accepted: 723 tokens, Drafted: 1110 tokens, Per-position acceptance rate: 0.769, 0.533, Avg Draft acceptance rate: 65.1% +(APIServer pid=1) INFO: 172.21.0.1:39208 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:55292 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:39224 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:51:52 [loggers.py:323] Engine 000: Avg prompt throughput: 15.8 tokens/s, Avg generation throughput: 125.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:51:52 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.29, Accepted throughput: 71.00 tokens/s, Drafted throughput: 109.99 tokens/s, Accepted: 710 tokens, Drafted: 1100 tokens, Per-position acceptance rate: 0.744, 0.547, Avg Draft acceptance rate: 64.5% +(APIServer pid=1) INFO 09-18 14:52:02 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 120.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:52:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.15, Accepted throughput: 64.39 tokens/s, Drafted throughput: 111.59 tokens/s, Accepted: 644 tokens, Drafted: 1116 tokens, Per-position acceptance rate: 0.697, 0.457, Avg Draft acceptance rate: 57.7% +(APIServer pid=1) INFO: 172.21.0.1:53966 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:52:12 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 123.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:52:12 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.23, Accepted throughput: 67.80 tokens/s, Drafted throughput: 110.59 tokens/s, Accepted: 678 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.722, 0.505, Avg Draft acceptance rate: 61.3% +(APIServer pid=1) INFO: 127.0.0.1:53434 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:52:22 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 121.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:52:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.18, Accepted throughput: 66.00 tokens/s, Drafted throughput: 111.39 tokens/s, Accepted: 660 tokens, Drafted: 1114 tokens, Per-position acceptance rate: 0.715, 0.470, Avg Draft acceptance rate: 59.2% +(APIServer pid=1) INFO: 172.21.0.1:58938 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:52:32 [loggers.py:323] Engine 000: Avg prompt throughput: 7.9 tokens/s, Avg generation throughput: 121.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:52:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.19, Accepted throughput: 65.60 tokens/s, Drafted throughput: 110.60 tokens/s, Accepted: 656 tokens, Drafted: 1106 tokens, Per-position acceptance rate: 0.712, 0.474, Avg Draft acceptance rate: 59.3% +(APIServer pid=1) INFO: 172.21.0.1:59878 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59890 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59894 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59900 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59908 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59912 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59914 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59930 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59932 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59944 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59950 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:52:42 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 105.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 15.7%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO 09-18 14:52:42 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.20, Accepted throughput: 57.80 tokens/s, Drafted throughput: 96.20 tokens/s, Accepted: 578 tokens, Drafted: 962 tokens, Per-position acceptance rate: 0.711, 0.491, Avg Draft acceptance rate: 60.1% +(APIServer pid=1) INFO: 172.21.0.1:59952 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59958 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59968 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:59980 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:55976 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:52:52 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 69.2%, Prefix cache hit rate: 14.9%, MM cache hit rate: 94.6% +(APIServer pid=1) INFO: 172.21.0.1:32818 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:32834 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:32836 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:53:02 [loggers.py:323] Engine 000: Avg prompt throughput: 13599.1 tokens/s, Avg generation throughput: 35.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7% +(APIServer pid=1) INFO 09-18 14:53:02 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.79, Accepted throughput: 11.35 tokens/s, Drafted throughput: 12.70 tokens/s, Accepted: 227 tokens, Drafted: 254 tokens, Per-position acceptance rate: 0.913, 0.874, Avg Draft acceptance rate: 89.4% +(APIServer pid=1) INFO 09-18 14:53:12 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 76.9%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7% +(APIServer pid=1) INFO: 172.21.0.1:53668 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53676 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53688 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53696 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53700 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 127.0.0.1:44356 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53716 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53720 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:53:22 [loggers.py:323] Engine 000: Avg prompt throughput: 11222.0 tokens/s, Avg generation throughput: 125.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7% +(APIServer pid=1) INFO 09-18 14:53:22 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.77, Accepted throughput: 40.25 tokens/s, Drafted throughput: 45.40 tokens/s, Accepted: 805 tokens, Drafted: 908 tokens, Per-position acceptance rate: 0.943, 0.830, Avg Draft acceptance rate: 88.7% +(APIServer pid=1) INFO: 172.21.0.1:53732 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:53748 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 14:53:32 [loggers.py:323] Engine 000: Avg prompt throughput: 17.3 tokens/s, Avg generation throughput: 144.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 21.8%, MM cache hit rate: 94.7% +(APIServer pid=1) INFO 09-18 14:53:32 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 2.66, Accepted throughput: 90.39 tokens/s, Drafted throughput: 108.99 tokens/s, Accepted: 904 tokens, Drafted: 1090 tokens, Per-position acceptance rate: 0.899, 0.760, Avg Draft acceptance rate: 82.9% +(APIServer pid=1) INFO: 172.21.0.1:40154 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:40170 - "POST /v1/chat/completions HTTP/1.1" 200 OK diff --git a/audit/tuning/steps-123/mtp2-b2048/status.json b/audit/tuning/steps-123/mtp2-b2048/status.json new file mode 100644 index 0000000..256e394 --- /dev/null +++ b/audit/tuning/steps-123/mtp2-b2048/status.json @@ -0,0 +1,5 @@ +{ + "startup": true, + "records": 44, + "failures": [] +} \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp3-b2048/compose.yaml b/audit/tuning/steps-123/mtp3-b2048/compose.yaml new file mode 100644 index 0000000..ae91912 --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/compose.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: "no" + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":3}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,4]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/steps-123/mtp3-b2048/mtp-metrics.txt b/audit/tuning/steps-123/mtp3-b2048/mtp-metrics.txt new file mode 100644 index 0000000..62f4251 --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/mtp-metrics.txt @@ -0,0 +1,28 @@ +# HELP vllm:spec_decode_num_drafts_total Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_total counter +vllm:spec_decode_num_drafts_total{engine="0",model_name="qwen3.8-flash-next"} 8872.0 +# HELP vllm:spec_decode_num_drafts_created Number of spec decoding drafts. +# TYPE vllm:spec_decode_num_drafts_created gauge +vllm:spec_decode_num_drafts_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577659e+09 +# HELP vllm:spec_decode_num_draft_tokens_total Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_total counter +vllm:spec_decode_num_draft_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 26616.0 +# HELP vllm:spec_decode_num_draft_tokens_created Number of draft tokens. +# TYPE vllm:spec_decode_num_draft_tokens_created gauge +vllm:spec_decode_num_draft_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.7897438345776758e+09 +# HELP vllm:spec_decode_num_accepted_tokens_total Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_total counter +vllm:spec_decode_num_accepted_tokens_total{engine="0",model_name="qwen3.8-flash-next"} 14825.0 +# HELP vllm:spec_decode_num_accepted_tokens_created Number of accepted tokens. +# TYPE vllm:spec_decode_num_accepted_tokens_created gauge +vllm:spec_decode_num_accepted_tokens_created{engine="0",model_name="qwen3.8-flash-next"} 1.789743834577687e+09 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_total Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_total counter +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="0"} 6682.0 +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="1"} 4815.0 +vllm:spec_decode_num_accepted_tokens_per_pos_total{engine="0",model_name="qwen3.8-flash-next",position="2"} 3328.0 +# HELP vllm:spec_decode_num_accepted_tokens_per_pos_created Accepted tokens per draft position. +# TYPE vllm:spec_decode_num_accepted_tokens_per_pos_created gauge +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="0"} 1.789743834577702e+09 +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="1"} 1.7897438345777066e+09 +vllm:spec_decode_num_accepted_tokens_per_pos_created{engine="0",model_name="qwen3.8-flash-next",position="2"} 1.7897438345777097e+09 \ No newline at end of file diff --git a/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl b/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl new file mode 100644 index 0000000..b28bbe9 --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl @@ -0,0 +1,7 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228, "response": {"id": "chatcmpl-bbb7d55702a41460", "object": "chat.completion", "created": 1789744068, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"supplier\":\"North Star Ltd\",\"currency\":\"CNY\",\"total\":476,\"due_date\":\"2026-10-03\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants me to extract invoice data into JSON format. Let me parse the information:\n\n- Supplier: North Star Ltd\n- Currency: CNY\n- Rows: 3 items at 125 each = 375, 2 items at 48 each = 96\n- Shipping: 20\n- Discount: 15\n- Total: 375 + 96 + 20 - 15 = 476\n- Due date: 2026-10-03\n\nThe total must be an integer: 476\n\nKeys required: supplier, currency, total, due_date\n\nLet me format as JSON only.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 90, "total_tokens": 275, "completion_tokens": 185, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 150}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236, "response": {"id": "chatcmpl-8083ab34aa888310", "object": "chat.completion", "created": 1789744069, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me process the orders step by step.\n\nInput orders:\n1. A pending 12\n2. B paid 20\n3. A paid 15\n4. C cancelled 99\n5. D paid 8\n6. B paid 25\n\nStep 1: Keep the last occurrence of each ID.\n- A: last occurrence is \"A paid 15\" (order 3)\n- B: last occurrence is \"B paid 25\" (order 6)\n- C: last occurrence is \"C cancelled 99\" (order 4)\n- D: last occurrence is \"D paid 8\" (order 5)\n\nStep 2: Discard cancelled orders.\n- C is cancelled, so discard it.\n- Remaining: A paid 15, B paid 25, D paid 8\n\nStep 3: Sort IDs alphabetically.\n- A, B, D\n\nStep 4: Calculate total.\n- 15 + 25 + 8 = 48\n\nOutput: {\"ids\":[\"A\",\"B\",\"D\"],\"total\":48}\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 78, "total_tokens": 342, "completion_tokens": 264, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 247}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427, "response": {"id": "chatcmpl-ae45cca384883fdb", "object": "chat.completion", "created": 1789744071, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"earliest_finish\":10,\"critical_path\":[\"A\",\"B\",\"D\"]}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "Let me analyze this scheduling problem carefully.\n\nJobs and their durations:\n- A: 3 hours (starts at hour 0)\n- B: 5 hours, starts after A\n- C: 4 hours, starts after A\n- D: 2 hours, starts after both B and C\n\nLet me trace the timeline:\n\n- A starts at hour 0, finishes at hour 3.\n- B starts after A → starts at hour 3, finishes at hour 3+5 = 8.\n- C starts after A → starts at hour 3, finishes at hour 3+4 = 7.\n- D starts after both B and C → starts at max(8, 7) = 8, finishes at 8+2 = 10.\n\nEarliest finish: 10 hours.\n\nCritical path: The longest path through the dependency graph.\n- Path A → B → D: 3 + 5 + 2 = 10\n- Path A → C → D: 3 + 4 + 2 = 9\n\nThe critical path is A → B → D with total duration 10.\n\nSo critical_path = [\"A\", \"B\", \"D\"]\nearliest_finish = 10\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 68, "total_tokens": 359, "completion_tokens": 291, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 271}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661, "response": {"id": "chatcmpl-af2a24bebab98e35", "object": "chat.completion", "created": 1789744073, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"甲\":\"主管\",\"乙\":\"经理\",\"丙\":\"经理\",\"丁\":\"总监\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "让我仔细分析审批制度:\n\n- 金额不超过5000元:主管审批\n- 超过5000且不超过20000:经理审批\n- 超过20000:总监审批\n- 加急不改变审批人\n\n现在分析各申请:\n\n甲:5000元 → 不超过5000 → 主管\n乙:5001元加急 → 超过5000且不超过20000 → 经理(加急不改变审批人)\n丙:20000元 → 超过5000且不超过20000 → 经理\n丁:20001元 → 超过20000 → 总监\n\n输出JSON格式。\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 106, "total_tokens": 295, "completion_tokens": 189, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 169}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-9c2f3c0c1afec9f2", "object": "chat.completion", "created": 1789744074, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `merge_intervals` that merges overlapping integer intervals, including touching endpoints. Let me think about the requirements:\n\n1. Merge overlapping intervals, including touching endpoints (e.g., [1,3] and [3,5] should merge to [1,5])\n2. Return sorted merged intervals as lists\n3. Reply ONLY a JSON object with a code string\n4. No imports, type annotations, decorators, top-level statements or helper functions\n5. Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n6. Allowed methods: append, sort, copy\n\nLet me write the function:\n\n```python\ndef merge_intervals(intervals):\n if len(intervals) == 0:\n return []\n sorted_intervals = sorted(intervals, key=lambda x: x[0])\n merged = [sorted_intervals[0].copy()]\n for i in range(1, len(sorted_intervals)):\n current = sorted_intervals[i]\n last = merged[len(merged) - 1]\n if current[0] <= last[1]:\n last[1] = max(last[1], current[1])\n else:\n merged.append(current.copy())\n return merged\n```\n\nWait, I need to check - can I use lambda? The allowed builtins are: sorted, len, range, min, max, list, enumerate, sum. Lambda isn't explicitly listed but it's a language construct, not a builtin. Let me think... `sorted` with a key function requires a callable. I could use `sorted(intervals)` since lists sort lexicographically by default, which would sort by first element then second element. That works for intervals!\n\nActually, `sorted(intervals)` on a list of lists will sort by the first element, then second element. That's exactly what we need.\n\nLet me also check: can I use `lambda`? It's not a builtin, it's a keyword/expression. The restriction says \"no imports, type annotations, decorators, top-level statements or helper functions.\" Lambda isn't a helper function per se, it's an anonymous function expression. But to be safe, let me just use `sorted(intervals)` without a key since default sorting of lists works correctly for intervals.\n\nAlso, I need to be careful about `.copy()` - it's in the allowed methods list. And `.append()` and `.sort()` are allowed.\n\nLet me also check: `len(merged) - 1` to get the last index - that's fine.\n\nLet me also make sure I'm not using any disallowed things. No type annotations, no decorators, no top-level statements, no helper functions. The function itself is fine.\n\nLet me format as JSON with a \"code\" key.\n\n```json\n{\"code\": \"def merge_intervals(intervals):\\n if len(intervals) == 0:\\n return []\\n sorted_intervals = sorted(intervals)\\n merged = [sorted_intervals[0].copy()]\\n for i in range(1, len(sorted_intervals)):\\n current = sorted_intervals[i]\\n last = merged[len(merged) - 1]\\n if current[0] <= last[1]:\\n last[1] = max(last[1], current[1])\\n else:\\n merged.append(current.copy())\\n return merged\"}\n```\n\nWait, I should double-check: `sorted(intervals)` on a list of lists like `[[1,3],[2,6],[8,10]]` will sort correctly by first element. Yes, Python's default list comparison is lexicographic, so it compares element by element. This is correct for our use case.\n\nAlso checking: touching endpoints means [1,3] and [3,5] merge. My condition `current[0] <= last[1]` handles this since 3 <= 3 is true.\n\nThis looks good. Let me format the JSON properly.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 81, "total_tokens": 1071, "completion_tokens": 990, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 857}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": "", "response": {"id": "chatcmpl-a61f6a3fb52eef25", "object": "chat.completion", "created": 1789744080, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": "\n\n{\"code\": \"def lower_bound(nums, target):\\n lo = 0\\n hi = len(nums)\\n while lo < hi:\\n mid = (lo + hi) // 2\\n if nums[mid] < target:\\n lo = mid + 1\\n else:\\n hi = mid\\n return lo\"}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "reasoning": "The user wants a Python function `lower_bound` that finds the index of the first element >= target in a sorted list `nums`, or `len(nums)` if none exists.\n\nConstraints:\n- Return ONLY a JSON object with a \"code\" string\n- No imports, type annotations, decorators, top-level statements, or helper functions\n- Allowed builtins: sorted, len, range, min, max, list, enumerate, sum\n- Allowed methods: append, sort, copy\n\nLet me write a binary search implementation for lower_bound:\n\n```python\ndef lower_bound(nums, target):\n lo = 0\n hi = len(nums)\n while lo < hi:\n mid = (lo + hi) // 2\n if nums[mid] < target:\n lo = mid + 1\n else:\n hi = mid\n return lo\n```\n\nThis is a standard binary search for lower bound. Let me check constraints:\n- No imports ✓\n- No type annotations ✓\n- No decorators ✓\n- No top-level statements ✓\n- No helper functions ✓\n- Uses `len` which is allowed ✓\n- Uses basic operators and comparisons ✓\n\nNow I need to wrap this in a JSON object with a \"code\" string.\n"}, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 92, "total_tokens": 438, "completion_tokens": 346, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 269}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} +{"name": "tool-arguments", "passed": true, "response": {"id": "chatcmpl-9b2fb34a4ff8498a", "object": "chat.completion", "created": 1789744082, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-b78447dfbb2c5061", "type": "function", "function": {"name": "create_ticket", "arguments": "{\"title\": \"GPU cooling failure\", \"priority\": \"urgent\", \"asset\": \"GPU-092\"}"}}], "reasoning": "The user wants me to create a ticket with specific parameters:\n- title: \"GPU cooling failure\" (exactly)\n- priority: \"urgent\"\n- asset: \"GPU-092\" (exactly)\n\nThis is straightforward - I'll call the create_ticket function with these exact values.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-471201ef", "usage": {"prompt_tokens": 325, "total_tokens": 447, "completion_tokens": 122, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 65}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} diff --git a/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl.log b/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl.log new file mode 100644 index 0000000..18b52b3 --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/quality-medium.jsonl.log @@ -0,0 +1,8 @@ +{"name": "invoice-extraction", "passed": true, "elapsed_s": 1.1935799049970228} +{"name": "deduplicate", "passed": true, "elapsed_s": 1.5906357709973236} +{"name": "schedule", "passed": true, "elapsed_s": 1.9658004070006427} +{"name": "chinese-policy", "passed": true, "elapsed_s": 1.2027164360042661} +{"name": "merge-intervals", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "binary-search", "passed": true, "validation_stdout": "passed\n", "validation_stderr": ""} +{"name": "tool-arguments", "passed": true} +QUALITY_COMPLETE diff --git a/audit/tuning/steps-123/mtp3-b2048/screen.jsonl b/audit/tuning/steps-123/mtp3-b2048/screen.jsonl new file mode 100644 index 0000000..922136f --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/screen.jsonl @@ -0,0 +1,37 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6857045710057719, "elapsed_s": 1.1168529220012715, "decode_tps": 187.87036947485734, "max_stream_gap_s": 0.021299945998180192, "events": 23, "content": "\n\n3973", "reasoning": "I need to compute 137 * 29 + 0.\n\n137 * 29 = 137 * 30 - 137 = 4110 - 137 = 3973\n\nAdding 0 doesn't change anything, so the answer is 3973.\n", "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.09021790199767565, "elapsed_s": 0.30122662099893205, "decode_tps": 170.60906378842878, "max_stream_gap_s": 0.0215804309991654, "events": 10, "content": "\n\n5761", "reasoning": "160 * 36 = 5760. 5760 + 1 = 5761.\n", "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0900614569982281, "elapsed_s": 0.43906909199722577, "decode_tps": 186.2423439538412, "max_stream_gap_s": 0.021113369999511633, "events": 18, "content": "\n\n7871", "reasoning": "183 * 43 = 183 * 40 + 183 * 3 = 7320 + 549 = 7869. Then 7869 + 2 = 7871.\n", "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.09134426800301298, "elapsed_s": 0.293823849002365, "decode_tps": 187.67324493881486, "max_stream_gap_s": 0.020979296998120844, "events": 11, "content": "\n\n10303", "reasoning": "206 * 50 = 10300\n10300 + 3 = 10303\n", "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08995563200005563, "elapsed_s": 0.4795566350003355, "decode_tps": 182.23772385911695, "max_stream_gap_s": 0.021156058995984495, "events": 20, "content": "\n\n13057", "reasoning": "229 * 57 = 229 * 50 + 229 * 7 = 11450 + 1603 = 13053. Then 13053 + 4 = 13057.\n", "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08942628499789862, "elapsed_s": 0.3126287199993385, "decode_tps": 183.68975230819282, "max_stream_gap_s": 0.02103689500654582, "events": 12, "content": "\n\n16133", "reasoning": "252 * 64 = 16128. Then 16128 + 5 = 16133.\n", "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08937776899983874, "elapsed_s": 0.4788391099937144, "decode_tps": 179.73542591253175, "max_stream_gap_s": 0.02123644899984356, "events": 20, "content": "\n\n19531", "reasoning": "275 * 71 = 275 * 70 + 275 * 1 = 19250 + 275 = 19525. Then 19525 + 6 = 19531.\n", "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.09002831100224284, "elapsed_s": 0.6215811549991486, "decode_tps": 174.95908647709422, "max_stream_gap_s": 0.021011450000514742, "events": 26, "content": "\n\n23251", "reasoning": "298 * 78 = 298 * 78. Let me compute this step by step.\n\n298 * 78 = 298 * 80 - 298 * 2 = 23840 - 596 = 23244\n\n23244 + 7 = 23251\n", "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08917911599564832, "elapsed_s": 0.5392506789939944, "decode_tps": 186.6369859948443, "max_stream_gap_s": 0.021216852001089137, "events": 23, "content": "\n\n27293", "reasoning": "Let me compute 321 * 85 + 8.\n\n321 * 85 = 321 * 80 + 321 * 5 = 25680 + 1605 = 27285\n\n27285 + 8 = 27293\n", "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.09073446100228466, "elapsed_s": 0.539050845996826, "decode_tps": 185.13711025977918, "max_stream_gap_s": 0.020977412998036016, "events": 23, "content": "\n\n31657", "reasoning": "Let me compute 344 * 92 + 9.\n\n344 * 92 = 344 * 90 + 344 * 2 = 30960 + 688 = 31648\n\n31648 + 9 = 31657\n", "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08987670400529169, "elapsed_s": 0.5182491330051562, "decode_tps": 191.42221685799936, "max_stream_gap_s": 0.02107875300134765, "events": 22, "content": "\n\n36343", "reasoning": "Let me compute 367 * 99 + 10.\n\n367 * 99 = 367 * (100 - 1) = 36700 - 367 = 36333\n\n36333 + 10 = 36343\n", "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09002829199744156, "elapsed_s": 0.5601620039960835, "decode_tps": 189.3078452545796, "max_stream_gap_s": 0.021013982994190883, "events": 24, "content": "\n\n41351", "reasoning": "Let me compute 390 * 106 + 11.\n\n390 * 106 = 390 * 100 + 390 * 6 = 39000 + 2340 = 41340\n\n41340 + 11 = 41351\n", "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.1072728509971057, "elapsed_s": 0.40980977100116434, "decode_tps": 138.82603154496502, "max_stream_gap_s": 0.020985293995181564, "events": 16, "content": "\n\n{\"name\":\"设备A\",\"count\":7,\"enabled\":true}", "reasoning": "The user wants me to return exactly a specific JSON object with no markdown formatting. Let me provide exactly what they asked for.\n", "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.09010298399516614, "elapsed_s": 0.5101531999971485, "decode_tps": 147.60140011380784, "max_stream_gap_s": 0.020926092998706736, "events": 22, "content": "\n\nAQUA-58321", "reasoning": "The user is asking me to recall the code they previously told me to remember. They said \"Remember the code AQUA-58321.\" and now they're asking what it was, with the instruction to reply only with the code.\n", "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3285, "completion_tokens": 150, "completion_tokens_details": {"reasoning_tokens": 111}}, "ttft_s": 0.7640111340006115, "elapsed_s": 1.6888792590034427, "decode_tps": 161.10404929302098, "max_stream_gap_s": 0.026646995996998157, "events": 45, "content": "\n\nVerification code: SPARK-82741\nSum of amounts: 1368 (125 + 478 + 765)", "reasoning": "The user wants the verification code and the sum of three amounts from the invoice image. Reading the image: Verification code is SPARK-82741. Amounts are 125, 478, 765. Sum: 125 + 478 = 603; 603 + 765 = 1368. I need to provide a clean, self-contained response with the code and the sum. No need for elaborate explanation beyond the answer.\n", "sha256": "59acfe36a1bac3e2c16c79d5c3712068c6d77030f2161a2297c21a795d2a2f9c"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12191, "completion_tokens": 122, "completion_tokens_details": {"reasoning_tokens": 108}}, "ttft_s": 2.5409359719997155, "elapsed_s": 3.243877524000709, "decode_tps": 172.1337992548049, "max_stream_gap_s": 0.03231711799890036, "events": 35, "content": "\n\nRed, Green, Blue, Yellow, Purple, White", "reasoning": "The user wants the six background colors in chronological order, replying only with English color names. Looking at the frames: 0-19s red, 20-39s green, 40-59s blue, 60-79s yellow, 80-99s purple, 100-119s white. So the order is Red, Green, Blue, Yellow, Purple, White. I need to provide just the names, no extra text.\n", "sha256": "34f1ae68e14f16d3cbbeed12b616b538f07ebec97e9ad601ced414e453c29f65"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.463653916005569, "elapsed_s": 1.8863289360015187, "decode_tps": 165.61186890266376, "max_stream_gap_s": 0.020653829000366386, "events": 22, "content": "\n\nSPARK-82741", "reasoning": "The user wants the verification code from the invoice. All four copies show the same line: \"Verification: SPARK-82741\". The instruction is to reply only with the code. So the answer is SPARK-82741. No need for elaboration.\n", "sha256": "c4d4ac93f59330deca1f3c5fe9842a60a2ac31170fdbc40f311385c571ed9024"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08850599199649878, "elapsed_s": 2.1877699620017665, "decode_tps": 121.47114590803943, "max_stream_gap_s": 0.0206061229982879, "events": 106, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling,", "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08851322199916467, "elapsed_s": 16.576275200000964, "decode_tps": 124.152689900008, "max_stream_gap_s": 0.021029635005106684, "events": 824, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and stop? Or error row.\n REPLACE: if exists repo.update else insert; also seen? For within file, replace previous? Could update.\n- If no duplicate_keys: no duplicate detection.\n\nNeed repository transaction? Could have repository.begin/commit/rollback optional. But keep simple.\n\nError reporting:\n-", "sha256": "9e8927825d4c68bc3bb71bc5d5dc66a8a35e5961a52ff04a44c5211082f2d4c9"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.09001792799972463, "elapsed_s": 16.722888370000874, "decode_tps": 123.06955718424507, "max_stream_gap_s": 0.021017879000282846, "events": 831, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and continue? Or raise? Could add error.\n REPLACE: if exists repo.update else insert; also seen? For within file, if duplicate, update? Need seen_in_file maybe if REPLACE, update in-memory? If repository in-memory, update. But if external, seen_in_file still prevents? Could allow later row to", "sha256": "649658539a14ad95cd3aba4a031955cf646f9c2cef5e606a2f3f25be99a3962b"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07350860699807527, "elapsed_s": 16.641791551002825, "decode_tps": 123.54931448950835, "max_stream_gap_s": 0.021029645002272446, "events": 828, "content": "", "reasoning": "We need answer user's request: \"Write a production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, and a detailed explanation.\"\n\nNeed produce final answer with code, tests, explanation. Need likely include a complete utility. Need think through design.\n\nWe need satisfy: production-quality Python CSV import utility with validation, duplicate handling, error reporting, unit tests, detailed explanation.\n\nWe can provide a single-file module maybe csv_importer.py with dataclasses, exceptions, validators, duplicate strategies, error report, CLI? Unit tests using unittest or pytest. Need detailed explanation.\n\nNeed ensure code is correct. Let's design robust CSV import utility.\n\nRequirements:\n- Import CSV into records (maybe database? but generic). Could be utility that reads CSV, validates rows, handles duplicates, reports errors, returns result.\n- Production-quality: type hints, logging, config, encoding, delimiter, quoting, BOM, schema, required fields, types, constraints, custom validators, duplicate keys, strategies (skip, update, fail, replace?), error reporting with row numbers, field, code, message, line, raw row.\n- Unit tests: use unittest or pytest. Need maybe include tests for validation, duplicate handling, error reporting, encoding, malformed CSV, custom validators, CLI? Keep manageable.\n- Detailed explanation: architecture, usage, extension.\n\nNeed maybe avoid external dependencies beyond stdlib. Use csv, dataclasses, enum, typing, logging, pathlib, io, hashlib? For duplicate handling maybe in-memory set or callback. Could support database via repository interface.\n\nLet's design:\n\nModule csv_importer.py:\n\nImports:\nfrom __future__ import annotations\nimport csv, logging, re, unicodedata, hashlib, json, argparse, sys\nfrom dataclasses import dataclass, field\nfrom datetime import datetime, date\nfrom enum import Enum\nfrom pathlib import Path\nfrom typing import Any, Callable, Iterable, Iterator, Mapping, Sequence, TextIO, TypeVar, Optional, Union\nfrom collections.abc import Mapping? Use typing.\n\nExceptions:\nclass ImportError(Exception) maybe CSVImportError.\nclass SchemaError(CSVImportError)\nclass ValidationError(CSVImportError)\n\nEnums:\nclass DuplicateStrategy(str, Enum): SKIP, UPDATE, FAIL, REPLACE? For import into repository.\nclass OnError(str, Enum): FAIL, SKIP_ROW, SKIP_FILE? Maybe continue.\n\nDataclasses:\n@dataclass(frozen=True)\nclass FieldSpec:\n name: str\n required: bool = False\n type: str = \"str\" # str, int, float, bool, date, datetime, email, uuid?\n nullable: bool = False\n default: Any = None\n min_length: Optional[int] = None\n max_length: Optional[int] = None\n pattern: Optional[str] = None\n choices: Optional[Sequence[str]] = None\n min_value: Optional[float] = None\n max_value: Optional[float] = None\n trim: bool = True\n empty_to_null: bool = True\n validators: tuple[Callable[[Any], None], ...] = ()? dataclass with tuple.\n\nBut validators callable may raise ValidationError.\n\n@dataclass(frozen=True)\nclass Schema:\n fields: tuple[FieldSpec,...]\n duplicate_keys: tuple[str,...] = ()\n allow_extra_fields: bool = False\n required_headers: maybe derived.\n\n@dataclass(frozen=True)\nclass ImportOptions:\n encoding: str = \"utf-8-sig\"\n delimiter: str = \",\"\n quotechar: str = '\"'\n escapechar: Optional[str] = None\n doublequote: bool = True\n skipinitialspace: bool = False\n max_rows: Optional[int] = None\n max_errors: Optional[int] = 100\n on_error: OnError = OnError.SKIP_ROW? For production maybe fail? Let's choose SKIP_ROW to collect errors.\n duplicate_strategy: DuplicateStrategy = DuplicateStrategy.SKIP\n strict_headers: bool = True? maybe exact headers? Could allow extra.\n\n@dataclass(frozen=True)\nclass RowError:\n row_number: int # 1-based data row? include header? Use data_row_number and csv_line_number.\n csv_line_number: int\n field: Optional[str]\n code: str\n message: str\n value: Optional[str]\n raw_row: Optional[Mapping[str, str]]? For security maybe redact? Could include raw row dict.\n\n@dataclass\nclass ImportResult:\n imported: list[dict[str, Any]]\n skipped_duplicates: list[dict[str, Any]]\n updated: list[dict[str, Any]]\n errors: list[RowError]\n total_rows: int\n valid_rows: int\n invalid_rows: int\n duplicate_rows: int\n started_at, finished_at? maybe.\n def ok property.\n\nRepository protocol:\nclass Repository(Protocol):\n def exists(self, key: tuple[Any,...]) -> bool: ...\n def insert(self, record: dict[str, Any]) -> None: ...\n def update(self, record: dict[str, Any]) -> None: ...\n def upsert? Could use exists + insert/update.\n def begin/commit? Maybe transaction.\n\nMemoryRepository:\nclass InMemoryRepository:\n def __init__(self, key_fields): self._store = {}\n def key(self, record): tuple(record.get(k) for k in key_fields)\n exists, insert, update, all.\n\nBut duplicate handling can be done by importer with repository. Need handle duplicates within file too. Use seen_keys set.\n\nValidation:\n- parse headers: first row. Check required headers present. If strict_headers? maybe if allow_extra_fields false, unknown headers error.\n- For each row: csv.DictReader. Need line numbers. DictReader has line_num? csv.reader line_num. Could use csv.reader manually to get line numbers and handle header. Better:\n with open(path, newline='', encoding=...) as f:\n reader = csv.reader(f, **dialect)\n try: headers = next(reader)\n except StopIteration: raise SchemaError(\"empty\")\n header_line = reader.line_num\n normalize headers? strip.\n validate headers.\n for row in reader:\n line_num = reader.line_num\n if len(row) != len(headers): error? DictReader would restkey. We can handle.\n raw = dict(zip(headers, row)) if len row >=? If too few, missing fields. If too many, extra.\n process.\n\nNeed handle BOM: encoding utf-8-sig.\n\nField parsing:\ndef parse_value(raw, spec):\n if raw is None: value = None\n else:\n if spec.trim: raw = raw.strip()\n if raw == \"\" and spec.empty_to_null: value = None else raw\n if value is None:\n if spec.required and not spec.nullable: error REQUIRED\n return spec.default if value is None and spec.default is not None? But if required no.\n type conversion:\n str: ensure str\n int: int(value) maybe underscores? Use int(value, 10). bool: accept true/false/1/0/yes/no/y/n? Need case-insensitive.\n float: float(value)\n date: date.fromisoformat(value)\n datetime: datetime.fromisoformat(value)\n email: regex.\n uuid: uuid.UUID.\n constraints: min_length, max_length, pattern, choices, min_value, max_value.\n custom validators.\n\nNeed collect multiple errors per row? Could stop at first? Production maybe collect all field errors. But duplicate handling after validation. We can collect row_errors list. If any errors, add to result and continue based on on_error. If on_error FAIL, raise? But result? Could raise CSVImportError with errors. Simpler: importer collects errors; if on_error=FAIL and error occurs, stop and return? Maybe raise. But unit tests easier if returns errors. Let's define OnError: FAIL raises CSVImportError after first error? Or collect and stop. For production, fail fast. But error reporting wants errors. Could have max_errors. Let's implement: if on_error == FAIL, raise CSVImportError with errors list. But tests can catch. Or return result with ok false. Hmm.\n\nMaybe simpler: Importer always collects errors up to max_errors, then if on_error == FAIL and errors, raise CSVImportError. But if max_errors reached, stop reading. For SKIP_ROW, continue. For SKIP_FILE? not needed.\n\nDuplicate handling:\n- If schema.duplicate_keys: key = tuple(record.get(k) for k in keys). If any key None? If required fields validated, not None. But if nullable? Could skip duplicate check if any None? Maybe treat None as value.\n- If key in seen_in_file or repo.exists(key): duplicate.\n strategy:\n SKIP: add to skipped_duplicates, continue.\n UPDATE: if repo.exists: repo.update(record), add updated; else insert.\n FAIL: add error DUPLICATE_KEY and stop? Or error row.\n REPLACE: if exists repo.update else insert; also seen? For within file, replace previous? Could update.\n- If no duplicate_keys: no duplicate handling.\n\nNeed repository transaction? Could wrap in try/except. If repo insert raises, add error? Maybe repository errors.\n\n", "sha256": "4af2092c8a16885a6236f0c3adc0227eb9dd8079ec421ba9c2eb5cc404b3440b"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07293527999718208, "elapsed_s": 1.9857562760007568, "decode_tps": 133.31095828243588, "max_stream_gap_s": 0.020934283995302394, "events": 97, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08870576600020286, "elapsed_s": 15.336975530000927, "decode_tps": 134.24473934955645, "max_stream_gap_s": 0.020986846000596415, "events": 765, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过/拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员挂起、恢复、终止,并记录原因 |\n\n---\n\n## 4.2.2 需求项:审批", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "d4b632309eb489d41b5803e52e0ae083ffe5c0f801ac927326c6037eb95c6474"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08885356999962823, "elapsed_s": 15.44073144599679, "decode_tps": 133.33873657244945, "max_stream_gap_s": 0.02105036300054053, "events": 770, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过/拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 流程实例异常中断 | 支持管理员人工干预、重试、终止 |\n\n---\n\n## 4.2.2 需求项:审批提交与路由\n\n", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "795e37bad007369fe0ce6ce658b4ebf659222130ea24f338f6f342b4ce6aff47"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07399053900007857, "elapsed_s": 15.320730624000134, "decode_tps": 134.25820789152598, "max_stream_gap_s": 0.020978505002858583, "events": 765, "content": "\n\n# 企业采购系统需求说明书\n\n## 1. 文档说明\n\n### 1.1 文档目的\n\n本文档用于定义企业采购系统的完整业务需求,覆盖采购申请、审批、预算控制、供应商管理、采购执行、库存联动、审计追溯等核心能力,并针对每项关键需求说明异常处理机制,确保系统具备可落地、可审计、可追溯、可运营的业务闭环。\n\n### 1.2 适用范围\n\n本系统适用于企业内部采购业务全流程管理,包括但不限于:\n\n- 采购需求发起\n- 采购审批\n- 预算占用与释放\n- 供应商准入与评估\n- 询价比价、订单生成\n- 收货入库、库存更新\n- 对账、付款前置校验\n- 审计日志与合规追溯\n\n### 1.3 设计目标\n\n系统应满足以下目标:\n\n1. **流程可控**:采购全流程线上化,审批路径清晰可配置。\n2. **预算受控**:采购行为与预算强关联,避免超预算采购。\n3. **供应商可管**:供应商准入、评级、黑名单、合同、资质统一管理。\n4. **库存联动**:采购、收货、入库、退货、调拨与库存实时一致。\n5. **审计可查**:所有关键操作留痕,支持事后审计与责任追溯。\n6. **异常可处理**:对审批异常、预算异常、供应商异常、库存异常、审计异常均有明确处理机制。\n\n---\n\n## 2. 系统角色与权限\n\n### 2.1 主要角色\n\n| 角色 | 职责 |\n|---|---|\n| 申请人 | 发起采购申请、补充资料、跟踪进度 |\n| 部门主管 | 审核业务必要性、合理性 |\n| 财务/预算管理员 | 审核预算、控制预算占用 |\n| 采购员 | 询价、比价、下单、供应商协同 |\n| 采购经理 | 审批采购方案、供应商选择、订单审批 |\n| 仓库管理员 | 收货、入库、库存维护 |\n| 质量/验收人员 | 验收物料、处理质量异常 |\n| 审计员 | 查询日志、导出审计资料、发起合规检查 |\n| 系统管理员 | 配置流程、权限、基础数据、异常规则 |\n\n### 2.2 权限原则\n\n- 最小权限原则\n- 职责分离原则\n- 审批与执行分离\n- 预算控制与采购执行分离\n- 审计只读、不可篡改\n\n---\n\n## 3. 总体业务流程\n\n### 3.1 标准采购流程\n\n1. 申请人提交采购申请\n2. 系统校验预算、物料、供应商、合规规则\n3. 进入审批流\n4. 审批通过后生成采购需求单/采购订单\n5. 采购员选择供应商并下单\n6. 供应商发货\n7. 仓库收货、质检、入库\n8. 库存更新\n9. 对账、付款申请\n10. 全流程审计留痕\n\n### 3.2 异常流程总览\n\n- 审批驳回、撤回、超时、代理审批\n- 预算不足、预算冻结、预算释放失败\n- 供应商失效、资质过期、黑名单、价格异常\n- 收货差异、库存不足、入库失败、退货异常\n- 审计日志缺失、越权操作、数据篡改风险\n\n---\n\n## 4. 功能需求说明\n\n以下按模块逐项说明功能需求及异常处理。\n\n---\n\n# 4.1 采购申请模块\n\n## 4.1.1 需求项:采购申请创建\n\n### 功能说明\n\n系统应支持用户创建采购申请,填写以下信息:\n\n- 申请单号\n- 申请部门\n- 申请人\n- 申请日期\n- 期望到货日期\n- 采购类型(生产物料、办公用品、服务、固定资产等)\n- 物料明细\n- 数量\n- 单位\n- 预估单价\n- 预估金额\n- 预算科目\n- 成本中心\n- 项目编码(可选)\n- 申请原因\n- 附件(报价单、需求说明、图片等)\n\n### 业务规则\n\n- 申请金额 = 数量 × 预估单价\n- 同一申请单可包含多个物料行\n- 申请单保存后进入草稿状态\n- 提交后进入审批中状态\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 必填项缺失 | 前端拦截,提示缺失字段,不允许提交 |\n| 数量或单价为负数 | 禁止提交,提示“数量/单价必须大于0” |\n| 物料编码不存在 | 提示“物料不存在或已停用”,要求重新选择 |\n| 附件格式不支持 | 提示支持格式,拒绝上传 |\n| 附件大小超限 | 提示最大限制,拒绝上传 |\n| 重复提交 | 系统生成唯一提交令牌,防止重复创建 |\n| 网络中断导致提交失败 | 保留草稿,提示用户重新提交,避免数据丢失 |\n| 申请单被他人并发修改 | 采用乐观锁,提示“数据已变更,请刷新后重试” |\n\n---\n\n## 4.1.2 需求项:采购申请编辑与撤回\n\n### 功能说明\n\n- 草稿状态可编辑\n- 审批中状态可撤回(若流程允许)\n- 已审批通过状态不可直接编辑,需走变更流程\n\n### 业务规则\n\n- 仅申请人或授权人可编辑\n- 撤回后状态回到草稿\n- 撤回操作需记录日志\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 非申请人编辑 | 拒绝操作,提示“无权限” |\n| 审批中不允许撤回 | 提示“当前节点不允许撤回” |\n| 撤回时审批人已处理 | 提示“审批已处理,无法撤回” |\n| 撤回后再次提交 | 重新走完整审批流,并保留历史版本 |\n| 撤回失败 | 事务回滚,状态不变,记录失败日志 |\n\n---\n\n## 4.1.3 需求项:采购申请复制与引用\n\n### 功能说明\n\n- 支持从历史申请复制生成新申请\n- 支持从库存请领、项目需求、维修工单等来源生成采购申请\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 源单据已删除 | 提示“源单据不存在” |\n| 源单据无权限查看 | 拒绝复制 |\n| 复制后物料已停用 | 标记异常行,要求用户处理 |\n| 复制后预算科目失效 | 提示重新选择预算科目 |\n\n---\n\n# 4.2 审批模块\n\n## 4.2.1 需求项:审批流配置\n\n### 功能说明\n\n系统应支持按以下条件配置审批流:\n\n- 采购类型\n- 金额区间\n- 部门\n- 项目\n- 物料类别\n- 供应商类型\n- 是否紧急采购\n- 是否超预算\n- 是否指定供应商\n\n### 审批节点类型\n\n- 部门主管审批\n- 财务审批\n- 预算审批\n- 采购经理审批\n- 总经理审批\n- 会签\n- 或签\n- 条件分支\n- 自动通过\n- 自动拒绝\n\n### 业务规则\n\n- 审批流版本化管理\n- 已发起流程按发起时版本执行\n- 新流程仅对新单据生效\n- 支持紧急采购简化审批流\n\n### 异常处理\n\n| 异常场景 | 处理方式 |\n|---|---|\n| 审批流未配置 | 提交时提示“未配置审批流”,禁止提交 |\n| 审批流配置冲突 | 保存时校验,提示冲突规则 |\n| 审批流版本被删除 | 已发起流程继续执行,新单据禁止使用 |\n| 条件分支无匹配 | 进入默认审批流或提示配置错误 |\n| 审批节点为空 | 禁止发布流程 |\n| 审批人离职/停用 | 自动转交上级或管理员指定人员,并记录异常 |\n| 审批人长期未处理 | 触发超时提醒、升级、自动转交或自动通过/拒绝(按配置) |\n| 审批流死循环 | 发布前校验,禁止发布 |\n| 并发审批同一单据 | 仅允许一次有效审批,后提交者提示“单据已处理” |\n\n---\n\n## 4.2.2 需求", "reasoning": "我们需要回答用户中文请求:“请写一份企业采购系统的完整需求说明,覆盖审批、预算、供应商、库存和审计,并逐项说明异常处理。” 需要生成完整需求说明,覆盖审批、预算、供应商、库存、审计,并逐项说明异常处理。需要可能结构化:文档目的、范围、角色、总体流程、功能需求、非功能需求、数据模型、接口、权限、异常处理逐项。用户要求“逐项说明异常处理”,可能每个模块/需求项都要异常处理。需要详尽。无需外部信息。最终中文。确保覆盖审批、预算、供应商、库存、审计。可以包含需求编号、描述、异常处理。可能太长但可接受。需要“完整需求说明”。可以按模块列功能需求,每项异常处理。也可包含状态机、规则、验收标准。注意不要过度啰嗦但足够完整。\n", "sha256": "ece641cb2f77db4659a64c75f36366c11b83095a29fc22cb320a1b306b6da1f5"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.09028758799831849, "elapsed_s": 2.1541953229971114, "decode_tps": 123.55203465534235, "max_stream_gap_s": 0.02097705299820518, "events": 105, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsett", "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0739625900023384, "elapsed_s": 16.64171508599975, "decode_tps": 123.55327015505169, "max_stream_gap_s": 0.021100942998600658, "events": 828, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled. Good.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax", "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07516271799977403, "elapsed_s": 16.646220769005595, "decode_tps": 123.52862404436223, "max_stream_gap_s": 0.021613906996208243, "events": 828, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax.\n", "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07455741699959617, "elapsed_s": 16.66266934000305, "decode_tps": 123.40162699055197, "max_stream_gap_s": 0.021173682995140553, "events": 829, "content": "", "reasoning": "We need answer user: \"Explain and prove why Dijkstra works with nonnegative weights. Include a counterexample with negative edges and compare Bellman-Ford.\" Need provide explanation, proof, counterexample, comparison. Need final only. Need be thorough. Let's plan.\n\nWe need explain Dijkstra algorithm: single-source shortest paths in directed/undirected graph with nonnegative edge weights. Greedy: maintain dist labels, priority queue, settled set S. Initially dist(s)=0, others infinity. Repeatedly extract min unsettled vertex u, settle it, relax outgoing edges.\n\nWhy nonnegative weights matter: Once vertex u has minimum tentative distance among unsettled, no unsettled vertex can provide a shorter path to u because any path from s to u through unsettled vertex v has prefix to v with length at least dist[v] (or at least dist[u]?) Need formal proof.\n\nProof by induction: invariant: for each settled vertex x, dist[x]=δ(s,x) true shortest distance. Also for each unsettled vertex v, dist[v] is length of shortest path from s to v whose internal vertices are settled? Or upper bound. Need prove extraction min is correct.\n\nStandard proof: Let u be min tentative among unsettled. Suppose there is a shorter path P from s to u. Let y be first unsettled vertex on P (u itself if all previous settled? Actually if P shorter than dist[u], since s settled, there is first unsettled vertex y on P; let x be predecessor of y on P, settled. When x was settled, edge (x,y) relaxed, so dist[y] ≤ length(P[s..y]). Since weights nonnegative, length(P[s..y]) ≤ length(P) < dist[u]. Thus dist[y] < dist[u], contradicting u min. This proof uses nonnegative weights to ensure prefix length ≤ full path length. If negative edges, prefix can be longer than full path, so contradiction fails.\n\nAlternative proof: Dijkstra's invariant: settled vertices have final distances. For unsettled, dist[v] is shortest path using only settled internal vertices. When extract min u, any path to u not using only settled vertices has first unsettled vertex y; prefix to y has length ≤ total path due nonnegative, and dist[y] ≤ prefix, so dist[y] ≤ path length. Since u min, dist[u] ≤ dist[y] ≤ path length. Thus dist[u] is shortest.\n\nNeed include counterexample with negative edges. Need show Dijkstra fails. Classic: s->a weight 1, s->b weight 2, a->b weight -2. Shortest s->b = -1 via a, but Dijkstra settles a (dist 1) then b? Let's simulate: initial dist s=0, a=1, b=2. Extract a (1), relax a->b: dist b = -1. Then extract b (-1). Actually Dijkstra would find correct if b not settled before a. Need counterexample where Dijkstra settles vertex too early due negative edge from unsettled vertex. Need graph where vertex v has small tentative but later negative edge from another unsettled vertex reduces it after v settled. Example: s->a=1, s->b=2, b->a=-2. Shortest to a = 0 via b. Dijkstra: dist a=1, b=2. Extract a (1) settle. Then extract b (2), relax b->a gives -1? Wait 2-2=0, but a already settled, Dijkstra doesn't update settled. Final dist a=1, wrong (0). This works. Need ensure nonnegative? Negative edge b->a. Dijkstra picks a before b because 1<2, but path s-b-a length 0 shorter. Good. Could also use s->a=2, s->b=3, b->a=-2 => shortest a=1, Dijkstra settles a=2 wrong. But first example fine. Need maybe directed graph. If undirected negative edge? Dijkstra fails too. Use directed.\n\nNeed compare Bellman-Ford: handles negative weights (no negative cycles reachable from source) by relaxing all edges V-1 times; detects negative cycles. Complexity O(VE), slower than Dijkstra O((V+E) log V) with binary heap or O(E+V log V). Dijkstra requires nonnegative weights; BF can handle negative edges but not negative cycles (shortest paths undefined). BF dynamic programming: after k iterations, dist[v] shortest path with at most k edges. Proof by induction. If negative cycle reachable, distances can decrease indefinitely; BF detects if an edge can still relax on Vth iteration.\n\nNeed maybe mention Dijkstra with negative edges can fail even if no negative cycles. Counterexample no negative cycle. Compare: Dijkstra greedy local optimal; BF exhaustive relaxation. Dijkstra cannot recover from settled vertex; BF can update any vertex repeatedly.\n\nNeed maybe include pseudocode? Could include. Need proof rigorous. Need maybe discuss zero weights okay. Nonnegative includes zero. Need mention if negative edges but no negative cycles, Dijkstra may still work in some cases but not guaranteed. Need maybe mention Johnson reweighting? Not necessary.\n\nNeed final answer structured: 1. Algorithm. 2. Why nonnegative weights. 3. Proof. 4. Counterexample. 5. Bellman-Ford comparison. Maybe include table.\n\nLet's craft final with definitions:\nGraph G=(V,E), weight w:E→R, source s. δ(s,v) shortest path distance. Dijkstra maintains d[v]. S settled.\n\nPseudocode:\nd[s]=0; d[v]=∞; PQ all vertices; S=∅\nwhile PQ not empty:\n u=extract-min(PQ)\n S.add(u)\n for (u,v) in E:\n if d[v] > d[u]+w(u,v): d[v]=d[u]+w(u,v); decrease-key\n\nProof:\nLemma: At start of each iteration, for every settled x, d[x]=δ(s,x). For every unsettled v, d[v] is length of shortest s-v path whose internal vertices are in S (or ∞ if none). Actually after relaxing from settled vertices, d[v] equals min over paths from s to v with all internal vertices settled? Need prove. Initially S empty: d[s]=0 path no internal; others ∞. When settle u, relax edges: for any unsettled v, new paths with internal vertices S∪{u} either old (internal S) or path to u (internal S) + edge u-v. Since d[u]=δ(s,u) and path to u internal S? Need careful: d[u] is shortest path with internal S, but if u settled, d[u]=δ(s,u). For v, d[v] updated to min(old, d[u]+w(u,v)). This gives shortest path with internal S∪{u}. Good.\n\nThen correctness of extracting u: Let u be min d among unsettled. Need show d[u]=δ(s,u). Let P be any shortest s-u path. If all internal vertices settled, then by invariant d[u] ≤ length(P)=δ(s,u); and d[u] always upper bound so equality. If some internal unsettled, let y be first unsettled vertex on P, x predecessor settled (or s). Let P_y prefix s to y. By invariant d[y] ≤ length(P_y). Since weights nonnegative, length(P_y) ≤ length(P)=δ(s,u). Thus d[y] ≤ δ(s,u). If δ(s,u)δ(s,u). Let P shortest path. Let y first unsettled on P. Let x predecessor. Since x settled, edge relaxed when x settled, so d[y] ≤ length(P prefix to y). Nonnegative => length(prefix) ≤ length(P) < d[u]. So d[y] < d[u], contradiction. This is concise. Need mention d[y] may have been updated later? It is at most that value at all times after relaxation, so at extraction time still ≤. Good.\n\nNeed maybe discuss negative weights break proof: prefix can be longer than full path, so d[y] may not be < d[u]. In counterexample, shortest path to a is s-b-a length 0; first unsettled y=b? Let's map: Dijkstra extracts a first. Shortest path to a: s->b->a. First unsettled vertex on path is b (since s settled, a target unsettled? Actually path to a: s (settled), b (unsettled), a (target unsettled). First unsettled y=b. Prefix length s-b=2, full path length 0. Nonnegative condition fails: prefix 2 > full 0. d[b]=2, not < d[a]=1, so no contradiction. Dijkstra settles a at 1, later b relaxes to 0 but a settled.\n\nCounterexample details:\nVertices s,a,b. Edges: s→a weight 1, s→b weight 2, b→a weight -2. No negative cycle. Dijkstra:\ninit d[s]=0, d[a]=1, d[b]=2.\nextract s, relax.\n", "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126125, "completion_tokens": 130, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.516383726004278, "elapsed_s": 15.218214432999957, "decode_tps": 183.80500983237002, "max_stream_gap_s": 0.02093203100230312, "events": 36, "content": "\n\nNorthern access code: NORTH-49273\nSouthern access code: SOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (found between Item 000019 and Item 000021)\n- \"The southern access code is SOUTH-18362.\" (found between Item 002499 and Item 002501)\n", "sha256": "9a2e65aa6f54431f5ea3bfa29bcdd76256bef03427fca286f07cc3232bfdc529"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126119, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.510828671998752, "elapsed_s": 15.23376362700219, "decode_tps": 170.13978802479537, "max_stream_gap_s": 0.021043553999334108, "events": 37, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data they've provided.\n\nLooking through the data:\n- The northern access code is stated as: NORTH-49273 (found between Item 000019 and Item 000021)\n- The southern access code is stated as: SOUTH-18362 (found between Item 002499 and Item 002501)\n", "sha256": "1df90f428d94a645b711abef54df465c7fa0db783e3b3602ea5b4aa73f29aaab"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 14.506387795998307, "elapsed_s": 15.205817506997846, "decode_tps": 168.7088754513573, "max_stream_gap_s": 0.020960481000656728, "events": 36, "content": "\n\nNORTH-49273\nSOUTH-18362", "reasoning": "The user is asking me to provide both the northern and southern access codes from the data they've provided.\n\nLooking through the data, I can find:\n- \"The northern access code is NORTH-49273.\" (appears after Item 000019)\n- \"The southern access code is SOUTH-18362.\" (appears after Item 002499)\n\nThe user wants only these two codes.\n", "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 12.83541851100017, "elapsed_s": 13.968540412002767, "decode_tps": 167.67834054913806, "max_stream_gap_s": 0.021041211002739146, "events": 57, "content": "\n\n- **Northern access code:** NORTH-49273\n- **Southern access code:** SOUTH-18362\n- **Invoice verification code:** SPARK-82741", "reasoning": "The user wants three pieces of information: the northern access code, the southern access code, and the invoice verification code. These are embedded in the text and the image.\n\nScanning the text: \"The northern access code is NORTH-49273.\" appears after Item 000019. \"The southern access code is SOUTH-18362.\" appears after Item 002499.\n\nFrom the image: \"Verification: SPARK-82741\".\n\nSo the answer is straightforward extraction. No computation needed. I'll present all three clearly. The amounts A/B/C are distractors not requested.\n\nI need a self-contained response.\n", "sha256": "86bc2cd86b8219571bdb62f3c793782b1b6de944ea376483b0595f97b9bb55aa"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08649174699530704, "elapsed_s": 0.44836226699408144, "decode_tps": 174.09541954457458, "max_stream_gap_s": 0.020703647001937497, "events": 19, "content": "\n\n731", "reasoning": "The user wants me to compute 43 * 17 and reply with only the integer.\n\n43 * 17 = 43 * 10 + 43 * 7 = 430 + 301 = 731\n", "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.5161374739982421, "elapsed_s": 0.8586695900012273, "decode_tps": 183.9243593714609, "max_stream_gap_s": 0.021041070998762734, "events": 18, "content": "\n\n765", "reasoning": "The user wants me to compute 45 * 17 and reply with only the integer.\n\n45 * 17 = 45 * 10 + 45 * 7 = 450 + 315 = 765\n", "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.9265989400009857, "elapsed_s": 1.1674528340008692, "decode_tps": 174.37957635852183, "max_stream_gap_s": 0.02082027800497599, "events": 13, "content": "\n\n748", "reasoning": "44 * 17 = 44 * 10 + 44 * 7 = 440 + 308 = 748\n", "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.2341738940012874, "elapsed_s": 1.5963426840025932, "decode_tps": 173.95204042781506, "max_stream_gap_s": 0.02091294499405194, "events": 19, "content": "\n\n782", "reasoning": "The user wants me to compute 46 * 17 and reply with just the integer.\n\n46 * 17 = 46 * 10 + 46 * 7 = 460 + 322 = 782\n", "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} diff --git a/audit/tuning/steps-123/mtp3-b2048/screen.jsonl.log b/audit/tuning/steps-123/mtp3-b2048/screen.jsonl.log new file mode 100644 index 0000000..7f6403c --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/screen.jsonl.log @@ -0,0 +1,38 @@ +{"name": "arithmetic-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 141, "completion_tokens": 82, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.6857045710057719, "elapsed_s": 1.1168529220012715, "decode_tps": 187.87036947485734, "max_stream_gap_s": 0.021299945998180192, "events": 23, "sha256": "d929fd317464b07ece811a135b5937af8eb36ea0b1e88ec82bc5137102d0a62c"} +{"name": "arithmetic-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 96, "completion_tokens": 37, "completion_tokens_details": {"reasoning_tokens": 30}}, "ttft_s": 0.09021790199767565, "elapsed_s": 0.30122662099893205, "decode_tps": 170.60906378842878, "max_stream_gap_s": 0.0215804309991654, "events": 10, "sha256": "5cd050422245af84010bf1c71082001dcd6d880bdd6dd7002e69a8402ab6c73c"} +{"name": "arithmetic-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 125, "completion_tokens": 66, "completion_tokens_details": {"reasoning_tokens": 59}}, "ttft_s": 0.0900614569982281, "elapsed_s": 0.43906909199722577, "decode_tps": 186.2423439538412, "max_stream_gap_s": 0.021113369999511633, "events": 18, "sha256": "c6dabe0673abae92ebdac3d245deb7c9ca89c874d316c46b8ce09fde923a752f"} +{"name": "arithmetic-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 98, "completion_tokens": 39, "completion_tokens_details": {"reasoning_tokens": 31}}, "ttft_s": 0.09134426800301298, "elapsed_s": 0.293823849002365, "decode_tps": 187.67324493881486, "max_stream_gap_s": 0.020979296998120844, "events": 11, "sha256": "4f382303f23d66fb4abcb57f3ad914e7f87831db4a3757ab5fdab83d25b58b1b"} +{"name": "arithmetic-4", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 131, "completion_tokens": 72, "completion_tokens_details": {"reasoning_tokens": 64}}, "ttft_s": 0.08995563200005563, "elapsed_s": 0.4795566350003355, "decode_tps": 182.23772385911695, "max_stream_gap_s": 0.021156058995984495, "events": 20, "sha256": "5c8a761b69d32970d9e4021e9cddda2caf44a6201a9bf34e503844cd2ce1dc0c"} +{"name": "arithmetic-5", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 101, "completion_tokens": 42, "completion_tokens_details": {"reasoning_tokens": 34}}, "ttft_s": 0.08942628499789862, "elapsed_s": 0.3126287199993385, "decode_tps": 183.68975230819282, "max_stream_gap_s": 0.02103689500654582, "events": 12, "sha256": "38ecdb08b2fcabd6a6a91e18d08f90ca325aae7f5c9ee565aa360e4d3cb610e9"} +{"name": "arithmetic-6", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 130, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 63}}, "ttft_s": 0.08937776899983874, "elapsed_s": 0.4788391099937144, "decode_tps": 179.73542591253175, "max_stream_gap_s": 0.02123644899984356, "events": 20, "sha256": "7ac565b6322b904ff447bc9ef3ed069ebc4e48b8a51e72f9e18919bd8e3000d1"} +{"name": "arithmetic-7", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 153, "completion_tokens": 94, "completion_tokens_details": {"reasoning_tokens": 86}}, "ttft_s": 0.09002831100224284, "elapsed_s": 0.6215811549991486, "decode_tps": 174.95908647709422, "max_stream_gap_s": 0.021011450000514742, "events": 26, "sha256": "e7b73eb708d2c4c7d2fc5dca2aed40a571c8b0445cd1867d916b47f57ec4cac0"} +{"name": "arithmetic-8", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 144, "completion_tokens": 85, "completion_tokens_details": {"reasoning_tokens": 77}}, "ttft_s": 0.08917911599564832, "elapsed_s": 0.5392506789939944, "decode_tps": 186.6369859948443, "max_stream_gap_s": 0.021216852001089137, "events": 23, "sha256": "9d93671c15eac2005a27c4fe38e5d4285b3feca77fa1e90f138c71a2b5164db8"} +{"name": "arithmetic-9", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 59, "total_tokens": 143, "completion_tokens": 84, "completion_tokens_details": {"reasoning_tokens": 76}}, "ttft_s": 0.09073446100228466, "elapsed_s": 0.539050845996826, "decode_tps": 185.13711025977918, "max_stream_gap_s": 0.020977412998036016, "events": 23, "sha256": "6b95a8b82a4ba5a8335b5f10899663aeebe71fed7e5673836b82ec74cfa1cbee"} +{"name": "arithmetic-10", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 60, "total_tokens": 143, "completion_tokens": 83, "completion_tokens_details": {"reasoning_tokens": 75}}, "ttft_s": 0.08987670400529169, "elapsed_s": 0.5182491330051562, "decode_tps": 191.42221685799936, "max_stream_gap_s": 0.02107875300134765, "events": 22, "sha256": "662f32768d742ea67e45b8109190fc2ba55c429346f4be9b9492ff457cdacc48"} +{"name": "arithmetic-11", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 61, "total_tokens": 151, "completion_tokens": 90, "completion_tokens_details": {"reasoning_tokens": 82}}, "ttft_s": 0.09002829199744156, "elapsed_s": 0.5601620039960835, "decode_tps": 189.3078452545796, "max_stream_gap_s": 0.021013982994190883, "events": 24, "sha256": "cdbf70494a3484ed891137af4bad12b7417009befd921930fc1fa5e5e3907e27"} +{"name": "json", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 66, "total_tokens": 109, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 26}}, "ttft_s": 0.1072728509971057, "elapsed_s": 0.40980977100116434, "decode_tps": 138.82603154496502, "max_stream_gap_s": 0.020985293995181564, "events": 16, "sha256": "2acba5868b21fa90ff7d281b66898b4ed66b6ff2f42a3d5a1d5bd11569806949"} +{"name": "multiturn", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 82, "total_tokens": 145, "completion_tokens": 63, "completion_tokens_details": {"reasoning_tokens": 51}}, "ttft_s": 0.09010298399516614, "elapsed_s": 0.5101531999971485, "decode_tps": 147.60140011380784, "max_stream_gap_s": 0.020926092998706736, "events": 22, "sha256": "dd12a39a5152c7bd1ce0e00fedc430e10a83c25c8a7850e9be06dcc0b4fe0353"} +{"name": "image-ocr", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 3135, "total_tokens": 3285, "completion_tokens": 150, "completion_tokens_details": {"reasoning_tokens": 111}}, "ttft_s": 0.7640111340006115, "elapsed_s": 1.6888792590034427, "decode_tps": 161.10404929302098, "max_stream_gap_s": 0.026646995996998157, "events": 45, "sha256": "59acfe36a1bac3e2c16c79d5c3712068c6d77030f2161a2297c21a795d2a2f9c"} +{"name": "video-120s", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12069, "total_tokens": 12191, "completion_tokens": 122, "completion_tokens_details": {"reasoning_tokens": 108}}, "ttft_s": 2.5409359719997155, "elapsed_s": 3.243877524000709, "decode_tps": 172.1337992548049, "max_stream_gap_s": 0.03231711799890036, "events": 35, "sha256": "34f1ae68e14f16d3cbbeed12b616b538f07ebec97e9ad601ced414e453c29f65"} +{"name": "multi-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 12356, "total_tokens": 12427, "completion_tokens": 71, "completion_tokens_details": {"reasoning_tokens": 60}}, "ttft_s": 1.463653916005569, "elapsed_s": 1.8863289360015187, "decode_tps": 165.61186890266376, "max_stream_gap_s": 0.020653829000366386, "events": 22, "sha256": "c4d4ac93f59330deca1f3c5fe9842a60a2ac31170fdbc40f311385c571ed9024"} +{"name": "warmup-code", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 333, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.08850599199649878, "elapsed_s": 2.1877699620017665, "decode_tps": 121.47114590803943, "max_stream_gap_s": 0.0206061229982879, "events": 106, "sha256": "ae01d596ec7a45f3c398b1526e5f345c2a100644fe1e8d2975c98d406465b7ad"} +{"name": "perf-code-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.08851322199916467, "elapsed_s": 16.576275200000964, "decode_tps": 124.152689900008, "max_stream_gap_s": 0.021029635005106684, "events": 824, "sha256": "9e8927825d4c68bc3bb71bc5d5dc66a8a35e5961a52ff04a44c5211082f2d4c9"} +{"name": "perf-code-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.09001792799972463, "elapsed_s": 16.722888370000874, "decode_tps": 123.06955718424507, "max_stream_gap_s": 0.021017879000282846, "events": 831, "sha256": "649658539a14ad95cd3aba4a031955cf646f9c2cef5e606a2f3f25be99a3962b"} +{"name": "perf-code-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 77, "total_tokens": 2125, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07350860699807527, "elapsed_s": 16.641791551002825, "decode_tps": 123.54931448950835, "max_stream_gap_s": 0.021029645002272446, "events": 828, "sha256": "4af2092c8a16885a6236f0c3adc0227eb9dd8079ec421ba9c2eb5cc404b3440b"} +{"name": "warmup-chinese", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07293527999718208, "elapsed_s": 1.9857562760007568, "decode_tps": 133.31095828243588, "max_stream_gap_s": 0.020934283995302394, "events": 97, "sha256": "c7ecb8327c0218df2d04be1c67cec8ee285dd526519e8615316908f083855b7c"} +{"name": "perf-chinese-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08870576600020286, "elapsed_s": 15.336975530000927, "decode_tps": 134.24473934955645, "max_stream_gap_s": 0.020986846000596415, "events": 765, "sha256": "d4b632309eb489d41b5803e52e0ae083ffe5c0f801ac927326c6037eb95c6474"} +{"name": "perf-chinese-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.08885356999962823, "elapsed_s": 15.44073144599679, "decode_tps": 133.33873657244945, "max_stream_gap_s": 0.02105036300054053, "events": 770, "sha256": "795e37bad007369fe0ce6ce658b4ebf659222130ea24f338f6f342b4ce6aff47"} +{"name": "perf-chinese-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 182}}, "ttft_s": 0.07399053900007857, "elapsed_s": 15.320730624000134, "decode_tps": 134.25820789152598, "max_stream_gap_s": 0.020978505002858583, "events": 765, "sha256": "ece641cb2f77db4659a64c75f36366c11b83095a29fc22cb320a1b306b6da1f5"} +{"name": "warmup-reasoning", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 335, "completion_tokens": 256, "completion_tokens_details": {"reasoning_tokens": 256}}, "ttft_s": 0.09028758799831849, "elapsed_s": 2.1541953229971114, "decode_tps": 123.55203465534235, "max_stream_gap_s": 0.02097705299820518, "events": 105, "sha256": "d557c88f4473c4da0f26d8a2bfdb8aae0734ac1d790d7d3f74c26151b5cbe509"} +{"name": "perf-reasoning-0", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.0739625900023384, "elapsed_s": 16.64171508599975, "decode_tps": 123.55327015505169, "max_stream_gap_s": 0.021100942998600658, "events": 828, "sha256": "eec62c20975cfeaada4ad14ef0f726db52eaff6b498dcbbf3f5f515c2c5725c8"} +{"name": "perf-reasoning-1", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07516271799977403, "elapsed_s": 16.646220769005595, "decode_tps": 123.52862404436223, "max_stream_gap_s": 0.021613906996208243, "events": 828, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "perf-reasoning-2", "passed": true, "finish": "length", "usage": {"prompt_tokens": 79, "total_tokens": 2127, "completion_tokens": 2048, "completion_tokens_details": {"reasoning_tokens": 2048}}, "ttft_s": 0.07455741699959617, "elapsed_s": 16.66266934000305, "decode_tps": 123.40162699055197, "max_stream_gap_s": 0.021173682995140553, "events": 829, "sha256": "10121a80459a4b8f8a0773b132bcdb6398463e07aa0645ee478ad590cd9def46"} +{"name": "long-context-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126125, "completion_tokens": 130, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.516383726004278, "elapsed_s": 15.218214432999957, "decode_tps": 183.80500983237002, "max_stream_gap_s": 0.02093203100230312, "events": 36, "sha256": "9a2e65aa6f54431f5ea3bfa29bcdd76256bef03427fca286f07cc3232bfdc529"} +{"name": "long-context-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126119, "completion_tokens": 124, "completion_tokens_details": {"reasoning_tokens": 104}}, "ttft_s": 14.510828671998752, "elapsed_s": 15.23376362700219, "decode_tps": 170.13978802479537, "max_stream_gap_s": 0.021043553999334108, "events": 37, "sha256": "1df90f428d94a645b711abef54df465c7fa0db783e3b3602ea5b4aa73f29aaab"} +{"name": "long-context-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 125995, "total_tokens": 126114, "completion_tokens": 119, "completion_tokens_details": {"reasoning_tokens": 99}}, "ttft_s": 14.506387795998307, "elapsed_s": 15.205817506997846, "decode_tps": 168.7088754513573, "max_stream_gap_s": 0.020960481000656728, "events": 36, "sha256": "109b3120d1bf2f56bd0c1ab331e7692f8a0a37b091eb2adb8d7a0a54731dba54"} +{"name": "long-context-image", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 111670, "total_tokens": 111861, "completion_tokens": 191, "completion_tokens_details": {"reasoning_tokens": 146}}, "ttft_s": 12.83541851100017, "elapsed_s": 13.968540412002767, "decode_tps": 167.67834054913806, "max_stream_gap_s": 0.021041211002739146, "events": 57, "sha256": "86bc2cd86b8219571bdb62f3c793782b1b6de944ea376483b0595f97b9bb55aa"} +{"name": "queued-0", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.08649174699530704, "elapsed_s": 0.44836226699408144, "decode_tps": 174.09541954457458, "max_stream_gap_s": 0.020703647001937497, "events": 19, "sha256": "c775e82fbe7367311970d698f4880b3e86da93aa21640d45d95ee9475e335166"} +{"name": "queued-2", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 0.5161374739982421, "elapsed_s": 0.8586695900012273, "decode_tps": 183.9243593714609, "max_stream_gap_s": 0.021041070998762734, "events": 18, "sha256": "4e0dac459752edb516d9c610865ec672d6805ddc71af097d21b28f7c53179b47"} +{"name": "queued-1", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 98, "completion_tokens": 43, "completion_tokens_details": {"reasoning_tokens": 37}}, "ttft_s": 0.9265989400009857, "elapsed_s": 1.1674528340008692, "decode_tps": 174.37957635852183, "max_stream_gap_s": 0.02082027800497599, "events": 13, "sha256": "577bfda055d6b1f136ea30fd3c42b3954036dd654d2eb7cace48396c49aa115a"} +{"name": "queued-3", "passed": true, "finish": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 119, "completion_tokens": 64, "completion_tokens_details": {"reasoning_tokens": 58}}, "ttft_s": 1.2341738940012874, "elapsed_s": 1.5963426840025932, "decode_tps": 173.95204042781506, "max_stream_gap_s": 0.02091294499405194, "events": 19, "sha256": "fbb92f72869a68308179711e06612346b886a880c1f5fbd390eb0731790dbd0c"} +SUITE_COMPLETE diff --git a/audit/tuning/steps-123/mtp3-b2048/server.log b/audit/tuning/steps-123/mtp3-b2048/server.log new file mode 100644 index 0000000..ece5d6c --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/server.log @@ -0,0 +1,345 @@ +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █ █ █▄ ▄█ +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] █▄█▀ █ █ █ █ model /model +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 15:01:22 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 3}, 'compilation_config': {'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:01:22 [model.py:691] Resolved architecture: Qwen4ExpForConditionalGeneration +(APIServer pid=1) INFO 09-18 15:01:22 [model.py:2024] Using max model len 131072 +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:01:25 [cache.py:345] Using fp8 data type to store kv cache. It reduces the GPU memory footprint and boosts the performance. Meanwhile, it may cause accuracy drop without a proper scaling factor +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) INFO 09-18 15:01:25 [model.py:691] Resolved architecture: Qwen4ExpMTP +(APIServer pid=1) INFO 09-18 15:01:25 [model.py:2024] Using max model len 262144 +(APIServer pid=1) WARNING 09-18 15:01:25 [speculative.py:1360] Enabling num_speculative_tokens > 1 will run multiple times of forward on same MTP layer,which may result in lower acceptance rate +(APIServer pid=1) INFO 09-18 15:01:25 [speculative.py:1653] Overriding draft model max model len from 262144 to 131072 +(APIServer pid=1) INFO 09-18 15:01:25 [config.py:625] Mamba cache mode is set to 'align' for Qwen4ExpForConditionalGeneration by default when prefix caching is enabled +(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:1271] Resolved Engram configuration: EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False) +(APIServer pid=1) INFO 09-18 15:01:25 [vllm.py:781] Auto-enabling VLLM_USE_BREAKABLE_CUDAGRAPH=1. Set VLLM_USE_BREAKABLE_CUDAGRAPH=0 to opt out. +(APIServer pid=1) INFO 09-18 15:01:25 [kernel.py:408] Final IR op priority after setting platform defaults: IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']) +(APIServer pid=1) WARNING 09-18 15:01:25 [vllm.py:2152] max_num_scheduled_tokens is set to 2048 based on the speculative decoding settings. This may lead to suboptimal performance. Consider increasing max_num_batched_tokens to accommodate the additional draft token slots, or decrease num_speculative_tokens. +(APIServer pid=1) INFO 09-18 15:01:25 [compilation.py:331] Enabled custom fusions: norm_quant, act_quant +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) INFO 09-18 15:01:38 [core.py:123] Initializing a V1 LLM engine (v0.3.1.dev3+g0bfc7a15d) with config: model='/model', speculative_config=SpeculativeConfig(method='mtp', model='/model', num_spec_tokens=3), tokenizer='/model', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=131072, download_dir=None, load_format=safetensors, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, decode_context_parallel_size=1, dcp_comm_backend=ag_rs, disable_custom_all_reduce=False, quantization=modelopt_mixed, quantization_config=None, enforce_eager=False, enable_return_routed_experts=False, kv_cache_dtype=fp8, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='qwen3', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, per_request_spec_decode_metrics='none', kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False, enable_mfu_metrics=False, enable_mm_processor_stats=False, enable_logging_iteration_details=False, jit_monitor_mode='warn', jit_monitor_verbose=False), seed=0, served_model_name=qwen3.8-flash-next, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['+quant_fp8', 'all', '+quant_fp8'], 'ir_enable_torch_wrap': False, 'splitting_ops': [], 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': [], 'compile_ranges_endpoints': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 4], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': True, 'fuse_act_quant': True, 'fuse_attn_quant': False, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False, 'enable_qk_norm_rope_fusion': False, 'fuse_rope_kvcache_cat_mla': False, 'fuse_act_padding': False, 'fuse_qk_norm_rope_kvcache': False}, 'max_cudagraph_capture_size': 4, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': False, 'static_all_moe_layers': []}, kernel_config=KernelConfig(ir_op_priority=IrOpPriorityConfig(rms_norm=['vllm_c', 'native'], fused_add_rms_norm=['vllm_c', 'native'], gelu_and_mul_sparse=['triton', 'native']), enable_flashinfer_autotune=False, enable_cutedsl_warmup=True, enable_jit_warmup=True, moe_backend='auto', sparse_indexer_topk_backend='auto', linear_backend='auto', linear_backend_per_quant=None) +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:1825] world_size=1 rank=0 local_rank=0 distributed_init_method=file:///tmp/vllm_dist_f7efa1f40cb14f80867628d23379f752 backend=nccl +(EngineCore pid=112) INFO 09-18 15:01:39 [parallel_state.py:2269] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, ETP rank 0, EP rank 0, EPLB rank N/A +(EngineCore pid=112) INFO 09-18 15:01:39 [gpu_worker.py:441] Using V2 Model Runner +(EngineCore pid=112) INFO 09-18 15:01:40 [model_runner.py:387] Loading model from scratch... +(EngineCore pid=112) INFO 09-18 15:01:40 [cuda.py:595] Using backend AttentionBackendEnum.FLASH_ATTN for vit attention +(EngineCore pid=112) INFO 09-18 15:01:40 [mm_encoder_attention.py:372] Using AttentionBackendEnum.FLASH_ATTN for MMEncoderAttention. +(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:176] Using FlashInfer GDN prefill kernel (requested=auto, head_k_dim=128). +(EngineCore pid=112) INFO 09-18 15:01:40 [qwen_gdn_linear_attn.py:528] GDN decode kernel: cuda +(EngineCore pid=112) INFO 09-18 15:01:42 [nvfp4.py:302] Using 'FLASHINFER_CUTLASS' NvFp4 MoE backend out of potential backends: ['FLASHINFER_TRTLLM', 'FLASHINFER_CUTEDSL', 'FLASHINFER_CUTEDSL_BATCHED', 'FLASHINFER_CUTLASS', 'VLLM_CUTLASS', 'MARLIN', 'HUMMING', 'EMULATION']. +(APIServer pid=1) [transformers] Qwen3VL video processing does not apply the per-frame pixel cap the reference implementation (qwen-vl-utils) applies, so some videos cost far more tokens than they would there. In v5.22 the capped behavior will become the default and `cap_pixels_per_frame` will be removed. Pass `cap_pixels_per_frame=True` to adopt the reference behavior now, or `False` to keep the current behavior and silence this warning. +(APIServer pid=1) INFO 09-18 15:01:44 [base.py:261] Multi-modal warmup completed in 12.297s +(APIServer pid=1) INFO 09-18 15:01:45 [base.py:261] Readonly multi-modal warmup completed in 1.333s +(EngineCore pid=112) INFO 09-18 15:02:16 [ngram_embedding.py:720] Initialized PLE embedding language_model.model.layers.1.ple.ple_embedding.ngram_embedding: quantization_method=Qwen4ExpPLEFp8EmbeddingMethod, weight_dtype=torch.float8_e4m3fn, weight_device=cpu, pinned=True +(EngineCore pid=112) INFO 09-18 15:02:16 [flash_attn.py:1115] Using FlashAttention version 2 +(EngineCore pid=112) WARNING 09-18 15:02:16 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect +(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.27 GiB. +(EngineCore pid=112) INFO 09-18 15:02:16 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch. +(EngineCore pid=112) +Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00= mamba page size. +(EngineCore pid=112) INFO 09-18 15:03:06 [interface.py:942] Padding mamba page size by 0.25% to ensure that mamba page size and attention page size are exactly equal. +(EngineCore pid=112) INFO 09-18 15:03:06 [utils.py:320] Using BLNHC KV cache layout. +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] The `use_fast` parameter is deprecated and will be removed in a future version. Use `backend="torchvision"` instead of `use_fast=True`, or `backend="pil"` instead of `use_fast=False`. +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_section', 'mrope_interleaved'} +(EngineCore pid=112) INFO 09-18 15:03:11 [encoder_runner.py:131] Encoder cache will be initialized with a budget of 16384 tokens, and profiled with 1 image items of the maximum feature size. +(EngineCore pid=112) WARNING 09-18 15:03:33 [kv_cache_utils.py:2219] Speculative decoding (method=mtp) is enabled but no KV cache group could be identified as the draft model's. +(EngineCore pid=112) WARNING 09-18 15:03:33 [compilation.py:1415] CUDAGraphMode.FULL is not supported with GDNAttentionBackend backend (support: AttentionCGSupport.UNIFORM_BATCH); setting cudagraph_mode=FULL_DECODE_ONLY +(EngineCore pid=112) INFO 09-18 15:03:33 [speculator.py:119] Fused multi-step draft decode is not supported by attention backend(s) QWEN4_EXP_EXP_QSA_STATE; falling back to rebuilding attention metadata between draft steps. +(EngineCore pid=112) +Capturing CUDA graphs (FULL): 0%| | 0/1 [00:00 262144). Running this sequence through the model will result in indexing errors +(APIServer pid=1) INFO: 172.21.0.1:47280 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47290 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47300 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47314 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47322 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47324 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47340 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47356 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47372 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47378 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47382 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47384 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47398 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47414 - "POST /tokenize HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:47418 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:06:55 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 70.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO: 127.0.0.1:40192 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:41722 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:07:05 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 13.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 48.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 15:07:05 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.69, Accepted throughput: 4.70 tokens/s, Drafted throughput: 5.25 tokens/s, Accepted: 94 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.943, 0.943, 0.800, Avg Draft acceptance rate: 89.5% +(APIServer pid=1) INFO 09-18 15:07:15 [loggers.py:323] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 90.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO: 172.21.0.1:45834 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:07:25 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.9 tokens/s, Avg generation throughput: 12.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 68.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 66.7% +(APIServer pid=1) INFO 09-18 15:07:25 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.47, Accepted throughput: 4.45 tokens/s, Drafted throughput: 5.40 tokens/s, Accepted: 89 tokens, Drafted: 108 tokens, Per-position acceptance rate: 0.917, 0.833, 0.722, Avg Draft acceptance rate: 82.4% +(APIServer pid=1) INFO: 127.0.0.1:36836 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:36034 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:07:35 [loggers.py:323] Engine 000: Avg prompt throughput: 12599.4 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:07:35 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.43, Accepted throughput: 8.50 tokens/s, Drafted throughput: 10.50 tokens/s, Accepted: 85 tokens, Drafted: 105 tokens, Per-position acceptance rate: 0.886, 0.857, 0.686, Avg Draft acceptance rate: 81.0% +(APIServer pid=1) INFO 09-18 15:07:45 [loggers.py:323] Engine 000: Avg prompt throughput: 11167.1 tokens/s, Avg generation throughput: 2.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 81.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:07:45 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.33, Accepted throughput: 1.40 tokens/s, Drafted throughput: 1.80 tokens/s, Accepted: 14 tokens, Drafted: 18 tokens, Per-position acceptance rate: 0.833, 0.833, 0.667, Avg Draft acceptance rate: 77.8% +(APIServer pid=1) INFO: 172.21.0.1:48276 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48278 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48294 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48308 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48318 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48332 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48334 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:48342 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO 09-18 15:07:55 [loggers.py:323] Engine 000: Avg prompt throughput: 64.3 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 71.4% +(APIServer pid=1) INFO 09-18 15:07:55 [metrics.py:120] SpecDecoding metrics: Mean acceptance length: 3.42, Accepted throughput: 110.00 tokens/s, Drafted throughput: 136.50 tokens/s, Accepted: 1100 tokens, Drafted: 1365 tokens, Per-position acceptance rate: 0.923, 0.804, 0.690, Avg Draft acceptance rate: 80.6% +(APIServer pid=1) INFO: 127.0.0.1:49226 - "GET /health HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37666 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37670 - "POST /v1/chat/completions HTTP/1.1" 200 OK +(APIServer pid=1) INFO: 172.21.0.1:37676 - "POST /v1/chat/completions HTTP/1.1" 200 OK diff --git a/audit/tuning/steps-123/mtp3-b2048/status.json b/audit/tuning/steps-123/mtp3-b2048/status.json new file mode 100644 index 0000000..256e394 --- /dev/null +++ b/audit/tuning/steps-123/mtp3-b2048/status.json @@ -0,0 +1,5 @@ +{ + "startup": true, + "records": 44, + "failures": [] +} \ No newline at end of file diff --git a/audit/tuning/steps-123/original.yaml b/audit/tuning/steps-123/original.yaml new file mode 100644 index 0000000..693de66 --- /dev/null +++ b/audit/tuning/steps-123/original.yaml @@ -0,0 +1,46 @@ +name: qwen38-flash-rtx6000d +services: + vllm: + image: local/qwen38-flash-6000d:0bfc7a15 + build: . + container_name: qwen38-flash-6000d + gpus: all + shm_size: 8g + mem_limit: 210g + memswap_limit: 210g + restart: unless-stopped + ports: + - "${BIND_ADDRESS:-127.0.0.1}:${API_PORT:-8000}:8000" + environment: + HF_HUB_OFFLINE: "1" + VLLM_USE_V2_MODEL_RUNNER: "1" + VLLM_WORKER_MULTIPROC_METHOD: spawn + OMP_NUM_THREADS: "8" + TORCHINDUCTOR_COMPILE_THREADS: "2" + CUTE_DSL_ARCH: sm_120a + volumes: + - /data/flash-next/models/nvidia-Qwen3.8-Flash-Next-NVFP4:/model:ro + - /data/flash-next/cache:/root/.cache + - ./secrets/api-key:/run/secrets/api-key:ro + entrypoint: ["/bin/bash", "-lc"] + command: + - >- + exec vllm serve /model + --served-model-name qwen3.8-flash-next + --host 0.0.0.0 --port 8000 + --tensor-parallel-size 1 --dtype bfloat16 + --engram-config '{"cpu_offload":true}' + --kv-cache-dtype fp8 --gpu-memory-utilization 0.985 + --max-model-len 131072 --max-num-seqs 1 --max-num-batched-tokens 2048 + --enable-chunked-prefill --enable-prefix-caching + --speculative-config '{"method":"mtp","num_speculative_tokens":2}' + --compilation-config '{"mode":0,"cudagraph_mode":"FULL","cudagraph_capture_sizes":[1,3]}' --no-enable-flashinfer-autotune + --load-format safetensors + --reasoning-parser qwen3 --tool-call-parser qwen3_xml --enable-auto-tool-choice + --api-key "$$(cat /run/secrets/api-key)" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 5 + start_period: 30m diff --git a/audit/tuning/steps-123/resources.jsonl b/audit/tuning/steps-123/resources.jsonl new file mode 100644 index 0000000..03a0fe7 --- /dev/null +++ b/audit/tuning/steps-123/resources.jsonl @@ -0,0 +1,624 @@ +{"time": 1789742993.8930721, "available_gib": 172.44721603393555, "oom_kill": 0, "gpu": "84797, 85651, 0, 34", "stop": false} +{"time": 1789742995.9187403, "available_gib": 172.4483184814453, "oom_kill": 0, "gpu": "84797, 85651, 100, 40", "stop": false} +{"time": 1789742997.9435408, "available_gib": 172.43352890014648, "oom_kill": 0, "gpu": "84797, 85651, 88, 41", "stop": false} +{"time": 1789742999.969222, "available_gib": 172.44828033447266, "oom_kill": 0, "gpu": "84797, 85651, 100, 42", "stop": false} +{"time": 1789743001.994626, "available_gib": 172.44638061523438, "oom_kill": 0, "gpu": "84797, 85651, 100, 43", "stop": false} +{"time": 1789743004.0204384, "available_gib": 172.44339752197266, "oom_kill": 0, "gpu": "84797, 85651, 100, 43", "stop": false} +{"time": 1789743006.0460222, "available_gib": 172.41333389282227, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false} +{"time": 1789743008.0727768, "available_gib": 172.41802978515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false} +{"time": 1789743010.0991132, "available_gib": 172.4132423400879, "oom_kill": 0, "gpu": "84797, 85651, 100, 44", "stop": false} +{"time": 1789743012.1248758, "available_gib": 172.42271041870117, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false} +{"time": 1789743014.15081, "available_gib": 172.42755508422852, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false} +{"time": 1789743016.175784, "available_gib": 172.4231948852539, "oom_kill": 0, "gpu": "84797, 85651, 100, 45", "stop": false} +{"time": 1789743018.2006702, "available_gib": 172.4263038635254, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false} +{"time": 1789743020.2257159, "available_gib": 172.4193344116211, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false} +{"time": 1789743022.2507591, "available_gib": 172.41914749145508, "oom_kill": 0, "gpu": "84797, 85651, 100, 46", "stop": false} +{"time": 1789743024.2756846, "available_gib": 172.4089813232422, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false} +{"time": 1789743026.3011591, "available_gib": 172.41101837158203, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false} +{"time": 1789743028.3262846, "available_gib": 172.41080474853516, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false} +{"time": 1789743030.3521566, "available_gib": 172.41079330444336, "oom_kill": 0, "gpu": "84797, 85651, 100, 47", "stop": false} +{"time": 1789743032.3783853, "available_gib": 172.41777420043945, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false} +{"time": 1789743034.403828, "available_gib": 172.42018127441406, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false} +{"time": 1789743036.4298394, "available_gib": 172.41797256469727, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false} +{"time": 1789743038.4559314, "available_gib": 172.42116928100586, "oom_kill": 0, "gpu": "84797, 85651, 100, 48", "stop": false} +{"time": 1789743040.4814112, "available_gib": 172.42036056518555, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743042.5068018, "available_gib": 172.41886138916016, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743044.5317273, "available_gib": 172.41489791870117, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743046.5577147, "available_gib": 172.41412353515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743048.5837293, "available_gib": 172.41419219970703, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743050.6099472, "available_gib": 172.40557861328125, "oom_kill": 0, "gpu": "84797, 85651, 100, 49", "stop": false} +{"time": 1789743052.635809, "available_gib": 172.40662002563477, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743054.6614828, "available_gib": 172.40903854370117, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743056.686891, "available_gib": 172.40744018554688, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743058.7122042, "available_gib": 172.4039192199707, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743060.737155, "available_gib": 172.4096450805664, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743062.762888, "available_gib": 172.41356658935547, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743064.789021, "available_gib": 172.41518020629883, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743066.8147395, "available_gib": 172.4062843322754, "oom_kill": 0, "gpu": "84797, 85651, 100, 50", "stop": false} +{"time": 1789743068.8408678, "available_gib": 172.40911483764648, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743070.8669722, "available_gib": 172.4068145751953, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743072.893002, "available_gib": 172.40983963012695, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743074.9192512, "available_gib": 172.40817642211914, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743076.945159, "available_gib": 172.40537643432617, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743078.9712658, "available_gib": 172.40187454223633, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743080.9970002, "available_gib": 172.39935302734375, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743083.0224788, "available_gib": 172.40352249145508, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743085.0485544, "available_gib": 172.4073257446289, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743087.0739734, "available_gib": 172.40684509277344, "oom_kill": 0, "gpu": "84797, 85651, 100, 51", "stop": false} +{"time": 1789743089.0995305, "available_gib": 172.40550994873047, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743091.125159, "available_gib": 172.4019660949707, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743093.151235, "available_gib": 172.40221786499023, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743095.1767147, "available_gib": 172.40421295166016, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743097.2028806, "available_gib": 172.4161262512207, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743099.2291465, "available_gib": 172.4316177368164, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743101.2551146, "available_gib": 172.43370056152344, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743103.2816598, "available_gib": 172.43728256225586, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743105.307239, "available_gib": 172.43943786621094, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743107.3330648, "available_gib": 172.44949340820312, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743109.359068, "available_gib": 172.44104385375977, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743111.385109, "available_gib": 172.43330764770508, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743113.4113765, "available_gib": 172.43706893920898, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743115.4370093, "available_gib": 172.44180297851562, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743117.4625585, "available_gib": 172.44410705566406, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743119.4885128, "available_gib": 172.43635177612305, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743121.5149198, "available_gib": 172.44395446777344, "oom_kill": 0, "gpu": "84797, 85651, 100, 52", "stop": false} +{"time": 1789743123.5413435, "available_gib": 172.44974899291992, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743125.5679505, "available_gib": 172.44622421264648, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743127.5935173, "available_gib": 172.44979095458984, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743129.6191273, "available_gib": 172.45713424682617, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743131.6455028, "available_gib": 172.44577407836914, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743133.672497, "available_gib": 172.45316314697266, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743135.697847, "available_gib": 172.45311737060547, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743137.7242105, "available_gib": 172.44537353515625, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743139.7507234, "available_gib": 172.4357032775879, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743141.7774155, "available_gib": 172.44175720214844, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743143.8043108, "available_gib": 172.43401336669922, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743145.8301265, "available_gib": 172.44019317626953, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743147.855514, "available_gib": 172.45041275024414, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743149.8819315, "available_gib": 172.4467430114746, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743151.9083953, "available_gib": 172.45286178588867, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743153.934035, "available_gib": 172.4552459716797, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743155.9600441, "available_gib": 172.44754028320312, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743157.987035, "available_gib": 172.4474754333496, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743160.012713, "available_gib": 172.45933532714844, "oom_kill": 0, "gpu": "84797, 85651, 100, 53", "stop": false} +{"time": 1789743162.0334756, "available_gib": 172.38756561279297, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false} +{"time": 1789743164.0600572, "available_gib": 172.37825393676758, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false} +{"time": 1789743166.085538, "available_gib": 172.386474609375, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false} +{"time": 1789743168.1112113, "available_gib": 172.39423751831055, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false} +{"time": 1789743170.135376, "available_gib": 172.39533615112305, "oom_kill": 0, "gpu": "84797, 85651, 100, 59", "stop": false} +{"time": 1789743172.1629245, "available_gib": 172.40753555297852, "oom_kill": 0, "gpu": "84797, 85651, 100, 59", "stop": false} +{"time": 1789743174.1864367, "available_gib": 172.4146385192871, "oom_kill": 0, "gpu": "84797, 85651, 100, 60", "stop": false} +{"time": 1789743176.2113686, "available_gib": 172.4199676513672, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false} +{"time": 1789743178.235966, "available_gib": 172.41577529907227, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false} +{"time": 1789743180.2613661, "available_gib": 172.42212295532227, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false} +{"time": 1789743182.2844982, "available_gib": 172.41865921020508, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false} +{"time": 1789743184.3087924, "available_gib": 172.41510391235352, "oom_kill": 0, "gpu": "84797, 85651, 100, 61", "stop": false} +{"time": 1789743186.3337767, "available_gib": 172.42054748535156, "oom_kill": 0, "gpu": "84797, 85651, 100, 62", "stop": false} +{"time": 1789743188.3582373, "available_gib": 172.41769409179688, "oom_kill": 0, "gpu": "84797, 85651, 100, 62", "stop": false} +{"time": 1789743190.3825104, "available_gib": 172.41735076904297, "oom_kill": 0, "gpu": "84797, 85651, 100, 63", "stop": false} +{"time": 1789743192.4065006, "available_gib": 172.42127227783203, "oom_kill": 0, "gpu": "84797, 85651, 100, 63", "stop": false} +{"time": 1789743194.4317603, "available_gib": 172.42076873779297, "oom_kill": 0, "gpu": "84797, 85651, 100, 58", "stop": false} +{"time": 1789743196.4574146, "available_gib": 172.41965103149414, "oom_kill": 0, "gpu": "84797, 85651, 99, 58", "stop": false} +{"time": 1789743198.4829125, "available_gib": 172.43689346313477, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false} +{"time": 1789743200.508612, "available_gib": 172.41334915161133, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false} +{"time": 1789743202.5339994, "available_gib": 172.41661071777344, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false} +{"time": 1789743204.55841, "available_gib": 172.39996337890625, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false} +{"time": 1789743206.58313, "available_gib": 172.39534378051758, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false} +{"time": 1789743208.6087117, "available_gib": 172.3869171142578, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false} +{"time": 1789743210.6334622, "available_gib": 172.39297485351562, "oom_kill": 0, "gpu": "84797, 85651, 100, 56", "stop": false} +{"time": 1789743212.6587076, "available_gib": 172.39804077148438, "oom_kill": 0, "gpu": "84797, 85651, 100, 57", "stop": false} +{"time": 1789743214.679407, "available_gib": 174.36180877685547, "oom_kill": 0, "gpu": "84797, 85651, 0, 51", "stop": false} +{"time": 1789743216.6994534, "available_gib": 175.1852035522461, "oom_kill": 0, "gpu": "84797, 85651, 0, 50", "stop": false} +{"time": 1789743218.7198522, "available_gib": 175.1703224182129, "oom_kill": 0, "gpu": "84797, 85651, 0, 49", "stop": false} +{"time": 1789743220.740253, "available_gib": 175.17662811279297, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false} +{"time": 1789743222.7602718, "available_gib": 175.17460250854492, "oom_kill": 0, "gpu": "84797, 85651, 0, 48", "stop": false} +{"time": 1789743229.6084123, "available_gib": 190.91661071777344, "oom_kill": 0, "gpu": "3, 85651, 0, 45", "stop": false} +{"time": 1789743231.630355, "available_gib": 242.44229888916016, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false} +{"time": 1789743233.6530468, "available_gib": 242.35516738891602, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false} +{"time": 1789743235.6799445, "available_gib": 242.30181884765625, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false} +{"time": 1789743237.7048993, "available_gib": 242.2751808166504, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789743239.7298052, "available_gib": 242.2560272216797, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789743241.7549934, "available_gib": 242.20417404174805, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789743243.779685, "available_gib": 242.14367294311523, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false} +{"time": 1789743245.8037167, "available_gib": 242.09040069580078, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false} +{"time": 1789743247.8466423, "available_gib": 241.7714080810547, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789743249.8889477, "available_gib": 241.40981674194336, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789743251.9310393, "available_gib": 241.1487274169922, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789743253.9736116, "available_gib": 240.97250366210938, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789743256.0156634, "available_gib": 240.2674903869629, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789743258.0573394, "available_gib": 240.00533294677734, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789743260.1017191, "available_gib": 239.7233428955078, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789743262.1220598, "available_gib": 239.19343948364258, "oom_kill": 0, "gpu": "683, 85651, 2, 37", "stop": false} +{"time": 1789743264.1423042, "available_gib": 238.81260681152344, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789743266.1623287, "available_gib": 234.10787963867188, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789743268.1824176, "available_gib": 229.910400390625, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743270.2026012, "available_gib": 225.91469192504883, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743272.2226171, "available_gib": 221.90094757080078, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743274.2424407, "available_gib": 217.90217971801758, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743276.2622116, "available_gib": 213.95528411865234, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743278.2818873, "available_gib": 210.04045486450195, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743280.302443, "available_gib": 206.0911102294922, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743282.3223915, "available_gib": 202.18833541870117, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743284.342848, "available_gib": 198.23841857910156, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743286.3631046, "available_gib": 194.30224227905273, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743288.3831742, "available_gib": 190.38965225219727, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743290.4034154, "available_gib": 186.47941970825195, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743292.4239209, "available_gib": 182.56446838378906, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743294.4439209, "available_gib": 178.62585067749023, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743296.4635558, "available_gib": 174.67407989501953, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743298.484325, "available_gib": 173.4862289428711, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false} +{"time": 1789743300.5052922, "available_gib": 173.47106552124023, "oom_kill": 0, "gpu": "76995, 85651, 7, 36", "stop": false} +{"time": 1789743302.5263877, "available_gib": 173.44770050048828, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743304.5477655, "available_gib": 173.44190216064453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743306.572195, "available_gib": 173.42721557617188, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743308.5933337, "available_gib": 173.46649551391602, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743310.614528, "available_gib": 173.50944900512695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743312.6357765, "available_gib": 173.54594039916992, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743314.6571312, "available_gib": 173.57780838012695, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743316.6779072, "available_gib": 173.67039108276367, "oom_kill": 0, "gpu": "76995, 85651, 8, 36", "stop": false} +{"time": 1789743318.6988745, "available_gib": 173.6916618347168, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false} +{"time": 1789743320.7197983, "available_gib": 173.72816467285156, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false} +{"time": 1789743322.7410035, "available_gib": 173.74269485473633, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743324.7621593, "available_gib": 173.79139709472656, "oom_kill": 0, "gpu": "76995, 85651, 3, 36", "stop": false} +{"time": 1789743326.7830343, "available_gib": 173.8392562866211, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743328.8042507, "available_gib": 173.84291458129883, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743330.8253624, "available_gib": 173.9206657409668, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743332.846458, "available_gib": 173.9108123779297, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743334.8676717, "available_gib": 173.95557403564453, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743336.8889298, "available_gib": 173.98338317871094, "oom_kill": 0, "gpu": "76995, 85651, 6, 35", "stop": false} +{"time": 1789743338.9100575, "available_gib": 174.0068702697754, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false} +{"time": 1789743340.9305177, "available_gib": 174.01351165771484, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false} +{"time": 1789743342.9574733, "available_gib": 174.00801849365234, "oom_kill": 0, "gpu": "76995, 85651, 0, 35", "stop": false} +{"time": 1789743344.9781544, "available_gib": 173.8778190612793, "oom_kill": 0, "gpu": "82531, 85651, 0, 35", "stop": false} +{"time": 1789743346.9989328, "available_gib": 173.89619827270508, "oom_kill": 0, "gpu": "82531, 85651, 50, 36", "stop": false} +{"time": 1789743349.0198088, "available_gib": 173.94148635864258, "oom_kill": 0, "gpu": "82559, 85651, 0, 35", "stop": false} +{"time": 1789743351.040807, "available_gib": 173.9337158203125, "oom_kill": 0, "gpu": "79121, 85651, 0, 35", "stop": false} +{"time": 1789743353.0616393, "available_gib": 173.59209823608398, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false} +{"time": 1789743355.0827384, "available_gib": 173.4678955078125, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false} +{"time": 1789743357.104013, "available_gib": 172.85984802246094, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743359.1243138, "available_gib": 173.2513542175293, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743361.1446815, "available_gib": 173.23069763183594, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743363.1656852, "available_gib": 173.2288589477539, "oom_kill": 0, "gpu": "82317, 85651, 100, 43", "stop": false} +{"time": 1789743365.1863644, "available_gib": 173.2270965576172, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false} +{"time": 1789743367.206995, "available_gib": 173.21285247802734, "oom_kill": 0, "gpu": "82317, 85651, 64, 42", "stop": false} +{"time": 1789743369.227728, "available_gib": 173.15103149414062, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743371.2485144, "available_gib": 173.17145538330078, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743373.2686968, "available_gib": 173.10255432128906, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743375.289341, "available_gib": 173.1105842590332, "oom_kill": 0, "gpu": "79683, 85651, 63, 42", "stop": false} +{"time": 1789743377.310061, "available_gib": 173.2584686279297, "oom_kill": 0, "gpu": "80157, 85651, 0, 37", "stop": false} +{"time": 1789743379.330606, "available_gib": 173.22555923461914, "oom_kill": 0, "gpu": "80239, 85651, 0, 36", "stop": false} +{"time": 1789743381.3509958, "available_gib": 173.3180694580078, "oom_kill": 0, "gpu": "83047, 85651, 1, 36", "stop": false} +{"time": 1789743383.370858, "available_gib": 173.2816619873047, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false} +{"time": 1789743385.3907626, "available_gib": 173.23712921142578, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false} +{"time": 1789743387.4104686, "available_gib": 173.25224685668945, "oom_kill": 0, "gpu": "83047, 85651, 0, 36", "stop": false} +{"time": 1789743389.4309165, "available_gib": 173.26059341430664, "oom_kill": 0, "gpu": "83049, 85651, 1, 36", "stop": false} +{"time": 1789743391.450981, "available_gib": 173.16680526733398, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false} +{"time": 1789743393.4715426, "available_gib": 173.06145477294922, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false} +{"time": 1789743395.492277, "available_gib": 173.10271835327148, "oom_kill": 0, "gpu": "83055, 85651, 0, 36", "stop": false} +{"time": 1789743397.5132124, "available_gib": 173.11201095581055, "oom_kill": 0, "gpu": "83157, 85651, 0, 36", "stop": false} +{"time": 1789743399.5333133, "available_gib": 173.06784439086914, "oom_kill": 0, "gpu": "83157, 85651, 0, 36", "stop": false} +{"time": 1789743401.5540614, "available_gib": 173.05779266357422, "oom_kill": 0, "gpu": "83163, 85651, 0, 36", "stop": false} +{"time": 1789743403.579573, "available_gib": 173.06655502319336, "oom_kill": 0, "gpu": "83163, 85651, 84, 41", "stop": false} +{"time": 1789743405.6054838, "available_gib": 173.08317184448242, "oom_kill": 0, "gpu": "83163, 85651, 100, 42", "stop": false} +{"time": 1789743407.6315184, "available_gib": 173.0826072692871, "oom_kill": 0, "gpu": "83163, 85651, 100, 43", "stop": false} +{"time": 1789743409.6563766, "available_gib": 173.06651306152344, "oom_kill": 0, "gpu": "83165, 85651, 77, 43", "stop": false} +{"time": 1789743411.6810968, "available_gib": 172.9534683227539, "oom_kill": 0, "gpu": "83835, 85651, 100, 44", "stop": false} +{"time": 1789743413.705223, "available_gib": 172.33104705810547, "oom_kill": 0, "gpu": "84753, 85651, 100, 48", "stop": false} +{"time": 1789743415.7302306, "available_gib": 172.4871826171875, "oom_kill": 0, "gpu": "84755, 85651, 100, 44", "stop": false} +{"time": 1789743417.75524, "available_gib": 172.52772903442383, "oom_kill": 0, "gpu": "84755, 85651, 100, 46", "stop": false} +{"time": 1789743419.7805443, "available_gib": 172.54979705810547, "oom_kill": 0, "gpu": "84759, 85651, 100, 45", "stop": false} +{"time": 1789743421.806648, "available_gib": 172.57553100585938, "oom_kill": 0, "gpu": "84759, 85651, 100, 45", "stop": false} +{"time": 1789743423.8317566, "available_gib": 172.58489227294922, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false} +{"time": 1789743425.8571992, "available_gib": 172.58294677734375, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false} +{"time": 1789743427.8820453, "available_gib": 172.57914352416992, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false} +{"time": 1789743429.9071174, "available_gib": 172.58489227294922, "oom_kill": 0, "gpu": "84759, 85651, 100, 46", "stop": false} +{"time": 1789743431.9314394, "available_gib": 172.58298873901367, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false} +{"time": 1789743433.9564836, "available_gib": 172.58153915405273, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false} +{"time": 1789743435.9823027, "available_gib": 172.58718490600586, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false} +{"time": 1789743438.0072143, "available_gib": 172.58710861206055, "oom_kill": 0, "gpu": "84759, 85651, 100, 47", "stop": false} +{"time": 1789743440.0316548, "available_gib": 172.58600234985352, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743442.0570004, "available_gib": 172.58583068847656, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743444.081906, "available_gib": 172.5858039855957, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743446.1073487, "available_gib": 172.58935546875, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743448.1323597, "available_gib": 172.58035278320312, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743450.15754, "available_gib": 172.58575820922852, "oom_kill": 0, "gpu": "84759, 85651, 100, 48", "stop": false} +{"time": 1789743452.1833196, "available_gib": 172.5850830078125, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743454.208151, "available_gib": 172.5833282470703, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743456.2327328, "available_gib": 172.58715438842773, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743458.2573335, "available_gib": 172.58446502685547, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743460.2828667, "available_gib": 172.5850715637207, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743462.3078372, "available_gib": 172.5800018310547, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743464.33269, "available_gib": 172.58169174194336, "oom_kill": 0, "gpu": "84759, 85651, 100, 49", "stop": false} +{"time": 1789743466.3571842, "available_gib": 172.57829666137695, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743468.3811376, "available_gib": 172.57852172851562, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743470.4063337, "available_gib": 172.59275436401367, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743472.431251, "available_gib": 172.59864044189453, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743474.4564416, "available_gib": 172.60233306884766, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743476.481559, "available_gib": 172.59457778930664, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743478.5064373, "available_gib": 172.6103286743164, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743480.5307262, "available_gib": 172.62775039672852, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743482.5562549, "available_gib": 172.61786270141602, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743484.5813863, "available_gib": 172.62933731079102, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743486.6072564, "available_gib": 172.62554931640625, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743488.63282, "available_gib": 172.62833404541016, "oom_kill": 0, "gpu": "84759, 85651, 100, 50", "stop": false} +{"time": 1789743490.6583722, "available_gib": 172.62057876586914, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743492.684071, "available_gib": 172.62283325195312, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743494.7092838, "available_gib": 172.6273536682129, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743496.7351062, "available_gib": 172.62580108642578, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743498.7601368, "available_gib": 172.62276077270508, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743500.78492, "available_gib": 172.62115859985352, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743502.8107402, "available_gib": 172.6251106262207, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743504.8364482, "available_gib": 172.6214942932129, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743506.8617234, "available_gib": 172.6231918334961, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743508.8863335, "available_gib": 172.62788772583008, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743510.9119735, "available_gib": 172.62989044189453, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743512.9368422, "available_gib": 172.63046264648438, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743514.9624221, "available_gib": 172.62270736694336, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743516.987541, "available_gib": 172.62264251708984, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743519.0120552, "available_gib": 172.61489868164062, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743521.0366497, "available_gib": 172.61508560180664, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743523.0619605, "available_gib": 172.61526107788086, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743525.0872154, "available_gib": 172.61988067626953, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743527.1127303, "available_gib": 172.62136459350586, "oom_kill": 0, "gpu": "84759, 85651, 100, 51", "stop": false} +{"time": 1789743529.137909, "available_gib": 172.61513137817383, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743531.1632333, "available_gib": 172.62389373779297, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743533.188662, "available_gib": 172.6221923828125, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743535.2138846, "available_gib": 172.62632751464844, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743537.2387412, "available_gib": 172.6260643005371, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743539.2632637, "available_gib": 172.62511825561523, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743541.2876751, "available_gib": 172.62701797485352, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743543.3130155, "available_gib": 172.62727737426758, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743545.3385234, "available_gib": 172.63215255737305, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743547.3637378, "available_gib": 172.6363410949707, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743549.3886216, "available_gib": 172.6325225830078, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743551.4133384, "available_gib": 172.6255989074707, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743553.438609, "available_gib": 172.62480926513672, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743555.4636686, "available_gib": 172.62656784057617, "oom_kill": 0, "gpu": "84759, 85651, 86, 52", "stop": false} +{"time": 1789743557.4889739, "available_gib": 172.62308883666992, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743559.5141568, "available_gib": 172.6314353942871, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743561.539715, "available_gib": 172.62345504760742, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743563.565002, "available_gib": 172.63013458251953, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743565.5897946, "available_gib": 172.61325073242188, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743567.614607, "available_gib": 172.60382843017578, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743569.6399143, "available_gib": 172.61295318603516, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743571.665013, "available_gib": 172.6142234802246, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743573.691099, "available_gib": 172.61298370361328, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743575.716717, "available_gib": 172.61080169677734, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743577.741419, "available_gib": 172.60554885864258, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743579.7667089, "available_gib": 172.60740661621094, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743581.7916934, "available_gib": 172.6053695678711, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743583.8170035, "available_gib": 172.60088729858398, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743585.8422914, "available_gib": 172.6147918701172, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743587.8678956, "available_gib": 172.61311721801758, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743589.893016, "available_gib": 172.61607360839844, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743591.9179227, "available_gib": 172.6152458190918, "oom_kill": 0, "gpu": "84759, 85651, 100, 52", "stop": false} +{"time": 1789743593.9394135, "available_gib": 172.49266815185547, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false} +{"time": 1789743595.96263, "available_gib": 172.47529983520508, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false} +{"time": 1789743597.9865532, "available_gib": 172.47649383544922, "oom_kill": 0, "gpu": "84759, 85651, 100, 58", "stop": false} +{"time": 1789743600.011351, "available_gib": 172.47513961791992, "oom_kill": 0, "gpu": "84759, 85651, 100, 58", "stop": false} +{"time": 1789743602.0350573, "available_gib": 172.47442626953125, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false} +{"time": 1789743604.0597992, "available_gib": 172.47433471679688, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false} +{"time": 1789743606.0856547, "available_gib": 172.47597885131836, "oom_kill": 0, "gpu": "84759, 85651, 100, 59", "stop": false} +{"time": 1789743608.1109443, "available_gib": 172.4771270751953, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false} +{"time": 1789743610.136922, "available_gib": 172.47300720214844, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743612.1622267, "available_gib": 172.46633911132812, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743614.1874115, "available_gib": 172.46381378173828, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743616.210489, "available_gib": 172.47092056274414, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false} +{"time": 1789743618.2337203, "available_gib": 172.46821975708008, "oom_kill": 0, "gpu": "84759, 85651, 100, 61", "stop": false} +{"time": 1789743620.2577622, "available_gib": 172.47286987304688, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false} +{"time": 1789743622.2841165, "available_gib": 172.46925735473633, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false} +{"time": 1789743624.3088126, "available_gib": 172.4719467163086, "oom_kill": 0, "gpu": "84759, 85651, 100, 62", "stop": false} +{"time": 1789743626.3321114, "available_gib": 172.4673614501953, "oom_kill": 0, "gpu": "84759, 85651, 100, 63", "stop": false} +{"time": 1789743628.3577669, "available_gib": 172.45921325683594, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false} +{"time": 1789743630.381975, "available_gib": 172.45735931396484, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false} +{"time": 1789743632.4074128, "available_gib": 172.47180557250977, "oom_kill": 0, "gpu": "84759, 85651, 100, 57", "stop": false} +{"time": 1789743634.433758, "available_gib": 172.4716796875, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false} +{"time": 1789743636.4596326, "available_gib": 172.47660446166992, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false} +{"time": 1789743638.4853396, "available_gib": 172.4737548828125, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false} +{"time": 1789743640.5112894, "available_gib": 172.4694061279297, "oom_kill": 0, "gpu": "84759, 85651, 100, 56", "stop": false} +{"time": 1789743642.5373013, "available_gib": 172.4714698791504, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743644.5635104, "available_gib": 172.4744110107422, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743646.5894952, "available_gib": 172.47317504882812, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743648.6155689, "available_gib": 172.47101211547852, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743650.6409676, "available_gib": 172.46884155273438, "oom_kill": 0, "gpu": "84759, 85651, 100, 55", "stop": false} +{"time": 1789743652.6616406, "available_gib": 174.40550231933594, "oom_kill": 0, "gpu": "84759, 85651, 0, 50", "stop": false} +{"time": 1789743654.6818483, "available_gib": 174.40292358398438, "oom_kill": 0, "gpu": "84759, 85651, 0, 50", "stop": false} +{"time": 1789743656.7018807, "available_gib": 175.23258590698242, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false} +{"time": 1789743658.7224524, "available_gib": 175.22650909423828, "oom_kill": 0, "gpu": "84759, 85651, 0, 48", "stop": false} +{"time": 1789743660.7429912, "available_gib": 175.2299690246582, "oom_kill": 0, "gpu": "84759, 85651, 0, 47", "stop": false} +{"time": 1789743667.684546, "available_gib": 189.53302764892578, "oom_kill": 0, "gpu": "3, 85651, 0, 45", "stop": false} +{"time": 1789743669.7074604, "available_gib": 242.46288299560547, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false} +{"time": 1789743671.7295043, "available_gib": 242.38314819335938, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false} +{"time": 1789743673.7539766, "available_gib": 242.30734252929688, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false} +{"time": 1789743675.77852, "available_gib": 242.25893783569336, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789743677.8033016, "available_gib": 242.2384796142578, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789743679.8281374, "available_gib": 242.18122482299805, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789743681.8531134, "available_gib": 242.11988830566406, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false} +{"time": 1789743683.877929, "available_gib": 242.06714248657227, "oom_kill": 0, "gpu": "0, 85651, 0, 40", "stop": false} +{"time": 1789743685.9199724, "available_gib": 241.74395751953125, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789743687.9635887, "available_gib": 241.41803741455078, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789743690.005523, "available_gib": 241.17269897460938, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789743692.048056, "available_gib": 240.93939590454102, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789743694.0926096, "available_gib": 240.32037353515625, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789743696.134711, "available_gib": 240.09738159179688, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789743698.1778443, "available_gib": 239.82658004760742, "oom_kill": 0, "gpu": "3, 85651, 0, 37", "stop": false} +{"time": 1789743700.1996496, "available_gib": 239.30536651611328, "oom_kill": 0, "gpu": "683, 85651, 2, 37", "stop": false} +{"time": 1789743702.2195978, "available_gib": 239.08550262451172, "oom_kill": 0, "gpu": "2915, 85651, 0, 37", "stop": false} +{"time": 1789743704.239852, "available_gib": 234.2517204284668, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743706.2606213, "available_gib": 230.40094375610352, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743708.280376, "available_gib": 226.4360122680664, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743710.3012598, "available_gib": 222.46880340576172, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743712.32121, "available_gib": 218.4713897705078, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743714.341048, "available_gib": 214.47138214111328, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743716.360451, "available_gib": 210.4367904663086, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743718.3802285, "available_gib": 206.42337799072266, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743720.4000227, "available_gib": 202.46843719482422, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743722.4194143, "available_gib": 198.42879104614258, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789743724.4388602, "available_gib": 194.38553619384766, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743726.4584868, "available_gib": 190.3592414855957, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743728.4785507, "available_gib": 186.39047622680664, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743730.4990554, "available_gib": 182.43556594848633, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743732.5190928, "available_gib": 178.44296646118164, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743734.5388536, "available_gib": 174.4428482055664, "oom_kill": 0, "gpu": "4285, 85651, 0, 36", "stop": false} +{"time": 1789743736.5602393, "available_gib": 173.5626335144043, "oom_kill": 0, "gpu": "76995, 85651, 0, 36", "stop": false} +{"time": 1789743738.581365, "available_gib": 173.58563232421875, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743740.602393, "available_gib": 173.59395599365234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743742.623244, "available_gib": 173.592041015625, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743744.644461, "available_gib": 173.58086395263672, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743746.6653383, "available_gib": 173.5805892944336, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743748.6863484, "available_gib": 173.57102584838867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743750.7072456, "available_gib": 173.6135025024414, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743752.7283082, "available_gib": 173.61641693115234, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743754.7494485, "available_gib": 173.60739517211914, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743756.770568, "available_gib": 173.62178421020508, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743758.791231, "available_gib": 173.6512794494629, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743760.8119733, "available_gib": 173.6450080871582, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743762.8327234, "available_gib": 173.65488052368164, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743764.8535624, "available_gib": 173.63638305664062, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false} +{"time": 1789743766.8746958, "available_gib": 173.65426635742188, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743768.89607, "available_gib": 173.6699333190918, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743770.9170897, "available_gib": 173.67607879638672, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789743772.9382615, "available_gib": 173.70412826538086, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false} +{"time": 1789743774.959427, "available_gib": 173.7211685180664, "oom_kill": 0, "gpu": "76995, 85651, 4, 35", "stop": false} +{"time": 1789743776.9805934, "available_gib": 173.74411010742188, "oom_kill": 0, "gpu": "76995, 85651, 37, 35", "stop": false} +{"time": 1789743779.0047975, "available_gib": 173.68999481201172, "oom_kill": 0, "gpu": "76995, 85651, 0, 35", "stop": false} +{"time": 1789743781.0252128, "available_gib": 173.55987930297852, "oom_kill": 0, "gpu": "82531, 85651, 0, 35", "stop": false} +{"time": 1789743783.0469394, "available_gib": 173.57495498657227, "oom_kill": 0, "gpu": "82531, 85651, 10, 35", "stop": false} +{"time": 1789743785.0683353, "available_gib": 173.6102638244629, "oom_kill": 0, "gpu": "82559, 85651, 0, 35", "stop": false} +{"time": 1789743787.0894022, "available_gib": 173.57339477539062, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false} +{"time": 1789743789.109543, "available_gib": 173.24087524414062, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false} +{"time": 1789743791.13055, "available_gib": 173.11621475219727, "oom_kill": 0, "gpu": "79131, 85651, 0, 35", "stop": false} +{"time": 1789743793.152408, "available_gib": 172.55051803588867, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743795.1728363, "available_gib": 172.77595138549805, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743797.1929188, "available_gib": 172.77360916137695, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789743799.214182, "available_gib": 172.7921257019043, "oom_kill": 0, "gpu": "82317, 85651, 100, 44", "stop": false} +{"time": 1789743801.2358875, "available_gib": 172.80052947998047, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false} +{"time": 1789743803.2571647, "available_gib": 172.80226516723633, "oom_kill": 0, "gpu": "82317, 85651, 0, 39", "stop": false} +{"time": 1789743805.2780788, "available_gib": 172.79526901245117, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743807.299253, "available_gib": 172.80913162231445, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743809.3191426, "available_gib": 172.6723861694336, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789743811.3417072, "available_gib": 173.06355667114258, "oom_kill": 0, "gpu": "80519, 85651, 2, 37", "stop": false} +{"time": 1789743813.3622966, "available_gib": 173.13029861450195, "oom_kill": 0, "gpu": "80189, 85651, 0, 37", "stop": false} +{"time": 1789743815.3829036, "available_gib": 173.09708404541016, "oom_kill": 0, "gpu": "80241, 85651, 0, 36", "stop": false} +{"time": 1789743817.4042423, "available_gib": 173.31471633911133, "oom_kill": 0, "gpu": "80197, 85651, 0, 36", "stop": false} +{"time": 1789743819.4245021, "available_gib": 173.30271530151367, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false} +{"time": 1789743821.4452205, "available_gib": 173.21561431884766, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false} +{"time": 1789743823.4653258, "available_gib": 173.16995239257812, "oom_kill": 0, "gpu": "83003, 85651, 0, 36", "stop": false} +{"time": 1789743825.4860725, "available_gib": 173.21529006958008, "oom_kill": 0, "gpu": "83005, 85651, 0, 36", "stop": false} +{"time": 1789743827.5069244, "available_gib": 173.12132263183594, "oom_kill": 0, "gpu": "83011, 85651, 2, 36", "stop": false} +{"time": 1789743829.5271885, "available_gib": 172.99286651611328, "oom_kill": 0, "gpu": "83011, 85651, 0, 36", "stop": false} +{"time": 1789743831.549038, "available_gib": 173.04050064086914, "oom_kill": 0, "gpu": "83011, 85651, 5, 36", "stop": false} +{"time": 1789743833.570074, "available_gib": 173.05452728271484, "oom_kill": 0, "gpu": "83149, 85651, 0, 36", "stop": false} +{"time": 1789743835.5897286, "available_gib": 173.02277755737305, "oom_kill": 0, "gpu": "83149, 85651, 0, 36", "stop": false} +{"time": 1789743837.6093662, "available_gib": 173.02072143554688, "oom_kill": 0, "gpu": "83149, 85651, 0, 35", "stop": false} +{"time": 1789743839.6298525, "available_gib": 173.01683044433594, "oom_kill": 0, "gpu": "83155, 85651, 0, 36", "stop": false} +{"time": 1789743841.6556835, "available_gib": 173.02434158325195, "oom_kill": 0, "gpu": "83155, 85651, 100, 42", "stop": false} +{"time": 1789743843.681355, "available_gib": 173.0242156982422, "oom_kill": 0, "gpu": "83155, 85651, 100, 43", "stop": false} +{"time": 1789743845.7065227, "available_gib": 173.01176071166992, "oom_kill": 0, "gpu": "83155, 85651, 100, 44", "stop": false} +{"time": 1789743847.7316642, "available_gib": 172.9279899597168, "oom_kill": 0, "gpu": "83827, 85651, 100, 43", "stop": false} +{"time": 1789743849.7550159, "available_gib": 172.27711868286133, "oom_kill": 0, "gpu": "84693, 85651, 100, 48", "stop": false} +{"time": 1789743851.779772, "available_gib": 172.35914993286133, "oom_kill": 0, "gpu": "84693, 85651, 100, 48", "stop": false} +{"time": 1789743853.8058956, "available_gib": 172.42580795288086, "oom_kill": 0, "gpu": "84697, 85651, 100, 45", "stop": false} +{"time": 1789743855.83155, "available_gib": 172.41658782958984, "oom_kill": 0, "gpu": "84697, 85651, 100, 45", "stop": false} +{"time": 1789743857.8563097, "available_gib": 172.4184684753418, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false} +{"time": 1789743859.8813326, "available_gib": 172.41818618774414, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false} +{"time": 1789743861.9067163, "available_gib": 172.41766738891602, "oom_kill": 0, "gpu": "84697, 85651, 100, 46", "stop": false} +{"time": 1789743863.932184, "available_gib": 172.41712951660156, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false} +{"time": 1789743865.956786, "available_gib": 172.42663192749023, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false} +{"time": 1789743867.982782, "available_gib": 172.42996215820312, "oom_kill": 0, "gpu": "84697, 85651, 100, 47", "stop": false} +{"time": 1789743870.0088744, "available_gib": 172.4624481201172, "oom_kill": 0, "gpu": "84697, 85651, 100, 48", "stop": false} +{"time": 1789743872.0358696, "available_gib": 172.49190139770508, "oom_kill": 0, "gpu": "84697, 85651, 80, 48", "stop": false} +{"time": 1789743874.0616338, "available_gib": 172.50891876220703, "oom_kill": 0, "gpu": "84697, 85651, 100, 48", "stop": false} +{"time": 1789743876.0869436, "available_gib": 172.508544921875, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false} +{"time": 1789743878.1125872, "available_gib": 172.5115089416504, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false} +{"time": 1789743880.1391616, "available_gib": 172.51123046875, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false} +{"time": 1789743882.1654341, "available_gib": 172.51721954345703, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false} +{"time": 1789743884.1917255, "available_gib": 172.52043914794922, "oom_kill": 0, "gpu": "84697, 85651, 100, 49", "stop": false} +{"time": 1789743886.2185214, "available_gib": 172.52492904663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false} +{"time": 1789743888.2450676, "available_gib": 172.5250587463379, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false} +{"time": 1789743890.2708652, "available_gib": 172.5223503112793, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false} +{"time": 1789743892.2970574, "available_gib": 172.5244483947754, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false} +{"time": 1789743894.323345, "available_gib": 172.52897262573242, "oom_kill": 0, "gpu": "84697, 85651, 100, 50", "stop": false} +{"time": 1789743896.3477945, "available_gib": 172.5200309753418, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743898.3742619, "available_gib": 172.52397537231445, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743900.4001124, "available_gib": 172.53362274169922, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743902.4257197, "available_gib": 172.53076934814453, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743904.4522634, "available_gib": 172.53030395507812, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743906.4786067, "available_gib": 172.5284767150879, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743908.5038824, "available_gib": 172.53790283203125, "oom_kill": 0, "gpu": "84697, 85651, 100, 51", "stop": false} +{"time": 1789743910.5290341, "available_gib": 172.55227279663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743912.5544894, "available_gib": 172.5442771911621, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743914.580608, "available_gib": 172.5365219116211, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743916.6071196, "available_gib": 172.54441452026367, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743918.6325831, "available_gib": 172.5501251220703, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743920.6588407, "available_gib": 172.54729080200195, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743922.6839206, "available_gib": 172.56165313720703, "oom_kill": 0, "gpu": "84697, 85651, 94, 52", "stop": false} +{"time": 1789743924.710274, "available_gib": 172.56248474121094, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743926.7364788, "available_gib": 172.56631088256836, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743928.7624378, "available_gib": 172.57713317871094, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743930.7882645, "available_gib": 172.57723236083984, "oom_kill": 0, "gpu": "84697, 85651, 100, 52", "stop": false} +{"time": 1789743932.8145838, "available_gib": 172.57740783691406, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743934.8404663, "available_gib": 172.57373809814453, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743936.866788, "available_gib": 172.57367324829102, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743938.8923123, "available_gib": 172.57363891601562, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743940.9182467, "available_gib": 172.57357788085938, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743942.943654, "available_gib": 172.57180404663086, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743944.9701402, "available_gib": 172.5783462524414, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743946.996388, "available_gib": 172.5782928466797, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743949.0224254, "available_gib": 172.58384323120117, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743951.0482817, "available_gib": 172.586669921875, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743953.0737877, "available_gib": 172.59230041503906, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743955.1002982, "available_gib": 172.59587478637695, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743957.125969, "available_gib": 172.58476638793945, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743959.152024, "available_gib": 172.58797073364258, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743961.1781206, "available_gib": 172.5951156616211, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743963.2033892, "available_gib": 172.58736038208008, "oom_kill": 0, "gpu": "84697, 85651, 100, 53", "stop": false} +{"time": 1789743965.2290244, "available_gib": 172.58730697631836, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743967.2558095, "available_gib": 172.58724212646484, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743969.2818677, "available_gib": 172.58523559570312, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743971.30743, "available_gib": 172.58755493164062, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743973.3339229, "available_gib": 172.58671188354492, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743975.3602042, "available_gib": 172.59050369262695, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743977.3864458, "available_gib": 172.58787536621094, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743979.412776, "available_gib": 172.59032821655273, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743981.4392207, "available_gib": 172.58793258666992, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743983.4650486, "available_gib": 172.59032821655273, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743985.4914, "available_gib": 172.58974838256836, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743987.5175338, "available_gib": 172.57958984375, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743989.5431314, "available_gib": 172.57130432128906, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743991.5687299, "available_gib": 172.5556182861328, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743993.5943153, "available_gib": 172.5477066040039, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743995.6200294, "available_gib": 172.55400466918945, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743997.6459754, "available_gib": 172.5567626953125, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789743999.6716835, "available_gib": 172.5649642944336, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789744001.6973622, "available_gib": 172.5669174194336, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789744003.7233665, "available_gib": 172.56511688232422, "oom_kill": 0, "gpu": "84697, 85651, 100, 54", "stop": false} +{"time": 1789744005.7436235, "available_gib": 172.42804336547852, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false} +{"time": 1789744007.767596, "available_gib": 172.40644073486328, "oom_kill": 0, "gpu": "84697, 85651, 100, 58", "stop": false} +{"time": 1789744009.7923508, "available_gib": 172.39516830444336, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744011.8161004, "available_gib": 172.38743209838867, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744013.839925, "available_gib": 172.37675857543945, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false} +{"time": 1789744015.8648205, "available_gib": 172.38694763183594, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false} +{"time": 1789744017.8896184, "available_gib": 172.39286422729492, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false} +{"time": 1789744019.913305, "available_gib": 172.3947982788086, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false} +{"time": 1789744021.9392803, "available_gib": 172.4147834777832, "oom_kill": 0, "gpu": "84697, 85651, 100, 57", "stop": false} +{"time": 1789744023.9630058, "available_gib": 172.41676330566406, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false} +{"time": 1789744025.9881828, "available_gib": 172.40471267700195, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false} +{"time": 1789744028.012432, "available_gib": 172.42510604858398, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false} +{"time": 1789744030.036954, "available_gib": 172.42960357666016, "oom_kill": 0, "gpu": "84697, 85651, 100, 62", "stop": false} +{"time": 1789744032.061324, "available_gib": 172.42523956298828, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false} +{"time": 1789744034.0878584, "available_gib": 172.42697143554688, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false} +{"time": 1789744036.1117172, "available_gib": 172.42796325683594, "oom_kill": 0, "gpu": "84697, 85651, 100, 63", "stop": false} +{"time": 1789744038.136802, "available_gib": 172.43342208862305, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744040.160596, "available_gib": 172.43236541748047, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744042.1858406, "available_gib": 172.44089889526367, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744044.210708, "available_gib": 172.44932556152344, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744046.2357752, "available_gib": 172.44157791137695, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744048.2619452, "available_gib": 172.4429702758789, "oom_kill": 0, "gpu": "84697, 85651, 100, 64", "stop": false} +{"time": 1789744050.2856293, "available_gib": 172.44015502929688, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744052.3109236, "available_gib": 172.44293594360352, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false} +{"time": 1789744054.3350937, "available_gib": 172.43926620483398, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744056.3591633, "available_gib": 172.44363403320312, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744058.3845017, "available_gib": 172.43565368652344, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744060.409055, "available_gib": 172.41972732543945, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744062.4337656, "available_gib": 172.42286682128906, "oom_kill": 0, "gpu": "84697, 85651, 100, 66", "stop": false} +{"time": 1789744064.460525, "available_gib": 172.43537521362305, "oom_kill": 0, "gpu": "84697, 85651, 100, 65", "stop": false} +{"time": 1789744066.4872258, "available_gib": 172.42713928222656, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false} +{"time": 1789744068.512237, "available_gib": 172.4395408630371, "oom_kill": 0, "gpu": "84697, 85651, 100, 61", "stop": false} +{"time": 1789744070.538391, "available_gib": 172.43132400512695, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false} +{"time": 1789744072.5637136, "available_gib": 172.43741989135742, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false} +{"time": 1789744074.590121, "available_gib": 172.4353141784668, "oom_kill": 0, "gpu": "84697, 85651, 100, 60", "stop": false} +{"time": 1789744076.6162558, "available_gib": 172.43359756469727, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744078.6413624, "available_gib": 172.43523025512695, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744080.6638365, "available_gib": 172.4354362487793, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744082.689977, "available_gib": 172.4274673461914, "oom_kill": 0, "gpu": "84697, 85651, 100, 59", "stop": false} +{"time": 1789744084.7105415, "available_gib": 174.46597290039062, "oom_kill": 0, "gpu": "84697, 85651, 0, 53", "stop": false} +{"time": 1789744086.7306116, "available_gib": 174.4582176208496, "oom_kill": 0, "gpu": "84697, 85651, 0, 52", "stop": false} +{"time": 1789744088.7536929, "available_gib": 175.251953125, "oom_kill": 0, "gpu": "84697, 85651, 0, 51", "stop": false} +{"time": 1789744090.7746408, "available_gib": 175.2590446472168, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false} +{"time": 1789744092.7950041, "available_gib": 175.26241302490234, "oom_kill": 0, "gpu": "84697, 85651, 0, 50", "stop": false} +{"time": 1789744099.8720243, "available_gib": 189.0756950378418, "oom_kill": 0, "gpu": "3, 85651, 0, 47", "stop": false} +{"time": 1789744101.8936296, "available_gib": 242.46984481811523, "oom_kill": 0, "gpu": "0, 85651, 0, 46", "stop": false} +{"time": 1789744103.9150841, "available_gib": 242.3806266784668, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false} +{"time": 1789744105.9400728, "available_gib": 242.3127899169922, "oom_kill": 0, "gpu": "0, 85651, 0, 45", "stop": false} +{"time": 1789744107.9647262, "available_gib": 242.27752685546875, "oom_kill": 0, "gpu": "0, 85651, 0, 44", "stop": false} +{"time": 1789744109.989206, "available_gib": 242.25381088256836, "oom_kill": 0, "gpu": "0, 85651, 0, 43", "stop": false} +{"time": 1789744112.0140321, "available_gib": 242.22451782226562, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789744114.0383747, "available_gib": 242.15388107299805, "oom_kill": 0, "gpu": "0, 85651, 0, 42", "stop": false} +{"time": 1789744116.0627222, "available_gib": 242.1028938293457, "oom_kill": 0, "gpu": "0, 85651, 0, 41", "stop": false} +{"time": 1789744118.1058602, "available_gib": 241.75963973999023, "oom_kill": 0, "gpu": "3, 85651, 0, 41", "stop": false} +{"time": 1789744120.149718, "available_gib": 241.42746353149414, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false} +{"time": 1789744122.1952055, "available_gib": 241.17255783081055, "oom_kill": 0, "gpu": "3, 85651, 0, 40", "stop": false} +{"time": 1789744124.2375867, "available_gib": 240.98880767822266, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789744126.2794218, "available_gib": 240.34492874145508, "oom_kill": 0, "gpu": "3, 85651, 0, 39", "stop": false} +{"time": 1789744128.3214467, "available_gib": 240.07582092285156, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744130.3638144, "available_gib": 239.7653465270996, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744132.4073355, "available_gib": 239.3140525817871, "oom_kill": 0, "gpu": "3, 85651, 0, 38", "stop": false} +{"time": 1789744134.4280436, "available_gib": 239.03064346313477, "oom_kill": 0, "gpu": "2915, 85651, 0, 39", "stop": false} +{"time": 1789744136.447735, "available_gib": 234.87797164916992, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744138.4675038, "available_gib": 231.21246337890625, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744140.487026, "available_gib": 227.27257919311523, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744142.5069096, "available_gib": 223.3726921081543, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744144.5267863, "available_gib": 219.43412399291992, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744146.5464373, "available_gib": 215.4873390197754, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744148.566585, "available_gib": 211.5369987487793, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744150.5869632, "available_gib": 207.59728240966797, "oom_kill": 0, "gpu": "4285, 85651, 0, 38", "stop": false} +{"time": 1789744152.60729, "available_gib": 203.638671875, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744154.6276596, "available_gib": 199.62559509277344, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744156.6482484, "available_gib": 195.57702255249023, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744158.668383, "available_gib": 191.5196304321289, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744160.6888082, "available_gib": 187.50440216064453, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744162.708834, "available_gib": 183.54587936401367, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744164.7295425, "available_gib": 179.56959915161133, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744166.7498033, "available_gib": 175.61302947998047, "oom_kill": 0, "gpu": "4285, 85651, 0, 37", "stop": false} +{"time": 1789744168.7754214, "available_gib": 173.72442626953125, "oom_kill": 0, "gpu": "4413, 85651, 8, 37", "stop": false} +{"time": 1789744170.7964997, "available_gib": 173.6231575012207, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744172.8178098, "available_gib": 173.62652969360352, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744174.8396492, "available_gib": 173.62175369262695, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744176.8605325, "available_gib": 173.6126823425293, "oom_kill": 0, "gpu": "76995, 85651, 4, 37", "stop": false} +{"time": 1789744178.8817682, "available_gib": 173.62749099731445, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744180.9026644, "available_gib": 173.63713836669922, "oom_kill": 0, "gpu": "76995, 85651, 6, 36", "stop": false} +{"time": 1789744182.9237735, "available_gib": 173.64041900634766, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744184.944897, "available_gib": 173.63914108276367, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744186.9659956, "available_gib": 173.65515899658203, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744188.9869823, "available_gib": 173.68014907836914, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744191.0077991, "available_gib": 173.66809844970703, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744193.0285199, "available_gib": 173.6796112060547, "oom_kill": 0, "gpu": "76995, 85651, 2, 36", "stop": false} +{"time": 1789744195.049452, "available_gib": 173.67911529541016, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744197.0703564, "available_gib": 173.70183181762695, "oom_kill": 0, "gpu": "76995, 85651, 8, 36", "stop": false} +{"time": 1789744199.0914025, "available_gib": 173.7275505065918, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744201.1120594, "available_gib": 173.74087524414062, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744203.1332436, "available_gib": 173.74426651000977, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744205.1543298, "available_gib": 173.76029586791992, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744207.1753323, "available_gib": 173.78207397460938, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744209.1961696, "available_gib": 173.77024459838867, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744211.2166562, "available_gib": 173.79063415527344, "oom_kill": 0, "gpu": "76995, 85651, 4, 36", "stop": false} +{"time": 1789744213.2376473, "available_gib": 173.60406875610352, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false} +{"time": 1789744215.2581506, "available_gib": 173.57421112060547, "oom_kill": 0, "gpu": "82531, 85651, 0, 36", "stop": false} +{"time": 1789744217.2796466, "available_gib": 173.60267639160156, "oom_kill": 0, "gpu": "82559, 85651, 0, 36", "stop": false} +{"time": 1789744219.301794, "available_gib": 173.5945167541504, "oom_kill": 0, "gpu": "82613, 85651, 0, 36", "stop": false} +{"time": 1789744221.3223813, "available_gib": 173.4375114440918, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false} +{"time": 1789744223.3421645, "available_gib": 173.10824966430664, "oom_kill": 0, "gpu": "79131, 85651, 0, 36", "stop": false} +{"time": 1789744225.3638763, "available_gib": 172.49812698364258, "oom_kill": 0, "gpu": "81095, 85651, 0, 36", "stop": false} +{"time": 1789744227.3846838, "available_gib": 172.73027420043945, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789744229.4048204, "available_gib": 172.87688064575195, "oom_kill": 0, "gpu": "81095, 85651, 0, 35", "stop": false} +{"time": 1789744231.4258337, "available_gib": 172.88410186767578, "oom_kill": 0, "gpu": "82317, 85651, 100, 43", "stop": false} +{"time": 1789744233.448499, "available_gib": 173.24000930786133, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false} +{"time": 1789744235.4696748, "available_gib": 173.2450294494629, "oom_kill": 0, "gpu": "82317, 85651, 100, 45", "stop": false} +{"time": 1789744237.490613, "available_gib": 173.23291778564453, "oom_kill": 0, "gpu": "82319, 85651, 0, 38", "stop": false} +{"time": 1789744239.511178, "available_gib": 173.19620895385742, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789744241.5316548, "available_gib": 173.20362854003906, "oom_kill": 0, "gpu": "82319, 85651, 0, 37", "stop": false} +{"time": 1789744243.5523272, "available_gib": 173.18954849243164, "oom_kill": 0, "gpu": "79465, 85651, 11, 37", "stop": false} +{"time": 1789744245.5724537, "available_gib": 172.98081588745117, "oom_kill": 0, "gpu": "80553, 85651, 0, 37", "stop": false} +{"time": 1789744247.5929437, "available_gib": 173.15055084228516, "oom_kill": 0, "gpu": "80239, 85651, 0, 37", "stop": false} +{"time": 1789744249.6134512, "available_gib": 173.34084701538086, "oom_kill": 0, "gpu": "80335, 85651, 0, 37", "stop": false} +{"time": 1789744251.6334639, "available_gib": 173.3362808227539, "oom_kill": 0, "gpu": "83033, 85651, 0, 37", "stop": false} +{"time": 1789744253.6532722, "available_gib": 173.2673683166504, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744255.6729264, "available_gib": 173.26757431030273, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744257.6932175, "available_gib": 173.25699615478516, "oom_kill": 0, "gpu": "83033, 85651, 0, 36", "stop": false} +{"time": 1789744259.7134147, "available_gib": 173.12919235229492, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false} +{"time": 1789744261.733558, "available_gib": 173.114990234375, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false} +{"time": 1789744263.7532177, "available_gib": 173.02079010009766, "oom_kill": 0, "gpu": "83041, 85651, 0, 36", "stop": false} +{"time": 1789744265.773419, "available_gib": 173.05761337280273, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} +{"time": 1789744267.793305, "available_gib": 173.0266342163086, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} +{"time": 1789744269.8131666, "available_gib": 173.04352188110352, "oom_kill": 0, "gpu": "83175, 85651, 0, 36", "stop": false} diff --git a/audit/tuning/steps-123/restored.json b/audit/tuning/steps-123/restored.json new file mode 100644 index 0000000..4ca9b12 --- /dev/null +++ b/audit/tuning/steps-123/restored.json @@ -0,0 +1 @@ +{"healthy": true, "time": 1789744271.2613404} \ No newline at end of file diff --git a/audit/tuning/summary.json b/audit/tuning/summary.json new file mode 100644 index 0000000..d47ce21 --- /dev/null +++ b/audit/tuning/summary.json @@ -0,0 +1,202 @@ +{ + "batch-1024/mtp2-b1024": { + "startup": true, + "records": 44, + "failures": [], + "screen.jsonl": { + "count": 37, + "failed": [] + }, + "decode_tps": { + "code": 123.50695983978235, + "chinese": 128.67815702769335, + "reasoning": 122.09499443849444 + }, + "latency": { + "video-120s": { + "ttft_s": 2.827090976003092, + "elapsed_s": 3.731427850005275, + "passed": true + }, + "multi-image": { + "ttft_s": 1.7833786369956215, + "elapsed_s": 2.3456459599983646, + "passed": true + }, + "long-context-0": { + "ttft_s": 17.993200959004753, + "elapsed_s": 18.83731006600283, + "passed": true + }, + "long-context-1": { + "ttft_s": 0.8536870109965093, + "elapsed_s": 1.7167949249997037, + "passed": true + }, + "long-context-image": { + "ttft_s": 15.955126308996114, + "elapsed_s": 17.181451591997757, + "passed": true + } + }, + "quality-medium.jsonl": { + "count": 7, + "failed": [] + }, + "memory_log": [ + "(EngineCore pid=113) INFO 09-18 15:13:31 [model_runner.py:419] Model loading took 76.23 GiB memory and 86.731724 seconds", + "(EngineCore pid=113) INFO 09-18 15:14:01 [gpu_worker.py:641] Available KV cache memory: 2.96 GiB", + "(EngineCore pid=113) INFO 09-18 15:14:01 [kv_cache_utils.py:2404] GPU KV cache size: 155,509 tokens, Maximum concurrency for 131,072 tokens per request: 1.19x" + ] + }, + "steps-123/mtp1-b2048": { + "startup": true, + "records": 44, + "failures": [], + "screen.jsonl": { + "count": 37, + "failed": [] + }, + "decode_tps": { + "code": 109.6317739856207, + "chinese": 112.60621062069599, + "reasoning": 108.27878210561818 + }, + "latency": { + "video-120s": { + "ttft_s": 2.490573155002494, + "elapsed_s": 3.5669434599985834, + "passed": true + }, + "multi-image": { + "ttft_s": 1.4522779820035794, + "elapsed_s": 2.0256833830062533, + "passed": true + }, + "long-context-0": { + "ttft_s": 14.440546402998734, + "elapsed_s": 15.590460605999397, + "passed": true + }, + "long-context-1": { + "ttft_s": 0.856336849006766, + "elapsed_s": 1.9747926430063671, + "passed": true + }, + "long-context-image": { + "ttft_s": 12.791386062002857, + "elapsed_s": 14.462968651001574, + "passed": true + } + }, + "quality-medium.jsonl": { + "count": 7, + "failed": [] + }, + "memory_log": [ + "(EngineCore pid=112) INFO 09-18 14:55:51 [model_runner.py:419] Model loading took 76.36 GiB memory and 89.100896 seconds", + "(EngineCore pid=112) INFO 09-18 14:56:21 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", + "(EngineCore pid=112) INFO 09-18 14:56:21 [kv_cache_utils.py:2404] GPU KV cache size: 159,669 tokens, Maximum concurrency for 131,072 tokens per request: 1.22x" + ] + }, + "steps-123/mtp2-b2048": { + "startup": true, + "records": 44, + "failures": [], + "screen.jsonl": { + "count": 37, + "failed": [] + }, + "decode_tps": { + "code": 123.72166768011161, + "chinese": 128.75257173627188, + "reasoning": 122.10854572395078 + }, + "latency": { + "video-120s": { + "ttft_s": 0.8094746360002318, + "elapsed_s": 1.7980530589993577, + "passed": true + }, + "multi-image": { + "ttft_s": 1.4533180090002134, + "elapsed_s": 1.9994508850068087, + "passed": true + }, + "long-context-0": { + "ttft_s": 14.42599186499865, + "elapsed_s": 15.14422588799789, + "passed": true + }, + "long-context-1": { + "ttft_s": 0.7020562550023897, + "elapsed_s": 1.4927438169979723, + "passed": true + }, + "long-context-image": { + "ttft_s": 12.815331379999407, + "elapsed_s": 14.372689276002347, + "passed": true + } + }, + "quality-medium.jsonl": { + "count": 7, + "failed": [] + }, + "memory_log": [ + "(EngineCore pid=113) INFO 09-18 14:15:53 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.461433 seconds", + "(EngineCore pid=113) INFO 09-18 14:16:24 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", + "(EngineCore pid=113) INFO 09-18 14:16:24 [kv_cache_utils.py:2404] GPU KV cache size: 146,622 tokens, Maximum concurrency for 131,072 tokens per request: 1.12x" + ] + }, + "steps-123/mtp3-b2048": { + "startup": true, + "records": 44, + "failures": [], + "screen.jsonl": { + "count": 37, + "failed": [] + }, + "decode_tps": { + "code": 123.54931448950835, + "chinese": 134.24473934955645, + "reasoning": 123.52862404436223 + }, + "latency": { + "video-120s": { + "ttft_s": 2.5409359719997155, + "elapsed_s": 3.243877524000709, + "passed": true + }, + "multi-image": { + "ttft_s": 1.463653916005569, + "elapsed_s": 1.8863289360015187, + "passed": true + }, + "long-context-0": { + "ttft_s": 14.516383726004278, + "elapsed_s": 15.218214432999957, + "passed": true + }, + "long-context-1": { + "ttft_s": 14.510828671998752, + "elapsed_s": 15.23376362700219, + "passed": true + }, + "long-context-image": { + "ttft_s": 12.83541851100017, + "elapsed_s": 13.968540412002767, + "passed": true + } + }, + "quality-medium.jsonl": { + "count": 7, + "failed": [] + }, + "memory_log": [ + "(EngineCore pid=112) INFO 09-18 15:03:06 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.716523 seconds", + "(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", + "(EngineCore pid=112) INFO 09-18 15:03:37 [kv_cache_utils.py:2404] GPU KV cache size: 137,414 tokens, Maximum concurrency for 131,072 tokens per request: 1.05x" + ] + } +} diff --git a/docs/mtp-tuning.md b/docs/mtp-tuning.md new file mode 100644 index 0000000..52b5ac8 --- /dev/null +++ b/docs/mtp-tuning.md @@ -0,0 +1,61 @@ +# MTP 参数优化试验计划 + +基准提交:918b78a;日期:2026-09-18。当前默认 MTP 2、128K、FP8 KV、prefill batch 2048、GPU memory utilization 0.985、单活动请求。 + +## 顺序与固定变量 + +1. 保存当前 Compose 和运行状态。首先在当前配置跑筛选集,获得本轮对照。 +2. 分别仅把预测步数改为 1、3;Graph capture sizes 随验证长度改为 [1,2]、[1,4]。保持官方镜像 digest、模型 revision、128K、视觉编码器、FP8 KV、显存预算和并发不变。 +3. 对最合适的步数,仅把 max-num-batched-tokens 从 2048 改到 1024,再跑同一筛选集。 +4. 候选若有明确收益,补做完整套件,包括 16384-token 输出和至少 600 秒混合负载,及 xhigh/medium 质量题。只有通过后才替换默认;否则恢复原 MTP 2。 + +## 筛选集 + +沿用 qualification.py 的算术、JSON、多轮、2048×1536 OCR、多图、120 秒视频、三类各三次 2048-token 输出、接近 128K 检索、长文本加图片及排队请求。筛选阶段跳过 16384-token 单次长输出及持续负载,完整阶段再补足。另运行 medium 正常 EOS 质量题,包括生成代码的固定测试。 + +记录:模型和 KV 显存、KV 容量、MTP 接受率、TTFT、解码速度、请求总耗时、错误、OOM、温度和输出正确性。不同输出文本不声称质量等价;固定长度只作为吞吐测试。缓存状态影响长输入,分别记录首次和重复结果,不宣称未控制缓存状态的冷启动对照。 + +## 判定 + +- 任一功能/质量回归、OOM、服务退出:不提升为默认,保存失败证据。 +- 启动要求仍能容纳 128K,不通过降低上下文或关闭视觉让候选通过。 +- 同类计时三次中位数;约 5% 以内变化视为证据不足,优先保留原配置。 +- 选择综合收益,不只取峰值速度。步数增加但速度下降、首 token 明显恶化或缓存更少,应记录并拒绝。 +- batch 1024 仅在 KV/长上下文或显存收益可见、且生成速度没有明显下降时进入完整验证;如纯粹降低 prefill 吞吐则恢复 2048。 +- 不再提高 0.985 显存比例,不增加并发。实验中关闭自动重启,避免失败循环加载;正式配置使用 unless-stopped。 + +状态:四组筛选完成,各 44 条记录通过;默认配置保持不变。所有回执存入 audit/tuning/,旧验收证据保留。 + +## 步数筛选的中间结论 + +三类各 3 次 2048-token 计时,中位数 token/s: + +| 步数 | 代码 | 中文 | 推理 | +|---|---:|---:|---:| +| 1 | 109.63 | 112.61 | 108.28 | +| 2 | 123.72 | 128.75 | 122.11 | +| 3 | 123.55 | 134.24 | 123.53 | + +MTP 1 明显更慢;MTP 3 的最大提升约 4.3%,其余基本相同,未达到事先设定的约 5% 明显收益门槛。继续以 MTP 2 做 batch 1024 筛选,不因短算术请求的高速度换成 MTP 3。 + +## batch 1024 筛选结论 + +保持 MTP 2 后,KV 预算从 2.78 GiB 增至 2.96 GiB,容量从 146622 增至 155509 token(约 +6.1%)。三类解码速度基本不变,但长输入首 token 变慢: + +| 请求 TTFT | batch 2048 | batch 1024 | +|---|---:|---:| +| 接近 128K 首次检索 | 14.426 秒 | 17.993 秒 | +| 相同前缀再次检索 | 0.702 秒 | 0.854 秒 | +| 长文本+图片 | 12.815 秒 | 15.955 秒 | + +额外缓存没有改善本组长上下文请求,首 token 反而约慢 22%–25%。因此不将 1024 提升为默认,也不因理论上的缓存增量继续完整压力验证。未测试不同于本组的真实业务分布,不能宣称 1024 在所有场景都更差。 + +## 最终选择 + +继续保持 **MTP 2、batch 2048、128K、FP8 KV、0.985 GPU 预算、单活动请求、保留视觉能力**。MTP 1 更慢;MTP 3 长输出收益不明显且缓存更少;batch 1024 缓存略增但长输入更慢。本轮没有发现足以替代现有默认值的配置。 + +完整持续负载只对收益明确的候选开展。本轮候选在筛选阶段没有达到收益标准,所以保留此前已完成完整验证的配置,不将筛选测试描述成新的完整压力测试。 + +KV 容量补充:MTP 1/2/3 在相同 2.78 GiB KV 预算下分别为 159669 / 146622 / 137414 token。MTP 3 的 128K 余量最小,因此没有理由为最多约 4.3% 的单项计时增益替换默认步数。 + +验收归档:4 组各 44 条记录,共 176 条筛选/质量记录通过;两段资源监测均未发生主机 OOM 或触发保护。试验结束后配置与试验前逐字一致,容器 healthy、OOMKilled=false、restart=unless-stopped。详见 `audit/tuning/summary.json`、各组原始记录和 `audit/tuning/final-state.json`。 diff --git a/scripts/qualification.py b/scripts/qualification.py index 7e9b7b3..f42205b 100644 --- a/scripts/qualification.py +++ b/scripts/qualification.py @@ -1,7 +1,7 @@ """Paired API qualification. Run identically on baseline and MTP; synthetic data only.""" import argparse,base64,concurrent.futures,hashlib,json,random,time,urllib.request from pathlib import Path -p=argparse.ArgumentParser();p.add_argument('--output',type=Path,required=True);p.add_argument('--soak-seconds',type=int,default=600);p.add_argument('--assets',type=Path,default=Path('/data/flash-next/cache/qualification'));a=p.parse_args() +p=argparse.ArgumentParser();p.add_argument('--output',type=Path,required=True);p.add_argument('--soak-seconds',type=int,default=600);p.add_argument('--screen',action='store_true',help='Skip 16K output and soak for parameter screening');p.add_argument('--assets',type=Path,default=Path('/data/flash-next/cache/qualification'));a=p.parse_args() key=(Path(__file__).resolve().parents[1]/'secrets/api-key').read_text().strip();url='http://127.0.0.1:8000';model='qwen3.8-flash-next' def req(path,body=None): return urllib.request.urlopen(urllib.request.Request(url+path,data=None if body is None else json.dumps(body).encode(),headers={'Authorization':'Bearer '+key,'Content-Type':'application/json'}),timeout=1200) @@ -65,7 +65,7 @@ with a.output.open('x') as out: for tag,prompt in prompts: record(run('warmup-'+tag,text(prompt),256,forced=True,effort='xhigh')) for i in range(3):record(run(f'perf-{tag}-{i}',text(prompt),2048,forced=True,effort='xhigh')) - record(run('long-output-16384',text('Write an extensive technical textbook chapter explaining algorithms, with many worked examples and Python implementations. Continue with additional examples throughout.'),16384,forced=True,effort='xhigh')) + if not a.screen:record(run('long-output-16384',text('Write an extensive technical textbook chapter explaining algorithms, with many worked examples and Python implementations. Continue with additional examples throughout.'),16384,forced=True,effort='xhigh')) # Exact token count, fresh random filler, two needles at separated positions. rng=random.Random(927);lines=[f'Item {i:06d}: batch {rng.randrange(10**8):08d}, archived in shelf {rng.randrange(1000):03d}.\n' for i in range(16000)] lines[20]='The northern access code is NORTH-49273.\n';lines[2500]='The southern access code is SOUTH-18362.\n' @@ -87,7 +87,7 @@ with a.output.open('x') as out: fs=[pool.submit(run,f'queued-{i}',text(f'Compute {43+i} * 17. Reply only with the integer.'),2048,str((43+i)*17)) for i in range(4)] for f in concurrent.futures.as_completed(fs):record(f.result()) start=time.monotonic();i=0 - while time.monotonic()-start')) + try: + req=urllib.request.Request('http://127.0.0.1:8000/metrics',headers={'Authorization':'Bearer '+key}) + with urllib.request.urlopen(req,timeout=5) as r:metrics=r.read().decode() + (folder/'mtp-metrics.txt').write_text('\n'.join(l for l in metrics.splitlines() if 'spec_decode' in l)) + except Exception:pass +monlog=(audit/'resources.jsonl').open('x');monitor=subprocess.Popen(['python3','scripts/monitor.py'],stdout=monlog,stderr=subprocess.STDOUT) +(audit/'monitor.pid').write_text(str(monitor.pid)) +try: + for n in a.steps: + folder=audit/f'mtp{n}-b{a.batch}';folder.mkdir() + config=original.replace('num_speculative_tokens":2','num_speculative_tokens":'+str(n)).replace('"cudagraph_capture_sizes":[1,3]','"cudagraph_capture_sizes":[1,'+str(n+1)+']').replace('--max-num-batched-tokens 2048','--max-num-batched-tokens '+str(a.batch)).replace('restart: unless-stopped','restart: "no"') + (folder/'compose.yaml').write_text(config) + # Run current MTP2 as-is for fresh paired control; all later candidates restart. + same=n==2 and a.batch==2048 and a.steps.index(n)==0 + if not same: + (root/'compose.yaml').write_text(config);cmd(['docker','compose','--progress','plain','up','-d','--force-recreate'],timeout=120) + if not ready(): + capture(folder);(folder/'status.json').write_text(json.dumps({'startup':False}));print('STARTUP_FAILED',folder,flush=True);continue + for script,name,extra in [('qualification.py','screen.jsonl',['--screen']),('quality_tasks.py','quality-medium.jsonl',['--effort','medium'])]: + with (folder/(name+'.log')).open('x') as log: + result=subprocess.run(['python3','scripts/'+script,'--output',str(folder/name)]+extra,stdout=log,stderr=subprocess.STDOUT,timeout=1200) + if result.returncode!=0:print('SCRIPT_FAILED',folder,script,result.returncode,flush=True) + capture(folder);rows=[] + for name in ['screen.jsonl','quality-medium.jsonl']: + f=folder/name + if f.exists():rows += [json.loads(l) for l in f.read_text().splitlines()] + status={'startup':True,'records':len(rows),'failures':[r['name'] for r in rows if not r['passed']]} + (folder/'status.json').write_text(json.dumps(status,indent=2));print('SCREEN_COMPLETE',folder,status,flush=True) +finally: + (root/'compose.yaml').write_text(original) + cmd(['docker','compose','--progress','plain','up','-d','--force-recreate'],timeout=120) + restored=ready();(audit/'restored.json').write_text(json.dumps({'healthy':restored,'time':time.time()})) + monitor.terminate();monitor.wait(timeout=15);monlog.close() + print('RESTORED',restored,flush=True)