Files
qwen38-flash-next-rtx6000d/audit/tuning/summary.json
T

203 lines
5.9 KiB
JSON

{
"batch-1024/mtp2-b1024": {
"startup": true,
"records": 44,
"failures": [],
"screen.jsonl": {
"count": 37,
"failed": []
},
"decode_tps": {
"code": 123.50695983978235,
"chinese": 128.67815702769335,
"reasoning": 122.09499443849444
},
"latency": {
"video-120s": {
"ttft_s": 2.827090976003092,
"elapsed_s": 3.731427850005275,
"passed": true
},
"multi-image": {
"ttft_s": 1.7833786369956215,
"elapsed_s": 2.3456459599983646,
"passed": true
},
"long-context-0": {
"ttft_s": 17.993200959004753,
"elapsed_s": 18.83731006600283,
"passed": true
},
"long-context-1": {
"ttft_s": 0.8536870109965093,
"elapsed_s": 1.7167949249997037,
"passed": true
},
"long-context-image": {
"ttft_s": 15.955126308996114,
"elapsed_s": 17.181451591997757,
"passed": true
}
},
"quality-medium.jsonl": {
"count": 7,
"failed": []
},
"memory_log": [
"(EngineCore pid=113) INFO 09-18 15:13:31 [model_runner.py:419] Model loading took 76.23 GiB memory and 86.731724 seconds",
"(EngineCore pid=113) INFO 09-18 15:14:01 [gpu_worker.py:641] Available KV cache memory: 2.96 GiB",
"(EngineCore pid=113) INFO 09-18 15:14:01 [kv_cache_utils.py:2404] GPU KV cache size: 155,509 tokens, Maximum concurrency for 131,072 tokens per request: 1.19x"
]
},
"steps-123/mtp1-b2048": {
"startup": true,
"records": 44,
"failures": [],
"screen.jsonl": {
"count": 37,
"failed": []
},
"decode_tps": {
"code": 109.6317739856207,
"chinese": 112.60621062069599,
"reasoning": 108.27878210561818
},
"latency": {
"video-120s": {
"ttft_s": 2.490573155002494,
"elapsed_s": 3.5669434599985834,
"passed": true
},
"multi-image": {
"ttft_s": 1.4522779820035794,
"elapsed_s": 2.0256833830062533,
"passed": true
},
"long-context-0": {
"ttft_s": 14.440546402998734,
"elapsed_s": 15.590460605999397,
"passed": true
},
"long-context-1": {
"ttft_s": 0.856336849006766,
"elapsed_s": 1.9747926430063671,
"passed": true
},
"long-context-image": {
"ttft_s": 12.791386062002857,
"elapsed_s": 14.462968651001574,
"passed": true
}
},
"quality-medium.jsonl": {
"count": 7,
"failed": []
},
"memory_log": [
"(EngineCore pid=112) INFO 09-18 14:55:51 [model_runner.py:419] Model loading took 76.36 GiB memory and 89.100896 seconds",
"(EngineCore pid=112) INFO 09-18 14:56:21 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB",
"(EngineCore pid=112) INFO 09-18 14:56:21 [kv_cache_utils.py:2404] GPU KV cache size: 159,669 tokens, Maximum concurrency for 131,072 tokens per request: 1.22x"
]
},
"steps-123/mtp2-b2048": {
"startup": true,
"records": 44,
"failures": [],
"screen.jsonl": {
"count": 37,
"failed": []
},
"decode_tps": {
"code": 123.72166768011161,
"chinese": 128.75257173627188,
"reasoning": 122.10854572395078
},
"latency": {
"video-120s": {
"ttft_s": 0.8094746360002318,
"elapsed_s": 1.7980530589993577,
"passed": true
},
"multi-image": {
"ttft_s": 1.4533180090002134,
"elapsed_s": 1.9994508850068087,
"passed": true
},
"long-context-0": {
"ttft_s": 14.42599186499865,
"elapsed_s": 15.14422588799789,
"passed": true
},
"long-context-1": {
"ttft_s": 0.7020562550023897,
"elapsed_s": 1.4927438169979723,
"passed": true
},
"long-context-image": {
"ttft_s": 12.815331379999407,
"elapsed_s": 14.372689276002347,
"passed": true
}
},
"quality-medium.jsonl": {
"count": 7,
"failed": []
},
"memory_log": [
"(EngineCore pid=113) INFO 09-18 14:15:53 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.461433 seconds",
"(EngineCore pid=113) INFO 09-18 14:16:24 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB",
"(EngineCore pid=113) INFO 09-18 14:16:24 [kv_cache_utils.py:2404] GPU KV cache size: 146,622 tokens, Maximum concurrency for 131,072 tokens per request: 1.12x"
]
},
"steps-123/mtp3-b2048": {
"startup": true,
"records": 44,
"failures": [],
"screen.jsonl": {
"count": 37,
"failed": []
},
"decode_tps": {
"code": 123.54931448950835,
"chinese": 134.24473934955645,
"reasoning": 123.52862404436223
},
"latency": {
"video-120s": {
"ttft_s": 2.5409359719997155,
"elapsed_s": 3.243877524000709,
"passed": true
},
"multi-image": {
"ttft_s": 1.463653916005569,
"elapsed_s": 1.8863289360015187,
"passed": true
},
"long-context-0": {
"ttft_s": 14.516383726004278,
"elapsed_s": 15.218214432999957,
"passed": true
},
"long-context-1": {
"ttft_s": 14.510828671998752,
"elapsed_s": 15.23376362700219,
"passed": true
},
"long-context-image": {
"ttft_s": 12.83541851100017,
"elapsed_s": 13.968540412002767,
"passed": true
}
},
"quality-medium.jsonl": {
"count": 7,
"failed": []
},
"memory_log": [
"(EngineCore pid=112) INFO 09-18 15:03:06 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.716523 seconds",
"(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB",
"(EngineCore pid=112) INFO 09-18 15:03:37 [kv_cache_utils.py:2404] GPU KV cache size: 137,414 tokens, Maximum concurrency for 131,072 tokens per request: 1.05x"
]
}
}