{ "batch-1024/mtp2-b1024": { "startup": true, "records": 44, "failures": [], "screen.jsonl": { "count": 37, "failed": [] }, "decode_tps": { "code": 123.50695983978235, "chinese": 128.67815702769335, "reasoning": 122.09499443849444 }, "latency": { "video-120s": { "ttft_s": 2.827090976003092, "elapsed_s": 3.731427850005275, "passed": true }, "multi-image": { "ttft_s": 1.7833786369956215, "elapsed_s": 2.3456459599983646, "passed": true }, "long-context-0": { "ttft_s": 17.993200959004753, "elapsed_s": 18.83731006600283, "passed": true }, "long-context-1": { "ttft_s": 0.8536870109965093, "elapsed_s": 1.7167949249997037, "passed": true }, "long-context-image": { "ttft_s": 15.955126308996114, "elapsed_s": 17.181451591997757, "passed": true } }, "quality-medium.jsonl": { "count": 7, "failed": [] }, "memory_log": [ "(EngineCore pid=113) INFO 09-18 15:13:31 [model_runner.py:419] Model loading took 76.23 GiB memory and 86.731724 seconds", "(EngineCore pid=113) INFO 09-18 15:14:01 [gpu_worker.py:641] Available KV cache memory: 2.96 GiB", "(EngineCore pid=113) INFO 09-18 15:14:01 [kv_cache_utils.py:2404] GPU KV cache size: 155,509 tokens, Maximum concurrency for 131,072 tokens per request: 1.19x" ] }, "steps-123/mtp1-b2048": { "startup": true, "records": 44, "failures": [], "screen.jsonl": { "count": 37, "failed": [] }, "decode_tps": { "code": 109.6317739856207, "chinese": 112.60621062069599, "reasoning": 108.27878210561818 }, "latency": { "video-120s": { "ttft_s": 2.490573155002494, "elapsed_s": 3.5669434599985834, "passed": true }, "multi-image": { "ttft_s": 1.4522779820035794, "elapsed_s": 2.0256833830062533, "passed": true }, "long-context-0": { "ttft_s": 14.440546402998734, "elapsed_s": 15.590460605999397, "passed": true }, "long-context-1": { "ttft_s": 0.856336849006766, "elapsed_s": 1.9747926430063671, "passed": true }, "long-context-image": { "ttft_s": 12.791386062002857, "elapsed_s": 14.462968651001574, "passed": true } }, "quality-medium.jsonl": { "count": 7, "failed": [] }, "memory_log": [ "(EngineCore pid=112) INFO 09-18 14:55:51 [model_runner.py:419] Model loading took 76.36 GiB memory and 89.100896 seconds", "(EngineCore pid=112) INFO 09-18 14:56:21 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", "(EngineCore pid=112) INFO 09-18 14:56:21 [kv_cache_utils.py:2404] GPU KV cache size: 159,669 tokens, Maximum concurrency for 131,072 tokens per request: 1.22x" ] }, "steps-123/mtp2-b2048": { "startup": true, "records": 44, "failures": [], "screen.jsonl": { "count": 37, "failed": [] }, "decode_tps": { "code": 123.72166768011161, "chinese": 128.75257173627188, "reasoning": 122.10854572395078 }, "latency": { "video-120s": { "ttft_s": 0.8094746360002318, "elapsed_s": 1.7980530589993577, "passed": true }, "multi-image": { "ttft_s": 1.4533180090002134, "elapsed_s": 1.9994508850068087, "passed": true }, "long-context-0": { "ttft_s": 14.42599186499865, "elapsed_s": 15.14422588799789, "passed": true }, "long-context-1": { "ttft_s": 0.7020562550023897, "elapsed_s": 1.4927438169979723, "passed": true }, "long-context-image": { "ttft_s": 12.815331379999407, "elapsed_s": 14.372689276002347, "passed": true } }, "quality-medium.jsonl": { "count": 7, "failed": [] }, "memory_log": [ "(EngineCore pid=113) INFO 09-18 14:15:53 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.461433 seconds", "(EngineCore pid=113) INFO 09-18 14:16:24 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", "(EngineCore pid=113) INFO 09-18 14:16:24 [kv_cache_utils.py:2404] GPU KV cache size: 146,622 tokens, Maximum concurrency for 131,072 tokens per request: 1.12x" ] }, "steps-123/mtp3-b2048": { "startup": true, "records": 44, "failures": [], "screen.jsonl": { "count": 37, "failed": [] }, "decode_tps": { "code": 123.54931448950835, "chinese": 134.24473934955645, "reasoning": 123.52862404436223 }, "latency": { "video-120s": { "ttft_s": 2.5409359719997155, "elapsed_s": 3.243877524000709, "passed": true }, "multi-image": { "ttft_s": 1.463653916005569, "elapsed_s": 1.8863289360015187, "passed": true }, "long-context-0": { "ttft_s": 14.516383726004278, "elapsed_s": 15.218214432999957, "passed": true }, "long-context-1": { "ttft_s": 14.510828671998752, "elapsed_s": 15.23376362700219, "passed": true }, "long-context-image": { "ttft_s": 12.83541851100017, "elapsed_s": 13.968540412002767, "passed": true } }, "quality-medium.jsonl": { "count": 7, "failed": [] }, "memory_log": [ "(EngineCore pid=112) INFO 09-18 15:03:06 [model_runner.py:419] Model loading took 76.36 GiB memory and 86.716523 seconds", "(EngineCore pid=112) INFO 09-18 15:03:37 [gpu_worker.py:641] Available KV cache memory: 2.78 GiB", "(EngineCore pid=112) INFO 09-18 15:03:37 [kv_cache_utils.py:2404] GPU KV cache size: 137,414 tokens, Maximum concurrency for 131,072 tokens per request: 1.05x" ] } }