diff --git a/audit/runtime/acceptance-after-mtp.jsonl b/audit/runtime/acceptance-after-mtp.jsonl new file mode 100644 index 0000000..8a9afa2 --- /dev/null +++ b/audit/runtime/acceptance-after-mtp.jsonl @@ -0,0 +1,4 @@ +{"models": {"object": "list", "data": [{"id": "qwen3.8-flash-next", "object": "model", "created": 1789700529, "owned_by": "vllm", "root": "/model", "parent": null, "max_model_len": 131072, "permission": [{"id": "modelperm-933247eb769488b6", "object": "model_permission", "created": 1789700529, "allow_create_engine": false, "allow_sampling": true, "allow_logprobs": true, "allow_search_indices": false, "allow_view": true, "allow_fine_tuning": false, "organization": "*", "group": null, "is_blocking": false}]}]}} +{"test": "math", "passed": true, "content": "\n\n323", "reasoning_chars": 111, "ttft_s": 0.20431093200022588, "elapsed_s": 0.9834625690000394, "finish_reason": "stop", "usage": {"prompt_tokens": 55, "total_tokens": 115, "completion_tokens": 60, "completion_tokens_details": {"reasoning_tokens": 54}}} +{"test": "chinese", "passed": true, "content": "\n\n模型已就绪", "reasoning_chars": 53, "ttft_s": 0.07846190500004013, "elapsed_s": 0.5019190799994249, "finish_reason": "stop", "usage": {"prompt_tokens": 47, "total_tokens": 80, "completion_tokens": 33, "completion_tokens_details": {"reasoning_tokens": 27}}} +{"test": "tool", "passed": true, "response": {"id": "chatcmpl-b6b3720854c14b05", "object": "chat.completion", "created": 1789700530, "model": "qwen3.8-flash-next", "choices": [{"index": 0, "message": {"role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [{"id": "chatcmpl-tool-a6205fc98e7ef570", "type": "function", "function": {"name": "get_weather", "arguments": "{\"city\": \"Shanghai\"}"}}], "reasoning": "The user wants me to check the weather in Shanghai using the get_weather tool. This is a straightforward single tool call.\n"}, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null, "routed_experts": null}], "service_tier": null, "system_fingerprint": "vllm-0.3.1.dev3+g0bfc7a15d-9c4a3436", "usage": {"prompt_tokens": 301, "total_tokens": 355, "completion_tokens": 54, "prompt_tokens_details": null, "completion_tokens_details": {"reasoning_tokens": 25}}, "prompt_logprobs": null, "prompt_token_ids": null, "prompt_text": null, "kv_transfer_params": null, "ec_transfer_params": null, "metrics": null}} diff --git a/audit/runtime/mtp2-0985-server.log b/audit/runtime/mtp2-0985-server.log index a5ed3e9..04f0fc0 100644 --- a/audit/runtime/mtp2-0985-server.log +++ b/audit/runtime/mtp2-0985-server.log @@ -1,9 +1,9 @@ -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] █ █ █▄ ▄█ -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] █▄█▀ █ █ █ █ model /model -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ -(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] █ █ █▄ ▄█ +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] ▄▄ ▄█ █ █ █ ▀▄▀ █ version 0.3.1.dev3+g0bfc7a15d +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] █▄█▀ █ █ █ █ model /model +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] ▀▀ ▀▀▀▀▀ ▀▀▀▀▀ ▀ ▀ +(APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:347] (APIServer pid=1) INFO 09-18 02:55:12 [api_utils.py:286] non-default args: {'model_tag': '/model', 'enable_auto_tool_choice': True, 'tool_call_parser': 'qwen3_xml', 'host': '0.0.0.0', 'api_key': '***', 'model': '/model', 'dtype': 'bfloat16', 'max_model_len': 131072, 'served_model_name': ['qwen3.8-flash-next'], 'load_format': 'safetensors', 'reasoning_parser': 'qwen3', 'gpu_memory_utilization': 0.985, 'kv_cache_dtype': 'fp8', 'enable_prefix_caching': True, 'max_num_batched_tokens': 2048, 'max_num_seqs': 1, 'enable_chunked_prefill': True, 'enable_flashinfer_autotune': False, 'speculative_config': {'method': 'mtp', 'num_speculative_tokens': 2}, 'compilation_config': {'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': [], 'ir_enable_torch_wrap': None, 'splitting_ops': None, 'compile_mm_encoder': False, 'cudagraph_mm_encoder': False, 'encoder_cudagraph_token_budgets': [], 'encoder_cudagraph_max_vision_items_per_batch': 0, 'encoder_cudagraph_max_frames_per_batch': None, 'compile_sizes': None, 'compile_ranges_endpoints': None, 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 0, 'cudagraph_capture_sizes': [1, 3], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': None, 'pass_config': {}, 'max_cudagraph_capture_size': None, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False, 'assume_32_bit_indexing': False}, 'local_cache_dir': None, 'fast_moe_cold_start': None, 'static_all_moe_layers': []}, 'engram_config': EngramConfig(cpu_offload=True, embedding_across_dp=False, dp_shared_memory=False)} (APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} (APIServer pid=1) [transformers] Unrecognized keys in `rope_parameters` for 'rope_type'='default': {'mrope_interleaved', 'mrope_section'} @@ -63,19 +63,31 @@ (EngineCore pid=112) WARNING 09-18 02:56:06 [compilation.py:1350] Op 'quant_fp8' not present in model, enabling with '+quant_fp8' has no effect (EngineCore pid=112) INFO 09-18 02:56:06 [weight_utils.py:900] Filesystem type for checkpoints: EXT4. Checkpoint size: 123.57 GiB. Available RAM: 142.29 GiB. (EngineCore pid=112) INFO 09-18 02:56:06 [weight_utils.py:923] Auto-prefetch is disabled because the filesystem (EXT4) is not a recognized network FS (NFS/Lustre). If you want to force prefetching, start vLLM with --safetensors-load-strategy=prefetch. -(EngineCore pid=112) Loading safetensors checkpoint shards: 0% Completed | 0/11 [00:00