docs(perf): profile CUDA graph coverage and PLE costs on Spark

This commit is contained in:
2026-09-18 01:22:13 +08:00
parent f1a8964072
commit d7e1e745c3
23 changed files with 8384 additions and 1 deletions
+67
View File
@@ -0,0 +1,67 @@
import argparse,json,hashlib,time,urllib.request,signal
from pathlib import Path
parser=argparse.ArgumentParser();parser.add_argument('--label',required=True);args=parser.parse_args()
base='http://127.0.0.1:8000'
key=Path('/run/secrets/qwen_api_key').read_text().strip()
headers={'Authorization':'Bearer '+key,'Content-Type':'application/json'}
def emit(r):print(json.dumps({'label':args.label,'time':time.time(),**r},ensure_ascii=False),flush=True)
def request(path,payload=None):
req=urllib.request.Request(base+path,headers=headers,data=json.dumps(payload).encode() if payload is not None else None)
with urllib.request.urlopen(req,timeout=60) as res:return res.read()
def metrics():
result={}
for l in request('/metrics').decode().splitlines():
if l.startswith('vllm:ple_mmap_') or l.startswith('vllm:prefix_cache_'):
name=l.split('{')[0].split()[0]
if name.endswith('_total'):result[name]=result.get(name,0)+float(l.split()[-1])
return result
def deadline(*_):raise TimeoutError('Request exceeded 180 seconds')
signal.signal(signal.SIGALRM,deadline)
def run(name,prompt,expected=None,profile=False):
m=[{'role':'user','content':prompt}]
count=json.loads(request('/tokenize',{'model':'qwen3.8-flash-next','messages':m,'chat_template_kwargs':{'reasoning_effort':'low'}}))['count']
assert count+2048<=131072
before=metrics();payload={'model':'qwen3.8-flash-next','messages':m,'temperature':0,'seed':42,'reasoning_effort':'low','max_tokens':2048 if expected else 512,'stream':True,'stream_options':{'include_usage':True}}
req=urllib.request.Request(base+'/v1/chat/completions',headers=headers,data=json.dumps(payload,ensure_ascii=False).encode())
emit({'event':'REQUEST_START','name':name,'profile':profile,'prompt_tokens_preflight':count})
if profile:request('/start_profile',{})
t=time.monotonic();first=last=visible=None;content=reasoning='';usage={};finish=None
signal.alarm(180)
try:
with urllib.request.urlopen(req,timeout=180) as res:
for raw in res:
line=raw.decode().strip()
if not line.startswith('data: ') or line=='data: [DONE]':continue
x=json.loads(line[6:])
if x.get('error'):raise RuntimeError('Streaming error')
if x.get('usage'):usage=x['usage']
for c in x.get('choices',[]):
d=c.get('delta',{});a=d.get('content') or '';r=d.get('reasoning') or d.get('reasoning_content') or ''
if a or r:
now=time.monotonic();first=first if first is not None else now;last=now
if a:visible=visible if visible is not None else now
content+=a;reasoning+=r;finish=c.get('finish_reason') or finish
finally:
signal.alarm(0)
if profile:request('/stop_profile',{})
elapsed=time.monotonic()-t
after=metrics()
correct=bool(content.strip()) and (expected is None or content.strip()==expected) and finish=='stop'
emit({'event':'RESULT','name':name,'profile':profile,'correct':correct,'content':content,'finish_reason':finish,'usage':usage,'max_tokens':payload['max_tokens'],'prompt_sha256':hashlib.sha256(prompt.encode()).hexdigest(),'output_sha256':hashlib.sha256((reasoning+'\0'+content).encode()).hexdigest(),'ttft_s':first-t if first else None,'first_content_s':visible-t if visible else None,'elapsed_s':elapsed,'decode_tps_approx':(usage.get('completion_tokens',0)-1)/(last-first) if last and first and last>first else None,'metrics_delta':{k:after[k]-before.get(k,0) for k in after}})
assert correct and usage['prompt_tokens']==count,name
return prompt
short='用中文写一段约150字的说明,解释数据库索引为什么能加快查询,以及它对写入有什么影响。'
run('warmup','计算17乘19,只输出结果。','323')
# Same prefix fixtures across configurations; phase/round identity is before the document.
fixtures={n:json.loads(Path(f'/profiles/data/ctx-{n}-1.json').read_text())['cases'][1] for n in [8192,32768]}
for i in range(3):
run(f'short-{i}',short)
for n,c in fixtures.items():
prompt=f'测试文档批次 BENCH-{i}-{n}\n'+c['messages'][0]['content']
run(f'prefill-{n}-{i}',prompt,c['expected'])
run(f'reuse-{n}-{i}',prompt,c['expected'])
run('trace-short',short,profile=True)
c=fixtures[32768];prompt='测试文档批次 TRACE-32768。\n'+c['messages'][0]['content']
run('trace-prefill-32768',prompt,c['expected'],profile=True)
run('trace-reuse-32768',prompt,c['expected'],profile=True)
emit({'event':'PROFILE_SUITE_PASS'})