Test ordered one-slot DFlash2 placements and preserve results
This commit is contained in:
1 parent
3d5931146b
commit
4007242709
25 files changed
+360
-5
No files matched your search
@@ -55,3 +55,32 @@ Target loaded; draft allocation of1079.61MiB onCUDA0 failed with CUDA malloc
|
||||
OOM. No benchmark inference executed. Supervisor restored original production;
|
||||
health/smoke/kernel checks passed. No performance or quality conclusion can be
|
||||
drawn. See results/ and restore-verification.json.
|
||||
|
||||
## User-ordered one-slot follow-up
|
||||
|
||||
The user authorized this order after the initial two-slot allocation failure:
|
||||
|
||||
1. One slot, target80:20, draft on5080.
|
||||
2. One slot, target80:20, draft on3060, regardless of whether the first works.
|
||||
3. Only if step2 fails: one slot, target70:30, draft back on5080. Moving both
|
||||
additional target layers and the draft onto3060 would compete for its memory;
|
||||
this fallback instead makes space on5080 for the draft.
|
||||
|
||||
Context remains160000 and other quick-probe settings are unchanged. These tests
|
||||
are temporary; production containers retain their original two-slot arguments.
|
||||
Case JSON now controls slot count and draft device (historical quick.json keeps
|
||||
its original two-slot meaning). Failed attempts include an explicit error field.
|
||||
|
||||
An active production request postpones startup for up to180 seconds; no live
|
||||
request is deliberately interrupted. The existing drain check still protects
|
||||
requests racing with the transition. Each attempt has its own result directory;
|
||||
previous results and the frozen Qwen reference are never overwritten.
|
||||
|
||||
## One-slot results
|
||||
|
||||
Priority1 fails during draft graph initialization: output.weight resides onCUDA1
|
||||
but draft backends are restricted toCUDA0. Priority2 works; priority3 therefore
|
||||
not run. German37.79 tok/s, code75.51 tok/s, 4196-token prefill1437.56 tok/s,
|
||||
recall3/3. Allocation160000 is not full-context validation. Original two-slot
|
||||
production restored and checked after both cases. See
|
||||
../../docs/DFLASH2_ONE_SLOT_20260920.md for comparison caveats and quality findings.
|
||||
@@ -0,0 +1,20 @@
|
||||
import ast,asyncio,json,pathlib,re,typing,gzip
|
||||
p=pathlib.Path(__file__).resolve().parent/'results/medium-slot1-priority2-draft-cuda1/result.json.gz'
|
||||
d=json.loads(gzip.decompress(p.read_bytes()));content=d['decode'][1]['choices'][0]['message']['content']
|
||||
source=re.search(r'```python\s*\n(.*?)```',content,re.S).group(1)
|
||||
# Reviewed above: only asyncio task orchestration and exception handling, no I/O.
|
||||
node=next(n for n in ast.parse(source).body if isinstance(n,ast.AsyncFunctionDef) and n.name=='first_success')
|
||||
scope={'asyncio':asyncio,'Any':typing.Any,'List':typing.List,'Awaitable':typing.Awaitable}
|
||||
exec(compile(ast.Module(body=[node],type_ignores=[]),'<reviewed-dflash-answer>','exec'),scope)
|
||||
async def check(all_fail):
|
||||
async def task(delay,fail):
|
||||
await asyncio.sleep(delay)
|
||||
if fail:raise ValueError('synthetic failure')
|
||||
return 7
|
||||
try:
|
||||
r=await asyncio.wait_for(scope['first_success']([task(.001,True),task(.005,all_fail)]),timeout=1)
|
||||
return {'returned':r,'pass':not all_fail and r==7}
|
||||
except Exception as e:
|
||||
return {'error':type(e).__name__+': '+str(e),'pass':all_fail and type(e).__name__=='ExceptionGroup'}
|
||||
out={'scope':'Two checks of manually reviewed decode answer, not broad quality evaluation','fast_failure_then_success':asyncio.run(check(False)),'all_fail':asyncio.run(check(True))}
|
||||
print(json.dumps(out,indent=2))
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"scope": "Two checks of manually reviewed decode answer, not broad quality evaluation",
|
||||
"fast_failure_then_success": {
|
||||
"returned": 7,
|
||||
"pass": true
|
||||
},
|
||||
"all_fail": {
|
||||
"error": "NameError: name 'builtins' is not defined",
|
||||
"pass": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"checked_at": 1789931025.6105647,
|
||||
"uptime": "21:03:45 up 3 days, 9:58, 1 user, load average: 1.34, 0.91, 0.73",
|
||||
"containers": {
|
||||
"mike-ai-llama-medium": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"started": "2026-09-20T19:03:00.899996682Z"
|
||||
},
|
||||
"mike-ai-router": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
|
||||
"started": "2026-09-20T19:03:11.359298419Z"
|
||||
},
|
||||
"mike-ai-profile-controller": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
|
||||
"started": "2026-09-20T19:03:11.213886021Z"
|
||||
},
|
||||
"mike-ai-wireguard-gateway": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
|
||||
"started": "2026-09-17T09:05:07.164533904Z"
|
||||
},
|
||||
"mike-ai-qwen3-tts": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
|
||||
"started": "2026-09-19T13:47:00.227813668Z"
|
||||
}
|
||||
},
|
||||
"router": {
|
||||
"/health": {
|
||||
"status": "ok",
|
||||
"router": "alive"
|
||||
},
|
||||
"/ready": {
|
||||
"status": "ok",
|
||||
"router": "alive",
|
||||
"upstream": "ready"
|
||||
}
|
||||
},
|
||||
"smoke": {
|
||||
"content": "OK",
|
||||
"usage": {
|
||||
"completion_tokens": 2,
|
||||
"prompt_tokens": 19,
|
||||
"total_tokens": 21,
|
||||
"prompt_tokens_details": {
|
||||
"cached_tokens": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
"kernel_errors": [],
|
||||
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10930 MiB, 12288 MiB, 56\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 61",
|
||||
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data"
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
[
|
||||
{
|
||||
"label": "medium-slot1-priority1-draft-cuda0",
|
||||
"model": "pure",
|
||||
"ctx": 160000,
|
||||
"single": false,
|
||||
"split": "80,20",
|
||||
"ubatch": 128,
|
||||
"parallel": 1,
|
||||
"draft_max": 7,
|
||||
"draft_kv": "q4_0",
|
||||
"draft_device": "CUDA0",
|
||||
"quality": false,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 512
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"checked_at": 1789931253.8391275,
|
||||
"uptime": "21:07:33 up 3 days, 10:02, 1 user, load average: 0.93, 0.95, 0.80",
|
||||
"containers": {
|
||||
"mike-ai-llama-medium": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"started": "2026-09-20T19:07:06.261497776Z"
|
||||
},
|
||||
"mike-ai-router": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
|
||||
"started": "2026-09-20T19:07:16.676662725Z"
|
||||
},
|
||||
"mike-ai-profile-controller": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
|
||||
"started": "2026-09-20T19:07:16.548271903Z"
|
||||
},
|
||||
"mike-ai-wireguard-gateway": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
|
||||
"started": "2026-09-17T09:05:07.164533904Z"
|
||||
},
|
||||
"mike-ai-qwen3-tts": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
|
||||
"started": "2026-09-19T13:47:00.227813668Z"
|
||||
}
|
||||
},
|
||||
"router": {
|
||||
"/health": {
|
||||
"status": "ok",
|
||||
"router": "alive"
|
||||
},
|
||||
"/ready": {
|
||||
"status": "ok",
|
||||
"router": "alive",
|
||||
"upstream": "ready"
|
||||
}
|
||||
},
|
||||
"smoke": {
|
||||
"content": "OK",
|
||||
"usage": {
|
||||
"completion_tokens": 2,
|
||||
"prompt_tokens": 19,
|
||||
"total_tokens": 21,
|
||||
"prompt_tokens_details": {
|
||||
"cached_tokens": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
"kernel_errors": [],
|
||||
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10920 MiB, 12288 MiB, 45\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 49",
|
||||
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data"
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
[
|
||||
{
|
||||
"label": "medium-slot1-priority2-draft-cuda1",
|
||||
"model": "pure",
|
||||
"ctx": 160000,
|
||||
"single": false,
|
||||
"split": "80,20",
|
||||
"ubatch": 128,
|
||||
"parallel": 1,
|
||||
"draft_max": 7,
|
||||
"draft_kv": "q4_0",
|
||||
"draft_device": "CUDA1",
|
||||
"quality": false,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 512
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,19 @@
|
||||
[
|
||||
{
|
||||
"label": "medium-slot1-priority3-draft-cuda0",
|
||||
"model": "pure",
|
||||
"ctx": 160000,
|
||||
"single": false,
|
||||
"split": "70,30",
|
||||
"ubatch": 128,
|
||||
"parallel": 1,
|
||||
"draft_max": 7,
|
||||
"draft_kv": "q4_0",
|
||||
"draft_device": "CUDA0",
|
||||
"quality": false,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 512
|
||||
}
|
||||
]
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
@@ -54,8 +54,8 @@ def run_case(case):
|
||||
save(out/'config.json',case)
|
||||
print('START',label,flush=True)
|
||||
single=case.get('single',False)
|
||||
args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel','2','--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3']
|
||||
args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device','CUDA0','--spec-draft-ngl','all']
|
||||
args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel',str(case.get('parallel',2)),'--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3']
|
||||
args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device',case.get('draft_device','CUDA0'),'--spec-draft-ngl','all']
|
||||
save(out/'server-args.json',args)
|
||||
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
|
||||
stop=threading.Event(); samples=[]
|
||||
@@ -117,6 +117,9 @@ def run_case(case):
|
||||
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
|
||||
print(label,'prefill',n,r.get('timings'),flush=True)
|
||||
result['finished']=time.time()
|
||||
except Exception as exc:
|
||||
result['error']=str(exc)
|
||||
raise
|
||||
finally:
|
||||
stop.set(); thread.join(5)
|
||||
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
|
||||
|
||||
@@ -10,8 +10,12 @@ signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_sign
|
||||
# Refuse if the known production state has changed, or if requests are active.
|
||||
for name in NAMES:
|
||||
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
assert not any(s['is_processing'] for s in slots),'Production request active'
|
||||
for attempt in range(60):
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
if not any(s['is_processing'] for s in slots): break
|
||||
if attempt==0: print('WAIT production request active; no interruption',flush=True)
|
||||
time.sleep(3)
|
||||
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
|
||||
assert not run('docker','ps','-q','--filter','name=^mike-ai-dflash2-test$').stdout.strip(),'Existing experiment'
|
||||
child=None
|
||||
try:
|
||||
@@ -36,5 +40,11 @@ finally:
|
||||
for _ in range(150):
|
||||
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored Medium did not become healthy')
|
||||
run('docker','start',NAMES[1],NAMES[0])
|
||||
print('RESTORED existing medium/controller/router',flush=True)
|
||||
for _ in range(60):
|
||||
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
|
||||
if all(status=='healthy' for status in statuses): break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored router/controller did not become healthy')
|
||||
print('RESTORED existing medium/controller/router; all healthy',flush=True)
|
||||
Reference in new issue
Block a user