Test ordered one-slot DFlash2 placements and preserve results

This commit is contained in:
Mikei386
2026-09-20 21:10:08 +02:00
parent 3d5931146b
commit 4007242709
25 changed files with 360 additions and 5 deletions
@@ -55,3 +55,32 @@ Target loaded; draft allocation of1079.61MiB onCUDA0 failed with CUDA malloc
OOM. No benchmark inference executed. Supervisor restored original production;
health/smoke/kernel checks passed. No performance or quality conclusion can be
drawn. See results/ and restore-verification.json.
## User-ordered one-slot follow-up
The user authorized this order after the initial two-slot allocation failure:
1. One slot, target80:20, draft on5080.
2. One slot, target80:20, draft on3060, regardless of whether the first works.
3. Only if step2 fails: one slot, target70:30, draft back on5080. Moving both
additional target layers and the draft onto3060 would compete for its memory;
this fallback instead makes space on5080 for the draft.
Context remains160000 and other quick-probe settings are unchanged. These tests
are temporary; production containers retain their original two-slot arguments.
Case JSON now controls slot count and draft device (historical quick.json keeps
its original two-slot meaning). Failed attempts include an explicit error field.
An active production request postpones startup for up to180 seconds; no live
request is deliberately interrupted. The existing drain check still protects
requests racing with the transition. Each attempt has its own result directory;
previous results and the frozen Qwen reference are never overwritten.
## One-slot results
Priority1 fails during draft graph initialization: output.weight resides onCUDA1
but draft backends are restricted toCUDA0. Priority2 works; priority3 therefore
not run. German37.79 tok/s, code75.51 tok/s, 4196-token prefill1437.56 tok/s,
recall3/3. Allocation160000 is not full-context validation. Original two-slot
production restored and checked after both cases. See
../../docs/DFLASH2_ONE_SLOT_20260920.md for comparison caveats and quality findings.
@@ -0,0 +1,20 @@
import ast,asyncio,json,pathlib,re,typing,gzip
p=pathlib.Path(__file__).resolve().parent/'results/medium-slot1-priority2-draft-cuda1/result.json.gz'
d=json.loads(gzip.decompress(p.read_bytes()));content=d['decode'][1]['choices'][0]['message']['content']
source=re.search(r'```python\s*\n(.*?)```',content,re.S).group(1)
# Reviewed above: only asyncio task orchestration and exception handling, no I/O.
node=next(n for n in ast.parse(source).body if isinstance(n,ast.AsyncFunctionDef) and n.name=='first_success')
scope={'asyncio':asyncio,'Any':typing.Any,'List':typing.List,'Awaitable':typing.Awaitable}
exec(compile(ast.Module(body=[node],type_ignores=[]),'<reviewed-dflash-answer>','exec'),scope)
async def check(all_fail):
async def task(delay,fail):
await asyncio.sleep(delay)
if fail:raise ValueError('synthetic failure')
return 7
try:
r=await asyncio.wait_for(scope['first_success']([task(.001,True),task(.005,all_fail)]),timeout=1)
return {'returned':r,'pass':not all_fail and r==7}
except Exception as e:
return {'error':type(e).__name__+': '+str(e),'pass':all_fail and type(e).__name__=='ExceptionGroup'}
out={'scope':'Two checks of manually reviewed decode answer, not broad quality evaluation','fast_failure_then_success':asyncio.run(check(False)),'all_fail':asyncio.run(check(True))}
print(json.dumps(out,indent=2))
@@ -0,0 +1,11 @@
{
"scope": "Two checks of manually reviewed decode answer, not broad quality evaluation",
"fast_failure_then_success": {
"returned": 7,
"pass": true
},
"all_fail": {
"error": "NameError: name 'builtins' is not defined",
"pass": false
}
}
@@ -0,0 +1,61 @@
{
"checked_at": 1789931025.6105647,
"uptime": "21:03:45 up 3 days, 9:58, 1 user, load average: 1.34, 0.91, 0.73",
"containers": {
"mike-ai-llama-medium": {
"running": true,
"health": "healthy",
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"started": "2026-09-20T19:03:00.899996682Z"
},
"mike-ai-router": {
"running": true,
"health": "healthy",
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
"started": "2026-09-20T19:03:11.359298419Z"
},
"mike-ai-profile-controller": {
"running": true,
"health": "healthy",
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
"started": "2026-09-20T19:03:11.213886021Z"
},
"mike-ai-wireguard-gateway": {
"running": true,
"health": "healthy",
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
"started": "2026-09-17T09:05:07.164533904Z"
},
"mike-ai-qwen3-tts": {
"running": true,
"health": "healthy",
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
"started": "2026-09-19T13:47:00.227813668Z"
}
},
"router": {
"/health": {
"status": "ok",
"router": "alive"
},
"/ready": {
"status": "ok",
"router": "alive",
"upstream": "ready"
}
},
"smoke": {
"content": "OK",
"usage": {
"completion_tokens": 2,
"prompt_tokens": 19,
"total_tokens": 21,
"prompt_tokens_details": {
"cached_tokens": 0
}
}
},
"kernel_errors": [],
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10930 MiB, 12288 MiB, 56\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 61",
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data"
}
@@ -0,0 +1,19 @@
[
{
"label": "medium-slot1-priority1-draft-cuda0",
"model": "pure",
"ctx": 160000,
"single": false,
"split": "80,20",
"ubatch": 128,
"parallel": 1,
"draft_max": 7,
"draft_kv": "q4_0",
"draft_device": "CUDA0",
"quality": false,
"prompts": [
4096
],
"minimum_headroom_mib": 512
}
]
@@ -0,0 +1,61 @@
{
"checked_at": 1789931253.8391275,
"uptime": "21:07:33 up 3 days, 10:02, 1 user, load average: 0.93, 0.95, 0.80",
"containers": {
"mike-ai-llama-medium": {
"running": true,
"health": "healthy",
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"started": "2026-09-20T19:07:06.261497776Z"
},
"mike-ai-router": {
"running": true,
"health": "healthy",
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
"started": "2026-09-20T19:07:16.676662725Z"
},
"mike-ai-profile-controller": {
"running": true,
"health": "healthy",
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
"started": "2026-09-20T19:07:16.548271903Z"
},
"mike-ai-wireguard-gateway": {
"running": true,
"health": "healthy",
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
"started": "2026-09-17T09:05:07.164533904Z"
},
"mike-ai-qwen3-tts": {
"running": true,
"health": "healthy",
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
"started": "2026-09-19T13:47:00.227813668Z"
}
},
"router": {
"/health": {
"status": "ok",
"router": "alive"
},
"/ready": {
"status": "ok",
"router": "alive",
"upstream": "ready"
}
},
"smoke": {
"content": "OK",
"usage": {
"completion_tokens": 2,
"prompt_tokens": 19,
"total_tokens": 21,
"prompt_tokens_details": {
"cached_tokens": 0
}
}
},
"kernel_errors": [],
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10920 MiB, 12288 MiB, 45\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 49",
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data"
}
@@ -0,0 +1,19 @@
[
{
"label": "medium-slot1-priority2-draft-cuda1",
"model": "pure",
"ctx": 160000,
"single": false,
"split": "80,20",
"ubatch": 128,
"parallel": 1,
"draft_max": 7,
"draft_kv": "q4_0",
"draft_device": "CUDA1",
"quality": false,
"prompts": [
4096
],
"minimum_headroom_mib": 512
}
]
@@ -0,0 +1,19 @@
[
{
"label": "medium-slot1-priority3-draft-cuda0",
"model": "pure",
"ctx": 160000,
"single": false,
"split": "70,30",
"ubatch": 128,
"parallel": 1,
"draft_max": 7,
"draft_kv": "q4_0",
"draft_device": "CUDA0",
"quality": false,
"prompts": [
4096
],
"minimum_headroom_mib": 512
}
]
+5 -2
View File
@@ -54,8 +54,8 @@ def run_case(case):
save(out/'config.json',case)
print('START',label,flush=True)
single=case.get('single',False)
args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel','2','--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3']
args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device','CUDA0','--spec-draft-ngl','all']
args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel',str(case.get('parallel',2)),'--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3']
args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device',case.get('draft_device','CUDA0'),'--spec-draft-ngl','all']
save(out/'server-args.json',args)
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
stop=threading.Event(); samples=[]
@@ -117,6 +117,9 @@ def run_case(case):
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
print(label,'prefill',n,r.get('timings'),flush=True)
result['finished']=time.time()
except Exception as exc:
result['error']=str(exc)
raise
finally:
stop.set(); thread.join(5)
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
@@ -10,8 +10,12 @@ signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_sign
# Refuse if the known production state has changed, or if requests are active.
for name in NAMES:
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
assert not any(s['is_processing'] for s in slots),'Production request active'
for attempt in range(60):
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
if not any(s['is_processing'] for s in slots): break
if attempt==0: print('WAIT production request active; no interruption',flush=True)
time.sleep(3)
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
assert not run('docker','ps','-q','--filter','name=^mike-ai-dflash2-test$').stdout.strip(),'Existing experiment'
child=None
try:
@@ -36,5 +40,11 @@ finally:
for _ in range(150):
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
time.sleep(2)
else: raise RuntimeError('Restored Medium did not become healthy')
run('docker','start',NAMES[1],NAMES[0])
print('RESTORED existing medium/controller/router',flush=True)
for _ in range(60):
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
if all(status=='healthy' for status in statuses): break
time.sleep(2)
else: raise RuntimeError('Restored router/controller did not become healthy')
print('RESTORED existing medium/controller/router; all healthy',flush=True)