From 3d5931146b12eddc9258a2c46cd31d9171906c98 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sun, 20 Sep 2026 20:58:09 +0200 Subject: [PATCH] Record bounded Medium DFlash2 feasibility test and VRAM limit --- docs/DFLASH2_MEDIUM_QUICK_20260920.md | 51 ++++++ docs/INFERENCE_OPTIMIZATION_TODO_20260920.md | 4 + experiments/dflash2-medium-20260920/README.md | 57 +++++++ .../dflash2-medium-20260920/quick.json | 19 +++ .../restore-verification.json | 61 +++++++ .../results/config.json.gz | Bin 0 -> 207 bytes .../results/gpu.json.gz | Bin 0 -> 231 bytes .../results/result.json.gz | Bin 0 -> 246 bytes .../results/server-args.json.gz | Bin 0 -> 439 bytes .../results/server.log.gz | Bin 0 -> 969 bytes experiments/dflash2-medium-20260920/run.py | 161 ++++++++++++++++++ .../dflash2-medium-20260920/supervise.py | 40 +++++ .../dflash2-medium-20260920/verify_restore.py | 29 ++++ 13 files changed, 422 insertions(+) create mode 100644 docs/DFLASH2_MEDIUM_QUICK_20260920.md create mode 100644 experiments/dflash2-medium-20260920/README.md create mode 100644 experiments/dflash2-medium-20260920/quick.json create mode 100644 experiments/dflash2-medium-20260920/restore-verification.json create mode 100644 experiments/dflash2-medium-20260920/results/config.json.gz create mode 100644 experiments/dflash2-medium-20260920/results/gpu.json.gz create mode 100644 experiments/dflash2-medium-20260920/results/result.json.gz create mode 100644 experiments/dflash2-medium-20260920/results/server-args.json.gz create mode 100644 experiments/dflash2-medium-20260920/results/server.log.gz create mode 100644 experiments/dflash2-medium-20260920/run.py create mode 100644 experiments/dflash2-medium-20260920/supervise.py create mode 100644 experiments/dflash2-medium-20260920/verify_restore.py diff --git a/docs/DFLASH2_MEDIUM_QUICK_20260920.md b/docs/DFLASH2_MEDIUM_QUICK_20260920.md new file mode 100644 index 0000000..1134452 --- /dev/null +++ b/docs/DFLASH2_MEDIUM_QUICK_20260920.md @@ -0,0 +1,51 @@ +# DFlash2: kurzer Medium-Test auf Athena + +20.09.2026. **Ergebnis: Laden fehlgeschlagen, keine Durchsatzmessung möglich.** + +Die vorhandene llama.cpp-Version unterstützt DFlash2 einschließlich Selector und +Convolution. Das offizielle Q4_K_M-Draft-Modell (1,14 GB) wurde heruntergeladen +und per SHA256 geprüft. Hauptmodell blieb das bisherige Qwen Pure IQ4_XS. + +| Einstellung | Kurztest | +|---|---| +| Kontext / Slots | 160.000 gemeinsam / 2 | +| Hauptmodell GPUs | 5080:3060 = 80:20, Layer-Splitting | +| Draft | ausschließlich 5080, maximal 7 Draft-Tokens | +| KV / Microbatch | q4_0 für beide Modelle / 128 | +| Weitere Last | TTS auf 3060 blieb resident | +| Umfang | Text ohne Vision, keine parallelen Anfragen | + +Abweichungen vom bisherigen Medium: 80:20 statt 85:15, Microbatch 128 statt 512, +DFlash2 statt MTP3, kein Vision-Projektor, kein RAM-Promptcache. Es wurde weder +eine vollständige Medium-Funktionsgleichheit noch Langkontextnutzung geprüft. + +Beim Laden des Draft-Modells schlug `cudaMalloc` für **1079,61 MiB auf CUDA0** +fehl. Der Server beendete sich regulär mit einem Modellladefehler. Die geplanten +kurzen Deutsch-, Code- und Prefill-Aufgaben wurden nicht ausgeführt. Prefill, +Generierung und Antwortqualität sind für DFlash2 hier daher **unbekannt**. + +Die nach dem Laden vorgesehene 512-MiB-Reserveprüfung wurde nicht erreicht. +Das belegt eine Grenze dieses konkreten Speicherlayouts, nicht die allgemeine +Untauglichkeit oder Geschwindigkeit von DFlash2. Kein Versuch mit weiter +reduzierter Reserve, keine Serie von Speichergrenztests. + +## Lohnt ein weiterer Test? + +Als direkter Ersatz im getesteten Medium-Layout passt DFlash2 nicht. Ein weiterer +gezielter Versuch wäre nur mit angepasster Speicherverteilung sinnvoll: mehr +Hauptmodellschichten auf die 3060, Draft auf anderes Gerät, oder weniger +Slots/Kontext. Ob eine solche Variante schneller wäre, ist noch offen. +Für eine Geschwindigkeitsaussage liegen keinerlei DFlash2-Messwerte vor. + +Der Supervisor hat die bisherigen Container unverändert wieder gestartet. +Medium, Router, Controller, Gateway und TTS sind gesund. Router readiness und +minimale Modellantwort geprüft; keine Xid-, Kernel-Panic-, OOM-Kill- oder +GPU-fallen-off-Meldung im geprüften Kerneljournal seit Testbeginn. Kein Reboot, +keine Treiber-, Kernel- oder Netzwerkänderung. Die Qwen-Benchmarkreferenz wurde +nicht erneut gemessen. + +Rohdaten, Konfiguration, Testskripte und Wiederherstellungsnachweis liegen im +Repository unter `experiments/dflash2-medium-20260920/` und auf Athena unter +`/data/benchmarks/dflash2-medium-20260920/`. + +Quelle des Draft-Modells: [IncoAI DFlash2 GGUF](https://huggingface.co/incoai/Qwen3.8-27B-DFlash2-GGUF). diff --git a/docs/INFERENCE_OPTIMIZATION_TODO_20260920.md b/docs/INFERENCE_OPTIMIZATION_TODO_20260920.md index 5565c38..1954b9f 100644 --- a/docs/INFERENCE_OPTIMIZATION_TODO_20260920.md +++ b/docs/INFERENCE_OPTIMIZATION_TODO_20260920.md @@ -31,3 +31,7 @@ Stand: 20. September 2026. Modellfamilie bleibt Qwen3.8-27B. [Messbericht](QWEN38_BYTESHAPE_AB_20260920.md): mehr Single-GPU-Kontext mit ByteShape, aber kein allgemeiner Geschwindigkeitsgewinn und keine belegte Qualitätsgleichheit. Produktivmodelle bleiben unverändert. Punkte 2–4 sind offen. [Feste Qwen-Referenz](../benchmarks/athena-qwen38-reference-20260920/README.md) für alle weiteren Kandidaten verwenden; Qwen nicht automatisch neu testen. + +## DFlash2-Kurztest + +20.09.2026: [Medium-Ladetest](DFLASH2_MEDIUM_QUICK_20260920.md) mit 160.000 Kontext/zwei Slots scheitert an zusätzlichem Draft-VRAM auf der 5080. Keine Durchsatz- oder Qualitätswerte. Bisheriges Medium wiederhergestellt; Punkt 3 bleibt offen und benötigt ein anderes Speicherlayout. diff --git a/experiments/dflash2-medium-20260920/README.md b/experiments/dflash2-medium-20260920/README.md new file mode 100644 index 0000000..6ec9022 --- /dev/null +++ b/experiments/dflash2-medium-20260920/README.md @@ -0,0 +1,57 @@ +# DFlash2 Medium-only feasibility probe, 2026-09-20 + +User requested one quick indication of whether deeper tests are worthwhile. +No new Qwen baseline run. Reference: `benchmarks/athena-qwen38-reference-20260920/`. + +Target: existing Pure IQ4_XS, exact existing llama.cpp image b29c606. Runtime +help advertises draft-dflash; libllama contains build_dflash2_conv and +build_dflash2_selector. No rebuild, driver or kernel changes. + +Draft: incoai/Qwen3.8-27B-DFlash2-GGUF, revision +`51962825493a48b846b40126d35c799ac4093ad0`, Q4_K_M, 1,143,006,816 bytes. +SHA256: `1a25c56858e1ebe93f2718ac1d49d1151f9323325c1bbfd6209370f4db131ebd`. +Source: https://huggingface.co/incoai/Qwen3.8-27B-DFlash2-GGUF +Runtime source inspected: +https://github.com/ggml-org/llama.cpp/blob/b29c606/common/speculative.cpp + +## Scope + +160000 shared context, two configured slots, **one sequential request at a time**. +Pure target split 80:20, draft entirely on CUDA0/5080, q4_0 KV for target and +draft, microbatch128, draft7, batch2048, text only/no projector, cache RAM0. +TTS remains resident on 3060. These are changes from production Medium's +85:15/MTP3/ub512/vision/cache RAM32768. The test is not full Medium equivalence. +Maximum context is allocated, not validated by a full-length input in this quick +probe. No concurrency, vision or broad quality validation is claimed. + +Two existing decode prompts (German/code,768 output-token cap) and one frozen +4196-input-token retrieval/prefill prompt (512 output-token cap), plus health +smoke. Sampling matches the frozen requests: temp1/top-p.95/top-k20/min-p0, +seed42 (code43), no reasoning. Save all responses, timings and GPU samples. + +The saved dual-GPU Pure/MTP reference is a useful orientation, but uses 262144 +context and one slot. The single-GPU reference uses 32768 context/ub512/MTP3. +Any ratios are whole-configuration comparisons, not isolated DFlash speedups. +Same seed under speculative sampling does not guarantee identical output text. + +## Safety / restoration + +Adapted from the prior bounded isolated harness. No changes to production +container arguments or images. Supervisor drains and stops router/controller/ +Medium only and restores the same containers in finally. Ten-minute child limit; +read-only models,26GiB container RAM limit without extra swap, no privileges, +no core files. At least512MiB free per used GPU required before inference; +85C temperature and3GiB host RAM limits monitored. A failed case is not retried +with increasingly aggressive settings. No automated invocation on repo checkout. + +Commands, case configuration and results are archived here. Scripts have fixed +Athena paths; do not run them as generic unit tests or rerun without a concrete +benchmark task. This probe does not prove quality equivalence even if responses +look reasonable; that requires subsequent verification. + +## Observed outcome + +Target loaded; draft allocation of1079.61MiB onCUDA0 failed with CUDA malloc +OOM. No benchmark inference executed. Supervisor restored original production; +health/smoke/kernel checks passed. No performance or quality conclusion can be +drawn. See results/ and restore-verification.json. diff --git a/experiments/dflash2-medium-20260920/quick.json b/experiments/dflash2-medium-20260920/quick.json new file mode 100644 index 0000000..0f4eb15 --- /dev/null +++ b/experiments/dflash2-medium-20260920/quick.json @@ -0,0 +1,19 @@ +[ + { + "label": "medium-text-dflash2", + "model": "pure", + "ctx": 160000, + "single": false, + "split": "80,20", + "ubatch": 128, + "parallel": 2, + "draft_max": 7, + "draft_kv": "q4_0", + "draft_device": "CUDA0", + "quality": false, + "prompts": [ + 4096 + ], + "minimum_headroom_mib": 512 + } +] diff --git a/experiments/dflash2-medium-20260920/restore-verification.json b/experiments/dflash2-medium-20260920/restore-verification.json new file mode 100644 index 0000000..c3770f5 --- /dev/null +++ b/experiments/dflash2-medium-20260920/restore-verification.json @@ -0,0 +1,61 @@ +{ + "checked_at": 1789930622.3630264, + "uptime": "20:57:02 up 3 days, 9:52, 1 user, load average: 0.21, 0.30, 0.57", + "containers": { + "mike-ai-llama-medium": { + "running": true, + "health": "healthy", + "image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907", + "started": "2026-09-20T18:56:29.410526902Z" + }, + "mike-ai-router": { + "running": true, + "health": "healthy", + "image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb", + "started": "2026-09-20T18:56:39.861006576Z" + }, + "mike-ai-profile-controller": { + "running": true, + "health": "healthy", + "image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f", + "started": "2026-09-20T18:56:39.709347989Z" + }, + "mike-ai-wireguard-gateway": { + "running": true, + "health": "healthy", + "image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0", + "started": "2026-09-17T09:05:07.164533904Z" + }, + "mike-ai-qwen3-tts": { + "running": true, + "health": "healthy", + "image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98", + "started": "2026-09-19T13:47:00.227813668Z" + } + }, + "router": { + "/health": { + "status": "ok", + "router": "alive" + }, + "/ready": { + "status": "ok", + "router": "alive", + "upstream": "ready" + } + }, + "smoke": { + "content": "OK", + "usage": { + "completion_tokens": 2, + "prompt_tokens": 19, + "total_tokens": 21, + "prompt_tokens_details": { + "cached_tokens": 0 + } + } + }, + "kernel_errors": [], + "gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10920 MiB, 12288 MiB, 51\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 49", + "disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data" +} diff --git a/experiments/dflash2-medium-20260920/results/config.json.gz b/experiments/dflash2-medium-20260920/results/config.json.gz new file mode 100644 index 0000000000000000000000000000000000000000..09b9e9491940ae2814bc9d1c73d0d273f5303e71 GIT binary patch literal 207 zcmV;=05Ja_iwFP!00002|5cCO4uT*Q$M1a##G6AS|{

68nRY^A{Ah0tX0x4^ZoU4&+@okTzf0pMU?h?-@e411Io1y2m_L3RO5QyvCst^b^l6xp4JGSzXu)?wGN5ch1}rEwEustJ7JA=kNSmrZ z?Dvu^&qeccoH-X4*`*Wk=l9pgt9cF(uekj?&Dc>J?;<@+0HK1^<&+-R*X2p-pqoJe wx8tZAE!(Ky=QQ>oNrasT>jh80RIILdWY2bQIjcxfl-VPD0`Y8bEp!0@0Gg6>+yDRo literal 0 HcmV?d00001 diff --git a/experiments/dflash2-medium-20260920/results/server-args.json.gz b/experiments/dflash2-medium-20260920/results/server-args.json.gz new file mode 100644 index 0000000000000000000000000000000000000000..460a3db4db6998a73ea3fe4e7a548522f311c143 GIT binary patch literal 439 zcmV;o0Z9HIiwFP!00002|9z89Z=)~}hVS_mRnHkhXwsy|YWK3oRaI%F)vB7n0B(%! z*oIB^*Du8%Av7sS2tV(Pc|9}yp55Ku12nQAen0!Suz%#%%g408f>v@gHK4V^5h-F8 z<~4ymDFVOKVfB_?QM+5J#>RojZLJ9l@AGj&VTU%aC)&V9?z42Y4O%-Dms!*lYUDgr zI16~+K`U6vs0iepiFXW(-iMKbiOKxtK~ay;_``?M95|I z7XG<}6gGq3G3@>hgn>dBwrO~A92B1qx;Gpgrx&i{NQHM;C=CVgG|!pk1%BJ{&EfO- zxCE5V3BKOj584dvzFx# zW;p}2Rhq-3aBXd8GS{t8(|$v+6x^z68w9P~F^J?{=XQP5xIvu6@iBVz#$60Zc&u=v zSV3D=@Fp9dg{)wFg&TLg8~1U?Wk^;|kZv6D7&%YI>3Ypi#_xWO=%1g)aZ*)%c{9<) zaX>I(OmV`jnSoyR2Ft~AX|RAc%>zjyg8L?^RJtWQE~?tX!ic{(7JT4ED-Sk%%gquO z1){=PnB)HJVHgErm|VhN&%}TDSMup94lbp?j%LDd zuYpf|ZzjCzZP(k^9Db7eesYl{nJd9rk$o?S{+l51=W1{*L3){>EG4b*RDF#erD3#A zFO*Rd2J;vSwUbve%v?j?U56jds8q?HtR%d~Ua7u(rC=}7tY_0)cVggKJ(r=rk|9av zGMJiaUeKE3iT_rzfUKjC1pew^UQWHi*n6-|YQ*l8YR2LBAn7z@5Iq@SB9g{2Nmru> znhcNWu$?x{iY=!@6xH1HjeY4xg=VEa+mlvKQ8vuffjiE5X*@CU{1r*V1b%>_kcHeP2IEDOqBpZDLnwf#s_tG_I26&!!O!9Kz5{r4Ym^T$L6uu&8A4uKc1 z&?-HnqTp%ZA?noZ$IYO_MrFXYM!`>7c!t$5n69r99ifK)8B|(o)zT7^rK0Vcm^l`CKdLlU`9Z`{(szFYm7UGJI}88-+%@H% literal 0 HcmV?d00001 diff --git a/experiments/dflash2-medium-20260920/run.py b/experiments/dflash2-medium-20260920/run.py new file mode 100644 index 0000000..3c21696 --- /dev/null +++ b/experiments/dflash2-medium-20260920/run.py @@ -0,0 +1,161 @@ +#!/usr/bin/env python3 +"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production.""" +import json, pathlib, subprocess, sys, time, urllib.request, threading, signal +ROOT = pathlib.Path('/data/benchmarks/dflash2-medium-20260920') +NAME = 'mike-ai-dflash2-test' +BASE = 'http://127.0.0.1:5005' +GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe' +GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b' +IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907' +MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'} + +def cmd(*args, check=True, timeout=90): + r = subprocess.run(args, capture_output=True, text=True, timeout=timeout) + if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:]) + return r.stdout + +def api(path, data=None, timeout=900): + req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'}) + with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r) + +def save(path, data): + path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n') + +def gpu(): + rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu','--format=csv,noheader,nounits',timeout=15) + return [dict(zip(['name','used','total','temp','util'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()] + +def health_check(): + rows=gpu() + if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit') + mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines()) + if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB') + return rows + +def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None): + health_check() + p={'model':'benchmark','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False} + if tools: p.update(tools=tools,tool_choice='auto') + start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start + health_check() + return r + +def prefill(n, seed): + import gzip + payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)] + r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed']) + content=r['choices'][0]['message'].get('content','') + r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']} + return r + +def run_case(case): + label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True) + if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label) + save(out/'config.json',case) + print('START',label,flush=True) + single=case.get('single',False) + args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel','2','--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3'] + args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device','CUDA0','--spec-draft-ngl','all'] + save(out/'server-args.json',args) + cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args) + stop=threading.Event(); samples=[] + def monitor(): + while not stop.wait(2): + try: + rows=health_check() + samples.append({'time':time.time(),'gpus':rows}) + except RuntimeError as e: + samples.append({'error':str(e),'aborted':True}) + cmd('docker','stop','-t','10',NAME,check=False) + return + except Exception as e: samples.append({'error':str(e)}) + thread=threading.Thread(target=monitor,daemon=True); thread.start() + result={'case':case,'started':time.time()} + try: + for _ in range(150): + try: + if api('/health',timeout=3).get('status')=='ok': break + except Exception: pass + if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load') + time.sleep(2) + else: raise RuntimeError('Startup exceeded 300s') + result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots') + save(out/'loaded.json',result) + # Added after the 88:12 trial: model loading alone can succeed while + # the first real attention graph still needs more CUDA workspace. + minimum=case.get('minimum_headroom_mib',512) + used_devices=['5080'] if single else ['5080','3060'] + for g in result['idle_gpu']: + if any(device in g['name'] for device in used_devices): + free=int(g['total'])-int(g['used']) + if free=262144: break + previous=context + context=min(262144,context+growth) + final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]} + return run_case(final) + +if __name__=='__main__': + for case in json.loads(pathlib.Path(sys.argv[1]).read_text()): + if case.get('capacity_search'): capacity_case(case) + else: run_case(case) diff --git a/experiments/dflash2-medium-20260920/supervise.py b/experiments/dflash2-medium-20260920/supervise.py new file mode 100644 index 0000000..d035920 --- /dev/null +++ b/experiments/dflash2-medium-20260920/supervise.py @@ -0,0 +1,40 @@ +#!/usr/bin/env python3 +"""Stop only existing router/controller/model; always restore the same containers.""" +import json, pathlib, subprocess, sys, time, signal +ROOT=pathlib.Path('/data/benchmarks/dflash2-medium-20260920') +NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium'] +def run(*args,check=True,timeout=90): + return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout) +def stop_signal(*_): raise RuntimeError('Supervisor interrupted') +signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal) +# Refuse if the known production state has changed, or if requests are active. +for name in NAMES: + assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name +slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout) +assert not any(s['is_processing'] for s in slots),'Production request active' +assert not run('docker','ps','-q','--filter','name=^mike-ai-dflash2-test$').stdout.strip(),'Existing experiment' +child=None +try: + run('docker','stop','-t','30',*NAMES[:2]) + # Drain requests already handed to the model, before unloading it. + for _ in range(120): + slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout) + if not any(s['is_processing'] for s in slots): break + time.sleep(2) + else: raise RuntimeError('Model did not drain') + run('docker','stop','-t','30',NAMES[-1]) + child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]]) + code=child.wait(timeout=600) + if code: raise RuntimeError('Benchmark failed: '+str(code)) +finally: + if child is not None and child.poll() is None: + child.terminate() + try: child.wait(timeout=20) + except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10) + run('docker','rm','-f','mike-ai-dflash2-test',check=False) + run('docker','start',NAMES[-1]) + for _ in range(150): + if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break + time.sleep(2) + run('docker','start',NAMES[1],NAMES[0]) + print('RESTORED existing medium/controller/router',flush=True) diff --git a/experiments/dflash2-medium-20260920/verify_restore.py b/experiments/dflash2-medium-20260920/verify_restore.py new file mode 100644 index 0000000..107172d --- /dev/null +++ b/experiments/dflash2-medium-20260920/verify_restore.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python3 +"""Read-only restoration verification plus a two-token model smoke request.""" +import json,pathlib,re,subprocess,time +ROOT=pathlib.Path('/data/benchmarks/dflash2-medium-20260920') +def run(*args):return subprocess.check_output(args,text=True,timeout=30) +report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}} +for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']: + d=json.loads(run('docker','inspect',name))[0] + report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']} + assert d['State']['Running'],name + if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']: + assert d['State'].get('Health',{}).get('Status')=='healthy',name +assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907' +assert not run('docker','ps','-q','--filter','name=^mike-ai-dflash2-test$').strip() +probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))' +report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe)) +payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0} +r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions')) +report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')} +assert report['smoke']['content'].strip()=='OK',report['smoke'] +started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json')) +journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager') +pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I) +report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)] +report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip() +report['disk']=run('df','-h','/','/data').strip() +(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n') +print(json.dumps(report,indent=2)) +assert not report['kernel_errors'],'Kernel/GPU errors require review'