Refine dual GPU allocation in individual layer steps

This commit is contained in:
Mikei386
2026-09-28 21:53:53 +02:00
parent a77266c7bd
commit 37a757fd9f
4 changed files with 127 additions and 29 deletions
+28 -1
View File
@@ -2,7 +2,7 @@ import tempfile,unittest,json
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import Mock,patch
from auto_test import AutoTests
from auto_test import AutoTests,model_layers,fine_splits
from inference import Scheduler,InferenceError
class AutoTestsTests(unittest.TestCase):
@@ -36,4 +36,31 @@ class AutoTestsTests(unittest.TestCase):
with self.assertRaises(ValueError):self.auto.save(dict(job_id='a',index=0,name='x'))
def test_prompt_filled_by_token_count(self):
self.auto.request=Mock(side_effect=[{'tokens':list(range(4000))},{'timings':{'prompt_n':3072,'predicted_n':64,'predicted_per_second':12}}]);r=self.auto.benchmark(4096);self.assertEqual(r['prompt_n'],3072);self.assertEqual(len(self.auto.request.call_args.args[1]['prompt']),3072)
class FineTests(unittest.TestCase):
def test_single_layer_steps_between_coarse_candidates(self):
import math
values=fine_splits(66,[85,15],90)
self.assertEqual([k for k,_ in values],[58,59])
for k,split in values:self.assertEqual(math.ceil(66*split[0]/sum(split)),k)
self.assertEqual(fine_splits(66,[95,5],100)[-1][0],65)
def test_gguf_metadata_mtp_and_output(self):
import struct
def text(s):b=s.encode();return struct.pack('<Q',len(b))+b
blob=b'GGUF'+struct.pack('<IQQ',3,0,3)+text('general.architecture')+struct.pack('<I',8)+text('qwen35')
blob+=text('qwen35.block_count')+struct.pack('<II',4,65)+text('qwen35.nextn_predict_layers')+struct.pack('<II',4,1)
with tempfile.TemporaryDirectory() as d:
p=Path(d)/'model.gguf';p.write_bytes(blob);self.assertEqual(model_layers(p,True),66);self.assertEqual(model_layers(p,False),65)
p.write_bytes(blob[:-2])
with self.assertRaises(ValueError):model_layers(p,True)
def test_refinement_integrated_after_first_fit(self):
fixture=AutoTestsTests();fixture.setUp()
try:
a=fixture.auto;a.job=dict(id='j',model_id='m',max_context=4096,mtp=True,attempts=0,results=[],started_at=0)
seen=[]
def candidate(context,tier,devices,split,ratio,offload,layer_count=None):
seen.append((context,ratio,layer_count));return [True] if ratio and ratio[0]<=90 else []
a.test_candidate=candidate;fixture.worker.catalog.root=Path('/unused')
with patch('auto_test.model_layers',return_value=66):a.run('first','second')
self.assertEqual([v[2] for v in seen if v[2] is not None],[61,61])
finally:fixture.tearDown()
if __name__=='__main__':unittest.main()