Files
AI-Profile-Router/benchmarks/athena-qwen38-reference-20260920/manifest.json
T

161 lines
4.5 KiB
JSON

{
"reference_id": "athena-qwen38-reference-20260920-v1",
"created": "2026-09-20",
"status": "frozen",
"source_commit_before_benchmarks": "82c50962",
"runtime_image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"runtime": "llama.cpp 0.4.1 b29c606",
"environment_file": "reference-environment.json",
"requests_file": "frozen-requests.json.gz",
"request_provenance": "Reconstructed from measured run.py with the same Pure tokenizer using tokenization only; no new model inference. Decode/quality/tool request literals copied exactly. Original responses and timing are immutable measured data.",
"gpu_order": [
"RTX 5080",
"RTX 3060"
],
"shared_settings": {
"slots": 1,
"vision": false,
"flash_attention": true,
"kv_k": "q4_0",
"kv_v": "q4_0",
"batch": 2048,
"threads": 6,
"gpu_layers": "all",
"split_mode": "layer",
"cache_ram": 0,
"temperature": 1,
"top_p": 0.95,
"top_k": 20,
"min_p": 0,
"cache_prompt": false,
"spec_draft_p_min": 0.05,
"spec_draft_k": "f16",
"spec_draft_v": "f16"
},
"cases": [
{
"id": "mix-single-76800-ub64",
"configuration": {
"label": "mix-single-76800-ub64",
"model": "mix",
"ctx": 76800,
"single": true,
"ubatch": 64,
"mtp": 2,
"prompts": [
49152,
75776
]
},
"result": "cases/mix-single-76800-ub64/result.json.gz",
"started": 1789928595.806369,
"finished": 1789928753.7311614
},
{
"id": "pure-dual-262144-80-20",
"configuration": {
"label": "pure-dual-262144-80-20",
"model": "pure",
"ctx": 262144,
"single": false,
"split": "80,20",
"ubatch": 128,
"mtp": 2,
"prompts": [
49152,
261120
]
},
"result": "cases/pure-dual-262144-80-20/result.json.gz",
"started": 1789927147.4575458,
"finished": 1789927634.4379501
},
{
"id": "pure-single-32768",
"configuration": {
"label": "pure-single-32768",
"model": "pure",
"ctx": 32768,
"single": true,
"quality": true,
"prompts": [
4096,
24576
]
},
"result": "cases/pure-single-32768/result.json.gz",
"started": 1789925728.6442063,
"finished": 1789925970.5065877
},
{
"id": "pure-single-32768-repeat",
"configuration": {
"label": "pure-single-32768-repeat",
"model": "pure",
"ctx": 32768,
"single": true,
"prompts": [
4096
]
},
"result": "cases/pure-single-32768-repeat/result.json.gz",
"started": 1789928349.0414968,
"finished": 1789928379.4761648
},
{
"id": "pure-single-57344",
"configuration": {
"label": "pure-single-57344",
"model": "pure",
"ctx": 57344,
"single": true,
"prompts": [
49152,
56320
],
"quality_followup": true
},
"result": "cases/pure-single-57344/result.json.gz",
"started": 1789926319.9184752,
"finished": 1789926515.7629743
},
{
"id": "pure-single-61440-ub128",
"configuration": {
"label": "pure-single-61440-ub128",
"model": "pure",
"ctx": 61440,
"single": true,
"ubatch": 128,
"prompts": [
60416
]
},
"result": "cases/pure-single-61440-ub128/result.json.gz",
"started": 1789928380.1341271,
"finished": 1789928454.5977044
},
{
"id": "pure-single-ub128-validated-50176",
"configuration": {
"label": "pure-single-ub128-validated-50176",
"model": "pure",
"ctx": 50176,
"single": true,
"ubatch": 128,
"capacity_search": true,
"load_only": false,
"prompts": [
49152,
49152
]
},
"result": "cases/pure-single-ub128-validated-50176/result.json.gz",
"started": 1789926718.6930196,
"finished": 1789926820.7396717
}
],
"quality_scope": "Pure: nine initial tasks, three followups, one native tool call. MIX has throughput and three-needle retrieval only; no independent full quality battery. No BF16 reference.",
"reuse_policy": "Reuse this frozen baseline by default. Do not automatically rerun Qwen for another candidate. Document material environment/protocol differences. If remeasurement is justified, create a new version and retain this bundle."
}