{ "reference_id": "athena-qwen38-reference-20260920-v1", "created": "2026-09-20", "status": "frozen", "source_commit_before_benchmarks": "82c50962", "runtime_image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907", "runtime": "llama.cpp 0.4.1 b29c606", "environment_file": "reference-environment.json", "requests_file": "frozen-requests.json.gz", "request_provenance": "Reconstructed from measured run.py with the same Pure tokenizer using tokenization only; no new model inference. Decode/quality/tool request literals copied exactly. Original responses and timing are immutable measured data.", "gpu_order": [ "RTX 5080", "RTX 3060" ], "shared_settings": { "slots": 1, "vision": false, "flash_attention": true, "kv_k": "q4_0", "kv_v": "q4_0", "batch": 2048, "threads": 6, "gpu_layers": "all", "split_mode": "layer", "cache_ram": 0, "temperature": 1, "top_p": 0.95, "top_k": 20, "min_p": 0, "cache_prompt": false, "spec_draft_p_min": 0.05, "spec_draft_k": "f16", "spec_draft_v": "f16" }, "cases": [ { "id": "mix-single-76800-ub64", "configuration": { "label": "mix-single-76800-ub64", "model": "mix", "ctx": 76800, "single": true, "ubatch": 64, "mtp": 2, "prompts": [ 49152, 75776 ] }, "result": "cases/mix-single-76800-ub64/result.json.gz", "started": 1789928595.806369, "finished": 1789928753.7311614 }, { "id": "pure-dual-262144-80-20", "configuration": { "label": "pure-dual-262144-80-20", "model": "pure", "ctx": 262144, "single": false, "split": "80,20", "ubatch": 128, "mtp": 2, "prompts": [ 49152, 261120 ] }, "result": "cases/pure-dual-262144-80-20/result.json.gz", "started": 1789927147.4575458, "finished": 1789927634.4379501 }, { "id": "pure-single-32768", "configuration": { "label": "pure-single-32768", "model": "pure", "ctx": 32768, "single": true, "quality": true, "prompts": [ 4096, 24576 ] }, "result": "cases/pure-single-32768/result.json.gz", "started": 1789925728.6442063, "finished": 1789925970.5065877 }, { "id": "pure-single-32768-repeat", "configuration": { "label": "pure-single-32768-repeat", "model": "pure", "ctx": 32768, "single": true, "prompts": [ 4096 ] }, "result": "cases/pure-single-32768-repeat/result.json.gz", "started": 1789928349.0414968, "finished": 1789928379.4761648 }, { "id": "pure-single-57344", "configuration": { "label": "pure-single-57344", "model": "pure", "ctx": 57344, "single": true, "prompts": [ 49152, 56320 ], "quality_followup": true }, "result": "cases/pure-single-57344/result.json.gz", "started": 1789926319.9184752, "finished": 1789926515.7629743 }, { "id": "pure-single-61440-ub128", "configuration": { "label": "pure-single-61440-ub128", "model": "pure", "ctx": 61440, "single": true, "ubatch": 128, "prompts": [ 60416 ] }, "result": "cases/pure-single-61440-ub128/result.json.gz", "started": 1789928380.1341271, "finished": 1789928454.5977044 }, { "id": "pure-single-ub128-validated-50176", "configuration": { "label": "pure-single-ub128-validated-50176", "model": "pure", "ctx": 50176, "single": true, "ubatch": 128, "capacity_search": true, "load_only": false, "prompts": [ 49152, 49152 ] }, "result": "cases/pure-single-ub128-validated-50176/result.json.gz", "started": 1789926718.6930196, "finished": 1789926820.7396717 } ], "quality_scope": "Pure: nine initial tasks, three followups, one native tool call. MIX has throughput and three-needle retrieval only; no independent full quality battery. No BF16 reference.", "reuse_policy": "Reuse this frozen baseline by default. Do not automatically rerun Qwen for another candidate. Document material environment/protocol differences. If remeasurement is justified, create a new version and retain this bundle." }