Record bounded Medium GPU split and MTP comparison

This commit is contained in:
Mikei386 committed 2026-09-20 22:42:09 +02:00
1 parent 3cbcf41fab
commit 635b3c4ba9
32 files changed
+5107

No files matched your search

@@ -0,0 +1,307 @@
{
"case": {
"label": "85-mtp2",
"ubatch": 256,
"split": "85,15",
"mtp": 2,
"prompts": [
24576
],
"minimum_headroom_mib": 448
},
"started": 1789936655.4056249,
"idle_gpu": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10608",
"total": "12288",
"temp": "47",
"util": "2",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15574",
"total": "16303",
"temp": "53",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
],
"props": {
"default_generation_settings": {
"params": {
"seed": 4294967295,
"temperature": 1.0,
"dynatemp_range": 0.0,
"dynatemp_exponent": 1.0,
"top_k": 20,
"top_p": 0.949999988079071,
"min_p": 0.05000000074505806,
"top_n_sigma": -1.0,
"xtc_probability": 0.0,
"xtc_threshold": 0.10000000149011612,
"typical_p": 1.0,
"repeat_last_n": 64,
"repeat_penalty": 1.0,
"presence_penalty": 0.0,
"frequency_penalty": 0.0,
"dry_multiplier": 0.0,
"dry_base": 1.75,
"dry_allowed_length": 2,
"dry_penalty_last_n": 64,
"mirostat": 0,
"mirostat_tau": 5.0,
"mirostat_eta": 0.10000000149011612,
"adaptive_target": -1.0,
"adaptive_decay": 0.8999999761581421,
"max_tokens": -1,
"n_predict": -1,
"n_keep": 0,
"n_discard": 0,
"ignore_eos": false,
"stream": false,
"n_probs": 0,
"min_keep": 0,
"chat_format": "Content-only",
"reasoning_format": "none",
"reasoning_in_content": false,
"generation_prompt": "",
"samplers": [
"penalties",
"dry",
"top_n_sigma",
"top_k",
"typ_p",
"top_p",
"min_p",
"xtc",
"temperature"
],
"speculative.types": "none",
"timings_per_token": false,
"post_sampling_probs": false,
"backend_sampling": false,
"lora": []
},
"n_ctx": 160000
},
"total_slots": 2,
"model_alias": "qwen-medium",
"model_ftype": "IQ4_XS - 4.25 bpw",
"model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"modalities": {
"vision": true,
"video": true,
"audio": false
},
"media_marker": "<__media_uTU13sY3gUv1cIKm43dbCGG9ONDyqVyZ__>",
"endpoint_slots": true,
"endpoint_props": false,
"endpoint_metrics": true,
"ui": false,
"ui_settings": {},
"chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calLine truncated
"chat_template_caps": {
"supports_object_arguments": true,
"supports_parallel_tool_calls": true,
"supports_preserve_reasoning": true,
"supports_reasoning_effort": true,
"supports_string_content": true,
"supports_system_role": true,
"supports_tool_calls": true,
"supports_tools": true,
"supports_typed_content": true
},
"bos_token": "<|endoftext|>",
"eos_token": "<|im_end|>",
"build_info": "b10964-b29c606e2",
"is_sleeping": false,
"cors_proxy_enabled": false
},
"slots": [
{
"id": 0,
"n_ctx": 160000,
"speculative": true,
"is_processing": false
},
{
"id": 1,
"n_ctx": 160000,
"speculative": true,
"is_processing": false
}
],
"smoke": {
"choices": [
{
"finish_reason": "stop",
"index": 0,
"message": {
"role": "assistant",
"content": "OK"
}
}
],
"created": 1789936662,
"model": "qwen-medium",
"system_fingerprint": "b10964-b29c606e2",
"object": "chat.completion",
"usage": {
"completion_tokens": 2,
"prompt_tokens": 19,
"total_tokens": 21,
"prompt_tokens_details": {
"cached_tokens": 0
}
},
"id": "chatcmpl-H5kwhDH81uL8rJBQx1R7dHv1AmwUitq5",
"timings": {
"cache_n": 0,
"prompt_n": 19,
"prompt_ms": 781.219,
"prompt_per_token_ms": 41.116789473684214,
"prompt_per_second": 24.320965055893417,
"predicted_n": 2,
"predicted_ms": 286.485,
"predicted_per_token_ms": 286.485,
"predicted_per_second": 3.490584149257378,
"draft_n": 2,
"draft_n_accepted": 2
},
"wall_seconds": 1.0702720619738102
},
"decode": [
{
"choices": [
{
"finish_reason": "length",
"index": 0,
"message": {
"role": "assistant",
"content": "Ein Reverse Proxy ist ein zentrales Element in modernen IT-Infrastrukturen, das als Vermittler zwischen Clienten und Backend-Servern agiert. Im Gegensatz zu einem Forward Proxy, der Anfragen von internen Clients nach außen vertritt, stellt der Reverse Proxy eine Eindepunkt-Schnittstelle (Single Point of Entry) für externe Anfragen dar. Typische Vertreter sind nginx, HAProxy oder Apache HTTP Server. Die Grundprinzipie des Reverse Proxys besteht darin, eingehende HTTP- oder HTTPS-Anfragen entgegenzunehmen und diese basierend auf definierten Routing-Regeln an die geeigneten Backend-Server weiterzuleiten. Diese Architektur bietet erhebliche Vorteile wie Lastverteilung, SSL/TLS-Terminierung, Caching und eine vereinfachte Verwaltung mehrerer hinterlegter Dienste. Durch die Abstraktion der Backend-Infrastruktur wird sichergestellt, dass Clients die Details der internen Topologie, wie IP-Adressen oder Server-Namen, nicht kennen müssen.\n\nIn dynamischen Umgebungen, insbesondere bei der Nutzung von Container-Orchestrierungsplattformen wie Kubernetes oder Docker Compose, ist die Stabilität der Netzwerkanbindung eine große Herausforderung. Container sind per Definition flüchtig; sie können bei Updates, Skalierungsoperationen oder Fehlern neu gestartet werden. Bei jedem Neustart oder einer Neuanlage eines Containers kann die zugewiesene IP-Adresse im internen Docker- oder Kubernetes-Netzwerk wechseln. Wenn ein Reverse Proxy statische Konfigurationen verwendet, die auf bestimmte IP-Adressen oder Ports verweisen, führt dieser Wechsel direkt zu Verbindungsproblemen.\n\nDer häufigste Fehler in solchen Szenarien ist die \"Connection Refused\"- oder \"Connection Timeout\"-Situation. Wenn der Reverse Proxy versucht, eine Anfrage an eine IP-Adresse weiterzugeben, die einem Container zugeordnet war, dieser aber mittlerweile unter einer neuen Adresse erreichbar ist oder gar nicht mehr existiert, schlägt das TCP-Handshake-Protokoll fehl. Der Client sieht daraufhin typischerweise HTTP-Fehler 502 (Bad Gateway) oder 504 (Gateway Timeout). Ein 502-Fehler deutet oft darauf hin, dass das Backend-Server-Protokoll eine ungültige Antwort geliefert hat oder die Verbindung komplett abgelehnt wurde. Ein 504-Fehler tritt auf, wenn das Backend nicht rechtzeitig auf die Anfrage des Proxys antwortet, was bei einem hängenden oder falsch konfigurierten Container häufig vorkommt.\n\nUm solche Probleme anhand von Logs effektiv einzugrenzen, muss man systematisch vorgehen. Der erste Schritt ist die Analyse der Logs des Reverse Proxys selbst. Bei nginx befindet sich die Fehlerinformation im `error.log`. Hier sucht man nach Einträgen wie `connect() failed (111: Connection refused)` oder `upstream timed out`. Diese Meldungen bestätigen, dass der Proxy die Verbindung zum definierten Backend nicht herstellen konnte. Wichtig ist dabei, die Ziel-IP und den Port im Log-Eintrag zu notieren.\n\nIm zweiten Schritt muss man die Logs des Orchestrierers oder des Container-Runtimes (z. B. `docker logs` oder `kubectl describe pod`) prüfen. Man vergleicht die im Proxy-Log festgestellte IP-Adresse mit der aktuellen IP des laufenden Containers. Stimmt die IP nicht überein, liegt ein Problem mit der dynamischen DNS- oder Service-Auflösung vor. In Kubernetes-Umgebungen ist es daher ratsam, statt fester IPs die internen Service-Namen (z. B. `http://api-service`) in der Proxy-Konfiguration"
}
}
],
"created": 1789936674,
"model": "qwen-medium",
"system_fingerprint": "b10964-b29c606e2",
"object": "chat.completion",
"usage": {
"completion_tokens": 768,
"prompt_tokens": 59,
"total_tokens": 827,
"prompt_tokens_details": {
"cached_tokens": 0
}
},
"id": "chatcmpl-n5iknizKxzo961SEpq50X0OcfWWIPgyV",
"timings": {
"cache_n": 0,
"prompt_n": 59,
"prompt_ms": 118.721,
"prompt_per_token_ms": 2.012220338983051,
"prompt_per_second": 496.96346897347564,
"predicted_n": 768,
"predicted_ms": 12112.526,
"predicted_per_token_ms": 15.792080834419817,
"predicted_per_second": 63.322877490624165,
"draft_n": 732,
"draft_n_accepted": 401
},
"wall_seconds": 12.29133506200742
},
{
"choices": [
{
"finish_reason": "length",
"index": 0,
"message": {
"role": "assistant",
"content": "```python\nimport asyncio\nfrom typing import Any, Awaitable, Callable, List, Optional, Union\n\n\nasync def first_success(\n awaitables: List[Awaitable],\n return_exceptions: bool = True\n) -> Any:\n \"\"\"\n Start all awaitables concurrently, return the first successful result,\n cancel and await remaining tasks, and collect exceptions if all fail.\n\n Args:\n awaitables: A list of awaitables to run concurrently.\n return_exceptions: If True, when all awaitables fail, raise an\n ExceptionGroup (Python 3.11+) or re-raise the first exception\n with others attached. If False, simply re-raise the first exception.\n\n Returns:\n The result of the first successfully completed awaitable.\n\n Raises:\n ExceptionGroup (or the first exception if return_exceptions=False):\n If all awaitables fail.\n \"\"\"\n if not awaitables:\n raise ValueError(\"awaitables list must not be empty\")\n\n # Create tasks from all awaitables\n tasks = [asyncio.ensure_future(av) for av in awaitables]\n\n try:\n # Use asyncio.wait to get the first completed task\n # We need to keep polling until we find a successful one\n pending = set(tasks)\n\n while pending:\n done, pending = await asyncio.wait(pending, return_when=asyncio.FIRST_COMPLETED)\n\n for task in done:\n try:\n result = task.result()\n # Success! Cancel all remaining tasks and return result\n for t in pending:\n t.cancel()\n # Await the cancellation of remaining tasks to ensure clean shutdown\n if pending:\n await asyncio.gather(*pending, return_exceptions=True)\n return result\n except asyncio.CancelledError:\n # This task was cancelled, treat as failure\n continue\n except Exception:\n # Task raised an exception, continue checking others\n continue\n\n # If we get here, all done tasks failed; continue waiting for remaining\n\n # If we exhaust all tasks and none succeeded\n # Collect all exceptions\n exceptions = []\n for task in tasks:\n try:\n task.result()\n except Exception as e:\n exceptions.append(e)\n except asyncio.CancelledError:\n exceptions.append(asyncio.CancelledError())\n\n if not exceptions:\n # All tasks were cancelled somehow, but no exceptions captured\n raise RuntimeError(\"All awaitables were cancelled without explicit exceptions\")\n\n if return_exceptions:\n if hasattr(asyncio, 'ExceptionGroup'): # Python 3.11+\n raise asyncio.ExceptionGroup(\"All awaitables failed\", exceptions)\n else:\n # Fallback: raise the first exception, attach others\n first_exc = exceptions[0]\n # Add other exceptions as context\n for exc in exceptions[1:]:\n first_exc.__context__ = exc\n raise first_exc\n else:\n raise exceptions[0]\n\n except asyncio.CancelledError:\n # If the entire first_success coroutine is cancelled, cancel all tasks\n for t in tasks:\n t.cancel()\n await asyncio.gather(*tasks, return_exceptions=True)\n raise\n\n\n# Usage example\nasync def example():\n \"\"\"Demonstrate first_success with a practical example.\"\"\"\n\n async def slow_success(delay: float) -> str:\n \"\"\"Simulate a slow operation that eventually succeeds.\"\"\"\n await asyncio.sleep(delay)\n return f\"success after {delay}s\"\n\n"
}
}
],
"created": 1789936685,
"model": "qwen-medium",
"system_fingerprint": "b10964-b29c606e2",
"object": "chat.completion",
"usage": {
"completion_tokens": 768,
"prompt_tokens": 54,
"total_tokens": 822,
"prompt_tokens_details": {
"cached_tokens": 0
}
},
"id": "chatcmpl-JW9n3P53QA4c7sgHVdEmONCUEeTj2Dpv",
"timings": {
"cache_n": 0,
"prompt_n": 54,
"prompt_ms": 117.561,
"prompt_per_token_ms": 2.1770555555555555,
"prompt_per_second": 459.336004287136,
"predicted_n": 768,
"predicted_ms": 10178.693,
"predicted_per_token_ms": 13.270786179921773,
"predicted_per_second": 75.353485953452,
"draft_n": 614,
"draft_n_accepted": 460
},
"wall_seconds": 10.361086986027658
}
],
"prefill": [
{
"target": 24576,
"response": {
"choices": [
{
"finish_reason": "length",
"index": 0,
"message": {
"role": "assistant",
"content": "```json\n{\n \"alpha\": \"RAVEN-417\",\n \"beta\": \"CEDAR-928\",\n \"gamma\": \"ORBIT-563\"\n}\n```\n\nDie Verifizierung der identifizierten Nadelwerte (`RAVEN-417`, `CEDAR-928`, `ORBIT-563`) innerhalb des vorliegenden Datensatzes erfordert einen streng systematischen und evidenzbasierten Ansatz, der auf der präzisen Analyse der Textstruktur und der Positionierung der Markierungen beruht. Zunächst ist es entscheidend, die Integrität der Quelldaten zu wahren, indem man keine extrapolierten oder hypothetischen Datenpunkte einführt. Der vorliegende Datenset besteht überwiegend aus identischen, wiederholten Zeilen, die den String „cobalt lantern maple orbit quartz river silver tango.“ enthalten, wobei die Einträge von `Record 000000` bis `Record 001169` nummeriert sind. Die Nadeln wurden jedoch nicht als Teil des eigentlichen Datensatzinhalts (wie etwa in der Spalte der Werte) platziert, sondern als isolierte Metadaten-Zeilen zwischen den regulären Aufzeichnungen.\n\nUm `NEEDLE_ALPHA=RAVEN-417` zu verifizieren, muss der Fokus auf die Region um die Aufzeichnungen 000146 bis 000147 gelegt werden. Hier findet sich die Zeile, die mit „Record 000146: cob...“ beginnt und abgebrochen ist, gefolgt direkt von der isolierten Zeile „NEEDLE_ALPHA=RAVEN-417“. Die Verifikation erfolgt durch die Bestätigung, dass diese spezifische String-Kombination exakt an dieser Position im Textstrom erscheint und nicht in anderen Teilen des Dokuments vorkommt. Ein ähnlicher Prozess gilt für `NEEDLE_BETA=CEDAR-928`, der sich zwischen `Record 000584` und `Record 000585` befindet. Hier ist besondere Sorgfalt geboten, da die Zeile für Record 000585 aufgrund der Platzierung der Nadel ungewöhnlich formatiert oder fragmentiert erscheint („Recor... d 000585“). Die Verifikation muss daher sicherstellen,"
}
}
],
"created": 1789936706,
"model": "qwen-medium",
"system_fingerprint": "b10964-b29c606e2",
"object": "chat.completion",
"usage": {
"completion_tokens": 512,
"prompt_tokens": 24674,
"total_tokens": 25186,
"prompt_tokens_details": {
"cached_tokens": 0
}
},
"id": "chatcmpl-ri1lA3zBowOkArxgp0aNkq6PGla5VmBI",
"timings": {
"cache_n": 0,
"prompt_n": 24674,
"prompt_ms": 12942.719,
"prompt_per_token_ms": 0.5245488773607846,
"prompt_per_second": 1906.4000385081374,
"predicted_n": 512,
"predicted_ms": 8537.934,
"predicted_per_token_ms": 16.70828571428571,
"predicted_per_second": 59.850544639956226,
"draft_n": 441,
"draft_n_accepted": 290
},
"wall_seconds": 21.566215507977176,
"recall": {
"RAVEN-417": true,
"CEDAR-928": true,
"ORBIT-563": true
}
}
}
],
"finished": 1789936707.0146825
}