308 lines
27 KiB
JSON
308 lines
27 KiB
JSON
{
|
|
"case": {
|
|
"label": "control-85-mtp3",
|
|
"ubatch": 256,
|
|
"split": "85,15",
|
|
"mtp": 3,
|
|
"prompts": [
|
|
24576
|
|
],
|
|
"minimum_headroom_mib": 448
|
|
},
|
|
"started": 1789936595.6476705,
|
|
"idle_gpu": [
|
|
{
|
|
"name": "NVIDIA GeForce RTX 3060",
|
|
"used": "10640",
|
|
"total": "12288",
|
|
"temp": "43",
|
|
"util": "0",
|
|
"pcie_gen": "3",
|
|
"pcie_width": "4"
|
|
},
|
|
{
|
|
"name": "NVIDIA GeForce RTX 5080",
|
|
"used": "15842",
|
|
"total": "16303",
|
|
"temp": "49",
|
|
"util": "0",
|
|
"pcie_gen": "4",
|
|
"pcie_width": "16"
|
|
}
|
|
],
|
|
"props": {
|
|
"default_generation_settings": {
|
|
"params": {
|
|
"seed": 4294967295,
|
|
"temperature": 1.0,
|
|
"dynatemp_range": 0.0,
|
|
"dynatemp_exponent": 1.0,
|
|
"top_k": 20,
|
|
"top_p": 0.949999988079071,
|
|
"min_p": 0.05000000074505806,
|
|
"top_n_sigma": -1.0,
|
|
"xtc_probability": 0.0,
|
|
"xtc_threshold": 0.10000000149011612,
|
|
"typical_p": 1.0,
|
|
"repeat_last_n": 64,
|
|
"repeat_penalty": 1.0,
|
|
"presence_penalty": 0.0,
|
|
"frequency_penalty": 0.0,
|
|
"dry_multiplier": 0.0,
|
|
"dry_base": 1.75,
|
|
"dry_allowed_length": 2,
|
|
"dry_penalty_last_n": 64,
|
|
"mirostat": 0,
|
|
"mirostat_tau": 5.0,
|
|
"mirostat_eta": 0.10000000149011612,
|
|
"adaptive_target": -1.0,
|
|
"adaptive_decay": 0.8999999761581421,
|
|
"max_tokens": -1,
|
|
"n_predict": -1,
|
|
"n_keep": 0,
|
|
"n_discard": 0,
|
|
"ignore_eos": false,
|
|
"stream": false,
|
|
"n_probs": 0,
|
|
"min_keep": 0,
|
|
"chat_format": "Content-only",
|
|
"reasoning_format": "none",
|
|
"reasoning_in_content": false,
|
|
"generation_prompt": "",
|
|
"samplers": [
|
|
"penalties",
|
|
"dry",
|
|
"top_n_sigma",
|
|
"top_k",
|
|
"typ_p",
|
|
"top_p",
|
|
"min_p",
|
|
"xtc",
|
|
"temperature"
|
|
],
|
|
"speculative.types": "none",
|
|
"timings_per_token": false,
|
|
"post_sampling_probs": false,
|
|
"backend_sampling": false,
|
|
"lora": []
|
|
},
|
|
"n_ctx": 160000
|
|
},
|
|
"total_slots": 2,
|
|
"model_alias": "qwen-medium",
|
|
"model_ftype": "IQ4_XS - 4.25 bpw",
|
|
"model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
|
"modalities": {
|
|
"vision": true,
|
|
"video": true,
|
|
"audio": false
|
|
},
|
|
"media_marker": "<__media_CclCoD7CO3rlBd673CRwhz9GjHIXhKmN__>",
|
|
"endpoint_slots": true,
|
|
"endpoint_props": false,
|
|
"endpoint_metrics": true,
|
|
"ui": false,
|
|
"ui_settings": {},
|
|
"chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if merged_system %}\n {{- '\\n\\n' + merged_system }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if merged_system %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + merged_system + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if loop.index0 >= num_sys %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" or message.role == \"developer\" %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if tool_call.name is not defined or tool_call.name is none %}\n {{- raise_exception('Tool call is missing a function name.') }}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is mapping %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- elif tool_call.arguments is string %}\n {%- if tool_call.arguments|trim %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" were passed as a JSON string. Parse them into an object before calling apply_chat_template.') }}\n {%- endif %}\n {%- elif tool_call.arguments is defined and tool_call.arguments is not none %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" must be an object/mapping or a JSON string.') }}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- else %}\n {{- '<think>\\n' }}\n {%- endif %}\n{%- endif %}\n{#- Unsloth fixes - developer role, merged system messages, tool calling #}",
|
|
"chat_template_caps": {
|
|
"supports_object_arguments": true,
|
|
"supports_parallel_tool_calls": true,
|
|
"supports_preserve_reasoning": true,
|
|
"supports_reasoning_effort": true,
|
|
"supports_string_content": true,
|
|
"supports_system_role": true,
|
|
"supports_tool_calls": true,
|
|
"supports_tools": true,
|
|
"supports_typed_content": true
|
|
},
|
|
"bos_token": "<|endoftext|>",
|
|
"eos_token": "<|im_end|>",
|
|
"build_info": "b10964-b29c606e2",
|
|
"is_sleeping": false,
|
|
"cors_proxy_enabled": false
|
|
},
|
|
"slots": [
|
|
{
|
|
"id": 0,
|
|
"n_ctx": 160000,
|
|
"speculative": true,
|
|
"is_processing": false
|
|
},
|
|
{
|
|
"id": 1,
|
|
"n_ctx": 160000,
|
|
"speculative": true,
|
|
"is_processing": false
|
|
}
|
|
],
|
|
"smoke": {
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "OK"
|
|
}
|
|
}
|
|
],
|
|
"created": 1789936608,
|
|
"model": "qwen-medium",
|
|
"system_fingerprint": "b10964-b29c606e2",
|
|
"object": "chat.completion",
|
|
"usage": {
|
|
"completion_tokens": 2,
|
|
"prompt_tokens": 19,
|
|
"total_tokens": 21,
|
|
"prompt_tokens_details": {
|
|
"cached_tokens": 0
|
|
}
|
|
},
|
|
"id": "chatcmpl-M3uwqFa6U3Cy6PocFxJn9saK8NrAxAPr",
|
|
"timings": {
|
|
"cache_n": 0,
|
|
"prompt_n": 19,
|
|
"prompt_ms": 811.12,
|
|
"prompt_per_token_ms": 42.69052631578948,
|
|
"prompt_per_second": 23.42440082848407,
|
|
"predicted_n": 2,
|
|
"predicted_ms": 306.127,
|
|
"predicted_per_token_ms": 306.127,
|
|
"predicted_per_second": 3.2666181029442027,
|
|
"draft_n": 3,
|
|
"draft_n_accepted": 3
|
|
},
|
|
"wall_seconds": 1.1197090409696102
|
|
},
|
|
"decode": [
|
|
{
|
|
"choices": [
|
|
{
|
|
"finish_reason": "length",
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Ein Reverse Proxy ist ein zentrales Element in modernen IT-Infrastrukturen, das als Vermittler zwischen Clienten und Backend-Servern agiert. Im Gegensatz zu einem Forward Proxy, der Anfragen von internen Clients nach außen vertritt, stellt der Reverse Proxy eine Eindepunkt-Schnittstelle (Single Point of Entry) für externe Anfragen dar. Typische Vertreter sind nginx, HAProxy oder Apache HTTP Server. Die Grundprinzipie des Reverse Proxys besteht darin, eingehende HTTP- oder HTTPS-Anfragen entgegenzunehmen und diese basierend auf definierten Routing-Regeln an die geeigneten Backend-Server weiterzuleiten. Diese Architektur bietet erhebliche Vorteile wie Lastverteilung, SSL/TLS-Terminierung, Caching und eine vereinfachte Verwaltung mehrerer hinterlegter Dienste. Durch die Abstraktion der Backend-Infrastruktur wird sichergestellt, dass Clients die Details der internen Topologie, wie IP-Adressen oder Server-Namen, nicht kennen müssen.\n\nIn dynamischen Umgebungen, insbesondere bei der Nutzung von Container-Orchestrierungsplattformen wie Kubernetes oder Docker Compose, ist die Stabilität der IP-Adressen ein wesentlicher Faktor. Container sind per Design oft ephemer; sie können bei Neustarts, Skalierungsoperationen oder Fehlerbehebungen neu erstellt werden. Dies führt dazu, dass ihre zugewiesenen IP-Adressen ändern können. Wenn ein Reverse Proxy statisch konfiguriert ist und auf feste IP-Adressen oder Ports zeigt, die durch den Container-Wechsel ungültig werden, treten häufig Verbindungsfehler auf. Ein typisches Szenario ist, dass der Proxy weiterhin versucht, den Datenverkehr an eine alte, nicht mehr existierende IP-Adresse zu senden. In diesem Fall scheitert die Verbindung, da es keinen Host mit dieser Adresse im Netzwerksegment gibt.\n\nEin weiterer häufiger Fehlermechanismus entsteht durch das Timing von DNS- oder Service-Auflösungen. In Container-Netzwerken werden oft Dienstnamen anstelle fester IPs verwendet. Wenn sich ein Container neu registriert, kann es eine Zeitspanne geben, in der die DNS-Einträge noch nicht aktualisiert sind. Der Reverse Proxy könnte dann eine veraltete Adresse auflösen, was zu einem \"Connection Refused\"-Fehler oder einer Timeout-Situation führt. Zudem können Probleme auftreten, wenn der Backend-Container zwar erreichbar ist, aber der Dienst innerhalb des Containers noch nicht bereit ist, Anfragen anzunehmen (beispielsweise während des Boot-Prozesses). Der Proxy meldet dann oft einen \"Bad Gateway\" (HTTP 502), obwohl die Netzwerkverbindung technisch möglich wäre, aber die Applikation nicht antwortet.\n\nUm solche Fehler effizient zu eingrenzen, ist die Analyse der Logs sowohl des Proxys als auch der umgebenden Infrastruktur unerlässlich. Bei nginx beispielsweise beginnt die Fehlersuche meist im `error.log`. Hier sucht man nach Meldungen wie `connect() failed (111: Connection refused)`, was darauf hindeutet, dass der Proxy eine IP erreicht hat, aber kein Prozess auf dem angegebenen Port lauscht, oder dass die IP falsch ist. Ein weiterer Hinweis ist `recv() failed (104: Connection reset by peer)`, was oft auf Netzwerkprobleme oder einen Abbruch durch den Backend-Server hindeutet.\n\nEs ist jedoch wichtig zu beachten, dass ein Reverse Proxy nicht immer die IP-Adresse loggt, an die er sich wendet, sondern eher den Dienstnamen. Daher muss man die Logs mit den Events der Orchestrierungsplattform korrelieren. In Kubernetes etwa kann man"
|
|
}
|
|
}
|
|
],
|
|
"created": 1789936622,
|
|
"model": "qwen-medium",
|
|
"system_fingerprint": "b10964-b29c606e2",
|
|
"object": "chat.completion",
|
|
"usage": {
|
|
"completion_tokens": 768,
|
|
"prompt_tokens": 59,
|
|
"total_tokens": 827,
|
|
"prompt_tokens_details": {
|
|
"cached_tokens": 0
|
|
}
|
|
},
|
|
"id": "chatcmpl-vmyiY20xbf8VvEpwoqU8ZaRalfjB5agL",
|
|
"timings": {
|
|
"cache_n": 0,
|
|
"prompt_n": 59,
|
|
"prompt_ms": 119.305,
|
|
"prompt_per_token_ms": 2.0221186440677967,
|
|
"prompt_per_second": 494.53082435773854,
|
|
"predicted_n": 768,
|
|
"predicted_ms": 13364.148,
|
|
"predicted_per_token_ms": 17.423921773142112,
|
|
"predicted_per_second": 57.39236051561237,
|
|
"draft_n": 1057,
|
|
"draft_n_accepted": 414
|
|
},
|
|
"wall_seconds": 13.542396485980134
|
|
},
|
|
{
|
|
"choices": [
|
|
{
|
|
"finish_reason": "length",
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "```python\nimport asyncio\nfrom typing import Any, Awaitable, Callable, List, Optional, Union\n\n\nasync def first_success(\n awaitables: List[Awaitable],\n return_exceptions: bool = True\n) -> Any:\n \"\"\"\n Start all awaitables concurrently, return the first successful result,\n cancel and await remaining tasks, and collect exceptions if all fail.\n\n Args:\n awaitables: A list of awaitables to run concurrently.\n return_exceptions: If True, when all awaitables fail, raise an\n ExceptionGroup (Python 3.11+) or re-raise the first exception\n with others attached. If False, simply re-raise the first exception.\n\n Returns:\n The result of the first successfully completed awaitable.\n\n Raises:\n ExceptionGroup (or the first exception if return_exceptions=False):\n If all awaitables fail.\n \"\"\"\n if not awaitables:\n raise ValueError(\"awaitables list must not be empty\")\n\n # Create tasks from all awaitables\n tasks = [asyncio.ensure_future(av) for av in awaitables]\n\n try:\n # Use asyncio.wait to get the first completed task\n # We need to keep polling until we find a successful one\n pending = set(tasks)\n\n while pending:\n done, pending = await asyncio.wait(\n pending,\n return_when=asyncio.FIRST_COMPLETED\n )\n\n if not done:\n break\n\n for task in done:\n try:\n result = task.result()\n # Success: cancel all remaining pending tasks and return result\n for remaining in pending:\n remaining.cancel()\n # Await cancellation of remaining tasks\n if pending:\n await asyncio.wait(pending)\n return result\n except Exception:\n # This task failed; continue checking other done tasks\n # (if multiple completed simultaneously)\n pass\n # If we get here, all completed tasks in this batch failed\n # Continue looping to check if any other pending tasks complete\n # All tasks are done and none succeeded\n # Collect all exceptions\n exceptions = []\n for task in tasks:\n try:\n task.result()\n except Exception as e:\n exceptions.append(e)\n\n if exceptions:\n if return_exceptions:\n if hasattr(asyncio, 'ExceptionGroup'):\n # Python 3.11+\n raise ExceptionGroup(\"All awaitables failed\", exceptions)\n else:\n # Fallback for older Python: raise first exception\n # Attach other exceptions as context if possible\n raise exceptions[0] from exceptions[1:] if len(exceptions) > 1 else None\n else:\n raise exceptions[0]\n else:\n # No exceptions but no success either (shouldn't happen normally)\n raise RuntimeError(\"No successful result and no exceptions captured\")\n\n except (asyncio.CancelledError, KeyboardInterrupt):\n # Cancel all tasks on interruption\n for task in tasks:\n if not task.done():\n task.cancel()\n if tasks:\n await asyncio.gather(*tasks, return_exceptions=True)\n raise\n\n\n# Usage example\nif __name__ == \"__main__\":\n async def slow_success(delay: float, value: Any) -> Any:\n \"\"\"Simulate an async operation that succeeds after a delay.\"\"\"\n await asyncio.sleep(delay)\n return value\n\n async def failing_op(delay: float, msg: str) -> Any:\n "
|
|
}
|
|
}
|
|
],
|
|
"created": 1789936632,
|
|
"model": "qwen-medium",
|
|
"system_fingerprint": "b10964-b29c606e2",
|
|
"object": "chat.completion",
|
|
"usage": {
|
|
"completion_tokens": 768,
|
|
"prompt_tokens": 54,
|
|
"total_tokens": 822,
|
|
"prompt_tokens_details": {
|
|
"cached_tokens": 0
|
|
}
|
|
},
|
|
"id": "chatcmpl-KoYAO3KmvB8ZbjUzmpfY0gidKhQtmqXD",
|
|
"timings": {
|
|
"cache_n": 0,
|
|
"prompt_n": 54,
|
|
"prompt_ms": 117.614,
|
|
"prompt_per_token_ms": 2.178037037037037,
|
|
"prompt_per_second": 459.1290152532862,
|
|
"predicted_n": 768,
|
|
"predicted_ms": 10175.854,
|
|
"predicted_per_token_ms": 13.26708474576271,
|
|
"predicted_per_second": 75.37450910754026,
|
|
"draft_n": 803,
|
|
"draft_n_accepted": 499
|
|
},
|
|
"wall_seconds": 10.357252335990779
|
|
}
|
|
],
|
|
"prefill": [
|
|
{
|
|
"target": 24576,
|
|
"response": {
|
|
"choices": [
|
|
{
|
|
"finish_reason": "length",
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "```json\n{\n \"alpha\": \"RAVEN-417\",\n \"beta\": \"CEDAR-928\",\n \"gamma\": \"ORBIT-563\"\n}\n```\n\nDie Verifizierung der identifizierten Nadelwerte (`RAVEN-417`, `CEDAR-928`, `ORBIT-563`) innerhalb des vorliegenden Datensatzes erfordert einen streng systematischen und evidenzbasierten Ansatz, der auf der präzisen Analyse der Textstruktur und der Positionierung der Markierungen beruht. Zunächst ist es entscheidend, die Integrität der Quelldaten zu wahren, indem man keine extrapolierten oder hypothetischen Datenpunkte einführt. Der vorliegende Datenset besteht überwiegend aus identischen, wiederholten Zeilen, die den String „cobalt lantern maple orbit quartz river silver tango.“ enthalten, wobei die Einträge von `Record 000000` bis `Record 001169` nummeriert sind. Die Nadeln wurden jedoch nicht als Teil des eigentlichen Datensatzinhalts (wie etwa in der Spalte der Werte) platziert, sondern als isolierte Metadaten-Zeilen zwischen den regulären Aufzeichnungen.\n\nUm `NEEDLE_ALPHA=RAVEN-417` zu verifizieren, muss der Fokus auf die Region um die Aufzeichnungen 000146 bis 000147 gelegt werden. Hier findet sich die Zeile, die mit „Record 000146: cob...“ beginnt und abgebrochen ist, gefolgt direkt von der isolierten Zeile „NEEDLE_ALPHA=RAVEN-417“. Die Verifikation erfolgt durch die Bestätigung, dass diese spezifische String-Kombination exakt an dieser Position im Textstrom erscheint und nicht in anderen Teilen des Dokuments vorkommt. Ein ähnlicher Prozess gilt für `NEEDLE_BETA=CEDAR-928`, der sich zwischen `Record 000584` und `Record 000585` befindet. Hier ist besondere Sorgfalt geboten, da die Zeile für Record 000585 aufgrund der Platzierung der Nadel ungewöhnlich formatiert oder fragmentiert erscheint („Recor... d 000585“). Die Verifikation verlangt die sorgfält"
|
|
}
|
|
}
|
|
],
|
|
"created": 1789936654,
|
|
"model": "qwen-medium",
|
|
"system_fingerprint": "b10964-b29c606e2",
|
|
"object": "chat.completion",
|
|
"usage": {
|
|
"completion_tokens": 512,
|
|
"prompt_tokens": 24674,
|
|
"total_tokens": 25186,
|
|
"prompt_tokens_details": {
|
|
"cached_tokens": 0
|
|
}
|
|
},
|
|
"id": "chatcmpl-SdiBR3b0wLujVUJv7evtigcnXq0O6zVk",
|
|
"timings": {
|
|
"cache_n": 0,
|
|
"prompt_n": 24674,
|
|
"prompt_ms": 12936.629,
|
|
"prompt_per_token_ms": 0.5243020588473697,
|
|
"prompt_per_second": 1907.2974883951606,
|
|
"predicted_n": 512,
|
|
"predicted_ms": 8522.374,
|
|
"predicted_per_token_ms": 16.677835616438355,
|
|
"predicted_per_second": 59.95981870779199,
|
|
"draft_n": 579,
|
|
"draft_n_accepted": 317
|
|
},
|
|
"wall_seconds": 21.544975613011047,
|
|
"recall": {
|
|
"RAVEN-417": true,
|
|
"CEDAR-928": true,
|
|
"ORBIT-563": true
|
|
}
|
|
}
|
|
}
|
|
],
|
|
"finished": 1789936654.5839925
|
|
}
|