{ "status": "complete", "scope": "Exercise native serving behavior, near-limit prefill and greedy decode over HTTP.\nLocal diagnostics, not an official task benchmark or Unsloth Divergence-300.\n", "base_url": "http://127.0.0.1:8000", "model": "Qwen/Qwen3.5-9B", "behavior": [ { "id": "behavior-arithmetic", "pass": true, "seconds": 0.21067160599977797, "response": { "id": "chatcmpl-9be1b4746a841200", "object": "chat.completion", "created": 1788789796, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "391", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 29, "total_tokens": 33, "completion_tokens": 4, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-json", "pass": true, "seconds": 0.4851606769998398, "response": { "id": "chatcmpl-a6c6905c42faf98f", "object": "chat.completion", "created": 1788789796, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "{\"name\": \"Ada\", \"age\": 37}", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 32, "total_tokens": 46, "completion_tokens": 14, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-sort", "pass": true, "seconds": 0.4584858130001521, "response": { "id": "chatcmpl-a04e50506f2e6c60", "object": "chat.completion", "created": 1788789796, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "[-2, 0, 5, 8]", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 36, "total_tokens": 49, "completion_tokens": 13, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-french", "pass": true, "seconds": 0.1838943630000358, "response": { "id": "chatcmpl-bee1cdc0e067e878", "object": "chat.completion", "created": 1788789797, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "bonjour", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 28, "total_tokens": 31, "completion_tokens": 3, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-spanish", "pass": true, "seconds": 0.1836344849998568, "response": { "id": "chatcmpl-b0488f078a632b8e", "object": "chat.completion", "created": 1788789797, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "31", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 32, "total_tokens": 35, "completion_tokens": 3, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-japanese", "pass": true, "seconds": 0.15754658599962568, "response": { "id": "chatcmpl-ad1a8e57246db75f", "object": "chat.completion", "created": 1788789797, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "東京", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 26, "total_tokens": 28, "completion_tokens": 2, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-python", "pass": true, "seconds": 0.5394853650004734, "response": { "id": "chatcmpl-87167857bc862301", "object": "chat.completion", "created": 1788789797, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "def is_even(n):\n return n % 2 == 0", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 42, "total_tokens": 58, "completion_tokens": 16, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-extract", "pass": true, "seconds": 0.2104586579998795, "response": { "id": "chatcmpl-85980341d80673ac", "object": "chat.completion", "created": 1788789798, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "mira@example.org", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 32, "total_tokens": 36, "completion_tokens": 4, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-reasoning", "pass": true, "seconds": 0.18333528800030763, "response": { "id": "chatcmpl-a3f8227f04c83350", "object": "chat.completion", "created": 1788789798, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "14", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 48, "total_tokens": 51, "completion_tokens": 3, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-instruction", "pass": true, "seconds": 0.1830841800001508, "response": { "id": "chatcmpl-87d30df0db67081e", "object": "chat.completion", "created": 1788789798, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "violet", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 45, "total_tokens": 48, "completion_tokens": 3, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } }, { "id": "behavior-tool", "pass": true, "seconds": 0.9492834880002192, "response": { "id": "chatcmpl-aa002150e4430236", "object": "chat.completion", "created": 1788789799, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": null, "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [ { "id": "chatcmpl-tool-bc01bc3df6cd847c", "type": "function", "function": { "name": "get_weather", "arguments": "{\"city\": \"Oslo\"}" } } ], "reasoning": null }, "logprobs": null, "finish_reason": "tool_calls", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 281, "total_tokens": 308, "completion_tokens": 27, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null } } ], "long_context": [ { "id": "heldout-near-8k-retrieval", "pass": true, "seconds": 3.222331581999242, "response": { "id": "chatcmpl-aa9cfa5330c8c0d4", "object": "chat.completion", "created": 1788789799, "model": "Qwen/Qwen3.5-9B", "choices": [ { "index": 0, "message": { "role": "assistant", "content": "7391", "refusal": null, "annotations": null, "audio": null, "function_call": null, "tool_calls": [], "reasoning": null }, "logprobs": null, "finish_reason": "stop", "stop_reason": null, "token_ids": null } ], "service_tier": null, "system_fingerprint": null, "usage": { "prompt_tokens": 7888, "total_tokens": 7893, "completion_tokens": 5, "prompt_tokens_details": null }, "prompt_logprobs": null, "prompt_token_ids": null, "kv_transfer_params": null }, "prepared_prompt_tokens": 7888 } ], "decode_runs": [ { "text": "In the context of Large Language Model (LLM) inference, batch-one decoding represents a scenario where the model processes a single input sequence token-by-token, generating one output token at a time. This mode of operation is frequently memory-bandwidth limited because the computational work required to generate a single token is disproportionately small compared to the massive volume of data that must be moved from memory to the GPU. Unlike matrix multiplication operations in training or prefill stages, which can saturate compute units with dense arithmetic, decoding involves sparse, sequential operations where the GPU spends a significant portion of its cycle time waiting for data to arrive from High-Bandwidth Memory", "usage": { "prompt_tokens": 55, "total_tokens": 183, "completion_tokens": 128 }, "ttft_ms": 139.36781699976564, "decode_tokens_per_second": 36.57806033620924, "decode_seconds": 3.472026641999946, "timing_scope": "Client-observed first nonempty delta through finish event; excludes first output token" }, { "text": "In the context of Large Language Model (LLM) inference, batch-one decoding represents a scenario where the model processes a single input sequence token-by-token, generating one output token at a time. This mode of operation is frequently memory-bandwidth limited because the computational work required to generate a single token is disproportionately small compared to the massive volume of data that must be moved from memory to the GPU. Unlike matrix multiplication operations in training or prefill stages, which can saturate compute units with dense arithmetic, decoding involves sparse, sequential operations where the GPU spends a significant portion of its cycle time waiting for data to arrive from High-Bandwidth Memory", "usage": { "prompt_tokens": 55, "total_tokens": 183, "completion_tokens": 128 }, "ttft_ms": 126.2130730001445, "decode_tokens_per_second": 36.57407054866148, "decode_seconds": 3.4724053979998644, "timing_scope": "Client-observed first nonempty delta through finish event; excludes first output token" }, { "text": "In the context of Large Language Model (LLM) inference, batch-one decoding represents a scenario where the model processes a single input sequence token-by-token, generating one output token at a time. This mode of operation is frequently memory-bandwidth limited because the computational work required to generate a single token is disproportionately small compared to the massive volume of data that must be moved from memory to the GPU. Unlike matrix multiplication operations in training or prefill stages, which can saturate compute units with dense arithmetic, decoding involves sparse, sequential operations where the GPU spends a significant portion of its cycle time waiting for data to arrive from High-Bandwidth Memory", "usage": { "prompt_tokens": 55, "total_tokens": 183, "completion_tokens": 128 }, "ttft_ms": 126.25369200031855, "decode_tokens_per_second": 36.581673138552055, "decode_seconds": 3.471683744999609, "timing_scope": "Client-observed first nonempty delta through finish event; excludes first output token" } ], "initial_decode": { "text": "In the context of Large Language Model (LLM) inference, batch-one decoding represents a scenario where the model processes a single input sequence token-by-token, generating one output token at a time. This mode of operation is frequently memory-bandwidth limited because the computational work required to generate a single token is disproportionately small compared to the massive volume of data that must be moved from memory to the GPU. Unlike matrix multiplication operations in training or prefill stages, which can saturate compute units with dense arithmetic, decoding involves sparse, sequential operations where the GPU spends a significant portion of its cycle time waiting for data to arrive from High-Bandwidth Memory", "usage": { "prompt_tokens": 55, "total_tokens": 183, "completion_tokens": 128 }, "ttft_ms": 209.12037899961433, "decode_tokens_per_second": 36.55089353510533, "decode_seconds": 3.47460725899964, "timing_scope": "Client-observed first nonempty delta through finish event; excludes first output token" }, "behavior_passes": 11, "long_context_passes": 1, "median_decode_tokens_per_second": 36.57806033620924, "identical_greedy_completions": true, "request_isolation_passed": true }