{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 62.115,
        "median": 61.032,
        "minimum": 58.155,
        "p90": 62.115
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 68.529,
        "median": 67.349,
        "minimum": 63.722,
        "p90": 68.529
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.407,
        "median": 0.399,
        "minimum": 0.398,
        "p90": 0.407
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 62.115,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.721069,
              "decode_tokens_per_second": 68.529,
              "finish_reason": "length",
              "measured_sse_events": 251,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 995,
              "reasoning_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "response_tail_seconds": 0.000152,
              "stream_done_marker": true,
              "stream_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.120493,
              "ttft_seconds": 0.399424,
              "wall_seconds": 4.120645
            }
          ],
          "wall_seconds": 4.121419
        },
        {
          "aggregate_end_to_end_tokens_per_second": 61.032,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.786258,
              "decode_tokens_per_second": 67.349,
              "finish_reason": "length",
              "measured_sse_events": 251,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 995,
              "reasoning_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "response_tail_seconds": 0.000202,
              "stream_done_marker": true,
              "stream_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.193663,
              "ttft_seconds": 0.407404,
              "wall_seconds": 4.193865
            }
          ],
          "wall_seconds": 4.194505
        },
        {
          "aggregate_end_to_end_tokens_per_second": 58.155,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.00173,
              "decode_tokens_per_second": 63.722,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1043,
              "reasoning_sha256": "2a4e348aaa9a1f42c132d5cbb62ba5b97ce8470636f9137016b6503e4fc88e6f",
              "response_tail_seconds": 0.001922,
              "stream_done_marker": true,
              "stream_sha256": "2a4e348aaa9a1f42c132d5cbb62ba5b97ce8470636f9137016b6503e4fc88e6f",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.399588,
              "ttft_seconds": 0.397858,
              "wall_seconds": 4.40151
            }
          ],
          "wall_seconds": 4.402025
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 87.699,
        "median": 84.465,
        "minimum": 72.072,
        "p90": 87.699
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 50.074,
        "median": 47.458,
        "minimum": 39.749,
        "p90": 47.857
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.805,
        "median": 0.686,
        "minimum": 0.464,
        "p90": 0.745
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 72.072,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.85054,
              "decode_tokens_per_second": 43.586,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 993,
              "reasoning_sha256": "013fd5a358b83b7613dba7b5843b873f80d2288f5d3384b4ad5f27c94b2bd5ee",
              "response_tail_seconds": 0.000266,
              "stream_done_marker": true,
              "stream_sha256": "013fd5a358b83b7613dba7b5843b873f80d2288f5d3384b4ad5f27c94b2bd5ee",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.655043,
              "ttft_seconds": 0.804504,
              "wall_seconds": 6.65531
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 6.415206,
              "decode_tokens_per_second": 39.749,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 969,
              "reasoning_sha256": "ab062a86025e793687f8c11bbfe372007deee181ec2986e0f0641cf079a007f0",
              "response_tail_seconds": 0.001063,
              "stream_done_marker": true,
              "stream_sha256": "ab062a86025e793687f8c11bbfe372007deee181ec2986e0f0641cf079a007f0",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 7.101486,
              "ttft_seconds": 0.68628,
              "wall_seconds": 7.102549
            }
          ],
          "wall_seconds": 7.104023
        },
        {
          "aggregate_end_to_end_tokens_per_second": 87.699,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.372639,
              "decode_tokens_per_second": 47.463,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 1012,
              "reasoning_sha256": "18a802268b24fb5bcc264d6e6b5391b020ebc804c4ff2ffaac7fb52ad967da21",
              "response_tail_seconds": 0.000811,
              "stream_done_marker": true,
              "stream_sha256": "18a802268b24fb5bcc264d6e6b5391b020ebc804c4ff2ffaac7fb52ad967da21",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.836549,
              "ttft_seconds": 0.46391,
              "wall_seconds": 5.83736
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.328401,
              "decode_tokens_per_second": 47.857,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1058,
              "reasoning_sha256": "f7ac712a719254fa3c5feece800578c471261230fbebbc08a0295f2a5ea97c82",
              "response_tail_seconds": 0.000726,
              "stream_done_marker": true,
              "stream_sha256": "f7ac712a719254fa3c5feece800578c471261230fbebbc08a0295f2a5ea97c82",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.792282,
              "ttft_seconds": 0.463881,
              "wall_seconds": 5.793008
            }
          ],
          "wall_seconds": 5.838153
        },
        {
          "aggregate_end_to_end_tokens_per_second": 84.465,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.373759,
              "decode_tokens_per_second": 47.453,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 1065,
              "reasoning_sha256": "795f63912806e0683b9c5cc59a36484f27cd76a86423bbac715530949d8afcb7",
              "response_tail_seconds": 0.00071,
              "stream_done_marker": true,
              "stream_sha256": "795f63912806e0683b9c5cc59a36484f27cd76a86423bbac715530949d8afcb7",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.060022,
              "ttft_seconds": 0.686263,
              "wall_seconds": 6.060732
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.092466,
              "decode_tokens_per_second": 50.074,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1023,
              "reasoning_sha256": "13a424c2360e63263c95fb01b95f0bbab6c39cb424c5b209947043d1339cf01e",
              "response_tail_seconds": 0.000473,
              "stream_done_marker": true,
              "stream_sha256": "13a424c2360e63263c95fb01b95f0bbab6c39cb424c5b209947043d1339cf01e",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.837044,
              "ttft_seconds": 0.744578,
              "wall_seconds": 5.837517
            }
          ],
          "wall_seconds": 6.061704
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 116.094,
        "median": 114.693,
        "minimum": 112.103,
        "p90": 116.094
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 37.629,
        "median": 32.346,
        "minimum": 29.836,
        "p90": 36.136
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.875,
        "median": 0.585,
        "minimum": 0.572,
        "p90": 0.799
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 112.103,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.546586,
              "decode_tokens_per_second": 29.836,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1031,
              "reasoning_sha256": "ee6218cd43132fb27bd9ae75af79871019e722ae71a19f90a915e54b3d1a59b1",
              "response_tail_seconds": 0.000426,
              "stream_done_marker": true,
              "stream_sha256": "ee6218cd43132fb27bd9ae75af79871019e722ae71a19f90a915e54b3d1a59b1",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.131558,
              "ttft_seconds": 0.584973,
              "wall_seconds": 9.131985
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 6.776682,
              "decode_tokens_per_second": 37.629,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1019,
              "reasoning_sha256": "dedbf8c46d490570a13aeb75891db96aecd3c9e90f1f778b6b5af9aa8b61b33d",
              "response_tail_seconds": 7.8e-05,
              "stream_done_marker": true,
              "stream_sha256": "dedbf8c46d490570a13aeb75891db96aecd3c9e90f1f778b6b5af9aa8b61b33d",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 7.361002,
              "ttft_seconds": 0.58432,
              "wall_seconds": 7.36108
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.827016,
              "decode_tokens_per_second": 32.579,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 1024,
              "reasoning_sha256": "fba110c2d193cb8d83d1dfa301273446457e1c48f2541e65a7d96e4a9ac26659",
              "response_tail_seconds": 5.8e-05,
              "stream_done_marker": true,
              "stream_sha256": "fba110c2d193cb8d83d1dfa301273446457e1c48f2541e65a7d96e4a9ac26659",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.4119,
              "ttft_seconds": 0.584884,
              "wall_seconds": 8.411958
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.144286,
              "decode_tokens_per_second": 31.31,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1007,
              "reasoning_sha256": "13d4bc7792aba9858f60536de4e4d4668d4b28ecb8bcd59e4c4eeadc01fdc05b",
              "response_tail_seconds": 4.2e-05,
              "stream_done_marker": true,
              "stream_sha256": "13d4bc7792aba9858f60536de4e4d4668d4b28ecb8bcd59e4c4eeadc01fdc05b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.72912,
              "ttft_seconds": 0.584833,
              "wall_seconds": 8.729161
            }
          ],
          "wall_seconds": 9.134464
        },
        {
          "aggregate_end_to_end_tokens_per_second": 114.693,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.056579,
              "decode_tokens_per_second": 36.136,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1021,
              "reasoning_sha256": "e8603b3cef2b23c989c677b9e62c277aa7c9d002fca2ef1a3e8bd4331ebf3975",
              "response_tail_seconds": 0.000289,
              "stream_done_marker": true,
              "stream_sha256": "e8603b3cef2b23c989c677b9e62c277aa7c9d002fca2ef1a3e8bd4331ebf3975",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 7.855794,
              "ttft_seconds": 0.799215,
              "wall_seconds": 7.856083
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.049337,
              "decode_tokens_per_second": 31.68,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1087,
              "reasoning_sha256": "615d4a03633bbee39c2864e19267fa35695832a824a8240220c99e2dabe2bf9b",
              "response_tail_seconds": 0.000756,
              "stream_done_marker": true,
              "stream_sha256": "615d4a03633bbee39c2864e19267fa35695832a824a8240220c99e2dabe2bf9b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.924235,
              "ttft_seconds": 0.874898,
              "wall_seconds": 8.924991
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.124298,
              "decode_tokens_per_second": 31.387,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1036,
              "reasoning_sha256": "8319f52baa6d742018a1a579430b990e89963c2a75d041b989328fe0f3be7ced",
              "response_tail_seconds": 0.001784,
              "stream_done_marker": true,
              "stream_sha256": "8319f52baa6d742018a1a579430b990e89963c2a75d041b989328fe0f3be7ced",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.92379,
              "ttft_seconds": 0.799493,
              "wall_seconds": 8.925574
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.941065,
              "decode_tokens_per_second": 32.112,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1025,
              "reasoning_sha256": "a11f2ceff8d33bb8ee9d234100769938dda910c9ea63e1f719feb50e74edb2ea",
              "response_tail_seconds": 0.000194,
              "stream_done_marker": true,
              "stream_sha256": "a11f2ceff8d33bb8ee9d234100769938dda910c9ea63e1f719feb50e74edb2ea",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.740345,
              "ttft_seconds": 0.79928,
              "wall_seconds": 8.740538
            }
          ],
          "wall_seconds": 8.928187
        },
        {
          "aggregate_end_to_end_tokens_per_second": 116.094,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.243028,
              "decode_tokens_per_second": 30.935,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 976,
              "reasoning_sha256": "2040704ee3e65c256bb7dff690d2d2fddd3b02ff7f59e3cb49379c1ec7d59814",
              "response_tail_seconds": 0.001177,
              "stream_done_marker": true,
              "stream_sha256": "2040704ee3e65c256bb7dff690d2d2fddd3b02ff7f59e3cb49379c1ec7d59814",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.815872,
              "ttft_seconds": 0.572844,
              "wall_seconds": 8.817049
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.762766,
              "decode_tokens_per_second": 32.849,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1060,
              "reasoning_sha256": "d72acaae60636d4a258d54ce23e91938463ab51c80c2358813ea9e24922d76ff",
              "response_tail_seconds": 0.000617,
              "stream_done_marker": true,
              "stream_sha256": "d72acaae60636d4a258d54ce23e91938463ab51c80c2358813ea9e24922d76ff",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.33449,
              "ttft_seconds": 0.571724,
              "wall_seconds": 8.335107
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.763234,
              "decode_tokens_per_second": 32.847,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1062,
              "reasoning_sha256": "6eb154af63be49f7c3ecd3c15196b6719cd9d5dd92aca9a026226ffc4dbab7a6",
              "response_tail_seconds": 0.000257,
              "stream_done_marker": true,
              "stream_sha256": "6eb154af63be49f7c3ecd3c15196b6719cd9d5dd92aca9a026226ffc4dbab7a6",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.335082,
              "ttft_seconds": 0.571849,
              "wall_seconds": 8.33534
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.296684,
              "decode_tokens_per_second": 34.947,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1036,
              "reasoning_sha256": "87f6a846d606e572b2c4013f6466ba4132c36c05271b69e43ad36ff9421f964e",
              "response_tail_seconds": 0.000446,
              "stream_done_marker": true,
              "stream_sha256": "87f6a846d606e572b2c4013f6466ba4132c36c05271b69e43ad36ff9421f964e",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 7.869297,
              "ttft_seconds": 0.572613,
              "wall_seconds": 7.869744
            }
          ],
          "wall_seconds": 8.820443
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 75.764,
        "median": 71.783,
        "minimum": 66.099,
        "p90": 75.764
      },
      "pooled_decode_tokens_per_second": 71.262,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 10179,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 141.789196,
          "decode_tokens_per_second": 71.783,
          "finish_reason": "stop",
          "measured_sse_events": 10171,
          "output": "```go\n// ratelimit.go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time; inject a fake implementation in tests.\ntype Clock interface{ Now() time.Time }\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// Errors returned by New.\nvar (\n\tErrCapacity = errors.New(\"ratelimit: capacity must be finite and > 0\")\n\tErrRate     = errors.New(\"ratelimit: rate must be finite and > 0\")\n)\n\n// Limiter is a token bucket holding at most capacity tokens, refilling at\n// rate tokens per second. The zero value is not usable; use New.\n// It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a limiter that starts full. If clock is nil, the system clock\n// is used.\nfunc New(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tif invalid(capacity) {\n\t\treturn nil, ErrCapacity\n\t}\n\tif invalid(rate) {\n\t\treturn nil, ErrRate\n\t}\n\tif clock == nil {\n\t\tclock = systemClock{}\n\t}\n\treturn &Limiter{clock: clock, capacity: capacity, rate: rate, tokens: capacity, last: clock.Now()}, nil\n}\n\nfunc invalid(f float64) bool { return math.IsNaN(f) || math.IsInf(f, 0) || f <= 0 }\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at the current clock reading\n// and consumes them if so. Negative n returns false; n == 0 consumes nothing\n// and returns true.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now := l.clock.Now(); now.After(l.last) { // ignore clock regressions\n\t\tl.tokens = math.Min(l.capacity, l.tokens+now.Sub(l.last).Seconds()*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct{ now time.Time }\n\nfunc (f *fakeClock) Now() time.Time          { return f.now }\nfunc (f *fakeClock) Advance(d time.Duration) { f.now = f.now.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tclock    Clock\n\t\twantErr  error\n\t}{\n\t\t{\"ok defaults to system clock\", 1, 1, nil, nil},\n\t\t{\"ok with fake clock\", 10, 0.5, &fakeClock{}, nil},\n\t\t{\"zero capacity\", 0, 1, nil, ErrCapacity},\n\t\t{\"negative capacity\", -1, 1, nil, ErrCapacity},\n\t\t{\"NaN capacity\", math.NaN(), 1, nil, ErrCapacity},\n\t\t{\"Inf capacity\", math.Inf(1), 1, nil, ErrCapacity},\n\t\t{\"zero rate\", 1, 0, nil, ErrRate},\n\t\t{\"negative rate\", 1, -2, nil, ErrRate},\n\t\t{\"NaN rate\", 1, math.NaN(), nil, ErrRate},\n\t\t{\"-Inf rate\", 1, math.Inf(-1), nil, ErrRate},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.capacity, tt.rate, tt.clock)\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"New(%v, %v) error = %v, want %v\", tt.capacity, tt.rate, err, tt.wantErr)\n\t\t\t}\n\t\t\tif tt.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"New() returned nil limiter without error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tsteps    []step\n\t}{\n\t\t{\"burst up to capacity\", 5, 1, []step{\n\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t}},\n\t\t{\"fractional refill\", 2, 1, []step{\n\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t{500 * time.Millisecond, 1, false},\n\t\t\t{500 * time.Millisecond, 1, true},\n\t\t}},\n\t\t{\"refill capped at capacity\", 2, 1, []step{\n\t\t\t{0, 2, true}, {10 * time.Second, 3, false}, {0, 2, true},\n\t\t}},\n\t\t{\"clock regression ignored\", 1, 1, []step{\n\t\t\t{0, 1, true}, {0, 1, false}, {-5 * time.Second, 1, false},\n\t\t}},\n\t\t{\"zero and negative n\", 1, 1, []step{\n\t\t\t{0, 0, true}, {0, -1, false}, {0, 1, true},\n\t\t}},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclk := &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n\t\t\tl, err := New(tt.capacity, tt.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: after advancing %v, AllowN(%d) = %v, want %v\",\n\t\t\t\t\t\ti, s.advance, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowNConcurrent(t *testing.T) {\n\tclk := &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n\tl, err := New(50, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tconst goroutines = 100\n\tvar allowed atomic.Int64\n\tvar wg sync.WaitGroup\n\twg.Add(goroutines)\n\tfor i := 0; i < goroutines; i++ {\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\tallowed.Add(1)\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := allowed.Load(); got != 50 {\n\t\tt.Fatalf(\"allowed = %d, want exactly 50\", got)\n\t}\n}\n```",
          "output_characters": 5122,
          "output_sha256": "00d8eefdee299eac14d5c72194c69e7d3f46715fb63444675f8bb460cdbedc4b",
          "prompt_tokens": 132,
          "reasoning_characters": 27550,
          "reasoning_sha256": "5cf32f21ad374b9a7544d896246ccf764a73efbccc6b3bb0dbc03040c34c1cc9",
          "response_tail_seconds": 0.004304,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "821b3925a052aa1ca5b85468810b2b04a5da6eaf5a9207fdbef4389f70884b7b",
          "time_to_first_visible_seconds": 128.92078,
          "time_to_last_output_seconds": 142.196569,
          "ttft_seconds": 0.407373,
          "wall_seconds": 142.200873
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 11598,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 175.447713,
          "decode_tokens_per_second": 66.099,
          "finish_reason": "stop",
          "measured_sse_events": 11596,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time retrieval; inject a fake in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// ErrInvalidConfig is returned by New when rate or burst is not finite and > 0.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be finite and > 0\")\n\n// Limiter is a concurrency-safe token-bucket rate limiter. Tokens refill at\n// rate per second up to burst capacity; the bucket starts full. The zero\n// value is not usable; construct with New.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n}\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. A nil clock defaults to the system clock.\nfunc New(rate, burst float64, clock Clock) (*Limiter, error) {\n\tif !(rate > 0) || math.IsInf(rate, 0) || !(burst > 0) || math.IsInf(burst, 0) {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = systemClock{}\n\t}\n\treturn &Limiter{clock: clock, rate: rate, burst: burst, tokens: burst, last: clock.Now()}, nil\n}\n\n// Allow reports whether one token is available at the clock's current time.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.clock.Now(), 1)\n}\n\n// AllowN reports whether n tokens may be consumed at time now. It returns\n// false for n <= 0 or n > burst. A timestamp older than the previous call\n// is ignored (no negative refill).\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n <= 0 || float64(n) > l.burst {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif elapsed := now.Sub(l.last).Seconds(); elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > l.burst {\n\t\t\tl.tokens = l.burst\n\t\t}\n\t\tl.last = now\n\t}\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   float64\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1, 5, true},\n\t\t{\"NaN rate\", math.NaN(), 5, true},\n\t\t{\"Inf rate\", math.Inf(1), 5, true},\n\t\t{\"zero burst\", 10, 0, true},\n\t\t{\"negative burst\", 10, -3, true},\n\t\t{\"NaN burst\", 10, math.NaN(), true},\n\t\t{\"Inf burst\", 10, math.Inf(1), true},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.rate, tt.burst, nil)\n\t\t\tif tt.wantErr {\n\t\t\t\tif !errors.Is(err, ErrInvalidConfig) {\n\t\t\t\t\tt.Fatalf(\"New(%v, %v) error = %v, want ErrInvalidConfig\", tt.rate, tt.burst, err)\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif err != nil || l == nil {\n\t\t\t\tt.Fatalf(\"New(%v, %v) = %v, %v; want limiter, nil error\", tt.rate, tt.burst, l, err)\n\t\t\t}\n\t\t})\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration // clock shift applied before the call\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst float64\n\t\tsteps []step\n\t}{\n\t\t{\"drains then denies\", 1, 3, []step{{n: 3, want: true}, {n: 1, want: false}}},\n\t\t{\"refills over elapsed time\", 2, 2, []step{\n\t\t\t{n: 2, want: true}, {n: 1, want: false},\n\t\t\t{advance: 500 * time.Millisecond, n: 1, want: true}, {n: 1, want: false},\n\t\t}},\n\t\t{\"accumulates fractional refill\", 1, 1, []step{\n\t\t\t{n: 1, want: true},\n\t\t\t{advance: 250 * time.Millisecond, n: 1, want: false},\n\t\t\t{advance: 250 * time.Millisecond, n: 1, want: false},\n\t\t\t{advance: 500 * time.Millisecond, n: 1, want: true},\n\t\t}},\n\t\t{\"caps refill at burst\", 1, 2, []step{\n\t\t\t{n: 2, want: true}, {advance: time.Hour, n: 3, want: false}, {n: 2, want: true},\n\t\t}},\n\t\t{\"denies more than burst\", 1, 2, []step{{n: 3, want: false}, {n: 2, want: true}}},\n\t\t{\"denies non-positive n\", 1, 2, []step{{n: 0, want: false}, {n: -1, want: false}}},\n\t\t{\"ignores backwards timestamps\", 1, 2, []step{\n\t\t\t{n: 2, want: true},\n\t\t\t{advance: -2 * time.Second, n: 1, want: false},\n\t\t\t{advance: 3 * time.Second, n: 1, want: true}, {n: 1, want: false},\n\t\t}},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclock := newFakeClock(base)\n\t\t\tl, err := New(tt.rate, tt.burst, clock)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tif s.advance != 0 {\n\t\t\t\t\tclock.Advance(s.advance)\n\t\t\t\t}\n\t\t\t\tif got := l.AllowN(clock.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d %+v: AllowN(%d) = %v, want %v\", i, s, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tclock := newFakeClock(base)\n\tl, err := New(1, 1, clock)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"Allow() = true, want false\")\n\t}\n\tclock.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tclock := newFakeClock(base)\n\tl, err := New(100, 50, clock)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tvar allowed int64\n\tdone := make(chan struct{})\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\tgo func() { // refill 100ms worth of tokens while callers run\n\t\tdefer wg.Done()\n\t\tfor i := 0; i < 100; i++ {\n\t\t\tclock.Advance(time.Millisecond)\n\t\t}\n\t\tclose(done)\n\t}()\n\tfor i := 0; i < 8; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tatomic.AddInt64(&allowed, 1)\n\t\t\t\t}\n\t\t\t\tselect {\n\t\t\t\tcase <-done:\n\t\t\t\t\treturn\n\t\t\t\tdefault:\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := atomic.LoadInt64(&allowed); got < 1 || got > 60 {\n\t\tt.Errorf(\"allowed = %d, want in [1, 60] (burst 50 + 100ms at 100/s)\", got)\n\t}\n}\n```",
          "output_characters": 6118,
          "output_sha256": "d16e5160c8666bbdcd23a85bbebbaeb177c44c73fbf1ca15b5cc85bf5bfb86c0",
          "prompt_tokens": 130,
          "reasoning_characters": 32752,
          "reasoning_sha256": "5fcfb05982a545d8befcb88559ecac4a420c73d79f1e72a3cd79879314100280",
          "response_tail_seconds": 0.007468,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "a25afc687eedff019bcc0c88ab604efb592ed6414ce40800883386484af3a111",
          "time_to_first_visible_seconds": 159.179238,
          "time_to_last_output_seconds": 175.839716,
          "ttft_seconds": 0.392003,
          "wall_seconds": 175.847185
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 14005,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 184.83826,
          "decode_tokens_per_second": 75.764,
          "finish_reason": "stop",
          "measured_sse_events": 13995,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time so tests can inject a fake.\ntype Clock interface{ Now() time.Time }\n\n// SystemClock is the production Clock backed by time.Now.\ntype SystemClock struct{}\n\nfunc (SystemClock) Now() time.Time { return time.Now() }\n\nvar (\n\tErrNilClock = errors.New(\"ratelimit: nil clock\")\n\tErrCapacity = errors.New(\"ratelimit: capacity must be finite and > 0\")\n\tErrRate     = errors.New(\"ratelimit: rate must be finite and > 0\")\n)\n\n// Limiter is a concurrency-safe token-bucket limiter. Tokens accrue\n// continuously at rate per second up to capacity; it starts full.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New validates its arguments and returns a full Limiter.\nfunc New(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tswitch {\n\tcase clock == nil:\n\t\treturn nil, ErrNilClock\n\tcase !valid(capacity):\n\t\treturn nil, ErrCapacity\n\tcase !valid(rate):\n\t\treturn nil, ErrRate\n\t}\n\treturn &Limiter{clock: clock, capacity: capacity, rate: rate, tokens: capacity, last: clock.Now()}, nil\n}\n\nfunc valid(v float64) bool { return v > 0 && !math.IsNaN(v) && !math.IsInf(v, 0) }\n\n// Allow consumes one token at the clock's current time.\nfunc (l *Limiter) Allow() bool { return l.AllowN(l.clock.Now(), 1) }\n\n// AllowN reports whether n tokens may be consumed at time now. Time must not\n// move backwards between calls; a backwards read neither refills nor panics.\n// n <= 0 consumes nothing and returns true.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n <= 0 {\n\t\treturn true\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now.After(l.last) {\n\t\tl.tokens = min(l.capacity, l.tokens+now.Sub(l.last).Seconds()*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manual Clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, time.UTC)\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tsteps    []step\n\t}{\n\t\t{\"burst up to capacity\", 3, 1, []step{\n\t\t\t{0, 3, true},\n\t\t\t{0, 1, false},\n\t\t}},\n\t\t{\"refill proportional to elapsed time\", 5, 10, []step{\n\t\t\t{0, 5, true},\n\t\t\t{0, 1, false},\n\t\t\t{500 * time.Millisecond, 6, false}, // only 5 accrued\n\t\t\t{500 * time.Millisecond, 5, true},\n\t\t\t{0, 1, false},\n\t\t}},\n\t\t{\"denials neither consume nor reset accrual\", 2, 2, []step{\n\t\t\t{0, 2, true},\n\t\t\t{0, 1, false},\n\t\t\t{0, 1, false},\n\t\t\t{time.Second, 2, true},\n\t\t}},\n\t\t{\"accrual is capped at capacity\", 2, 100, []step{\n\t\t\t{0, 2, true},\n\t\t\t{time.Hour, 3, false},\n\t\t\t{0, 2, true},\n\t\t}},\n\t\t{\"backwards clock neither refills nor panics\", 1, 1, []step{\n\t\t\t{0, 1, true},\n\t\t\t{-time.Second, 1, false},\n\t\t}},\n\t\t{\"non-positive n is a free no-op\", 1, 1, []step{\n\t\t\t{0, 1, true},\n\t\t\t{0, 0, true},\n\t\t\t{0, -3, true},\n\t\t\t{0, 1, false},\n\t\t}},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := New(tt.capacity, tt.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(clk.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(now, %d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tclock    Clock\n\t\twantErr  error\n\t}{\n\t\t{\"nil clock\", 1, 1, nil, ErrNilClock},\n\t\t{\"zero capacity\", 0, 1, SystemClock{}, ErrCapacity},\n\t\t{\"negative capacity\", -1, 1, SystemClock{}, ErrCapacity},\n\t\t{\"NaN capacity\", math.NaN(), 1, SystemClock{}, ErrCapacity},\n\t\t{\"+Inf capacity\", math.Inf(1), 1, SystemClock{}, ErrCapacity},\n\t\t{\"zero rate\", 1, 0, SystemClock{}, ErrRate},\n\t\t{\"negative rate\", 1, -2, SystemClock{}, ErrRate},\n\t\t{\"NaN rate\", 1, math.NaN(), SystemClock{}, ErrRate},\n\t\t{\"-Inf rate\", 1, math.Inf(-1), SystemClock{}, ErrRate},\n\t\t{\"valid arguments\", 1, 1, SystemClock{}, nil},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.capacity, tt.rate, tt.clock)\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"New() error = %v, want %v\", err, tt.wantErr)\n\t\t\t}\n\t\t\tif err == nil && l == nil {\n\t\t\t\tt.Fatal(\"New() returned nil limiter without error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tclk := newFakeClock(base)\n\tl, err := New(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true for full bucket\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"Allow() = true, want false for empty bucket\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true after refill\")\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tconst workers, perWorker, capacity = 50, 20, 100\n\tl, err := New(capacity, 1e-9, newFakeClock(base)) // refill negligible\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tvar granted int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < workers; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < perWorker; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tatomic.AddInt64(&granted, 1)\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := atomic.LoadInt64(&granted); got != capacity {\n\t\tt.Errorf(\"granted = %d, want exactly %d\", got, capacity)\n\t}\n}\n```",
          "output_characters": 5883,
          "output_sha256": "3c6a2617b44317f00c7e0748e74aef0dafcaec63af6450fb6ae94ea64044f035",
          "prompt_tokens": 131,
          "reasoning_characters": 34500,
          "reasoning_sha256": "d042f5009b74e806192d1be73cc7ea7e7585377f40916f0b68f216de52f736d7",
          "response_tail_seconds": 0.005172,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "0a84c8ae22f00d33e3c1a8a48d4d54f3920adf36822ead715fb6acc9787b6b7f",
          "time_to_first_visible_seconds": 170.129777,
          "time_to_last_output_seconds": 185.238496,
          "ttft_seconds": 0.400236,
          "wall_seconds": 185.243668
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 185.238,
        "median": 175.84,
        "minimum": 142.197,
        "p90": 185.238
      },
      "ttft_seconds": {
        "maximum": 0.407,
        "median": 0.4,
        "minimum": 0.392,
        "p90": 0.407
      },
      "wall_seconds": {
        "maximum": 185.244,
        "median": 175.847,
        "minimum": 142.201,
        "p90": 185.244
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 49.672,
        "median": 48.65,
        "minimum": 46.849,
        "p90": 49.672
      },
      "pooled_decode_tokens_per_second": 48.573,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2078,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 44.333921,
          "decode_tokens_per_second": 46.849,
          "finish_reason": "stop",
          "measured_sse_events": 2076,
          "output": "**Internal Memo**\n\n**To:** Platform Engineering\n**From:** R. Okafor, Infrastructure Lead\n**Date:** 14 March\n**Subject:** Capacity decision \u2014 buying more GPUs versus using the ones we have\n\nOver the past two months our inference queue times have crept from seconds into minutes during peak hours, and we now face a fork we've been circling for a while: either purchase additional hardware or get more out of the eight GPUs we already own. I want to lay out how I'm thinking about the trade-off and propose a path, with the caveat that I'm working from imperfect data and I expect reasonable people to disagree with parts of this.\n\nBuying hardware is the obvious fix and it has real appeal. It restores headroom immediately, it insulates our latency from demand spikes, and it gives us redundancy we currently lack \u2014 right now a single failed node takes out a quarter of our serving capacity. But the cost is heavier than the sticker price. Beyond the capital outlay, we'd take on power, cooling, and rack commitments, and the hardware depreciates quickly in a market where next year's cards make this year's look slow. Procurement lead times also mean the purchase is not truly immediate anyway. Most importantly, it is the least reversible option on the table. Once we've spent the money and racked the machines, that decision is largely locked in.\n\nImproving utilization sounds cheaper, and mostly is, but we need to be honest about what it means. It does not mean running the fleet hotter. Queueing latency is brutally nonlinear: pushing a system from seventy to ninety percent utilization can multiply tail latencies by an order of magnitude, and our p99s are already the thing users complain about. What it means is eliminating waste \u2014 models held in memory while idle, batch sizes tuned for a workload we no longer have, jobs fragmented across devices when they could pack cleanly, and the significant fraction of capacity our dashboards suggest sits unused overnight. That reclaimed capacity is real, and it's the cheapest capacity we will ever get.\n\nReliability cuts both ways here, and I want to flag it because it's the argument I'm least confident about. Aggressive consolidation erodes the headroom we need for failover and maintenance; a fleet at ninety-five percent utilization has no room to absorb a dead node. On the other hand, some utilization work \u2014 better draining, cleaner scheduling, proper health-aware placement \u2014 would itself improve reliability, since our current ad hoc job assignment is part of why node failures are so disruptive. Any efficiency push has to preserve a deliberate reliability margin, not treat idle capacity as pure waste.\n\nThen there is opportunity cost, which is the quiet factor in both directions. Hardware spend is capital we can't put toward hiring or experiments, and it arrives with a depreciation clock ticking. But engineering time is also capital: the weeks we spend profiling and re-plumbing our serving stack are weeks not spent on the product work that actually justifies our existence. The honest framing is that both options consume something scarce, and the question is which resource we can better afford to spend and how easily we can get it back. On reversibility, the asymmetry is decisive. Software changes can be rolled back in an afternoon; a hardware purchase cannot be un-bought. When two options are close, the reversible one deserves the first shot.\n\nSo my recommendation is this: we spend the next six weeks on a timeboxed utilization effort before any procurement. Step one is measurement \u2014 I don't fully trust our current picture of idle capacity, and we should not optimize against guesses. Then we fix batching, reclaim idle allocations, and improve scheduling, with a hard guardrail that p99 latency must not degrade at current traffic. We define purchase triggers now, while we're thinking clearly: if sustained demand exceeds what we've reclaimed, or if we hit our latency guardrail at a healthy utilization level, or if the engineering cost of further optimization clearly exceeds the hardware cost, we buy, and we'll have earned a much sharper case for the budget.\n\nI want to be candid about the uncertainty. Our demand forecast is shaky, our utilization data is noisy, and it's possible we're already near the efficient frontier and the six weeks will tell us that \u2014 which is itself a valuable answer. If the work balloons past the timebox, we stop and revisit. Push back on any of this by Friday; I'd rather argue now than in a procurement meeting.",
          "output_characters": 4533,
          "output_sha256": "f4938db61741ef33931047a286cff8e08d4e2edac74599ed9709a664204b20bb",
          "prompt_tokens": 124,
          "reasoning_characters": 5818,
          "reasoning_sha256": "556698a2d4a111b58b697394788e1d6f67df053ea721269b47a4126050af4b53",
          "response_tail_seconds": 0.002208,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "43d1212c9ef8f699d48ba8986b7b6e0344c54355d87d38c8d33ea726adf5fda4",
          "time_to_first_visible_seconds": 25.178106,
          "time_to_last_output_seconds": 44.81579,
          "ttft_seconds": 0.481869,
          "wall_seconds": 44.817999
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2925,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 60.103305,
          "decode_tokens_per_second": 48.65,
          "finish_reason": "stop",
          "measured_sse_events": 2921,
          "output": "**INTERNAL MEMO**\n\n**To:** Platform Engineering\n**From:** Infra lead\n**Re:** Inference capacity \u2014 buy more GPUs, or use the ones we have better\n\nOver the last two quarters, request volume to our local inference fleet has roughly tripled, driven by the assistant rollout and the new batch embedding jobs. Our four GPUs now sit above 85% utilisation during business hours, p95 latency has drifted from under a second to over four, and at least two teams have started quietly routing around us to a hosted API \u2014 which costs real money every month and rather defeats the point of owning hardware. We owe ourselves a decision: expand the fleet, or get more out of what we already have.\n\nThe case for buying is straightforward, and I don't want to undersell it. Two additional cards would drop into our existing chassis and power budget, latency would improve the day they're racked, and the engineering effort is close to zero. But the purchase price is only the visible cost. There is power and cooling, spares, depreciation, and the awkward fact that accelerator pricing and availability are volatile enough that procurement is a gamble in itself. More subtly, buying now locks in our current workload shape. If demand shifts toward a different model size or a new modality, we may find ourselves owning exactly the wrong hardware, expensively.\n\nThe case for better utilisation rests on what our telemetry already shows. Dev and experiment jobs hold allocations over nights and weekends. The gateway doesn't batch, so we pay full forward-pass cost for every request, even trivially small ones. A large fraction of traffic is short prompts that a quantised small model could serve at a fraction of the cost. Embeddings and generation share the same cards with no priority scheduling, so a bulk job can starve interactive traffic. Queueing behaviour matters here: latency stays flat until utilisation climbs past roughly 80%, then degrades nonlinearly \u2014 which is precisely where we're operating. Shaping demand and batching properly could plausibly cut effective load by a quarter without anyone noticing a difference in quality.\n\nThere is a real tension with reliability, though, and it cuts against the naive version of the optimisation story. Pushing utilisation higher erodes headroom. At 90%-plus, a single stuck job or a driver update reboot becomes an outage rather than a slowdown, and we currently have no redundancy: if one card dies, we lose a quarter of capacity instantly. So \"improve utilisation\" cannot mean \"run hotter.\" It has to mean eliminating waste so the same demand is met at a lower, safer utilisation band \u2014 which also restores our ability to take a card offline for maintenance without paging everyone.\n\nThen there is opportunity cost, which is the currency we actually spend. The utilisation work is not free; realistically it's four to six engineer-weeks across batching, scheduling, and model routing, and that is time not spent on the retrieval overhaul the product team is waiting for. But that spend is largely reversible and informative either way: the scheduler work retains value even if we buy hardware later. Hardware is the opposite. It is a multi-year commitment, difficult to resell at anything near cost, and it consumes rack power and budget we would then not have for anything else. Software changes can be undone; a purchase order mostly cannot.\n\nMy recommendation is a timeboxed utilisation effort \u2014 six weeks, with explicit targets: p95 under 1.5 seconds at current volume, business-hours utilisation held in the 60\u201375% band, and batch queue waits under ten minutes. Critically, we should pre-approve the hardware purchase now, to be executed automatically if we miss those targets at the end of the window, or if volume grows more than about 30% before then. Framed this way, we are buying an option rather than an asset: a small, reversible outlay that teaches us whether the demand is durable and whether the waste is real, while keeping the escape hatch open.\n\nI want to be honest about the uncertainty. Our demand forecast is shaky; if the assistant launch doubles volume again, no amount of batching saves us and we should pull the trigger early. Conversely, the gains might disappoint \u2014 batching improves throughput more than tail latency, and if p95 is dominated by long generations, no scheduler will fix that. There is also a real risk the six weeks slips under product pressure. But I would rather discover that in six weeks, with a purchase order already signed and waiting, than own hardware we didn't need.\n\nHappy to debate this at Thursday's sync; the utilisation dashboard is linked from the team wiki for anyone who wants to check my numbers.",
          "output_characters": 4700,
          "output_sha256": "77513bb24addb30e27b7e26a4f1b1cffbbcbdf3e1e3fc348b603f73c09436efc",
          "prompt_tokens": 120,
          "reasoning_characters": 9047,
          "reasoning_sha256": "a3bb226eeea2da3477c1ae2301de6f987c0bdb64244f68129baf30981bf36da9",
          "response_tail_seconds": 0.002247,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "98f0973cc40577711d65b1d96ae46a49f12ff357473263500e346778bd0d112b",
          "time_to_first_visible_seconds": 44.641748,
          "time_to_last_output_seconds": 60.582243,
          "ttft_seconds": 0.478938,
          "wall_seconds": 60.58449
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 3248,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 65.36824,
          "decode_tokens_per_second": 49.672,
          "finish_reason": "stop",
          "measured_sse_events": 3245,
          "output": "**MEMO**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Date:** 12 March\n**Subject:** GPU capacity \u2014 buy more, or use what we own better?\n\nOver the past month the question of whether we buy more GPUs has stopped being hypothetical. Training jobs have queued behind inference twice, the p95 latency on our serving path has drifted well past target, and we have a live quote in hand for a new node. This memo lays out how I see the tradeoffs and makes a recommendation, with the caveat that several of the numbers below carry real uncertainty.\n\nStart with what the fleet actually looks like. We own eight GPUs across two nodes, and our telemetry says average utilization last quarter was 38 percent, which on its face suggests we own twice the hardware we need. Averages mislead, though. The older A100 node is effectively pinned during business hours, the newer node idles overnight, and some of what looks like idle capacity is deliberate headroom for bursty inference and failover. The honest reading is that we are genuinely constrained perhaps twenty to twenty-five hours a week, not around the clock, and the pain is concentrated in predictable windows.\n\nNow cost. The quote is $140k for a 4-way H100 node, plus roughly $1,200 a month in power and cooling, plus the depreciation hit that comes with any accelerator purchase \u2014 this class of hardware loses resale value quickly, so capital committed here is close to irrecoverable. Against that, the utilization option is mostly engineer time: two of us for about six weeks, call it $60k fully loaded, plus the opportunity cost of whatever feature work we would not ship in that window. On pure dollars the software path is cheaper, but the gap is not so wide that it decides the question alone.\n\nLatency is where the decision gets interesting. A meaningful share of our latency is not compute-bound; it comes from queueing, cold model loads, and naive batching. Continuous batching, quantizing the latency-critical model, keeping warm replicas, and smarter placement could plausibly bring p95 from around 900ms down toward our 400ms target without new silicon. But software has a floor. When every GPU is genuinely busy, no scheduler conjures capacity, and I want us to be honest about which regime we are in before assuming an engineering push can substitute for hardware.\n\nReliability cuts both ways. More hardware would improve our redundancy \u2014 today a single node failure removes half the fleet, and our failover story is thin. Conversely, squeezing utilization removes slack, and slack is precisely what absorbs failures: hotter-running machines, tighter failover margins, more brittleness under incident conditions. One of our nodes is three years old, and its failure probability is not negligible. Any utilization effort should preserve explicit headroom, not eliminate it in pursuit of a better dashboard number.\n\nThat leaves opportunity cost and reversibility, and here the asymmetry is stark. Buying hardware is a one-way door: capital committed, rack space and a power contract locked in, a depreciating asset on the books. The utilization work is a two-way door. If it fails, we have spent engineer-months, but we can still buy \u2014 and we will do so knowing exactly what we need rather than guessing. Engineer time is a real cost, so the effort needs a hard boundary, but the reversibility argument strongly favors trying the cheap, undoable thing first.\n\nMy recommendation: do not buy hardware now. Instead, run a focused six-week effort with three strands \u2014 continuous batching and quantization on the inference path, a scheduler change so batch training yields to interactive inference, and better telemetry so we measure truly available capacity rather than flattering averages. Set explicit tripwires: if p95 inference latency exceeds 750ms for five consecutive business days, or median training queue wait exceeds three hours, or forecast demand puts sustained utilization above 70 percent, we revisit the purchase with a pre-approved budget and buy without relitigating the whole question. We re-review in one quarter regardless of whether the tripwires trip.\n\nI want to be clear about the uncertainty in this position. Our utilization numbers likely overstate available capacity, our demand forecast is shaky, quantization may not clear our accuracy bar, and accelerator pricing shifts unpredictably. This is a deferral, not a denial \u2014 if the thresholds break, we buy. Please send comments or objections by Friday so we can start the following Monday.",
          "output_characters": 4538,
          "output_sha256": "0bd338611ac4ad1cd67cab57754cb8fde1ccc796d95c4f91120f855456bf187c",
          "prompt_tokens": 124,
          "reasoning_characters": 10554,
          "reasoning_sha256": "753b4519c68286ee07457a8b0d9e3491e446dc4244cca121c35c46b78ab9d9a4",
          "response_tail_seconds": 0.002533,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "14829b024a014bfccb15d998dc65f80d91806cd782a02138f1bd00bcf66783f2",
          "time_to_first_visible_seconds": 50.225189,
          "time_to_last_output_seconds": 65.847462,
          "ttft_seconds": 0.479222,
          "wall_seconds": 65.849995
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 65.847,
        "median": 60.582,
        "minimum": 44.816,
        "p90": 65.847
      },
      "ttft_seconds": {
        "maximum": 0.482,
        "median": 0.479,
        "minimum": 0.479,
        "p90": 0.482
      },
      "wall_seconds": {
        "maximum": 65.85,
        "median": 60.584,
        "minimum": 44.818,
        "p90": 65.85
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 102.47,
        "median": 102.447,
        "minimum": 101.877,
        "p90": 102.47
      },
      "pooled_decode_tokens_per_second": 102.27,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1026,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 10.002951,
          "decode_tokens_per_second": 102.47,
          "finish_reason": "stop",
          "measured_sse_events": 1021,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 111,
          "reasoning_characters": 624,
          "reasoning_sha256": "ef5c0467c52425683215717de976eb3af7647a5359188ad7139e8d2f79ea9a62",
          "response_tail_seconds": 0.002414,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "d60c15c090800fcd2be1124bf494d4a5bda60fc4304934d04fca946dcc848488",
          "time_to_first_visible_seconds": 4.476464,
          "time_to_last_output_seconds": 10.47039,
          "ttft_seconds": 0.467439,
          "wall_seconds": 10.472804
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 9.580189,
          "decode_tokens_per_second": 101.877,
          "finish_reason": "stop",
          "measured_sse_events": 972,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.002186,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 3.994323,
          "time_to_last_output_seconds": 10.046421,
          "ttft_seconds": 0.466231,
          "wall_seconds": 10.048607
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1027,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 10.014929,
          "decode_tokens_per_second": 102.447,
          "finish_reason": "stop",
          "measured_sse_events": 1022,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 623,
          "reasoning_sha256": "efe779badef235a76affb2148a0ffee3f78f250c1e369fe6b24271bd1cfee954",
          "response_tail_seconds": 0.000517,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "ca3e49b98781e7b79632b4947f225dd278dfac8fa62a99c57f9bc3a6051ffcf8",
          "time_to_first_visible_seconds": 4.511229,
          "time_to_last_output_seconds": 10.480168,
          "ttft_seconds": 0.465238,
          "wall_seconds": 10.480685
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 10.48,
        "median": 10.47,
        "minimum": 10.046,
        "p90": 10.48
      },
      "ttft_seconds": {
        "maximum": 0.467,
        "median": 0.466,
        "minimum": 0.465,
        "p90": 0.467
      },
      "wall_seconds": {
        "maximum": 10.481,
        "median": 10.473,
        "minimum": 10.049,
        "p90": 10.481
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2027.859,
          "median": 2026.525,
          "minimum": 2025.136,
          "p90": 2027.859
        },
        "runs": [
          {
            "cache_salt_sha256": "615ecffb34aafcc4460ecfb6a3c88edbb7c788bf601b4bb7d1468939b9c14e6d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062293,
            "decode_tokens_per_second": 112.372,
            "effective_prefill_tokens_per_second": 2026.525,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001636,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 16.169554,
            "time_to_last_output_seconds": 16.231847,
            "ttft_seconds": 16.169554,
            "wall_seconds": 16.233483
          },
          {
            "cache_salt_sha256": "6d43662f63cc54b84705a4a345b5410e93bfd636261c599ffbe4c4c7f2ce79ea",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054109,
            "decode_tokens_per_second": 129.367,
            "effective_prefill_tokens_per_second": 2027.859,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001324,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 16.158911,
            "time_to_last_output_seconds": 16.21302,
            "ttft_seconds": 16.158911,
            "wall_seconds": 16.214344
          },
          {
            "cache_salt_sha256": "a17f22ec39163886d609d52d3f4e778c7a64450c4a00307ff4727599337fffa1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062459,
            "decode_tokens_per_second": 112.074,
            "effective_prefill_tokens_per_second": 2025.136,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001017,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.180639,
            "time_to_last_output_seconds": 16.243097,
            "ttft_seconds": 16.180639,
            "wall_seconds": 16.244114
          }
        ],
        "ttft_seconds": {
          "maximum": 16.181,
          "median": 16.17,
          "minimum": 16.159,
          "p90": 16.181
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 764500.023,
          "median": 591747.178,
          "minimum": 570473.538,
          "p90": 764500.023
        },
        "runs": [
          {
            "cache_salt_sha256": "615ecffb34aafcc4460ecfb6a3c88edbb7c788bf601b4bb7d1468939b9c14e6d",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.062467,
            "decode_tokens_per_second": 112.059,
            "effective_prefill_tokens_per_second": 764500.023,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001797,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.042862,
            "time_to_last_output_seconds": 0.105329,
            "ttft_seconds": 0.042862,
            "wall_seconds": 0.107126
          },
          {
            "cache_salt_sha256": "6d43662f63cc54b84705a4a345b5410e93bfd636261c599ffbe4c4c7f2ce79ea",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.05414,
            "decode_tokens_per_second": 129.294,
            "effective_prefill_tokens_per_second": 570473.538,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001156,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 0.05744,
            "time_to_last_output_seconds": 0.11158,
            "ttft_seconds": 0.05744,
            "wall_seconds": 0.112736
          },
          {
            "cache_salt_sha256": "a17f22ec39163886d609d52d3f4e778c7a64450c4a00307ff4727599337fffa1",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060601,
            "decode_tokens_per_second": 115.51,
            "effective_prefill_tokens_per_second": 591747.178,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001009,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.055375,
            "time_to_last_output_seconds": 0.115976,
            "ttft_seconds": 0.055375,
            "wall_seconds": 0.116986
          }
        ],
        "ttft_seconds": {
          "maximum": 0.057,
          "median": 0.055,
          "minimum": 0.043,
          "p90": 0.057
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2003.547,
          "median": 2003.084,
          "minimum": 2001.55,
          "p90": 2003.547
        },
        "runs": [
          {
            "cache_salt_sha256": "76f3370e0ff6d1e6370dce807e1fb3e02cfe25518d068da6624f14a6ce336598",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064624,
            "decode_tokens_per_second": 108.318,
            "effective_prefill_tokens_per_second": 2003.084,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001898,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 32.717549,
            "time_to_last_output_seconds": 32.782173,
            "ttft_seconds": 32.717549,
            "wall_seconds": 32.784071
          },
          {
            "cache_salt_sha256": "7c50f37d9e60dde83663eac5125de76125632dc186a298e8ebc7aa24c6f952ba",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061808,
            "decode_tokens_per_second": 113.255,
            "effective_prefill_tokens_per_second": 2003.547,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001722,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 32.709984,
            "time_to_last_output_seconds": 32.771791,
            "ttft_seconds": 32.709984,
            "wall_seconds": 32.773513
          },
          {
            "cache_salt_sha256": "bf18a40a605e9c93596649598e4b2ddec95ad97c824d39b3433741b0cd45b6cd",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062258,
            "decode_tokens_per_second": 112.435,
            "effective_prefill_tokens_per_second": 2001.55,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ", and the next action. The",
            "output_characters": 26,
            "output_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001795,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "time_to_first_visible_seconds": 32.742625,
            "time_to_last_output_seconds": 32.804883,
            "ttft_seconds": 32.742625,
            "wall_seconds": 32.806678
          }
        ],
        "ttft_seconds": {
          "maximum": 32.743,
          "median": 32.718,
          "minimum": 32.71,
          "p90": 32.743
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1135333.657,
          "median": 1064483.644,
          "minimum": 940838.681,
          "p90": 1135333.657
        },
        "runs": [
          {
            "cache_salt_sha256": "76f3370e0ff6d1e6370dce807e1fb3e02cfe25518d068da6624f14a6ce336598",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064831,
            "decode_tokens_per_second": 107.973,
            "effective_prefill_tokens_per_second": 1064483.644,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001533,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.061566,
            "time_to_last_output_seconds": 0.126397,
            "ttft_seconds": 0.061566,
            "wall_seconds": 0.12793
          },
          {
            "cache_salt_sha256": "7c50f37d9e60dde83663eac5125de76125632dc186a298e8ebc7aa24c6f952ba",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.061676,
            "decode_tokens_per_second": 113.496,
            "effective_prefill_tokens_per_second": 1135333.657,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002576,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.057724,
            "time_to_last_output_seconds": 0.119401,
            "ttft_seconds": 0.057724,
            "wall_seconds": 0.121977
          },
          {
            "cache_salt_sha256": "bf18a40a605e9c93596649598e4b2ddec95ad97c824d39b3433741b0cd45b6cd",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062516,
            "decode_tokens_per_second": 111.97,
            "effective_prefill_tokens_per_second": 940838.681,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ", and the next action. The",
            "output_characters": 26,
            "output_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002639,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "time_to_first_visible_seconds": 0.069657,
            "time_to_last_output_seconds": 0.132173,
            "ttft_seconds": 0.069657,
            "wall_seconds": 0.134813
          }
        ],
        "ttft_seconds": {
          "maximum": 0.07,
          "median": 0.062,
          "minimum": 0.058,
          "p90": 0.07
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1999.868,
          "median": 1995.905,
          "minimum": 1995.221,
          "p90": 1999.868
        },
        "runs": [
          {
            "cache_salt_sha256": "f1c44da8502b23d82dbea1d56de582189212122c9bac1ed25325e9e7944a8882",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055889,
            "decode_tokens_per_second": 125.249,
            "effective_prefill_tokens_per_second": 1995.905,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001372,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 4.104404,
            "time_to_last_output_seconds": 4.160293,
            "ttft_seconds": 4.104404,
            "wall_seconds": 4.161665
          },
          {
            "cache_salt_sha256": "13eb1bf70160d835dfd2d2e6de11ae98d8d50faaa6f6a3310d22c2b3e468e986",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055262,
            "decode_tokens_per_second": 126.67,
            "effective_prefill_tokens_per_second": 1995.221,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". Each observation records behaviour, evidence,",
            "output_characters": 47,
            "output_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001583,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "time_to_first_visible_seconds": 4.105811,
            "time_to_last_output_seconds": 4.161073,
            "ttft_seconds": 4.105811,
            "wall_seconds": 4.162656
          },
          {
            "cache_salt_sha256": "94d5d04b98f5626fd7904df0ecd1c90ad23c62669e4123125193fede9625da03",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054573,
            "decode_tokens_per_second": 128.268,
            "effective_prefill_tokens_per_second": 1999.868,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001461,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 4.096271,
            "time_to_last_output_seconds": 4.150844,
            "ttft_seconds": 4.096271,
            "wall_seconds": 4.152306
          }
        ],
        "ttft_seconds": {
          "maximum": 4.106,
          "median": 4.104,
          "minimum": 4.096,
          "p90": 4.106
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 239406.161,
          "median": 238181.078,
          "minimum": 178455.506,
          "p90": 239406.161
        },
        "runs": [
          {
            "cache_salt_sha256": "f1c44da8502b23d82dbea1d56de582189212122c9bac1ed25325e9e7944a8882",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.056307,
            "decode_tokens_per_second": 124.318,
            "effective_prefill_tokens_per_second": 178455.506,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001356,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 0.045905,
            "time_to_last_output_seconds": 0.102212,
            "ttft_seconds": 0.045905,
            "wall_seconds": 0.103568
          },
          {
            "cache_salt_sha256": "13eb1bf70160d835dfd2d2e6de11ae98d8d50faaa6f6a3310d22c2b3e468e986",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.055936,
            "decode_tokens_per_second": 125.143,
            "effective_prefill_tokens_per_second": 239406.161,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". Each observation records behaviour, evidence,",
            "output_characters": 47,
            "output_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001478,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "time_to_first_visible_seconds": 0.034218,
            "time_to_last_output_seconds": 0.090154,
            "ttft_seconds": 0.034218,
            "wall_seconds": 0.091632
          },
          {
            "cache_salt_sha256": "94d5d04b98f5626fd7904df0ecd1c90ad23c62669e4123125193fede9625da03",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.05372,
            "decode_tokens_per_second": 130.304,
            "effective_prefill_tokens_per_second": 238181.078,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001458,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 0.034394,
            "time_to_last_output_seconds": 0.088114,
            "ttft_seconds": 0.034394,
            "wall_seconds": 0.089572
          }
        ],
        "ttft_seconds": {
          "maximum": 0.046,
          "median": 0.034,
          "minimum": 0.034,
          "p90": 0.046
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "incoai/GLM-5.3-Flash-DFlash2 @ bf582e4e (Mia v1.10 default settings)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3",
      "model_revision": "13ba3ef94d88bc30b27691efa29a6012f66e964d (AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3 on Hugging Face; the same files, served from a local copy)",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-dflash2-max",
    "finished_at": "2026-10-08T13:11:11.054415+00:00",
    "label": "tp2-ours-dflash2-max",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T12:55:46.307138+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 32768,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "max"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 3,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}