{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 49.457,
        "median": 48.224,
        "minimum": 46.799,
        "p90": 49.457
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 53.776,
        "median": 52.353,
        "minimum": 51.64,
        "p90": 53.776
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.521,
        "median": 0.437,
        "minimum": 0.434,
        "p90": 0.521
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 48.224,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.870819,
              "decode_tokens_per_second": 52.353,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1033,
              "reasoning_sha256": "547f61b49864e4c22cf5763bd8c880e96f2765fec623cf9641becb6bac0d4653",
              "response_tail_seconds": 0.000173,
              "stream_done_marker": true,
              "stream_sha256": "547f61b49864e4c22cf5763bd8c880e96f2765fec623cf9641becb6bac0d4653",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.307437,
              "ttft_seconds": 0.436618,
              "wall_seconds": 5.30761
            }
          ],
          "wall_seconds": 5.308577
        },
        {
          "aggregate_end_to_end_tokens_per_second": 46.799,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.938072,
              "decode_tokens_per_second": 51.64,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 988,
              "reasoning_sha256": "bcf14da209d7c0f78045b3c9a903008662aba01dc07bd12c65e987d74ddbce20",
              "response_tail_seconds": 0.000138,
              "stream_done_marker": true,
              "stream_sha256": "bcf14da209d7c0f78045b3c9a903008662aba01dc07bd12c65e987d74ddbce20",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.459419,
              "ttft_seconds": 0.521346,
              "wall_seconds": 5.459557
            }
          ],
          "wall_seconds": 5.470245
        },
        {
          "aggregate_end_to_end_tokens_per_second": 49.457,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.741855,
              "decode_tokens_per_second": 53.776,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 996,
              "reasoning_sha256": "bf5e3614332122f591be360447793a4e2af3ebb86d3a9b24a3d60a8c77a8cc46",
              "response_tail_seconds": 5.7e-05,
              "stream_done_marker": true,
              "stream_sha256": "bf5e3614332122f591be360447793a4e2af3ebb86d3a9b24a3d60a8c77a8cc46",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.1756,
              "ttft_seconds": 0.433745,
              "wall_seconds": 5.175657
            }
          ],
          "wall_seconds": 5.176226
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 54.549,
        "median": 52.276,
        "minimum": 50.324,
        "p90": 54.549
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 62.087,
        "median": 58.288,
        "minimum": 48.791,
        "p90": 60.519
      },
      "per_stream_ttft_seconds": {
        "maximum": 5.39,
        "median": 2.683,
        "minimum": 0.417,
        "p90": 5.171
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 50.324,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.107144,
              "decode_tokens_per_second": 62.087,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1034,
              "reasoning_sha256": "b936eed2752f1cac583de75fa1996fe2481cc2908d43b03cfde8e2bb879fe1f0",
              "response_tail_seconds": 0.000121,
              "stream_done_marker": true,
              "stream_sha256": "b936eed2752f1cac583de75fa1996fe2481cc2908d43b03cfde8e2bb879fe1f0",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.526487,
              "ttft_seconds": 0.419342,
              "wall_seconds": 4.526608
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.226374,
              "decode_tokens_per_second": 48.791,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 959,
              "reasoning_sha256": "5e78e172faa891de3aec72bcdb8476814cc426e29acb9d89def5bee6e1fb4ac7",
              "response_tail_seconds": 0.000128,
              "stream_done_marker": true,
              "stream_sha256": "5e78e172faa891de3aec72bcdb8476814cc426e29acb9d89def5bee6e1fb4ac7",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 10.172575,
              "ttft_seconds": 4.9462,
              "wall_seconds": 10.172702
            }
          ],
          "wall_seconds": 10.174074
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.549,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.346865,
              "decode_tokens_per_second": 58.663,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1060,
              "reasoning_sha256": "e5f5faa35dc4f0525d6e744cfb92e0859fcdc17f4891d37fcc19b5b0145530d7",
              "response_tail_seconds": 0.00028,
              "stream_done_marker": true,
              "stream_sha256": "e5f5faa35dc4f0525d6e744cfb92e0859fcdc17f4891d37fcc19b5b0145530d7",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.764025,
              "ttft_seconds": 0.41716,
              "wall_seconds": 4.764306
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.213567,
              "decode_tokens_per_second": 60.519,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 130,
              "reasoning_characters": 1003,
              "reasoning_sha256": "33ec2f570e3317794c8c93e165c58eb8ec5c2bdeeff20711dd81cca11ec25c89",
              "response_tail_seconds": 0.000164,
              "stream_done_marker": true,
              "stream_sha256": "33ec2f570e3317794c8c93e165c58eb8ec5c2bdeeff20711dd81cca11ec25c89",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.384651,
              "ttft_seconds": 5.171083,
              "wall_seconds": 9.384814
            }
          ],
          "wall_seconds": 9.386058
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.276,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.545954,
              "decode_tokens_per_second": 56.094,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1004,
              "reasoning_sha256": "a9a11c0655544e787fa8af44c977b691ee8466f81cf4d8721fdc32968e91a2f2",
              "response_tail_seconds": 0.000155,
              "stream_done_marker": true,
              "stream_sha256": "a9a11c0655544e787fa8af44c977b691ee8466f81cf4d8721fdc32968e91a2f2",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.966387,
              "ttft_seconds": 0.420433,
              "wall_seconds": 4.966542
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.403054,
              "decode_tokens_per_second": 57.914,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1055,
              "reasoning_sha256": "d1649556a1ff261a446f2bb8fd1942211af636d7e846387f4d8a212205bcd14f",
              "response_tail_seconds": 9e-05,
              "stream_done_marker": true,
              "stream_sha256": "d1649556a1ff261a446f2bb8fd1942211af636d7e846387f4d8a212205bcd14f",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.792928,
              "ttft_seconds": 5.389874,
              "wall_seconds": 9.793018
            }
          ],
          "wall_seconds": 9.79417
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 53.136,
        "median": 51.878,
        "minimum": 50.803,
        "p90": 53.136
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 60.183,
        "median": 57.578,
        "minimum": 47.849,
        "p90": 59.649
      },
      "per_stream_ttft_seconds": {
        "maximum": 15.864,
        "median": 7.686,
        "minimum": 0.43,
        "p90": 14.829
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 53.136,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.41666,
              "decode_tokens_per_second": 57.736,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 958,
              "reasoning_sha256": "ced515370cf5125efd6c4b5598ba042324f3b9a9add7d0b92e3079567d28b331",
              "response_tail_seconds": 0.000149,
              "stream_done_marker": true,
              "stream_sha256": "ced515370cf5125efd6c4b5598ba042324f3b9a9add7d0b92e3079567d28b331",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.846983,
              "ttft_seconds": 0.430323,
              "wall_seconds": 4.847131
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.47011,
              "decode_tokens_per_second": 57.046,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 1051,
              "reasoning_sha256": "c13a9ab656b82c8a7441a6133f6aaa75ad2f3bdd6f12b9bbb07e82cdc560a4c8",
              "response_tail_seconds": 0.000188,
              "stream_done_marker": true,
              "stream_sha256": "c13a9ab656b82c8a7441a6133f6aaa75ad2f3bdd6f12b9bbb07e82cdc560a4c8",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 19.267582,
              "ttft_seconds": 14.797472,
              "wall_seconds": 19.26777
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.335337,
              "decode_tokens_per_second": 58.819,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 135,
              "reasoning_characters": 936,
              "reasoning_sha256": "541ee91d6601cb2cec7325c90dc14e624cc9f0b4f21422cbad7caf062a8716fc",
              "response_tail_seconds": 0.000121,
              "stream_done_marker": true,
              "stream_sha256": "541ee91d6601cb2cec7325c90dc14e624cc9f0b4f21422cbad7caf062a8716fc",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.616221,
              "ttft_seconds": 5.280883,
              "wall_seconds": 9.616341
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.320202,
              "decode_tokens_per_second": 59.025,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1004,
              "reasoning_sha256": "a2891b7ff0993c37472b1aa7497cead607c75bc9c6acfc561403c643cdf1082b",
              "response_tail_seconds": 9.7e-05,
              "stream_done_marker": true,
              "stream_sha256": "a2891b7ff0993c37472b1aa7497cead607c75bc9c6acfc561403c643cdf1082b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 14.356232,
              "ttft_seconds": 10.03603,
              "wall_seconds": 14.356329
            }
          ],
          "wall_seconds": 19.271271
        },
        {
          "aggregate_end_to_end_tokens_per_second": 50.803,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.542096,
              "decode_tokens_per_second": 56.141,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1060,
              "reasoning_sha256": "b7884147f18ee68a2b23ca44534f74168bcaf434380a9454f8ed273631ce3bac",
              "response_tail_seconds": 0.000589,
              "stream_done_marker": true,
              "stream_sha256": "b7884147f18ee68a2b23ca44534f74168bcaf434380a9454f8ed273631ce3bac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.9765,
              "ttft_seconds": 0.434404,
              "wall_seconds": 4.977089
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.275012,
              "decode_tokens_per_second": 59.649,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1066,
              "reasoning_sha256": "91b345557777c4b53b58b86e01e5aeebe91d7f5d9f8187c5433165c3b18705b2",
              "response_tail_seconds": 0.000425,
              "stream_done_marker": true,
              "stream_sha256": "91b345557777c4b53b58b86e01e5aeebe91d7f5d9f8187c5433165c3b18705b2",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 15.435316,
              "ttft_seconds": 11.160304,
              "wall_seconds": 15.43574
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.329292,
              "decode_tokens_per_second": 47.849,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 135,
              "reasoning_characters": 1082,
              "reasoning_sha256": "f8c0d196b7bb486b9f5ae28597828c8267cfc33ca7e8562d5fa33e5bf5090c72",
              "response_tail_seconds": 0.000607,
              "stream_done_marker": true,
              "stream_sha256": "f8c0d196b7bb486b9f5ae28597828c8267cfc33ca7e8562d5fa33e5bf5090c72",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 10.741571,
              "ttft_seconds": 5.412279,
              "wall_seconds": 10.742178
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.289759,
              "decode_tokens_per_second": 59.444,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1058,
              "reasoning_sha256": "7b0254ccdc362f352f02aef5df7a40528f0579dba13f9890cd62457eb92fcf8c",
              "response_tail_seconds": 0.000109,
              "stream_done_marker": true,
              "stream_sha256": "7b0254ccdc362f352f02aef5df7a40528f0579dba13f9890cd62457eb92fcf8c",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 20.153417,
              "ttft_seconds": 15.863658,
              "wall_seconds": 20.153525
            }
          ],
          "wall_seconds": 20.156199
        },
        {
          "aggregate_end_to_end_tokens_per_second": 51.878,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.237072,
              "decode_tokens_per_second": 60.183,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 995,
              "reasoning_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "response_tail_seconds": 0.000185,
              "stream_done_marker": true,
              "stream_sha256": "d94a78a60c32424a7571f224c35f3672f76bd0791d2f83a27ec122c9f869f56b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.686294,
              "ttft_seconds": 0.449222,
              "wall_seconds": 4.686479
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.90637,
              "decode_tokens_per_second": 51.973,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1022,
              "reasoning_sha256": "1f14b9b0570998464c774c093cd4d67f0e441a3133fcaa32ea898e4a6b52711c",
              "response_tail_seconds": 0.000136,
              "stream_done_marker": true,
              "stream_sha256": "1f14b9b0570998464c774c093cd4d67f0e441a3133fcaa32ea898e4a6b52711c",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 19.735231,
              "ttft_seconds": 14.828862,
              "wall_seconds": 19.735367
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.440977,
              "decode_tokens_per_second": 57.42,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1073,
              "reasoning_sha256": "87d72449cac014e37964efc6b2d62416b0577e728e328d7cec3705992e1547b3",
              "response_tail_seconds": 6.6e-05,
              "stream_done_marker": true,
              "stream_sha256": "87d72449cac014e37964efc6b2d62416b0577e728e328d7cec3705992e1547b3",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.543907,
              "ttft_seconds": 5.102929,
              "wall_seconds": 9.543973
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.446163,
              "decode_tokens_per_second": 57.353,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1056,
              "reasoning_sha256": "57d22d3ed0e8e150dd3b9d7acac3732a01be0f9f29c343d39129fdabd2666b16",
              "response_tail_seconds": 0.000756,
              "stream_done_marker": true,
              "stream_sha256": "57d22d3ed0e8e150dd3b9d7acac3732a01be0f9f29c343d39129fdabd2666b16",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 14.40517,
              "ttft_seconds": 9.959007,
              "wall_seconds": 14.405926
            }
          ],
          "wall_seconds": 19.738545
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 66.35,
        "median": 64.642,
        "minimum": 64.088,
        "p90": 66.35
      },
      "pooled_decode_tokens_per_second": 65.143,
      "runs": [
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 9405,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 141.733861,
          "decode_tokens_per_second": 66.35,
          "finish_reason": "stop",
          "measured_sse_events": 9380,
          "output": "```go\n// ratelimit.go\n// Package ratelimit provides a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time retrieval so tests can inject a fake.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Create one with New; the zero\n// value is not usable. A Limiter is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64   // tokens credited per second\n\ttokens   float64   // available tokens\n\tlast     time.Time // time of last refill accounting\n}\n\n// New returns a token-bucket Limiter that starts full and refills at rate\n// tokens per second, holding at most capacity tokens. If clock is nil the\n// system clock is used. capacity must be positive and rate must be finite\n// and positive; otherwise New returns an error.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: capacity must be positive\")\n\t}\n\tif rate <= 0 || math.IsNaN(rate) || math.IsInf(rate, 0) {\n\t\treturn nil, errors.New(\"ratelimit: rate must be finite and positive\")\n\t}\n\tif clock == nil {\n\t\tclock = systemClock{}\n\t}\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\tcapacity: float64(capacity),\n\t\trate:     rate,\n\t\ttokens:   float64(capacity),\n\t\tlast:     clock.Now(),\n\t}, nil\n}\n\n// Allow reports whether one token is available at the current time,\n// consuming it if so.\nfunc (l *Limiter) Allow() bool { return l.AllowN(l.clock.Now(), 1) }\n\n// AllowN reports whether n tokens are available at time now, consuming them\n// if so. It returns false if n is negative, n exceeds the bucket capacity,\n// or fewer than n tokens are available. A now earlier than a previous call\n// credits no refill for the negative interval.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refillLocked(now)\n\tif n < 0 || float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Tokens returns the number of tokens currently available, accounting for\n// refill up to the current time.\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refillLocked(l.clock.Now())\n\treturn l.tokens\n}\n\nfunc (l *Limiter) refillLocked(now time.Time) {\n\tif now.After(l.last) {\n\t\tl.tokens = math.Min(l.capacity, l.tokens+now.Sub(l.last).Seconds()*l.rate)\n\t\tl.last = now\n\t}\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct{ now time.Time }\n\nfunc (f *fakeClock) Now() time.Time          { return f.now }\nfunc (f *fakeClock) Advance(d time.Duration) { f.now = f.now.Add(d) }\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt0 := time.Date(2024, time.January, 1, 0, 0, 0, 0, time.UTC)\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []step\n\t}{\n\t\t{\"full bucket allows burst\", 5, 1, []step{{0, 5, true}, {0, 1, false}}},\n\t\t{\"refills one per second\", 2, 1, []step{{0, 2, true}, {0, 1, false}, {time.Second, 1, true}, {0, 1, false}}},\n\t\t{\"fractional refill accumulates\", 1, 0.5, []step{{0, 1, true}, {1500 * time.Millisecond, 1, false}, {500 * time.Millisecond, 1, true}}},\n\t\t{\"refill capped at capacity\", 2, 10, []step{{time.Hour, 3, false}, {0, 2, true}}},\n\t\t{\"request larger than capacity\", 1, 1, []step{{0, 2, false}}},\n\t\t{\"negative request rejected\", 1, 1, []step{{0, -1, false}}},\n\t\t{\"no grant without elapsed time\", 1, 1, []step{{0, 1, true}, {0, 1, false}}},\n\t\t{\"backwards clock grants no refill\", 1, 1, []step{{0, 1, true}, {-time.Second, 1, false}}},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tfc := &fakeClock{now: t0}\n\t\t\tlim, err := New(tt.capacity, tt.rate, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New(%d, %v) error = %v\", tt.capacity, tt.rate, err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := lim.AllowN(fc.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(now, %d) = %v, want %v (tokens = %v)\", i, s.n, got, s.want, lim.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 1, 1, false},\n\t\t{\"zero capacity\", 0, 1, true},\n\t\t{\"negative capacity\", -1, 1, true},\n\t\t{\"zero rate\", 1, 0, true},\n\t\t{\"negative rate\", 1, -0.5, true},\n\t\t{\"NaN rate\", 1, math.NaN(), true},\n\t\t{\"Inf rate\", 1, math.Inf(1), true},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tlim, err := New(tt.capacity, tt.rate, nil)\n\t\t\tif (err != nil) != tt.wantErr {\n\t\t\t\tt.Fatalf(\"New(%d, %v) error = %v, wantErr = %v\", tt.capacity, tt.rate, err, tt.wantErr)\n\t\t\t}\n\t\t\tif tt.wantErr && lim != nil {\n\t\t\t\tt.Fatal(\"New returned non-nil Limiter with error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tfc := &fakeClock{now: time.Unix(0, 0)}\n\tlim, err := New(1, 1, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !lim.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true\")\n\t}\n\tif lim.Allow() {\n\t\tt.Fatal(\"Allow() = true, want false\")\n\t}\n\tfc.Advance(time.Second)\n\tif !lim.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true after refill\")\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tlim, err := New(100, 1e-9, nil)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tvar allowed atomic.Int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 100; j++ {\n\t\t\t\tif lim.Allow() {\n\t\t\t\t\tallowed.Add(1)\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := allowed.Load(); got > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= 100\", got)\n\t}\n}\n```",
          "output_characters": 5871,
          "output_sha256": "6ac6527692bdea2b775b0822cf646cb10d40623a63b20664915952d0fea862af",
          "prompt_tokens": 131,
          "reasoning_characters": 24619,
          "reasoning_sha256": "1468fc3ebb67a75ee801b18d1d64feae0c0f9ddcdcc052d92ccf7c96c8399477",
          "response_tail_seconds": 0.002734,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "e2bcffe248b289db2b50338df1cba8f8f50c22c772f6be64b0909df80b65702a",
          "time_to_first_visible_seconds": 126.960829,
          "time_to_last_output_seconds": 141.806098,
          "ttft_seconds": 0.072237,
          "wall_seconds": 141.808832
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 6375,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 98.604332,
          "decode_tokens_per_second": 64.642,
          "finish_reason": "stop",
          "measured_sse_events": 6373,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"fmt\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time so callers can inject a fake in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// RealClock is the default Clock, backed by time.Now.\nvar RealClock Clock = realClock{}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens accrue continuously at rate per second, capped at burst.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64 // tokens added per second\n\tburst  float64 // maximum tokens\n\ttokens float64 // currently available tokens\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter that starts full and refills at rate tokens/second.\n// rate must be finite and > 0; burst must be >= 1. A nil clock defaults to\n// RealClock.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif math.IsNaN(rate) || math.IsInf(rate, 0) || rate <= 0 {\n\t\treturn nil, fmt.Errorf(\"ratelimit: rate must be finite and > 0, got %v\", rate)\n\t}\n\tif burst < 1 {\n\t\treturn nil, fmt.Errorf(\"ratelimit: burst must be >= 1, got %d\", burst)\n\t}\n\tif clock == nil {\n\t\tclock = RealClock\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available, consuming it if so.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// n < 0 returns false; n == 0 returns true without consuming.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif elapsed := now.Sub(l.last); elapsed > 0 { // ignore backward clock steps\n\t\tl.last = now\n\t\tdelta := elapsed.Seconds() * l.rate\n\t\tif delta > l.burst { // guard against overflow after long idle periods\n\t\t\tdelta = l.burst\n\t\t}\n\t\tl.tokens = min(l.burst, l.tokens+delta)\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (c *fakeClock) Now() time.Time {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\treturn c.now\n}\n\nfunc (c *fakeClock) Advance(d time.Duration) {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\tc.now = c.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname: \"burst available immediately, then empty\",\n\t\t\trate: 1, burst: 3,\n\t\t\tsteps: []step{\n\t\t\t\t{advance: 0, n: 3, want: true},\n\t\t\t\t{advance: 0, n: 1, want: false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"partial refill rejects until enough accrues\",\n\t\t\trate: 0.5, burst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{advance: 0, n: 1, want: true},\n\t\t\t\t{advance: time.Second, n: 1, want: false}, // only 0.5 token\n\t\t\t\t{advance: time.Second, n: 1, want: true},  // 1.0 token\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at burst\",\n\t\t\trate: 10, burst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{advance: 0, n: 2, want: true},\n\t\t\t\t{advance: time.Hour, n: 3, want: false},\n\t\t\t\t{advance: 0, n: 2, want: true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"zero and negative n\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{advance: 0, n: 0, want: true},\n\t\t\t\t{advance: 0, n: -1, want: false},\n\t\t\t\t{advance: 0, n: 1, want: true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"backward clock step grants nothing\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{advance: 0, n: 1, want: true},\n\t\t\t\t{advance: -time.Second, n: 1, want: false},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclock := newFakeClock(base)\n\t\t\tl, err := New(tt.rate, tt.burst, clock)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tclock.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t}{\n\t\t{name: \"zero rate\", rate: 0, burst: 1},\n\t\t{name: \"negative rate\", rate: -1, burst: 1},\n\t\t{name: \"NaN rate\", rate: math.NaN(), burst: 1},\n\t\t{name: \"Inf rate\", rate: math.Inf(1), burst: 1},\n\t\t{name: \"zero burst\", rate: 1, burst: 0},\n\t\t{name: \"negative burst\", rate: 1, burst: -5},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tif _, err := New(tt.rate, tt.burst, nil); err == nil {\n\t\t\t\tt.Errorf(\"New(%v, %d) error = nil, want error\", tt.rate, tt.burst)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewDefaultsAndAllow(t *testing.T) {\n\tl, err := New(1000, 1, nil) // nil clock falls back to RealClock\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Error(\"Allow() = false, want true for fresh limiter\")\n\t}\n\tif l.Allow() {\n\t\tt.Error(\"Allow() = true, want false after consuming burst\")\n\t}\n}\n\nfunc TestAllowNConcurrent(t *testing.T) {\n\tl, err := New(1e9, 100, newFakeClock(base))\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tconst goroutines = 100\n\tvar (\n\t\twg sync.WaitGroup\n\t\tmu sync.Mutex\n\t\tok int\n\t)\n\tfor i := 0; i < goroutines; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tok++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif ok != goroutines {\n\t\tt.Errorf(\"admitted %d requests, want %d\", ok, goroutines)\n\t}\n}\n```",
          "output_characters": 5615,
          "output_sha256": "e67589bfcfdb6c594c4693652dd27a6664a540e59004c93d5d4f5eb7b9c6e8cd",
          "prompt_tokens": 133,
          "reasoning_characters": 15139,
          "reasoning_sha256": "c5e8b5f5fb2946f48d3b5a4f8b63bcfe7142f1866d5b4d0f8574a074fd0fb51b",
          "response_tail_seconds": 0.003048,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "db2d7c200df39af4121127d64d76fb3d08e7bab4ec10090146bc74c5391cc3f7",
          "time_to_first_visible_seconds": 80.732322,
          "time_to_last_output_seconds": 98.686523,
          "ttft_seconds": 0.082191,
          "wall_seconds": 98.689571
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 7381,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 115.154384,
          "decode_tokens_per_second": 64.088,
          "finish_reason": "stop",
          "measured_sse_events": 7376,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time; inject an implementation for determinism.\ntype Clock interface{ Now() time.Time }\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// Errors returned by New.\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be finite and > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be >= 1\")\n\tErrClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens accrue continuously at rate per second, capped at burst.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n}\n\n// New returns a Limiter admitting burst events instantly, refilling at rate\n// tokens per second. All parameters are validated.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tswitch {\n\tcase clock == nil:\n\t\treturn nil, ErrClock\n\tcase !(rate > 0) || math.IsInf(rate, 1): // also rejects NaN\n\t\treturn nil, ErrRate\n\tcase burst < 1:\n\t\treturn nil, ErrBurst\n\t}\n\treturn &Limiter{\n\t\tclock:  clock,\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t}, nil\n}\n\n// Allow reports whether one token is available at the injected clock's time.\nfunc (l *Limiter) Allow() bool { return l.AllowN(l.clock.Now(), 1) }\n\n// AllowN reports whether n tokens can be consumed at time now, refilling\n// the bucket first. n < 0 is rejected without state changes; n > burst is\n// always rejected because the bucket can never hold that many tokens.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tif d := now.Sub(l.last).Seconds(); d > 0 { // ignore clock regressions\n\t\tl.tokens = math.Min(l.burst, l.tokens+d*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a deterministic Clock advanced manually by tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tstart := time.Date(2024, 1, 1, 0, 0, 0, time.UTC)\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname: \"drain then refill over time\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{0, 1, false},\n\t\t\t\t{time.Second, 1, true},\n\t\t\t\t{time.Second, 2, false},\n\t\t\t\t{2 * time.Second, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at burst\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {time.Hour, 1, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname: \"fractional accrual\",\n\t\t\trate: 0.5, burst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {0, 1, false}, {time.Second, 1, false}, {time.Second, 1, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"no elapsed time means no refill\",\n\t\t\trate: 1000, burst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname: \"zero cost is free\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{{0, 0, true}, {0, 0, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"negative cost rejected without side effects\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{{0, -1, false}, {0, 1, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"request above burst never allowed\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []step{{0, 3, false}, {time.Minute, 3, false}},\n\t\t},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(start)\n\t\t\tl, err := New(tt.rate, tt.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New(%v, %d) unexpected error: %v\", tt.rate, tt.burst, err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(fc.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(now, %d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewValidation(t *testing.T) {\n\tfc := newFakeClock(time.Date(2024, 1, 1, 0, 0, 0, time.UTC))\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\tclock   Clock\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 1, fc, ErrRate},\n\t\t{\"negative rate\", -0.5, 1, fc, ErrRate},\n\t\t{\"NaN rate\", math.NaN(), 1, fc, ErrRate},\n\t\t{\"positive Inf rate\", math.Inf(1), 1, fc, ErrRate},\n\t\t{\"zero burst\", 1, 0, fc, ErrBurst},\n\t\t{\"negative burst\", 1, -3, fc, ErrBurst},\n\t\t{\"nil clock\", 1, 1, nil, ErrClock},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.rate, tt.burst, tt.clock)\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"New() error = %v, want %v\", err, tt.wantErr)\n\t\t\t}\n\t\t\tif l != nil {\n\t\t\t\tt.Fatalf(\"New() = %v, want nil limiter on error\", l)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tfc := newFakeClock(time.Date(2024, 1, 1, 0, 0, 0, time.UTC))\n\tl, err := New(1, 2, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tif !l.Allow() || !l.Allow() {\n\t\tt.Fatal(\"expected initial burst to be admitted\")\n\t}\n\tif l.Allow() {\n\t\tt.Error(\"expected empty bucket to reject\")\n\t}\n\tfc.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Error(\"expected refill after advancing fake clock\")\n\t}\n}\n\nfunc TestAllowConcurrent(t *testing.T) {\n\tfc := newFakeClock(time.Date(2024, 1, 1, 0, 0, 0, time.UTC))\n\tl, err := New(1, 10, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tvar allowed atomic.Int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.Allow() {\n\t\t\t\tallowed.Add(1)\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := allowed.Load(); got > 10 {\n\t\tt.Errorf(\"allowed = %d, want <= burst (10)\", got)\n\t}\n}\n```",
          "output_characters": 5885,
          "output_sha256": "2d10c1757a6a9a973fb31e56383177cfc528ac501104b20afe8f74f237bc51d7",
          "prompt_tokens": 132,
          "reasoning_characters": 18124,
          "reasoning_sha256": "61fdbe244541f24aae0e7a296794bd03490c22d3a595f931d59392b124070fb3",
          "response_tail_seconds": 0.003019,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "58cdfdbe0316d992d404970907c67bfaab996192f063424205ae4f7b5e261bc4",
          "time_to_first_visible_seconds": 94.60598,
          "time_to_last_output_seconds": 115.231812,
          "ttft_seconds": 0.077429,
          "wall_seconds": 115.234831
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 141.806,
        "median": 115.232,
        "minimum": 98.687,
        "p90": 141.806
      },
      "ttft_seconds": {
        "maximum": 0.082,
        "median": 0.077,
        "minimum": 0.072,
        "p90": 0.082
      },
      "wall_seconds": {
        "maximum": 141.809,
        "median": 115.235,
        "minimum": 98.69,
        "p90": 141.809
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 66.368,
        "median": 61.386,
        "minimum": 52.405,
        "p90": 66.368
      },
      "pooled_decode_tokens_per_second": 61.865,
      "runs": [
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 9807,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 159.742694,
          "decode_tokens_per_second": 61.386,
          "finish_reason": "stop",
          "measured_sse_events": 9791,
          "output": "**To:** Platform Engineering; Applied ML\n**From:** Engineering Lead\n**Date:** Friday\n**Subject:** Local AI capacity \u2014 buy more hardware, or use what we own better\n\nFor the past month we have circled one decision: whether to spend money on more GPUs for our local cluster, or to spend engineering time making better use of the eight cards we already own. This memo recommends the second path, with an explicit trigger to buy if we miss defined targets.\n\nThe data is worth restating. Over the trailing six weeks, average GPU utilization has been about thirty-four percent, but the average hides the shape of the problem. Weekday afternoons we saturate the cluster; fine-tuning jobs queue behind one another and the ninety-fifth percentile wait has crept toward forty-five minutes. Nights and weekends the same hardware sits essentially idle. Interactive inference shows the same concentration: p99 latency roughly triples during the ten-to-two window. The issue is not absolute capacity but concentration.\n\nBuying is the obvious fix. It is simple, fast, and predictable: place the order, rack the cards, and latency improves within days. But the cost is heavier than the invoice. There is power and cooling on a circuit we already know is marginal, depreciation on hardware that loses resale value quickly, and a larger failure surface to maintain. And it does nothing about concentration; we would still own idle GPUs most of the week.\n\nImproving utilization is the alternative, and the headroom is real. The concrete levers are a shared scheduler with priorities instead of per-project partitions, dynamic batching for our two hottest inference models, and quantization that lets each card serve meaningfully more traffic. We would also need to close gaps in telemetry, since the current exporter under-reports during driver resets. My estimate is two engineers for one quarter.\n\nLatency is where the two options are often confused. Utilization work is not merely a throughput play: better scheduling and batching cut queue time directly, and queue time is the dominant term in our p95. The risk runs the other way: aggressive batching can raise per-request latency at low load, and tighter packing erodes headroom, so a single node failure disrupts more concurrent work. We should treat some slack as a deliberate reliability reserve rather than waste.\n\nBuying has its own reliability arithmetic. More cards mean more driver and firmware drift to manage and more watts drawn, though a well-run expansion does add redundancy. Neither option is free of operational risk; they simply fail differently.\n\nOpportunity cost deserves honest treatment in both currencies. In capital: money sunk into GPUs depreciates fast, and the same budget held as a small cloud rental buffer buys flexibility we cannot get otherwise. In people: two engineers for a quarter is real roadmap capacity. But buying consumes people too \u2014 procurement, racking, maintenance \u2014 just less visibly. The difference is what compounds. Scheduler and batching work becomes durable capability that transfers to any hardware we own; a GPU purchase makes us better at exactly one thing: having more GPUs.\n\nReversibility is the decisive asymmetry for me. Hardware is close to a one-way door: resale value is poor, obsolescence is fast, and once cards are racked there is a quiet temptation to keep finding work that justifies them. Utilization work is reversible. If the scheduler project stalls in month two, we stop, and we have lost a quarter of two people's time rather than a capital commitment. Even the telemetry pays off regardless.\n\nMy recommendation: commit two engineers for a quarter to build the shared scheduler, add dynamic batching to the two hot models, and close the telemetry gaps. Define success before we start: ninety-fifth percentile queue wait under fifteen minutes and average utilization above fifty-five percent, with no regression in interactive p99. If we miss those thresholds, we buy \u2014 modestly, two cards rather than a rack, informed by a quarter of trustworthy data. Until then, a small monthly cloud rental budget can serve as an elastic buffer for peak bursts: more expensive per hour, but reversible, and it caps the damage of a wrong demand forecast.\n\nI want to name the uncertainties plainly. Our figures come from an exporter we know misbehaves, so true utilization could be meaningfully higher or lower than thirty-four percent. The demand forecast is soft; if the roadmap doubles inference traffic by Q3, the math changes and buying becomes right sooner. And utilization gains may hit diminishing returns if the real bottleneck is memory bandwidth or storage I/O rather than GPU compute. Any of these, once confirmed, strengthens the case for purchase.\n\nPlease send objections by Thursday. I would like to lock the decision at next week's planning meeting, with the success thresholds written down before the first line of scheduler code exists.",
          "output_characters": 4945,
          "output_sha256": "b2ed45878b0988681d901728ce756e097e4b3fb26907fd041c4bb02e6d1f48e8",
          "prompt_tokens": 121,
          "reasoning_characters": 38123,
          "reasoning_sha256": "749ad438cd70e523f6d1bad2339e8d2bd2d2628adb0c05d5df894e5552f30407",
          "response_tail_seconds": 0.003715,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "7145995c671ab727109956bf2398e625ec30ba4f9a5b5b73589ea832058173c0",
          "time_to_first_visible_seconds": 150.477033,
          "time_to_last_output_seconds": 159.97242,
          "ttft_seconds": 0.229726,
          "wall_seconds": 159.976134
        },
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 11479,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 172.943672,
          "decode_tokens_per_second": 66.368,
          "finish_reason": "stop",
          "measured_sse_events": 11477,
          "output": "**MEMORANDUM**\n\nTo: Platform Engineering\nFrom: R. Okafor, Infrastructure Lead\nDate: 12 March 2025\nSubject: Inference capacity \u2014 buy hardware, or use what we own better\n\nWe need to decide this quarter between buying two more GPU nodes for the inference cluster \u2014 roughly $180k installed \u2014 and investing about six engineer-weeks in using the hardware we already own more effectively. This memo gives my recommendation and reasoning across cost, latency, reliability, opportunity cost, and reversibility, and is explicit about where I am unsure.\n\nThe current picture is lumpy. Our twelve-GPU fleet averages 38% utilization, but overnight batch jobs and the midday surge of interactive traffic rarely overlap, so we idle capacity for hours and then queue for it. Median latency is a comfortable 180 milliseconds; the tail is not. P99 degrades past two seconds when the afternoon peak collides with a long batch, and twice last month a rollout slipped because no node was free to stage on. The problem is less a shortage of hardware than of coordination.\n\nOn cost: hardware is simple money \u2014 $180k of capital plus perhaps $400 a month in power and colocation, depreciated over three years. The utilization option costs less cash but is not free: six weeks of a senior engineer is real money, and the work may yield less than we hope. The honest framing is expected value. If better packing and time-shifting of batch work lifts average utilization to 55\u201360%, we defer the purchase by at least two quarters, worth more than the engineering time at our growth rate. If it stalls near 45%, the arithmetic flips and we should buy.\n\nLatency is where the options pull in opposite directions. New hardware buys headroom: queues shorten, the tail tightens, the afternoon peak stops being an event. Utilization work does the opposite by design \u2014 packing jobs tighter means less slack, and a badly tuned scheduler can turn a 180ms median into something noticeably worse. I believe we can hold the line, since moving batch work off the interactive peak is nearly free latency-wise, but any consolidation of interactive traffic should carry an explicit latency budget, not merely a utilization target.\n\nReliability cuts both ways. Twelve GPUs is already thin redundancy: losing one node removes eight percent of capacity, and today we absorb that with slack. Squeezing utilization erodes exactly that slack, and I do not want a fleet where a single hardware failure puts us over SLO for a day. More hardware, conversely, means more failure surface and more operational burden on a team of five. Either way, I want an explicit rule: size for one node being down, not one node being idle.\n\nOpportunity cost is the sharpest edge of the utilization option. Six weeks is roughly a quarter of one senior engineer's output, and the same person is the natural owner of the rollout-automation work we have deferred since January. The $180k has alternatives; it is about half the cost of a contractor for the data-pipeline backlog this year. Neither trade is catastrophic, but neither is free, and I do not want to pretend the money or the time is idle capacity we are merely using better.\n\nReversibility is what tips the balance. Hardware is a one-way door: once purchased, racked, and depreciating, the decision is made whether or not demand materializes, and reselling GPUs recovers perhaps half of what we paid. Utilization work is a sequence of small, reversible changes \u2014 a scheduler tweak can be reverted in an afternoon, a batch job rescheduled in minutes. If we learn something surprising mid-stream, we have spent time, not capital. Buying now would foreclose the cheaper experiment before we have run it.\n\nMy recommendation is to spend the six weeks first, strictly time-boxed, against two pre-agreed criteria: average utilization at or above 55%, and p99 latency under one second through the afternoon peak. Meeting both defers the purchase to next quarter's review with better data. Missing either by mid-May means we buy the nodes without relitigating \u2014 the trigger is agreed now precisely so future-us cannot talk itself out of it.\n\nFinally, the uncertainty. I do not know whether our utilization ceiling is 45% or 65% given our workload shapes; whether demand growth will hold; or whether the scheduler work will surface complications that make consolidation riskier than it looks on a whiteboard. The time-box and the purchase trigger are how we bound that uncertainty rather than resolve it. I am happy to argue thresholds at Thursday's sync, and Priya has offered to draft the latency budget before then.",
          "output_characters": 4596,
          "output_sha256": "f791539ee11dd01cdca622ad38fd831df8a856c0d4a1e47e23d7057d9de31bd7",
          "prompt_tokens": 121,
          "reasoning_characters": 41759,
          "reasoning_sha256": "275e244799d332173ae67129c4dffdf7d34703a499e5769d33a08395754706e9",
          "response_tail_seconds": 0.003502,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "8c51dea7be47cd6e581c154a5cd2d053daf434c55bdeb88e9624f1b12e68706d",
          "time_to_first_visible_seconds": 164.000347,
          "time_to_last_output_seconds": 173.171553,
          "ttft_seconds": 0.227882,
          "wall_seconds": 173.175055
        },
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 3891,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 74.230086,
          "decode_tokens_per_second": 52.405,
          "finish_reason": "stop",
          "measured_sse_events": 3886,
          "output": "**To:** Engineering Team\n**From:** R. Vance, Infrastructure Lead\n**Date:** 12 May 2025\n**Subject:** GPU capacity \u2014 buy a third node, or use the two we have better\n\nThis memo settles the question we've been circling for a month: whether to spend roughly $260k on a third GPU node, or invest six weeks of engineering time in getting more out of the two we already own. My recommendation is the latter, with a pre-agreed trigger to buy if the numbers don't move. The reasoning follows.\n\nFirst, an honest picture of where we stand. Our dashboards report average GPU utilization of about 40 percent across the cluster, but I want to be careful with that number, because kernel-level utilization says little about useful throughput. What we can say confidently is that the fine-tuning queue regularly backs up \u2014 p95 wait time has crept from forty minutes to nearly six hours this quarter \u2014 while overnight and weekend capacity sits largely idle. The pattern suggests this is less a raw capacity problem than a scheduling and packing problem: large jobs monopolise whole nodes, small interactive jobs queue behind them, and nobody has had time to tune batch sizes or shard work properly.\n\nOn cost, the comparison is not close in the near term. A new 8-way H100 node is roughly $260k of capital, plus about $1,800 a month in power and cooling, and it depreciates to near nothing on our books within three years. Renting equivalent capacity from a cloud provider would run $25\u201330 per GPU-hour \u2014 painful at sustained load, but entirely sensible as a bridge. Against that, the utilisation effort is two engineers for six weeks, call it $70k fully loaded. If the work defers the purchase by even six months, it pays for itself several times over.\n\nLatency deserves its own paragraph, because it is the complaint we actually hear. Nobody on this team cares about fleet-average utilization; they care that an interactive experiment takes three hours to start. Better queueing \u2014 preemption, priority classes, right-sizing jobs to fewer GPUs \u2014 would cut tail latency more reliably than adding hardware, because adding hardware without fixing scheduling tends to just relocate the queue. There is a genuine trade-off in batching: larger batches raise throughput but add per-request latency, and we should make that knob explicit rather than leaving it at defaults.\n\nReliability cuts both ways. Today we have no redundancy: if node one loses a DIMM or needs a driver rollback, the training queue stops and someone's week is ruined. A third node would give us N+1 headroom, which is genuinely valuable. But better scheduling buys much of the same benefit more cheaply \u2014 if jobs can drain and migrate between nodes, we can take a machine down for maintenance without drama. Buying more hardware also adds failure surface: more PSUs, more cables, more things for a four-person team to babysit.\n\nOpportunity cost is the quiet factor. The $260k is earmarked but not free \u2014 it is the same budget line that would otherwise fund the eval harness we keep deferring, and six weeks of Priya's and my time is six weeks not spent on the retrieval pipeline. Conversely, if we keep limping with six-hour queues, the opportunity cost shows up as slower research iteration, which appears on no invoice but is probably the largest cost of all.\n\nReversibility argues the same direction. Software changes can be reverted in an afternoon; a purchased node cannot be un-bought. Used GPU resale is a buyer's market, and a half-idle node on our rack is capital slowly converting into depreciation. If we later do need the hardware, the utilisation work \u2014 the scheduler config, the batching discipline, the honest metrics \u2014 remains valuable regardless.\n\nSo, concretely: I propose six weeks of focused utilisation work, owned by Priya, delivering a proper queue with preemption, right-sized job templates, and a weekly report of useful GPU-hours rather than raw utilization. We agree now on the trigger: if, two weeks after that work lands, p95 queue wait still exceeds thirty minutes, or we turn away more than ten percent of requested GPU-hours in any week, we execute the purchase immediately, no further debate. And if the fine-tuning programme lands the contract it is chasing, we buy regardless, because forecast demand trumps any optimisation.\n\nI want to be explicit about uncertainty. Our demand forecast is soft, the utilisation numbers may be flattering, and it is possible the binding constraint is GPU memory rather than compute, in which case scheduling tricks help less than hoped and the purchase case strengthens. There is also key-person risk in leaning on one engineer's scheduler work. The trigger mechanism exists precisely because I might be wrong.\n\nPlease come to Thursday's planning meeting with objections. If nobody argues, we start Monday.",
          "output_characters": 4821,
          "output_sha256": "ee2b621259058882a0ec94480744a7f8e3d6b081ee22e705bc878031b558fc7e",
          "prompt_tokens": 122,
          "reasoning_characters": 12970,
          "reasoning_sha256": "25c1e9b145331443015a8cd52d715b8bec99e237862926add330bb6128d91b95",
          "response_tail_seconds": 0.001328,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "3333a13de30b8b5a0c25b032b9ecc49407d02924de472c7a25a1d4c474f9b141",
          "time_to_first_visible_seconds": 62.328138,
          "time_to_last_output_seconds": 74.461031,
          "ttft_seconds": 0.230945,
          "wall_seconds": 74.462359
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 173.172,
        "median": 159.972,
        "minimum": 74.461,
        "p90": 173.172
      },
      "ttft_seconds": {
        "maximum": 0.231,
        "median": 0.23,
        "minimum": 0.228,
        "p90": 0.231
      },
      "wall_seconds": {
        "maximum": 173.175,
        "median": 159.976,
        "minimum": 74.462,
        "p90": 173.175
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 70.98,
        "median": 70.823,
        "minimum": 69.668,
        "p90": 70.98
      },
      "pooled_decode_tokens_per_second": 70.49,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1020,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 14.388038,
          "decode_tokens_per_second": 70.823,
          "finish_reason": "stop",
          "measured_sse_events": 1018,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 112,
          "reasoning_characters": 594,
          "reasoning_sha256": "10213f102a9629c5bab0d0ea45b7de2461d947f617dfbf98e54677edb91726ff",
          "response_tail_seconds": 0.000744,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "da362a73478222f26ecebb95e137d9ebddb8e1aeea2acd57f57c3bd9ccbe98b2",
          "time_to_first_visible_seconds": 5.968686,
          "time_to_last_output_seconds": 14.873574,
          "ttft_seconds": 0.485536,
          "wall_seconds": 14.874319
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 13.750368,
          "decode_tokens_per_second": 70.98,
          "finish_reason": "stop",
          "measured_sse_events": 973,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.000495,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 5.255524,
          "time_to_last_output_seconds": 14.233079,
          "ttft_seconds": 0.482711,
          "wall_seconds": 14.233574
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 14.009214,
          "decode_tokens_per_second": 69.668,
          "finish_reason": "stop",
          "measured_sse_events": 973,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 589,
          "reasoning_sha256": "ac6f8cdd5ad461fb61b5117ff60fe8dfcc47e1ab8a8c10ec9de798af95a377ab",
          "response_tail_seconds": 0.00016,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "05aa596be4fd4d8b2992b39454b6f9210237ce7951785fe0d3b3a1a3f737101e",
          "time_to_first_visible_seconds": 5.54206,
          "time_to_last_output_seconds": 14.495948,
          "ttft_seconds": 0.486735,
          "wall_seconds": 14.496109
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 14.874,
        "median": 14.496,
        "minimum": 14.233,
        "p90": 14.874
      },
      "ttft_seconds": {
        "maximum": 0.487,
        "median": 0.486,
        "minimum": 0.483,
        "p90": 0.487
      },
      "wall_seconds": {
        "maximum": 14.874,
        "median": 14.496,
        "minimum": 14.234,
        "p90": 14.874
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1990.108,
          "median": 1985.743,
          "minimum": 1985.112,
          "p90": 1990.108
        },
        "runs": [
          {
            "cache_salt_sha256": "4e85efdc5eda589b2b9ac4fdb5436645cac93cf4ae9091bd837d966d47d11d17",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053438,
            "decode_tokens_per_second": 130.992,
            "effective_prefill_tokens_per_second": 1990.108,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000204,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 16.465434,
            "time_to_last_output_seconds": 16.518872,
            "ttft_seconds": 16.465434,
            "wall_seconds": 16.519076
          },
          {
            "cache_salt_sha256": "32929c762628bde8a998765537f789b633aa756e9950054ae032f7dd2879bf2b",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.060706,
            "decode_tokens_per_second": 115.31,
            "effective_prefill_tokens_per_second": 1985.743,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000111,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.501632,
            "time_to_last_output_seconds": 16.562338,
            "ttft_seconds": 16.501632,
            "wall_seconds": 16.562449
          },
          {
            "cache_salt_sha256": "dc857e8bb4cf88995bc91983c4b1274dc655283f97397781ab03bc14ebe57392",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.060843,
            "decode_tokens_per_second": 115.051,
            "effective_prefill_tokens_per_second": 1985.112,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000112,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 16.506879,
            "time_to_last_output_seconds": 16.567722,
            "ttft_seconds": 16.506879,
            "wall_seconds": 16.567834
          }
        ],
        "ttft_seconds": {
          "maximum": 16.507,
          "median": 16.502,
          "minimum": 16.465,
          "p90": 16.507
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 791077.205,
          "median": 758430.737,
          "minimum": 740084.922,
          "p90": 791077.205
        },
        "runs": [
          {
            "cache_salt_sha256": "4e85efdc5eda589b2b9ac4fdb5436645cac93cf4ae9091bd837d966d47d11d17",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.05507,
            "decode_tokens_per_second": 127.112,
            "effective_prefill_tokens_per_second": 791077.205,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001069,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 0.041422,
            "time_to_last_output_seconds": 0.096491,
            "ttft_seconds": 0.041422,
            "wall_seconds": 0.09756
          },
          {
            "cache_salt_sha256": "32929c762628bde8a998765537f789b633aa756e9950054ae032f7dd2879bf2b",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.061435,
            "decode_tokens_per_second": 113.941,
            "effective_prefill_tokens_per_second": 740084.922,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000118,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.044276,
            "time_to_last_output_seconds": 0.105712,
            "ttft_seconds": 0.044276,
            "wall_seconds": 0.10583
          },
          {
            "cache_salt_sha256": "dc857e8bb4cf88995bc91983c4b1274dc655283f97397781ab03bc14ebe57392",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.059758,
            "decode_tokens_per_second": 117.139,
            "effective_prefill_tokens_per_second": 758430.737,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00074,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 0.043205,
            "time_to_last_output_seconds": 0.102963,
            "ttft_seconds": 0.043205,
            "wall_seconds": 0.103703
          }
        ],
        "ttft_seconds": {
          "maximum": 0.044,
          "median": 0.043,
          "minimum": 0.041,
          "p90": 0.044
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1958.794,
          "median": 1958.031,
          "minimum": 1873.506,
          "p90": 1958.794
        },
        "runs": [
          {
            "cache_salt_sha256": "90ea619099946722a6253e8e0a689eb0167f4d196ffe45b5eca7e7ff53a66247",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064802,
            "decode_tokens_per_second": 108.022,
            "effective_prefill_tokens_per_second": 1873.506,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000154,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 34.980414,
            "time_to_last_output_seconds": 35.045216,
            "ttft_seconds": 34.980414,
            "wall_seconds": 35.04537
          },
          {
            "cache_salt_sha256": "6607b58e6eae0cfa3bb4f6187ce171eb6b9e47c172f5679e67a10a56af315042",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.063177,
            "decode_tokens_per_second": 110.799,
            "effective_prefill_tokens_per_second": 1958.031,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000486,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 33.470362,
            "time_to_last_output_seconds": 33.53354,
            "ttft_seconds": 33.470362,
            "wall_seconds": 33.534026
          },
          {
            "cache_salt_sha256": "b95f3e4e3dce093613877b6b1f722eb151496d893990da8ccb8b7b214f8e6a7a",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.06461,
            "decode_tokens_per_second": 108.343,
            "effective_prefill_tokens_per_second": 1958.794,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00045,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 33.457326,
            "time_to_last_output_seconds": 33.521936,
            "ttft_seconds": 33.457326,
            "wall_seconds": 33.522386
          }
        ],
        "ttft_seconds": {
          "maximum": 34.98,
          "median": 33.47,
          "minimum": 33.457,
          "p90": 34.98
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1274077.53,
          "median": 1242977.715,
          "minimum": 1197004.566,
          "p90": 1274077.53
        },
        "runs": [
          {
            "cache_salt_sha256": "90ea619099946722a6253e8e0a689eb0167f4d196ffe45b5eca7e7ff53a66247",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064466,
            "decode_tokens_per_second": 108.584,
            "effective_prefill_tokens_per_second": 1242977.715,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001842,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.052725,
            "time_to_last_output_seconds": 0.117191,
            "ttft_seconds": 0.052725,
            "wall_seconds": 0.119033
          },
          {
            "cache_salt_sha256": "6607b58e6eae0cfa3bb4f6187ce171eb6b9e47c172f5679e67a10a56af315042",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062875,
            "decode_tokens_per_second": 111.332,
            "effective_prefill_tokens_per_second": 1197004.566,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00074,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.05475,
            "time_to_last_output_seconds": 0.117625,
            "ttft_seconds": 0.05475,
            "wall_seconds": 0.118365
          },
          {
            "cache_salt_sha256": "b95f3e4e3dce093613877b6b1f722eb151496d893990da8ccb8b7b214f8e6a7a",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.06403,
            "decode_tokens_per_second": 109.323,
            "effective_prefill_tokens_per_second": 1274077.53,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000533,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.051438,
            "time_to_last_output_seconds": 0.115469,
            "ttft_seconds": 0.051438,
            "wall_seconds": 0.116001
          }
        ],
        "ttft_seconds": {
          "maximum": 0.055,
          "median": 0.053,
          "minimum": 0.051,
          "p90": 0.055
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1958.943,
          "median": 1957.523,
          "minimum": 1488.91,
          "p90": 1958.943
        },
        "runs": [
          {
            "cache_salt_sha256": "9356c8a390e58b0a02990516a5928e7c19442428709e159caebfb0d1bf13199e",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053603,
            "decode_tokens_per_second": 130.591,
            "effective_prefill_tokens_per_second": 1488.91,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " observation records behaviour, evidence, constraints,",
            "output_characters": 54,
            "output_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000798,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "time_to_first_visible_seconds": 5.502013,
            "time_to_last_output_seconds": 5.555616,
            "ttft_seconds": 5.502013,
            "wall_seconds": 5.556414
          },
          {
            "cache_salt_sha256": "009218582ed81b71403d9cca0bad5d85fb88767f539bc9ce927e9c76a0dd53ea",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053937,
            "decode_tokens_per_second": 129.781,
            "effective_prefill_tokens_per_second": 1958.943,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000651,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 4.181847,
            "time_to_last_output_seconds": 4.235784,
            "ttft_seconds": 4.181847,
            "wall_seconds": 4.236434
          },
          {
            "cache_salt_sha256": "cc05b00f66a3171b5d86057be949b398d8bc90e60c23e24dc9a9c982d97f70d6",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053387,
            "decode_tokens_per_second": 131.118,
            "effective_prefill_tokens_per_second": 1957.523,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000932,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.184881,
            "time_to_last_output_seconds": 4.238268,
            "ttft_seconds": 4.184881,
            "wall_seconds": 4.2392
          }
        ],
        "ttft_seconds": {
          "maximum": 5.502,
          "median": 4.185,
          "minimum": 4.182,
          "p90": 5.502
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 238056.492,
          "median": 206373.599,
          "minimum": 191904.048,
          "p90": 238056.492
        },
        "runs": [
          {
            "cache_salt_sha256": "9356c8a390e58b0a02990516a5928e7c19442428709e159caebfb0d1bf13199e",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.052635,
            "decode_tokens_per_second": 132.992,
            "effective_prefill_tokens_per_second": 238056.492,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " observation records behaviour, evidence, constraints,",
            "output_characters": 54,
            "output_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000612,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "time_to_first_visible_seconds": 0.034412,
            "time_to_last_output_seconds": 0.087047,
            "ttft_seconds": 0.034412,
            "wall_seconds": 0.087659
          },
          {
            "cache_salt_sha256": "009218582ed81b71403d9cca0bad5d85fb88767f539bc9ce927e9c76a0dd53ea",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.054388,
            "decode_tokens_per_second": 128.705,
            "effective_prefill_tokens_per_second": 191904.048,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001325,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 0.042688,
            "time_to_last_output_seconds": 0.097075,
            "ttft_seconds": 0.042688,
            "wall_seconds": 0.0984
          },
          {
            "cache_salt_sha256": "cc05b00f66a3171b5d86057be949b398d8bc90e60c23e24dc9a9c982d97f70d6",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.051983,
            "decode_tokens_per_second": 134.66,
            "effective_prefill_tokens_per_second": 206373.599,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000515,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.039695,
            "time_to_last_output_seconds": 0.091677,
            "ttft_seconds": 0.039695,
            "wall_seconds": 0.092193
          }
        ],
        "ttft_seconds": {
          "maximum": 0.043,
          "median": 0.04,
          "minimum": 0.034,
          "p90": 0.043
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "checkpoint MTP head (Mia v1.10 DRAFTER=mtp)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3",
      "model_revision": "13ba3ef94d88bc30b27691efa29a6012f66e964d (AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3 on Hugging Face; the same files, served from a local copy)",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-mtp-max",
    "finished_at": "2026-10-08T11:17:14.886382+00:00",
    "label": "tp2-ours-mtp-max",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T10:59:16.657843+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 32768,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "max"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 3,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}