{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 58.897,
        "median": 58.865,
        "minimum": 58.529,
        "p90": 58.897
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 64.723,
        "median": 64.684,
        "minimum": 64.386,
        "p90": 64.723
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.411,
        "median": 0.405,
        "minimum": 0.405,
        "p90": 0.411
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 58.865,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.942233,
              "decode_tokens_per_second": 64.684,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.001357,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.347006,
              "ttft_seconds": 0.404773,
              "wall_seconds": 4.348363
            }
          ],
          "wall_seconds": 4.348954
        },
        {
          "aggregate_end_to_end_tokens_per_second": 58.529,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.960466,
              "decode_tokens_per_second": 64.386,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1007,
              "reasoning_sha256": "b21083223e8b7e6d0859f18ed8d9203d302fb1d772bf784aaa09f1ae2c6358cf",
              "response_tail_seconds": 0.001642,
              "stream_done_marker": true,
              "stream_sha256": "b21083223e8b7e6d0859f18ed8d9203d302fb1d772bf784aaa09f1ae2c6358cf",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.371861,
              "ttft_seconds": 0.411395,
              "wall_seconds": 4.373503
            }
          ],
          "wall_seconds": 4.373912
        },
        {
          "aggregate_end_to_end_tokens_per_second": 58.897,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.939876,
              "decode_tokens_per_second": 64.723,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1060,
              "reasoning_sha256": "6b297dc5e57a2e9eff1dd98d0a0541816e97e9cad5f7bcdcd65fd3ada5af8a67",
              "response_tail_seconds": 0.001486,
              "stream_done_marker": true,
              "stream_sha256": "6b297dc5e57a2e9eff1dd98d0a0541816e97e9cad5f7bcdcd65fd3ada5af8a67",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.344673,
              "ttft_seconds": 0.404797,
              "wall_seconds": 4.34616
            }
          ],
          "wall_seconds": 4.346577
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 91.789,
        "median": 77.078,
        "minimum": 76.138,
        "p90": 91.789
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 50.521,
        "median": 46.152,
        "minimum": 44.781,
        "p90": 50.104
      },
      "per_stream_ttft_seconds": {
        "maximum": 1.027,
        "median": 0.953,
        "minimum": 0.486,
        "p90": 1.027
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 76.138,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.694357,
              "decode_tokens_per_second": 44.781,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1004,
              "reasoning_sha256": "c77d12089c696f7ff5269117f5675f20151762bf163ed04f15a0ac6ec98c7205",
              "response_tail_seconds": 0.002237,
              "stream_done_marker": true,
              "stream_sha256": "c77d12089c696f7ff5269117f5675f20151762bf163ed04f15a0ac6ec98c7205",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.721475,
              "ttft_seconds": 1.027117,
              "wall_seconds": 6.723711
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.510909,
              "decode_tokens_per_second": 46.272,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1008,
              "reasoning_sha256": "4652763b695c4334369dc0cad8dc3e85cd0a6dd194c934f9720b99402e6cc708",
              "response_tail_seconds": 0.000649,
              "stream_done_marker": true,
              "stream_sha256": "4652763b695c4334369dc0cad8dc3e85cd0a6dd194c934f9720b99402e6cc708",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.537562,
              "ttft_seconds": 1.026653,
              "wall_seconds": 6.53821
            }
          ],
          "wall_seconds": 6.724659
        },
        {
          "aggregate_end_to_end_tokens_per_second": 77.078,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.53967,
              "decode_tokens_per_second": 46.032,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 941,
              "reasoning_sha256": "ab7080170a3d73a4c4e59cf499fa0add446ce6b48ab7b92bc42253ea8cbaf391",
              "response_tail_seconds": 0.0006,
              "stream_done_marker": true,
              "stream_sha256": "ab7080170a3d73a4c4e59cf499fa0add446ce6b48ab7b92bc42253ea8cbaf391",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.429559,
              "ttft_seconds": 0.889889,
              "wall_seconds": 6.430159
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.623825,
              "decode_tokens_per_second": 45.343,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1014,
              "reasoning_sha256": "cdbf65881629df2aeb85e2c87cdfd924ab9e58873d54f2548a100768881febbd",
              "response_tail_seconds": 0.001009,
              "stream_done_marker": true,
              "stream_sha256": "cdbf65881629df2aeb85e2c87cdfd924ab9e58873d54f2548a100768881febbd",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 6.640734,
              "ttft_seconds": 1.016909,
              "wall_seconds": 6.641743
            }
          ],
          "wall_seconds": 6.642615
        },
        {
          "aggregate_end_to_end_tokens_per_second": 91.789,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.04745,
              "decode_tokens_per_second": 50.521,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 1017,
              "reasoning_sha256": "bb007c6d68cbff0c9647af53b9438e822f8bcfacce39ddfcfe09086a4f09a384",
              "response_tail_seconds": 0.000562,
              "stream_done_marker": true,
              "stream_sha256": "bb007c6d68cbff0c9647af53b9438e822f8bcfacce39ddfcfe09086a4f09a384",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.533799,
              "ttft_seconds": 0.486349,
              "wall_seconds": 5.534362
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.089444,
              "decode_tokens_per_second": 50.104,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.001339,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.575872,
              "ttft_seconds": 0.486428,
              "wall_seconds": 5.577211
            }
          ],
          "wall_seconds": 5.578018
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 120.913,
        "median": 120.079,
        "minimum": 114.478,
        "p90": 120.913
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 40.14,
        "median": 33.657,
        "minimum": 31.314,
        "p90": 35.185
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.947,
        "median": 0.799,
        "minimum": 0.583,
        "p90": 0.947
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 120.913,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 6.352715,
              "decode_tokens_per_second": 40.14,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 953,
              "reasoning_sha256": "703fd2db31b9e039862b15838cebe57696bb8cc38a35436cd85656921358643b",
              "response_tail_seconds": 0.000543,
              "stream_done_marker": true,
              "stream_sha256": "703fd2db31b9e039862b15838cebe57696bb8cc38a35436cd85656921358643b",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 7.29939,
              "ttft_seconds": 0.946676,
              "wall_seconds": 7.299934
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.24743,
              "decode_tokens_per_second": 35.185,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 989,
              "reasoning_sha256": "aa81a726d3e8cc6506aa814037e229b6b5aea3ba969ba2b999432d2cb4bb2fb0",
              "response_tail_seconds": 0.001532,
              "stream_done_marker": true,
              "stream_sha256": "aa81a726d3e8cc6506aa814037e229b6b5aea3ba969ba2b999432d2cb4bb2fb0",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.193752,
              "ttft_seconds": 0.946323,
              "wall_seconds": 8.195284
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.494662,
              "decode_tokens_per_second": 34.024,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 1028,
              "reasoning_sha256": "4c79796864a0b17e0004ef01381dcc581550a98bd3c8a692d4fff2d19b8469c3",
              "response_tail_seconds": 0.001369,
              "stream_done_marker": true,
              "stream_sha256": "4c79796864a0b17e0004ef01381dcc581550a98bd3c8a692d4fff2d19b8469c3",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.292258,
              "ttft_seconds": 0.797596,
              "wall_seconds": 8.293627
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.519255,
              "decode_tokens_per_second": 33.913,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1042,
              "reasoning_sha256": "457bace4006ce8401440e7b010dc69326f529408d1d17fd180663f0adcab786c",
              "response_tail_seconds": 0.001407,
              "stream_done_marker": true,
              "stream_sha256": "457bace4006ce8401440e7b010dc69326f529408d1d17fd180663f0adcab786c",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.465839,
              "ttft_seconds": 0.946584,
              "wall_seconds": 8.467246
            }
          ],
          "wall_seconds": 8.468885
        },
        {
          "aggregate_end_to_end_tokens_per_second": 120.079,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.575595,
              "decode_tokens_per_second": 33.661,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 951,
              "reasoning_sha256": "5b998459c0170c7d6295b5b4cfb50f6f444703cb3f2ea4826c2a95c390cf766a",
              "response_tail_seconds": 0.000689,
              "stream_done_marker": true,
              "stream_sha256": "5b998459c0170c7d6295b5b4cfb50f6f444703cb3f2ea4826c2a95c390cf766a",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.158962,
              "ttft_seconds": 0.583367,
              "wall_seconds": 8.159651
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.941787,
              "decode_tokens_per_second": 32.109,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 962,
              "reasoning_sha256": "232a8fca2e8bc9fb65d47ea7e2b462e83098113237e216bcf172958fcfc4f8d8",
              "response_tail_seconds": 0.000663,
              "stream_done_marker": true,
              "stream_sha256": "232a8fca2e8bc9fb65d47ea7e2b462e83098113237e216bcf172958fcfc4f8d8",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.525633,
              "ttft_seconds": 0.583846,
              "wall_seconds": 8.526296
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.577449,
              "decode_tokens_per_second": 33.652,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 987,
              "reasoning_sha256": "86958b69cfa83f944c084bba351e934fca7f70c3e1ce61b8f587d3b6ce3f7e62",
              "response_tail_seconds": 0.000447,
              "stream_done_marker": true,
              "stream_sha256": "86958b69cfa83f944c084bba351e934fca7f70c3e1ce61b8f587d3b6ce3f7e62",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.160494,
              "ttft_seconds": 0.583046,
              "wall_seconds": 8.160941
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.753491,
              "decode_tokens_per_second": 32.888,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 969,
              "reasoning_sha256": "5c8ec75edc1bfb174c8cfc068a0a3a908abbef2b4641a0ce938f3a55a5dbfb26",
              "response_tail_seconds": 0.000827,
              "stream_done_marker": true,
              "stream_sha256": "5c8ec75edc1bfb174c8cfc068a0a3a908abbef2b4641a0ce938f3a55a5dbfb26",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.336838,
              "ttft_seconds": 0.583347,
              "wall_seconds": 8.337664
            }
          ],
          "wall_seconds": 8.527735
        },
        {
          "aggregate_end_to_end_tokens_per_second": 114.478,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.48583,
              "decode_tokens_per_second": 34.064,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1017,
              "reasoning_sha256": "eb8fdbfd0b66bf6e60944ac04febf34468f6bf60dce90deb747a820d539f7075",
              "response_tail_seconds": 0.000686,
              "stream_done_marker": true,
              "stream_sha256": "eb8fdbfd0b66bf6e60944ac04febf34468f6bf60dce90deb747a820d539f7075",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.285562,
              "ttft_seconds": 0.799733,
              "wall_seconds": 8.286249
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.700885,
              "decode_tokens_per_second": 33.113,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1016,
              "reasoning_sha256": "986da19753aa24fc19b2dcd1bbbbb7d5d5a2d519c6fcdd83cf75cd36b907d1a7",
              "response_tail_seconds": 0.001141,
              "stream_done_marker": true,
              "stream_sha256": "986da19753aa24fc19b2dcd1bbbbb7d5d5a2d519c6fcdd83cf75cd36b907d1a7",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.559544,
              "ttft_seconds": 0.858659,
              "wall_seconds": 8.560685
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.143218,
              "decode_tokens_per_second": 31.314,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 992,
              "reasoning_sha256": "7744c0273c17fdab8f47386006cafde430fc7e31c30277841677629c9a84c95f",
              "response_tail_seconds": 0.001249,
              "stream_done_marker": true,
              "stream_sha256": "7744c0273c17fdab8f47386006cafde430fc7e31c30277841677629c9a84c95f",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.941819,
              "ttft_seconds": 0.798601,
              "wall_seconds": 8.943068
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.89913,
              "decode_tokens_per_second": 32.282,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1031,
              "reasoning_sha256": "d29fcd5878ab64ccd9d8daf581d36b4fa8aa72f461d543d00983c02beac07ff4",
              "response_tail_seconds": 0.000771,
              "stream_done_marker": true,
              "stream_sha256": "d29fcd5878ab64ccd9d8daf581d36b4fa8aa72f461d543d00983c02beac07ff4",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 8.698317,
              "ttft_seconds": 0.799187,
              "wall_seconds": 8.699088
            }
          ],
          "wall_seconds": 8.944971
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 73.266,
        "median": 73.01,
        "minimum": 71.824,
        "p90": 73.266
      },
      "pooled_decode_tokens_per_second": 72.78,
      "runs": [
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 18788,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 256.422249,
          "decode_tokens_per_second": 73.266,
          "finish_reason": "stop",
          "measured_sse_events": 18769,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"fmt\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket limiter: tokens refill continuously at rate per\n// second, capped at burst. The zero value is unusable; construct with New.\n// All methods are safe for concurrent use.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter starting full. If clock is nil, the system clock is\n// used. rate must be finite and > 0; burst must be > 0.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif math.IsNaN(rate) || math.IsInf(rate, 0) || rate <= 0 {\n\t\treturn nil, fmt.Errorf(\"ratelimit: rate must be finite and > 0, got %v\", rate)\n\t}\n\tif burst <= 0 {\n\t\treturn nil, fmt.Errorf(\"ratelimit: burst must be > 0, got %d\", burst)\n\t}\n\tif clock == nil {\n\t\tclock = systemClock{}\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token may be consumed.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens may be consumed, refilling the bucket from\n// elapsed clock time first. n < 0 returns false; n == 0 returns true. If\n// fewer than n tokens are available, the bucket is left unchanged.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(l.clock.Now())\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the available balance after refilling.\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(l.clock.Now())\n\treturn l.tokens\n}\n\nfunc (l *Limiter) refill(now time.Time) {\n\tif d := now.Sub(l.last).Seconds(); d > 0 {\n\t\tl.tokens = math.Min(l.burst, l.tokens+d*l.rate)\n\t\tl.last = now\n\t}\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manual Clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock() *fakeClock { return &fakeClock{now: time.Unix(0, 0).UTC()} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 1, 1, false},\n\t\t{\"zero rate\", 0, 1, true},\n\t\t{\"negative rate\", -0.5, 1, true},\n\t\t{\"NaN rate\", math.NaN(), 1, true},\n\t\t{\"Inf rate\", math.Inf(1), 1, true},\n\t\t{\"zero burst\", 1, 0, true},\n\t\t{\"negative burst\", 1, -3, true},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.rate, tt.burst, newFakeClock())\n\t\t\tif (err != nil) != tt.wantErr {\n\t\t\t\tt.Fatalf(\"New() error = %v, wantErr %v\", err, tt.wantErr)\n\t\t\t}\n\t\t\tif !tt.wantErr && l.Tokens() != float64(tt.burst) {\n\t\t\t\tt.Fatalf(\"initial tokens = %v, want %v\", l.Tokens(), float64(tt.burst))\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNilClock(t *testing.T) {\n\tl, err := New(1, 1, nil)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true\")\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname:  \"burst drains then blocks\",\n\t\t\trate:  1,\n\t\t\tburst: 3,\n\t\t\tsteps: []step{{n: 3, want: true}, {n: 1, want: false}},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over elapsed time\",\n\t\t\trate:  2,\n\t\t\tburst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{n: 2, want: true},\n\t\t\t\t{n: 1, want: false},\n\t\t\t\t{advance: 500 * time.Millisecond, n: 1, want: true},\n\t\t\t\t{n: 1, want: false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill capped at burst\",\n\t\t\trate:  1,\n\t\t\tburst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{n: 2, want: true},\n\t\t\t\t{advance: time.Hour, n: 2, want: true},\n\t\t\t\t{n: 1, want: false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"partial refill rejected without deduction\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{n: 1, want: true},\n\t\t\t\t{advance: 100 * time.Millisecond, n: 1, want: false},\n\t\t\t\t{advance: 900 * time.Millisecond, n: 1, want: true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"zero and negative n\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{{n: 0, want: true}, {n: -1, want: false}, {n: 1, want: true}},\n\t\t},\n\t\t{\n\t\t\tname:  \"backwards clock does not refill\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{n: 1, want: true},\n\t\t\t\t{advance: -time.Second, n: 1, want: false},\n\t\t\t\t{advance: 2 * time.Second, n: 1, want: true},\n\t\t\t},\n\t\t},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock()\n\t\t\tl, err := New(tt.rate, tt.burst, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tl, err := New(1, 100, newFakeClock())\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tvar allowed atomic.Int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 100; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tallowed.Add(1)\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := allowed.Load(); got > 100 {\n\t\tt.Fatalf(\"allowed = %d, want <= burst (100)\", got)\n\t}\n}\n```",
          "output_characters": 5827,
          "output_sha256": "e732127893d14a4154962a20689a2b3af50aec30284b187f7692b065a70b82be",
          "prompt_tokens": 132,
          "reasoning_characters": 56460,
          "reasoning_sha256": "fbf47f91aa1b0cf3855246fce4ce7210f47d2ae9a94a61222cfd524a3f192e2a",
          "response_tail_seconds": 0.00814,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "2622771d77754097a00849bcb22faec1df63adc51a47afe003506d2be65aac06",
          "time_to_first_visible_seconds": 241.046261,
          "time_to_last_output_seconds": 256.494065,
          "ttft_seconds": 0.071816,
          "wall_seconds": 256.502205
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 18083,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 247.666124,
          "decode_tokens_per_second": 73.01,
          "finish_reason": "stop",
          "measured_sse_events": 18079,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time. Implement it to control time in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// SystemClock reports real wall-clock time.\ntype SystemClock struct{}\n\nfunc (SystemClock) Now() time.Time { return time.Now() }\n\n// Errors returned by New.\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be finite and > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// Limiter is a token-bucket rate limiter. The bucket starts full with burst\n// tokens and refills at rate tokens per second, capped at burst. All methods\n// are safe for concurrent use.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64 // tokens added per second\n\tburst  float64 // maximum tokens\n\ttokens float64 // available tokens\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter admitting bursts of up to burst events, refilling at\n// rate events per second. It returns ErrRate, ErrBurst, or ErrClock if the\n// corresponding parameter is invalid.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif !(rate > 0) || math.IsInf(rate, 0) { // !(rate > 0) is true for NaN too\n\t\treturn nil, ErrRate\n\t}\n\tif burst <= 0 {\n\t\treturn nil, ErrBurst\n\t}\n\tif clock == nil {\n\t\treturn nil, ErrClock\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single event is permitted.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n events are permitted at the clock's current time,\n// consuming n tokens if so. It returns false for negative n.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif d := now.Sub(l.last); d > 0 {\n\t\tl.tokens = math.Min(l.burst, l.tokens+d.Seconds()*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Tokens returns the tokens available as of the most recent AllowN call.\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct{ now time.Time }\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.now }\nfunc (f *fakeClock) Advance(d time.Duration) { f.now = f.now.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\tt.Parallel()\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\tclock   Clock\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 1, newFakeClock(), ErrRate},\n\t\t{\"negative rate\", -2.5, 1, newFakeClock(), ErrRate},\n\t\t{\"NaN rate\", math.NaN(), 1, newFakeClock(), ErrRate},\n\t\t{\"Inf rate\", math.Inf(1), 1, newFakeClock(), ErrRate},\n\t\t{\"zero burst\", 1, 0, newFakeClock(), ErrBurst},\n\t\t{\"negative burst\", 1, -3, newFakeClock(), ErrBurst},\n\t\t{\"nil clock\", 1, 1, nil, ErrClock},\n\t\t{\"valid\", 2, 10, newFakeClock(), nil},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tl, err := New(tt.rate, tt.burst, tt.clock)\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"New(%v, %d, _) error = %v, want %v\", tt.rate, tt.burst, err, tt.wantErr)\n\t\t\t}\n\t\t\tif tt.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"New() returned nil Limiter with nil error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\ntype step struct {\n\tadvance    time.Duration\n\tn          int\n\twant       bool\n\twantTokens float64\n\tcheckTok   bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt.Parallel()\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname: \"burst to capacity then deny\",\n\t\t\trate: 1, burst: 3,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 3, true, 0, false},\n\t\t\t\t{0, 1, false, 0, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\trate: 2, burst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 2, true, 0, false},\n\t\t\t\t{0, 1, false, 0, false},\n\t\t\t\t{time.Second, 2, true, 0, false},\n\t\t\t\t{250 * time.Millisecond, 1, false, 0.5, true},\n\t\t\t\t{250 * time.Millisecond, 1, true, 0, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at burst\",\n\t\t\trate: 10, burst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 2, true, 0, false},\n\t\t\t\t{time.Hour, 0, true, 2, true},\n\t\t\t\t{0, 2, true, 0, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"partial refill denies oversized request\",\n\t\t\trate: 1, burst: 5,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 5, true, 0, false},\n\t\t\t\t{500 * time.Millisecond, 2, false, 0.5, true},\n\t\t\t\t{500 * time.Millisecond, 1, true, 0, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"zero and negative requests\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 0, true, 1, true},\n\t\t\t\t{0, -1, false, 1, true},\n\t\t\t\t{0, 1, true, 0, false},\n\t\t\t},\n\t\t},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := New(tt.rate, tt.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New(%v, %d, _) error = %v\", tt.rate, tt.burst, err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t\tif s.checkTok && math.Abs(l.Tokens()-s.wantTokens) > 1e-9 {\n\t\t\t\t\tt.Fatalf(\"step %d: Tokens() = %v, want %v\", i, l.Tokens(), s.wantTokens)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tt.Parallel()\n\tl, err := New(1, 1, newFakeClock())\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"Allow() = true, want false\")\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tt.Parallel()\n\tconst burst, goroutines = 50, 200\n\tl, err := New(1, burst, newFakeClock())\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed atomic.Int64\n\tfor i := 0; i < goroutines; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\tallowed.Add(1)\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := allowed.Load(); got != burst {\n\t\tt.Fatalf(\"allowed = %d, want %d\", got, burst)\n\t}\n}\n```",
          "output_characters": 6227,
          "output_sha256": "394beb743b19e01f85d2ca1e8335e14cfe5a46a575d854e480cc2e23f714aa16",
          "prompt_tokens": 130,
          "reasoning_characters": 48811,
          "reasoning_sha256": "c361a6b6204eb043e430aa013ad718a03649a9c3f36c5dceeddc95ebc050a94b",
          "response_tail_seconds": 0.004966,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "15224156efec686c879bad1f1930ead13eb1980a2e3110837843c369ba019e8e",
          "time_to_first_visible_seconds": 231.447423,
          "time_to_last_output_seconds": 247.724933,
          "ttft_seconds": 0.058808,
          "wall_seconds": 247.729899
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 13639,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 189.880209,
          "decode_tokens_per_second": 71.824,
          "finish_reason": "stop",
          "measured_sse_events": 13629,
          "output": "**ratelimit.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"fmt\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time; inject a fake in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter. It starts full,\n// refills at rate tokens per second, and caps capacity at burst. Construct\n// with New; the zero value is not usable.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64 // tokens added per second\n\tburst  float64 // maximum token capacity\n\ttokens float64 // currently available tokens\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter with the given refill rate (tokens/second), burst\n// capacity, and clock. It errors if rate is not finite and > 0, burst < 1,\n// or clock is nil.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tswitch {\n\tcase clock == nil:\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\tcase math.IsNaN(rate) || math.IsInf(rate, 0) || rate <= 0:\n\t\treturn nil, fmt.Errorf(\"ratelimit: rate must be finite and > 0, got %v\", rate)\n\tcase burst < 1:\n\t\treturn nil, fmt.Errorf(\"ratelimit: burst must be >= 1, got %d\", burst)\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available, consuming it if so.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// n == 0 always returns true; n < 0 returns false.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif d := now.Sub(l.last).Seconds(); d > 0 {\n\t\tl.tokens = math.Min(l.burst, l.tokens+d*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (c *fakeClock) Now() time.Time {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\treturn c.now\n}\n\nfunc (c *fakeClock) Advance(d time.Duration) {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\tc.now = c.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\tcases := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\tclock   Clock\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 10, 5, newFakeClock(), false},\n\t\t{\"zero rate\", 0, 5, newFakeClock(), true},\n\t\t{\"negative rate\", -1, 5, newFakeClock(), true},\n\t\t{\"NaN rate\", math.NaN(), 5, newFakeClock(), true},\n\t\t{\"Inf rate\", math.Inf(1), 5, newFakeClock(), true},\n\t\t{\"zero burst\", 1, 0, newFakeClock(), true},\n\t\t{\"negative burst\", 1, -3, newFakeClock(), true},\n\t\t{\"nil clock\", 1, 5, nil, true},\n\t}\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, tc.clock)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %d) error = %v, wantErr %v\", tc.rate, tc.burst, err, tc.wantErr)\n\t\t\t}\n\t\t\tif err == nil && l == nil {\n\t\t\t\tt.Fatal(\"New returned nil limiter and nil error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tcases := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\"starts full, then denied when empty\", 10, 2, []step{\n\t\t\t{0, 2, true}, {0, 1, false},\n\t\t}},\n\t\t{\"partial refill insufficient\", 10, 2, []step{\n\t\t\t{0, 2, true}, {50 * time.Millisecond, 1, false},\n\t\t}},\n\t\t{\"full refill after burst period\", 10, 2, []step{\n\t\t\t{0, 2, true}, {200 * time.Millisecond, 2, true},\n\t\t}},\n\t\t{\"refill capped at burst\", 10, 2, []step{\n\t\t\t{0, 2, true}, {time.Hour, 3, false}, {0, 2, true},\n\t\t}},\n\t\t{\"zero request always allowed\", 10, 2, []step{\n\t\t\t{0, 2, true}, {0, 0, true},\n\t\t}},\n\t\t{\"negative request denied\", 10, 2, []step{\n\t\t\t{0, -1, false},\n\t\t}},\n\t\t{\"fractional accumulation\", 2, 1, []step{\n\t\t\t{0, 1, true}, {0, 1, false},\n\t\t\t{250 * time.Millisecond, 1, false}, {250 * time.Millisecond, 1, true},\n\t\t}},\n\t}\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := New(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New(%v, %d) error = %v\", tc.rate, tc.burst, err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tl, err := New(1, 10, newFakeClock())\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tvar allowed int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.Allow() {\n\t\t\t\tatomic.AddInt64(&allowed, 1)\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif got := atomic.LoadInt64(&allowed); got != 10 {\n\t\tt.Errorf(\"allowed = %d, want 10\", got)\n\t}\n}\n```",
          "output_characters": 5136,
          "output_sha256": "0d28bdac98320d3750874cfaf83df07a2d526deb468705ff83f3ff39ff8452c1",
          "prompt_tokens": 131,
          "reasoning_characters": 35737,
          "reasoning_sha256": "0bda248f2e2db217c7885a52b2e306d44480923fcf16036610799ea7da967ed8",
          "response_tail_seconds": 0.004017,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "4a98e5072307e5c6f0b366d07d6bc7580533f9cdcb557aeeb51ade9b7608c617",
          "time_to_first_visible_seconds": 176.011627,
          "time_to_last_output_seconds": 189.944687,
          "ttft_seconds": 0.064478,
          "wall_seconds": 189.948704
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 256.494,
        "median": 247.725,
        "minimum": 189.945,
        "p90": 256.494
      },
      "ttft_seconds": {
        "maximum": 0.072,
        "median": 0.064,
        "minimum": 0.059,
        "p90": 0.072
      },
      "wall_seconds": {
        "maximum": 256.502,
        "median": 247.73,
        "minimum": 189.949,
        "p90": 256.502
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 70.702,
        "median": 68.203,
        "minimum": 49.46,
        "p90": 70.702
      },
      "pooled_decode_tokens_per_second": 66.42,
      "runs": [
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 16670,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 235.765229,
          "decode_tokens_per_second": 70.702,
          "finish_reason": "stop",
          "measured_sse_events": 16665,
          "output": "**Internal memo**\n\n**To:** Platform & ML Engineering\n**From:** Infrastructure Lead\n**Date:** 12 May\n**Re:** GPU capacity \u2014 buy a third node, or raise utilization of the two we have\n\nOver the past quarter our two GPU nodes have gone from comfortably under-subscribed to visibly strained, and we face a genuine fork: order a third node at roughly $190,000 all-in, or spend six weeks of engineering time making better use of what we already own. My recommendation is to invest in utilization first, while keeping the purchase option genuinely open rather than treating it as a fallback we hope never to use.\n\nThe purchase is easy to price and hard to undo. A comparable four-GPU node runs about $190,000 in capital, plus power that at our rates comes to nearly a thousand dollars a month, plus rack space we have yet to confirm. The utilization project is not free either: two engineers for six weeks is roughly $60,000 of loaded time that cannot go toward product work. But that money is already inside the team, and unlike the node, the tooling keeps paying rent after the quarter ends.\n\nIt matters what we mean by utilization. Our dashboards show about thirty-five percent average compute utilization, but the more useful number is the phantom load: notebooks and dev servers that reserve an 80GB card and touch it a few percent of the time. Reclaiming that \u2014 idle timeouts, preemption with checkpointing, honest quotas \u2014 adds capacity without adding heat and without touching the latency of running jobs. That is not the same as packing the cluster hotter: queueing delay grows brutally near saturation, and our worst pain is the p95 afternoon wait, not the average. The goal is to remove waste, not to run at ninety percent.\n\nLatency is the reason we own this hardware. Local GPUs give us fast iteration loops, keep proprietary data on our own racks, and keep internal inference off someone else's rate limits. A third node buys latency headroom by adding capacity; the scheduling work buys it by clearing phantom load, which relieves the same peak-hour queues. Where the paths diverge is batching: raising throughput on the inference services means larger batches and slower individual responses. For anything user-facing I would keep batching conservative and accept the throughput we have.\n\nReliability cuts both ways. A new node is one more set of fans, power supplies and accelerators to RMA, one more machine to patch, and on a five-person team every addition is felt. It would also add failover capacity during maintenance, which we currently lack. The scheduler changes carry their own risk: a preemption bug that kills a twelve-hour training run would cost us more goodwill than an empty rack. So the work should ship behind flags, canary on the least critical queue, and assume long jobs are checkpointed before anything is evicted.\n\nThe asymmetry that decides this for me is reversibility. Capital spent on a node is committed for years: it depreciates, resells poorly, locks in power and rack commitments, and carries an opportunity cost \u2014 it is a hire we are not making or a dataset we are not licensing. The utilization investment is software \u2014 reversible, portable, and still valuable if we buy hardware later. The honest counterargument is that tuning delays capacity by weeks, and accelerator lead times make deferral costly in its own right. So we should do the reversible parts of procurement now \u2014 quotes, a power and cooling check, a budget line. That costs nothing and preserves the option.\n\nWe should be honest about what we do not know. Our telemetry may undercount idle reservations; the fine-tuning project planned for next quarter could shift demand in ways the averages hide; and the thirty percent of capacity we think we can reclaim is an estimate, not a measurement. Gains may not translate linearly into shorter peak waits, and may not materialize if the waste is structural rather than behavioral. We could be wrong in either direction, which is why this plan needs a checkpoint rather than a leap of faith.\n\nConcretely, then: two engineers, six weeks, one measurable target \u2014 cut peak-hour queue wait by half and lift effective capacity by at least a quarter. At the checkpoint we decide with data. If we hit the target, we defer the purchase and revisit next quarter; if we miss, or the demand forecast has firmed, we order the node without re-litigating. Either way we end the quarter with a cluster we understand better and a choice we can defend.",
          "output_characters": 4486,
          "output_sha256": "89888227024f9b6fb3e1db4d78eb3bcf2e1b7a6a79a9c58d60900c7943e6c90e",
          "prompt_tokens": 124,
          "reasoning_characters": 66212,
          "reasoning_sha256": "9c3026d416a76c5ebc4782ea59ac744154f61976856e6543e39582c8b80b3adc",
          "response_tail_seconds": 0.006973,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "a93defc992b295105f7c2541e371feea55b32eae34d4aa2e8dfcfa88e5e7369d",
          "time_to_first_visible_seconds": 229.316673,
          "time_to_last_output_seconds": 235.989956,
          "ttft_seconds": 0.224727,
          "wall_seconds": 235.996929
        },
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 10322,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 151.326999,
          "decode_tokens_per_second": 68.203,
          "finish_reason": "stop",
          "measured_sse_events": 10318,
          "output": "**Memo: GPU capacity \u2014 buy more, or use what we own better?**\n\nTo: Platform engineering (cc: Finance)\nFrom: Priya Natarajan, Infrastructure\nDate: 12 March\n\nOver the last two quarters, demand for our local GPU cluster has outgrown it in a specific way: training jobs queue for hours behind batch inference while the fleet sits mostly idle. Two proposals are on the table. One is to buy two more nodes, roughly $400k all-in once you include racks, power, and cooling. The other is to spend a quarter making better use of the hardware we already have. This memo recommends the second, with a pre-agreed condition under which we do the first.\n\nStart with cost, because the purchase is not just the invoice. New nodes bring depreciation, electricity, floor space, and ongoing integration work. The utilization project, by contrast, is about one engineer for one quarter \u2014 call it $90k fully loaded \u2014 plus the risk that it simply fails. The strongest argument for buying is also an argument against it: we average about 35% utilization, which means we already own more compute than we use in a typical hour. Until we fix the shape of our demand, the marginal value of another node is low. The caveat I take seriously is that averages hide peaks, and peaks are what people actually feel.\n\nWhat people feel is latency. Median wait for a fine-tuning job is four hours; the p95 is over a day, and the inference service misses its latency budget whenever a training job hogs a shared node. Much of this is queueing, which better scheduling attacks directly: backfill, preemption, and a clean separation between the serving and training pools. Some of it is not. Our serving path already batches aggressively, and pushing utilization higher there would trade throughput for per-request latency in a way product has explicitly ruled out. So the utilization work should target the queue, not squeeze the serving path. Buying hardware would fix queueing immediately, and it is the only fix if our jobs are genuinely concurrent at peak \u2014 a possibility we cannot yet rule out.\n\nReliability cuts both ways, and I want to be honest about that. More nodes mean more power supplies, NICs, fans, and ECC errors to babysit; our second node already generates a disproportionate share of tickets. Consolidating load onto fewer machines raises the blast radius of any single failure. But the same scheduling discipline that raises utilization \u2014 clean draining, live migration, honest health checks \u2014 is also what lets us take nodes down for maintenance without ceremony. Today we avoid reboots because we are afraid of the queue. That is not reliability; that is luck, and luck has been trending badly.\n\nThen there is opportunity cost, on both sides of the ledger. The engineer who does this work is not shipping product, a real cost even if it never hits a budget line. The $400k is real too, and it is not only money: GPUs depreciate quickly, and in eighteen months today's node will be worth a fraction of its price. The accelerator landscape is also shifting \u2014 new memory classes, new form factors \u2014 so capital spent now is a bet on next year's workloads made with this year's information. Deferring the purchase spends nothing irreversible and keeps that bet open.\n\nThat is the crux of my recommendation: reversibility. Software and policy changes are cheap to undo, and rented burst capacity is a genuinely reversible middle path for genuine peaks. Hardware is not reversible; resale is slow and lossy. So we should do the reversible things first and buy only what irreversibility actually justifies.\n\nConcretely: fund one engineer for one quarter to fix queueing policy, separate the serving and training pools, and instrument utilization properly, because our current numbers come from an exporter we only half trust. The success criteria are fleet utilization above 60% during business hours, training queue p95 under two hours, and no regression in inference latency. If we miss two of the three, we buy one node \u2014 agreed now, so we are not relitigating this in June. Let me name the uncertainty plainly: demand may spike when the fine-tuning push lands next month, utilization gains may not materialize at peak even if averages improve, and the scheduling work could stall on the storage bottleneck we've been ignoring. But the cost of being wrong here is a quarter of one engineer's time, not $400k of depreciating silicon. That asymmetry is the whole argument.\n\nI'd like a decision by Friday. Happy to walk through the numbers with anyone.",
          "output_characters": 4526,
          "output_sha256": "92538ae11c027a579afed467f9c0d98f18851df50c9233ee382cde107da88dcc",
          "prompt_tokens": 120,
          "reasoning_characters": 39884,
          "reasoning_sha256": "fdb19745e1bda1ca7462f882c77d1e87fd475e0e57aa08ebed3031fff4f07ca1",
          "response_tail_seconds": 0.006583,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "43b1435adb28a341041f0e80b8c87fecc8b6a64ad5a10123a57b8352a0556bb5",
          "time_to_first_visible_seconds": 144.116538,
          "time_to_last_output_seconds": 151.543562,
          "ttft_seconds": 0.216562,
          "wall_seconds": 151.550145
        },
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 3732,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 75.435219,
          "decode_tokens_per_second": 49.46,
          "finish_reason": "stop",
          "measured_sse_events": 3730,
          "output": "**Internal Memo**\n\n**To:** Compute & Platform team\n**From:** Dana Reyes, Engineering Lead\n**Date:** 14 March\n**Subject:** GPU capacity \u2014 buy more, or use what we own better?\n\nOver the past two months our four GPU nodes have gone from comfortably underused to visibly congested. Training jobs queue for hours midweek, two researchers have taken to running experiments at midnight to catch idle windows, and our inference latency has held up only because we keep one node effectively reserved. The obvious fix \u2014 order two more nodes \u2014 is on the table, and I want us to decide deliberately rather than by default.\n\nStart with cost, because the framing matters. A new node is roughly $180k all-in, plus about $1,500 a month in power and cooling, and it depreciates fast; the resale market for used accelerators is thin. The alternative path costs engineering time: one engineer for a quarter is roughly the same money. So the real comparison is not hardware versus nothing, but capital versus attention \u2014 and attention is our scarcer asset. That said, capital is not free either, and every dollar locked into silicon is a dollar unavailable for hiring or the next experiment.\n\nLatency cuts both ways, and it is the reason a single fleet-wide answer is wrong. Our interactive inference services need headroom; packing them tighter will degrade tail latency in ways users will feel before our dashboards do. Batch training, by contrast, tolerates queueing just fine. This suggests we should segment workloads: reserve capacity for latency-sensitive serving, and schedule everything else aggressively. That reframes the question from \"how many GPUs do we need\" to \"how much reserved capacity does serving actually require\" \u2014 a question we can answer with telemetry we already collect.\n\nReliability is the counterweight. Slack is what absorbs failures, and a fleet running at 85% utilization turns one dead node into a multi-day queue backlog. Buying more hardware adds failure modes, but it can also add redundancy if we deliberately keep an N+1 margin. Any utilization push needs a ceiling \u2014 I'd propose 70\u201375% fleet-wide \u2014 below which we retain enough headroom for maintenance windows and faults. If we do end up buying, one of the nodes should purchase resilience, not raw throughput.\n\nBoth options carry opportunity costs, and we should name them honestly. An engineer spent on scheduling is an engineer not spent on the product. But idle hardware overnight is value we are already paying for and discarding, and slow experiments have their own hidden price: researchers waiting in queue ship less, and iteration speed compounds. The cheapest capacity available to us is the capacity we already own and waste.\n\nReversibility tips the balance. Software and process changes can be undone; hardware cannot, not without loss. Power contracts, rack space, and depreciation make a purchase a multi-year commitment, and there is an organizational ratchet too \u2014 once teams grow used to generous capacity, taking it back is politically harder than never granting it. Utilization work, meanwhile, leaves durable assets behind: a proper priority queue, better telemetry, habits of right-sizing jobs. Those pay off even if we buy hardware later.\n\nSo here is my recommendation. We time-box a utilization effort: one engineer, one quarter, with explicit targets \u2014 raise fleet utilization from roughly 35% to 60%, cut p95 training queue time from hours to under fifteen minutes, and hold inference tail latency flat. Concretely, that means segmenting serving from batch, introducing a priority queue, shifting batch work to off-peak windows, and right-sizing the oversized jobs we've all been guilty of submitting. We defer the purchase decision to the end of that quarter, with tripwires: if our demand forecast exceeds capacity at the 70% ceiling, if we miss the mid-quarter checkpoint, or if a new project lands and changes the picture, we order one node immediately \u2014 vendor lead time is six weeks, so we can start procurement without waiting for the quarter to close.\n\nI want to be honest about the uncertainty here. Our demand forecast is essentially a guess; utilization gains may stall on jobs that cannot be preempted; and a promising model line could double our compute needs overnight. If that happens, buying is unambiguously right, and we will have spent a quarter of one engineer and some queue pain to learn it \u2014 cheap tuition. The asymmetry favors trying the reversible, cheaper option first, and it leaves us better informed whichever way we go.\n\nPlease send comments or objections by Thursday; I'd like to kick off the following Monday.",
          "output_characters": 4636,
          "output_sha256": "60b66fa931e084543f9dedbb5a42d779f85e1be2bc21509ff67420e6fc8dc502",
          "prompt_tokens": 124,
          "reasoning_characters": 12892,
          "reasoning_sha256": "b3ab7942793b74e6efc7799ae3fbfa048ab563943605afc6ddbe1cd3162e0f7b",
          "response_tail_seconds": 0.003701,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "b49eef9d711d20b8a0c7ece6457330e7ae662be090a362ea6e9de46d6d52dea3",
          "time_to_first_visible_seconds": 60.083775,
          "time_to_last_output_seconds": 75.65765,
          "ttft_seconds": 0.222431,
          "wall_seconds": 75.661351
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 235.99,
        "median": 151.544,
        "minimum": 75.658,
        "p90": 235.99
      },
      "ttft_seconds": {
        "maximum": 0.225,
        "median": 0.222,
        "minimum": 0.217,
        "p90": 0.225
      },
      "wall_seconds": {
        "maximum": 235.997,
        "median": 151.55,
        "minimum": 75.661,
        "p90": 235.997
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 102.326,
        "median": 102.167,
        "minimum": 101.624,
        "p90": 102.326
      },
      "pooled_decode_tokens_per_second": 102.033,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1027,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 10.042372,
          "decode_tokens_per_second": 102.167,
          "finish_reason": "stop",
          "measured_sse_events": 1022,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 111,
          "reasoning_characters": 623,
          "reasoning_sha256": "efe779badef235a76affb2148a0ffee3f78f250c1e369fe6b24271bd1cfee954",
          "response_tail_seconds": 0.002093,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "ca3e49b98781e7b79632b4947f225dd278dfac8fa62a99c57f9bc3a6051ffcf8",
          "time_to_first_visible_seconds": 4.461759,
          "time_to_last_output_seconds": 10.513697,
          "ttft_seconds": 0.471325,
          "wall_seconds": 10.515791
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 9.538129,
          "decode_tokens_per_second": 102.326,
          "finish_reason": "stop",
          "measured_sse_events": 972,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.003013,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 3.945513,
          "time_to_last_output_seconds": 10.006126,
          "ttft_seconds": 0.467997,
          "wall_seconds": 10.009139
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1027,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 10.096058,
          "decode_tokens_per_second": 101.624,
          "finish_reason": "stop",
          "measured_sse_events": 1022,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 623,
          "reasoning_sha256": "efe779badef235a76affb2148a0ffee3f78f250c1e369fe6b24271bd1cfee954",
          "response_tail_seconds": 0.002612,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "ca3e49b98781e7b79632b4947f225dd278dfac8fa62a99c57f9bc3a6051ffcf8",
          "time_to_first_visible_seconds": 4.514661,
          "time_to_last_output_seconds": 10.563931,
          "ttft_seconds": 0.467874,
          "wall_seconds": 10.566543
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 10.564,
        "median": 10.514,
        "minimum": 10.006,
        "p90": 10.564
      },
      "ttft_seconds": {
        "maximum": 0.471,
        "median": 0.468,
        "minimum": 0.468,
        "p90": 0.471
      },
      "wall_seconds": {
        "maximum": 10.567,
        "median": 10.516,
        "minimum": 10.009,
        "p90": 10.567
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2014.626,
          "median": 2009.469,
          "minimum": 2009.294,
          "p90": 2014.626
        },
        "runs": [
          {
            "cache_salt_sha256": "8a70de5811ec8a0ae9575bb10bfa3b6639c54e515053edefeef8bede0db56248",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064453,
            "decode_tokens_per_second": 108.607,
            "effective_prefill_tokens_per_second": 2009.469,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002315,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 16.306799,
            "time_to_last_output_seconds": 16.371252,
            "ttft_seconds": 16.306799,
            "wall_seconds": 16.373567
          },
          {
            "cache_salt_sha256": "91c4a80c98dbd154f71b3c53c2c7b7c26c449c87e43376ec8a8d2b6717a3284b",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055585,
            "decode_tokens_per_second": 125.933,
            "effective_prefill_tokens_per_second": 2014.626,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001478,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 16.265057,
            "time_to_last_output_seconds": 16.320642,
            "ttft_seconds": 16.265057,
            "wall_seconds": 16.322121
          },
          {
            "cache_salt_sha256": "d8ae6642a37f817fb0535566370e546dd39651070b1683ce74aef7bd4e852e72",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061069,
            "decode_tokens_per_second": 114.625,
            "effective_prefill_tokens_per_second": 2009.294,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001614,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.308212,
            "time_to_last_output_seconds": 16.369281,
            "ttft_seconds": 16.308212,
            "wall_seconds": 16.370894
          }
        ],
        "ttft_seconds": {
          "maximum": 16.308,
          "median": 16.307,
          "minimum": 16.265,
          "p90": 16.308
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 741659.499,
          "median": 677290.672,
          "minimum": 583569.304,
          "p90": 741659.499
        },
        "runs": [
          {
            "cache_salt_sha256": "8a70de5811ec8a0ae9575bb10bfa3b6639c54e515053edefeef8bede0db56248",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.064832,
            "decode_tokens_per_second": 107.972,
            "effective_prefill_tokens_per_second": 583569.304,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002372,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.056151,
            "time_to_last_output_seconds": 0.120983,
            "ttft_seconds": 0.056151,
            "wall_seconds": 0.123354
          },
          {
            "cache_salt_sha256": "91c4a80c98dbd154f71b3c53c2c7b7c26c449c87e43376ec8a8d2b6717a3284b",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.055513,
            "decode_tokens_per_second": 126.097,
            "effective_prefill_tokens_per_second": 741659.499,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00102,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 0.044182,
            "time_to_last_output_seconds": 0.099694,
            "ttft_seconds": 0.044182,
            "wall_seconds": 0.100714
          },
          {
            "cache_salt_sha256": "d8ae6642a37f817fb0535566370e546dd39651070b1683ce74aef7bd4e852e72",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.059598,
            "decode_tokens_per_second": 117.454,
            "effective_prefill_tokens_per_second": 677290.672,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002081,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.048381,
            "time_to_last_output_seconds": 0.107979,
            "ttft_seconds": 0.048381,
            "wall_seconds": 0.11006
          }
        ],
        "ttft_seconds": {
          "maximum": 0.056,
          "median": 0.048,
          "minimum": 0.044,
          "p90": 0.056
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1988.912,
          "median": 1987.248,
          "minimum": 1986.299,
          "p90": 1988.912
        },
        "runs": [
          {
            "cache_salt_sha256": "a2f71cc719fb0cd687d0523cd11a33e6c5a59d244b6dc6d7a30c05b4eed70f76",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.065128,
            "decode_tokens_per_second": 107.481,
            "effective_prefill_tokens_per_second": 1987.248,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002276,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 32.978274,
            "time_to_last_output_seconds": 33.043401,
            "ttft_seconds": 32.978274,
            "wall_seconds": 33.045677
          },
          {
            "cache_salt_sha256": "328742763b4fe369c3c5ba1a5d558cc6e4fc0370816dc75ab8aa3e44f245a56c",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061736,
            "decode_tokens_per_second": 113.387,
            "effective_prefill_tokens_per_second": 1986.299,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001767,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 32.994024,
            "time_to_last_output_seconds": 33.055759,
            "ttft_seconds": 32.994024,
            "wall_seconds": 33.057526
          },
          {
            "cache_salt_sha256": "f2b9e3cf3329507fcf349cec12a303e5aadc351e3c42d991cae72fcd267a0616",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061711,
            "decode_tokens_per_second": 113.432,
            "effective_prefill_tokens_per_second": 1988.912,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ", and the next action. The",
            "output_characters": 26,
            "output_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002087,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "time_to_first_visible_seconds": 32.950671,
            "time_to_last_output_seconds": 33.012382,
            "ttft_seconds": 32.950671,
            "wall_seconds": 33.014469
          }
        ],
        "ttft_seconds": {
          "maximum": 32.994,
          "median": 32.978,
          "minimum": 32.951,
          "p90": 32.994
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1163243.934,
          "median": 974831.915,
          "minimum": 968507.544,
          "p90": 1163243.934
        },
        "runs": [
          {
            "cache_salt_sha256": "a2f71cc719fb0cd687d0523cd11a33e6c5a59d244b6dc6d7a30c05b4eed70f76",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064841,
            "decode_tokens_per_second": 107.956,
            "effective_prefill_tokens_per_second": 968507.544,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001145,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.067667,
            "time_to_last_output_seconds": 0.132508,
            "ttft_seconds": 0.067667,
            "wall_seconds": 0.133653
          },
          {
            "cache_salt_sha256": "328742763b4fe369c3c5ba1a5d558cc6e4fc0370816dc75ab8aa3e44f245a56c",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062872,
            "decode_tokens_per_second": 111.338,
            "effective_prefill_tokens_per_second": 974831.915,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001888,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.067228,
            "time_to_last_output_seconds": 0.130099,
            "ttft_seconds": 0.067228,
            "wall_seconds": 0.131987
          },
          {
            "cache_salt_sha256": "f2b9e3cf3329507fcf349cec12a303e5aadc351e3c42d991cae72fcd267a0616",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062732,
            "decode_tokens_per_second": 111.585,
            "effective_prefill_tokens_per_second": 1163243.934,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ", and the next action. The",
            "output_characters": 26,
            "output_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00145,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3b0dd9516539cec218f5d1bcadfcc66d63a74171bab17c2c0b040abaae9f023d",
            "time_to_first_visible_seconds": 0.056339,
            "time_to_last_output_seconds": 0.119071,
            "ttft_seconds": 0.056339,
            "wall_seconds": 0.120521
          }
        ],
        "ttft_seconds": {
          "maximum": 0.068,
          "median": 0.067,
          "minimum": 0.056,
          "p90": 0.068
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1971.493,
          "median": 1966.697,
          "minimum": 1480.456,
          "p90": 1971.493
        },
        "runs": [
          {
            "cache_salt_sha256": "fb380d179454e1b99c1d052aa6b449bc5bb7bbd2959b95c6011de77c0d6a32cb",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054336,
            "decode_tokens_per_second": 128.827,
            "effective_prefill_tokens_per_second": 1480.456,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001994,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 5.533429,
            "time_to_last_output_seconds": 5.587766,
            "ttft_seconds": 5.533429,
            "wall_seconds": 5.58976
          },
          {
            "cache_salt_sha256": "90668c49ac5a01445393b7e34638601a1dde33cca78eddecf5782877b306d51d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.05661,
            "decode_tokens_per_second": 123.653,
            "effective_prefill_tokens_per_second": 1966.697,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". Each observation records behaviour, evidence,",
            "output_characters": 47,
            "output_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.00188,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "time_to_first_visible_seconds": 4.165359,
            "time_to_last_output_seconds": 4.221969,
            "ttft_seconds": 4.165359,
            "wall_seconds": 4.223849
          },
          {
            "cache_salt_sha256": "ef97e2a245a6dec341ea0af1b0894735688ac70a2ea54c65a58cf4f2badd6128",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053781,
            "decode_tokens_per_second": 130.157,
            "effective_prefill_tokens_per_second": 1971.493,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001397,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 4.155227,
            "time_to_last_output_seconds": 4.209008,
            "ttft_seconds": 4.155227,
            "wall_seconds": 4.210405
          }
        ],
        "ttft_seconds": {
          "maximum": 5.533,
          "median": 4.165,
          "minimum": 4.155,
          "p90": 5.533
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 239826.688,
          "median": 237690.411,
          "minimum": 208341.811,
          "p90": 239826.688
        },
        "runs": [
          {
            "cache_salt_sha256": "fb380d179454e1b99c1d052aa6b449bc5bb7bbd2959b95c6011de77c0d6a32cb",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.054349,
            "decode_tokens_per_second": 128.797,
            "effective_prefill_tokens_per_second": 237690.411,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.002316,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 0.034465,
            "time_to_last_output_seconds": 0.088814,
            "ttft_seconds": 0.034465,
            "wall_seconds": 0.091129
          },
          {
            "cache_salt_sha256": "90668c49ac5a01445393b7e34638601a1dde33cca78eddecf5782877b306d51d",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.056605,
            "decode_tokens_per_second": 123.664,
            "effective_prefill_tokens_per_second": 239826.688,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". Each observation records behaviour, evidence,",
            "output_characters": 47,
            "output_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001525,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "0b845b81e55b13e918b7fb93d9d0053b186b565a143bfcbabc3e5e031ff0c973",
            "time_to_first_visible_seconds": 0.034158,
            "time_to_last_output_seconds": 0.090763,
            "ttft_seconds": 0.034158,
            "wall_seconds": 0.092288
          },
          {
            "cache_salt_sha256": "ef97e2a245a6dec341ea0af1b0894735688ac70a2ea54c65a58cf4f2badd6128",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.052596,
            "decode_tokens_per_second": 133.091,
            "effective_prefill_tokens_per_second": 208341.811,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.002191,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 0.03932,
            "time_to_last_output_seconds": 0.091915,
            "ttft_seconds": 0.03932,
            "wall_seconds": 0.094107
          }
        ],
        "ttft_seconds": {
          "maximum": 0.039,
          "median": 0.034,
          "minimum": 0.034,
          "p90": 0.039
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "incoai/GLM-5.3-Flash-DFlash2 @ bf582e4e (Mia v1.10 default settings)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "Mia-AiLab/GLM-5.3-Flash-EXL3-4bpw-TensorFold",
      "model_revision": "078455ffe6472f9a52fbc1139f58b9db2881b25c",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-dflash2-max",
    "finished_at": "2026-10-08T11:22:46.386929+00:00",
    "label": "tp2-stock-dflash2-max",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T10:59:16.678448+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 32768,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "max"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 3,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}
