{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 52.594,
        "median": 51.852,
        "minimum": 50.739,
        "p90": 52.594
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 57.46,
        "median": 56.645,
        "minimum": 56.396,
        "p90": 57.46
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.523,
        "median": 0.435,
        "minimum": 0.429,
        "p90": 0.523
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 51.852,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.501687,
              "decode_tokens_per_second": 56.645,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.000119,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.936324,
              "ttft_seconds": 0.434637,
              "wall_seconds": 4.936442
            }
          ],
          "wall_seconds": 4.937113
        },
        {
          "aggregate_end_to_end_tokens_per_second": 50.739,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.521563,
              "decode_tokens_per_second": 56.396,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 129,
              "reasoning_characters": 981,
              "reasoning_sha256": "5c1fcb4487ce97b4d79a55bb5ab973948ccca1681c100f98fafb9d2e7e0f6864",
              "response_tail_seconds": 9e-05,
              "stream_done_marker": true,
              "stream_sha256": "5c1fcb4487ce97b4d79a55bb5ab973948ccca1681c100f98fafb9d2e7e0f6864",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.044535,
              "ttft_seconds": 0.522972,
              "wall_seconds": 5.044625
            }
          ],
          "wall_seconds": 5.045438
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.594,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.437881,
              "decode_tokens_per_second": 57.46,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 134,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.0001,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.866776,
              "ttft_seconds": 0.428895,
              "wall_seconds": 4.866877
            }
          ],
          "wall_seconds": 4.867499
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 54.534,
        "median": 53.411,
        "minimum": 52.05,
        "p90": 54.534
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 61.035,
        "median": 58.047,
        "minimum": 55.598,
        "p90": 58.53
      },
      "per_stream_ttft_seconds": {
        "maximum": 5.434,
        "median": 2.713,
        "minimum": 0.417,
        "p90": 5.189
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 53.411,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.395403,
              "decode_tokens_per_second": 58.015,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1010,
              "reasoning_sha256": "4c6d4fe6574ec80b11ab574da557430cb32ceb6cfea4b39909476663d9fa257d",
              "response_tail_seconds": 0.000116,
              "stream_done_marker": true,
              "stream_sha256": "4c6d4fe6574ec80b11ab574da557430cb32ceb6cfea4b39909476663d9fa257d",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.584247,
              "ttft_seconds": 5.188843,
              "wall_seconds": 9.584363
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.356747,
              "decode_tokens_per_second": 58.53,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1024,
              "reasoning_sha256": "b29f7c955609cb370a479f510eb4c733ffb0e1f6f31ada880ae2be43bf2c6916",
              "response_tail_seconds": 0.000106,
              "stream_done_marker": true,
              "stream_sha256": "b29f7c955609cb370a479f510eb4c733ffb0e1f6f31ada880ae2be43bf2c6916",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.773686,
              "ttft_seconds": 0.416939,
              "wall_seconds": 4.773792
            }
          ],
          "wall_seconds": 9.586026
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.534,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.177954,
              "decode_tokens_per_second": 61.035,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1010,
              "reasoning_sha256": "2e36ddec1360ab4901d3186ff7833f8f1d32788dc901811ebf26026dcdb0b7a2",
              "response_tail_seconds": 9e-05,
              "stream_done_marker": true,
              "stream_sha256": "2e36ddec1360ab4901d3186ff7833f8f1d32788dc901811ebf26026dcdb0b7a2",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.594538,
              "ttft_seconds": 0.416583,
              "wall_seconds": 4.594628
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.390558,
              "decode_tokens_per_second": 58.079,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 130,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 8.2e-05,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.387485,
              "ttft_seconds": 4.996926,
              "wall_seconds": 9.387567
            }
          ],
          "wall_seconds": 9.388611
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.05,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.400283,
              "decode_tokens_per_second": 57.951,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 990,
              "reasoning_sha256": "967267e819437aa79e4447486138e5e1d4e20f1cc108109f7f2b0f7dcf8a0e71",
              "response_tail_seconds": 0.000324,
              "stream_done_marker": true,
              "stream_sha256": "967267e819437aa79e4447486138e5e1d4e20f1cc108109f7f2b0f7dcf8a0e71",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.834768,
              "ttft_seconds": 5.434486,
              "wall_seconds": 9.835093
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.586521,
              "decode_tokens_per_second": 55.598,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 996,
              "reasoning_sha256": "c4f5c0b26899a3eaa0cb5d3041b0c558d86e04646c5046a126b330d0036f47ef",
              "response_tail_seconds": 9.1e-05,
              "stream_done_marker": true,
              "stream_sha256": "c4f5c0b26899a3eaa0cb5d3041b0c558d86e04646c5046a126b330d0036f47ef",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.014999,
              "ttft_seconds": 0.428477,
              "wall_seconds": 5.01509
            }
          ],
          "wall_seconds": 9.836732
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 52.849,
        "median": 52.551,
        "minimum": 52.543,
        "p90": 52.849
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 60.255,
        "median": 58.014,
        "minimum": 53.589,
        "p90": 59.316
      },
      "per_stream_ttft_seconds": {
        "maximum": 15.185,
        "median": 7.859,
        "minimum": 0.427,
        "p90": 15.142
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 52.543,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.758439,
              "decode_tokens_per_second": 53.589,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1008,
              "reasoning_sha256": "0aba6adcebe27ab93980400b8953d57562a0babf0a8020abe05bc58729aceb9a",
              "response_tail_seconds": 0.000476,
              "stream_done_marker": true,
              "stream_sha256": "0aba6adcebe27ab93980400b8953d57562a0babf0a8020abe05bc58729aceb9a",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 5.187899,
              "ttft_seconds": 0.429459,
              "wall_seconds": 5.188375
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.349462,
              "decode_tokens_per_second": 58.628,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 1010,
              "reasoning_sha256": "324c55ca351258ae0a3cbe97e84b9ebbb95b7312e5dc7f99672ef453c3065849",
              "response_tail_seconds": 0.000632,
              "stream_done_marker": true,
              "stream_sha256": "324c55ca351258ae0a3cbe97e84b9ebbb95b7312e5dc7f99672ef453c3065849",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 19.486,
              "ttft_seconds": 15.136538,
              "wall_seconds": 19.486632
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.358882,
              "decode_tokens_per_second": 58.501,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 135,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.000398,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.976165,
              "ttft_seconds": 5.617283,
              "wall_seconds": 9.976563
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.31109,
              "decode_tokens_per_second": 59.15,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 0.000103,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 14.703172,
              "ttft_seconds": 10.392082,
              "wall_seconds": 14.703276
            }
          ],
          "wall_seconds": 19.488897
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.551,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.299023,
              "decode_tokens_per_second": 59.316,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 8.6e-05,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 19.483768,
              "ttft_seconds": 15.184745,
              "wall_seconds": 19.483854
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.458959,
              "decode_tokens_per_second": 57.188,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 1017,
              "reasoning_sha256": "eb8fdbfd0b66bf6e60944ac04febf34468f6bf60dce90deb747a820d539f7075",
              "response_tail_seconds": 0.000473,
              "stream_done_marker": true,
              "stream_sha256": "eb8fdbfd0b66bf6e60944ac04febf34468f6bf60dce90deb747a820d539f7075",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.728604,
              "ttft_seconds": 5.269645,
              "wall_seconds": 9.729076
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.605794,
              "decode_tokens_per_second": 55.365,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 135,
              "reasoning_characters": 1052,
              "reasoning_sha256": "e26301cfb3d01e3d8ec7476b06ecdca7d4da1e5523621e4932d448c731f4d29c",
              "response_tail_seconds": 0.000445,
              "stream_done_marker": true,
              "stream_sha256": "e26301cfb3d01e3d8ec7476b06ecdca7d4da1e5523621e4932d448c731f4d29c",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 14.764266,
              "ttft_seconds": 10.158473,
              "wall_seconds": 14.764712
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.429509,
              "decode_tokens_per_second": 57.568,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1011,
              "reasoning_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "response_tail_seconds": 8.6e-05,
              "stream_done_marker": true,
              "stream_sha256": "86b338cc65b15c9a09a57d8a49f76db483f7ac97577c9582b4f024a2004b92ac",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.856974,
              "ttft_seconds": 0.427465,
              "wall_seconds": 4.85706
            }
          ],
          "wall_seconds": 19.485927
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.849,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.232027,
              "decode_tokens_per_second": 60.255,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 136,
              "reasoning_characters": 1017,
              "reasoning_sha256": "d967ae4c4739141c4448f9d2f344027523f28c98545a525286cc2d30ae22ae9a",
              "response_tail_seconds": 0.0001,
              "stream_done_marker": true,
              "stream_sha256": "d967ae4c4739141c4448f9d2f344027523f28c98545a525286cc2d30ae22ae9a",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 19.373586,
              "ttft_seconds": 15.141559,
              "wall_seconds": 19.373686
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.475551,
              "decode_tokens_per_second": 56.976,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 133,
              "reasoning_characters": 1016,
              "reasoning_sha256": "986da19753aa24fc19b2dcd1bbbbb7d5d5a2d519c6fcdd83cf75cd36b907d1a7",
              "response_tail_seconds": 0.00016,
              "stream_done_marker": true,
              "stream_sha256": "986da19753aa24fc19b2dcd1bbbbb7d5d5a2d519c6fcdd83cf75cd36b907d1a7",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 4.91138,
              "ttft_seconds": 0.435829,
              "wall_seconds": 4.91154
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.361963,
              "decode_tokens_per_second": 58.46,
              "finish_reason": "length",
              "measured_sse_events": 256,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 132,
              "reasoning_characters": 1021,
              "reasoning_sha256": "476eb649be2ce204a931bfee5fbd719bffcad37628e685fd733ff33dc6e1188a",
              "response_tail_seconds": 0.000754,
              "stream_done_marker": true,
              "stream_sha256": "476eb649be2ce204a931bfee5fbd719bffcad37628e685fd733ff33dc6e1188a",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 9.689217,
              "ttft_seconds": 5.327253,
              "wall_seconds": 9.68997
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.608537,
              "decode_tokens_per_second": 55.332,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "",
              "output_characters": 0,
              "output_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "prompt_tokens": 131,
              "reasoning_characters": 995,
              "reasoning_sha256": "484a1b785ef4789818ff8480d46586f5f5336b1b772be7a5c4db4f7962140e75",
              "response_tail_seconds": 8.7e-05,
              "stream_done_marker": true,
              "stream_sha256": "484a1b785ef4789818ff8480d46586f5f5336b1b772be7a5c4db4f7962140e75",
              "time_to_first_visible_seconds": null,
              "time_to_last_output_seconds": 14.708357,
              "ttft_seconds": 10.09982,
              "wall_seconds": 14.708444
            }
          ],
          "wall_seconds": 19.375997
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 65.286,
        "median": 63.359,
        "minimum": 61.977,
        "p90": 65.286
      },
      "pooled_decode_tokens_per_second": 63.608,
      "runs": [
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 12702,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 204.929446,
          "decode_tokens_per_second": 61.977,
          "finish_reason": "stop",
          "measured_sse_events": 12695,
          "output": "ratelimit.go\n```go\n// Package ratelimit provides a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype systemClock struct{}\n\nfunc (systemClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket limiter. The bucket starts full, holds at most\n// burst tokens, and refills at rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter that uses the real system clock.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn newLimiter(rate, burst, systemClock{})\n}\n\n// NewWithClock returns a Limiter driven by clock, for testing.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\treturn newLimiter(rate, burst, clock)\n}\n\nfunc newLimiter(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif !(rate > 0) || math.IsInf(rate, 0) {\n\t\treturn nil, errors.New(\"ratelimit: rate must be finite and > 0\")\n\t}\n\tif burst <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: burst must be positive\")\n\t}\n\tif clock == nil {\n\t\treturn nil, errors.New(\"ratelimit: clock must not be nil\")\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available and, if so, consumes them.\n// It returns false for n <= 0. Time is read from the limiter's clock.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n <= 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif elapsed := now.Sub(l.last); elapsed > 0 { // ignore backward clock steps\n\t\tl.tokens = min(l.burst, l.tokens+elapsed.Seconds()*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n```\n\nratelimit_test.go\n```go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock. Not safe for concurrent use.\ntype fakeClock struct {\n\tnow time.Time\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.now }\n\nfunc (f *fakeClock) Advance(d time.Duration) { f.now = f.now.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tclock Clock\n\t}{\n\t\t{\"zero rate\", 0, 1, &fakeClock{}},\n\t\t{\"negative rate\", -1, 1, &fakeClock{}},\n\t\t{\"NaN rate\", math.NaN(), 1, &fakeClock{}},\n\t\t{\"zero burst\", 1, 0, &fakeClock{}},\n\t\t{\"negative burst\", 1, -1, &fakeClock{}},\n\t\t{\"nil clock\", 1, 1, nil},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tif _, err := NewWithClock(tt.rate, tt.burst, tt.clock); err == nil {\n\t\t\t\tt.Errorf(\"NewWithClock(%v, %d, %v) error = nil, want error\", tt.rate, tt.burst, tt.clock)\n\t\t\t}\n\t\t})\n\t}\n\tif _, err := New(1, 1); err != nil {\n\t\tt.Errorf(\"New(1, 1) error = %v, want nil\", err)\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\"starts full, then empty\", 1, 3, []step{\n\t\t\t{0, 3, true}, {0, 1, false}}},\n\t\t{\"refills with time\", 1, 2, []step{\n\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t{time.Second, 1, true}, {0, 1, false}}},\n\t\t{\"partial refill rejected\", 2, 1, []step{\n\t\t\t{0, 1, true}, {250 * time.Millisecond, 1, false},\n\t\t\t{250 * time.Millisecond, 1, true}}},\n\t\t{\"refill capped at burst\", 10, 2, []step{\n\t\t\t{time.Hour, 2, true}, {0, 1, false}}},\n\t\t{\"n above burst never allowed\", 1, 2, []step{\n\t\t\t{time.Hour, 3, false}}},\n\t\t{\"non-positive n rejected\", 1, 1, []step{\n\t\t\t{0, 0, false}, {0, -1, false}}},\n\t\t{\"backward clock adds nothing\", 1, 1, []step{\n\t\t\t{0, 1, true}, {-time.Second, 1, false}, {time.Second, 1, true}}},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tclk := &fakeClock{now: time.Unix(0, 0)}\n\t\t\tl, err := NewWithClock(tt.rate, tt.burst, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock(%v, %d) error = %v\", tt.rate, tt.burst, err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tl, err := New(1000, 100)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tstart := time.Now()\n\tvar allowed atomic.Int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 64; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 100; j++ {\n\t\t\t\tif l.AllowN(1) {\n\t\t\t\t\tallowed.Add(1)\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tlimit := 100 + int64(time.Since(start).Seconds()*1000) + 1\n\tif got := allowed.Load(); got > limit {\n\t\tt.Errorf(\"allowed %d tokens, want <= %d\", got, limit)\n\t}\n\tif allowed.Load() == 0 {\n\t\tt.Error(\"allowed 0 tokens, want > 0\")\n\t}\n}\n```",
          "output_characters": 5068,
          "output_sha256": "883dc5ddd21462520bf5f164642971eb988dd01883f4cff49945656c0d4e1e53",
          "prompt_tokens": 131,
          "reasoning_characters": 33153,
          "reasoning_sha256": "38cc14037aaf859ae08e10d965b440252454fbe7748ad5f2f87630ba5702c689",
          "response_tail_seconds": 0.004589,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "467ad4f0c30e68932ea0b876741ba4b0831489cdb5c67c9372d74cb15ec84416",
          "time_to_first_visible_seconds": 188.079748,
          "time_to_last_output_seconds": 205.003696,
          "ttft_seconds": 0.07425,
          "wall_seconds": 205.008285
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 12592,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 198.725638,
          "decode_tokens_per_second": 63.359,
          "finish_reason": "stop",
          "measured_sse_events": 12578,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit provides a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time. Implement it to drive the limiter\n// deterministically in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// Errors returned by NewWithClock.\nvar (\n\tErrRate     = errors.New(\"ratelimit: rate must be a positive, finite number\")\n\tErrBurst    = errors.New(\"ratelimit: burst must be positive\")\n\tErrNilClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// Limiter is a token-bucket rate limiter. The bucket starts full, holds at\n// most burst tokens, and refills at rate tokens per second. All methods are\n// safe for concurrent use.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter using the real clock.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock returns a Limiter driven by c. It returns an error if rate is\n// not positive and finite, burst is not positive, or c is nil.\nfunc NewWithClock(rate float64, burst int, c Clock) (*Limiter, error) {\n\tswitch {\n\tcase c == nil:\n\t\treturn nil, ErrNilClock\n\tcase math.IsNaN(rate) || math.IsInf(rate, 0) || rate <= 0:\n\t\treturn nil, ErrRate\n\tcase burst <= 0:\n\t\treturn nil, ErrBurst\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   c.Now(),\n\t\tclock:  c,\n\t}, nil\n}\n\n// Allow reports whether a single token is available, consuming it if so.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// A negative n is rejected without consuming anything.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif d := now.Sub(l.last); d > 0 { // ignore clock regressions\n\t\tl.tokens = math.Min(l.burst, l.tokens+d.Seconds()*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"sync/atomic\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manual clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock() *fakeClock { return &fakeClock{now: time.Unix(0, 0).UTC()} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt.Parallel()\n\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname:  \"burst consumed then denied\",\n\t\t\trate:  1,\n\t\t\tburst: 3,\n\t\t\tsteps: []step{{0, 3, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill accrues fractionally\",\n\t\t\trate:  2,\n\t\t\tburst: 2,\n\t\t\tsteps: []step{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true}, {0, 1, false},\n\t\t\t\t{250 * time.Millisecond, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill capped at burst\",\n\t\t\trate:  1,\n\t\t\tburst: 2,\n\t\t\tsteps: []step{{0, 2, true}, {10 * time.Second, 2, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname:  \"no elapsed time means no refill\",\n\t\t\trate:  5,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname:  \"clock regression adds nothing\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {-time.Second, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill enables retry\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {time.Second, 1, true}},\n\t\t},\n\t}\n\n\tfor _, tt := range tests {\n\t\ttt := tt\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := NewWithClock(tt.rate, tt.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestNewValidation(t *testing.T) {\n\tt.Parallel()\n\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\tclock   Clock\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 1, newFakeClock(), ErrRate},\n\t\t{\"negative rate\", -1.5, 1, newFakeClock(), ErrRate},\n\t\t{\"NaN rate\", math.NaN(), 1, newFakeClock(), ErrRate},\n\t\t{\"Inf rate\", math.Inf(1), 1, newFakeClock(), ErrRate},\n\t\t{\"zero burst\", 1, 0, newFakeClock(), ErrBurst},\n\t\t{\"negative burst\", 1, -3, newFakeClock(), ErrBurst},\n\t\t{\"nil clock\", 1, 1, nil, ErrNilClock},\n\t}\n\n\tfor _, tt := range tests {\n\t\ttt := tt\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\t\t\tl, err := NewWithClock(tt.rate, tt.burst, tt.clock)\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"NewWithClock(%v, %d) error = %v, want %v\", tt.rate, tt.burst, err, tt.wantErr)\n\t\t\t}\n\t\t\tif l != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock() limiter = %v, want nil\", l)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tt.Parallel()\n\n\tfc := newFakeClock()\n\tl, err := NewWithClock(1000, 50, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\n\tvar allowed, denied atomic.Int64\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 200; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.Allow() {\n\t\t\t\tallowed.Add(1)\n\t\t\t} else {\n\t\t\t\tdenied.Add(1)\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\n\tif got := allowed.Load(); got != 50 {\n\t\tt.Errorf(\"allowed = %d, want exactly 50 (burst)\", got)\n\t}\n\tif got := denied.Load(); got != 150 {\n\t\tt.Errorf(\"denied = %d, want 150\", got)\n\t}\n}\n```",
          "output_characters": 5778,
          "output_sha256": "9fad9706c540d097d118bd3b305c623e385dddeb73f2fd7dbd71c966b4ea23a7",
          "prompt_tokens": 133,
          "reasoning_characters": 34087,
          "reasoning_sha256": "6e26f0f25a7ca2ee0d96c622663acb13c4d881f6195412d4dad4daa5bf301066",
          "response_tail_seconds": 0.004051,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "78e007debe92f16829c0188fd977a7331c47326ddeac418d3c755ff834252b55",
          "time_to_first_visible_seconds": 182.996377,
          "time_to_last_output_seconds": 198.808297,
          "ttft_seconds": 0.082659,
          "wall_seconds": 198.812348
        },
        {
          "cached_prompt_tokens": 128,
          "completion_tokens": 14924,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 228.577245,
          "decode_tokens_per_second": 65.286,
          "finish_reason": "stop",
          "measured_sse_events": 14911,
          "output": "ratelimit.go\n```go\n// Package ratelimit provides a concurrency-safe token-bucket rate limiter\n// with an injectable clock for deterministic testing.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock supplies the current time. Implement it to fake time in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// Limiter is a concurrency-safe token bucket. Tokens accrue at rate per\n// second, capped at burst capacity; the bucket starts full.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64\n\tburst  float64\n\ttokens float64\n\tlast   time.Time\n\tclock  Clock\n}\n\n// New returns a Limiter with the given refill rate (tokens/second) and burst\n// capacity. A nil clock selects the real system clock.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif math.IsNaN(rate) || rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available, consuming it if so.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// Negative n returns false; zero n returns true without consuming.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.advance(l.clock.Now())\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// RetryAfter returns how long until n tokens would be available assuming no\n// other calls, or 0 if they are available now. It consumes nothing.\nfunc (l *Limiter) RetryAfter(n int) time.Duration {\n\tif n < 0 {\n\t\treturn 0\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.advance(l.clock.Now())\n\tdeficit := float64(n) - l.tokens\n\tif deficit <= 0 {\n\t\treturn 0\n\t}\n\tsecs := deficit / l.rate\n\tif secs > math.MaxInt64/float64(time.Second) {\n\t\treturn time.Duration(math.MaxInt64)\n\t}\n\treturn time.Duration(secs * float64(time.Second))\n}\n\n// advance refills tokens up to now. The caller must hold l.mu.\nfunc (l *Limiter) advance(now time.Time) {\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed <= 0 {\n\t\treturn\n\t}\n\tl.tokens = math.Min(l.burst, l.tokens+elapsed*l.rate)\n\tl.last = now\n}\n```\n\nratelimit_test.go\n```go\npackage ratelimit\n\nimport (\n\t\"math\"\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\tt.Parallel()\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1.5, 5, true},\n\t\t{\"NaN rate\", math.NaN(), 5, true},\n\t\t{\"zero burst\", 10, 0, true},\n\t\t{\"negative burst\", 10, -3, true},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\t_, err := New(tt.rate, tt.burst, newFakeClock())\n\t\t\tif (err != nil) != tt.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %d) error = %v, wantErr %v\", tt.rate, tt.burst, err, tt.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\ntype step struct {\n\tadvance time.Duration\n\tn       int\n\twant    bool\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt.Parallel()\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []step\n\t}{\n\t\t{\n\t\t\tname: \"burst then empty\", rate: 10, burst: 2,\n\t\t\tsteps: []step{{0, 2, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\", rate: 10, burst: 1,\n\t\t\tsteps: []step{{0, 1, true}, {0, 1, false}, {100 * time.Millisecond, 1, true}, {0, 1, false}},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at burst\", rate: 1, burst: 2,\n\t\t\tsteps: []step{{0, 2, true}, {time.Second, 3, false}, {time.Second, 2, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"request exceeds burst\", rate: 10, burst: 5,\n\t\t\tsteps: []step{{0, 6, false}, {0, 5, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"partial refill insufficient\", rate: 10, burst: 5,\n\t\t\tsteps: []step{{0, 5, true}, {50 * time.Millisecond, 1, false}, {50 * time.Millisecond, 1, true}},\n\t\t},\n\t\t{\n\t\t\tname: \"zero and negative n\", rate: 10, burst: 1,\n\t\t\tsteps: []step{{0, 0, true}, {0, -1, false}, {0, 1, true}},\n\t\t},\n\t}\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := New(tt.rate, tt.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tt.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestRetryAfter(t *testing.T) {\n\tt.Parallel()\n\tfc := newFakeClock()\n\tl, err := New(10, 1, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !l.AllowN(1) {\n\t\tt.Fatal(\"AllowN(1) = false, want true\")\n\t}\n\tif got, want := l.RetryAfter(1), 100*time.Millisecond; got != want {\n\t\tt.Fatalf(\"RetryAfter(1) = %v, want %v\", got, want)\n\t}\n\tif got := l.RetryAfter(0); got != 0 {\n\t\tt.Fatalf(\"RetryAfter(0) = %v, want 0\", got)\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tt.Parallel()\n\tconst burst = 1000\n\tl, err := New(0.001, burst, newFakeClock())\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tvar granted int\n\tvar mu sync.Mutex\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < burst; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.Allow() {\n\t\t\t\tmu.Lock()\n\t\t\t\tgranted++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif granted != burst {\n\t\tt.Fatalf(\"granted %d, want %d\", granted, burst)\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"Allow() = true after exhaustion, want false\")\n\t}\n}\n```",
          "output_characters": 5992,
          "output_sha256": "9ef279ed21c98d35da880f25e26de318ee7eccefe294dc0806e01a97db8cf30c",
          "prompt_tokens": 132,
          "reasoning_characters": 37962,
          "reasoning_sha256": "a1eb1ae6414f4b441fbbce40f4b21abc294f8ac006ab57aaab1f6902cd4c25fb",
          "response_tail_seconds": 0.004995,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "83b6cdd6acfd210db8328a882f062d082a50d2c3fe1d4bab1f7d27e4dc96bfcc",
          "time_to_first_visible_seconds": 211.937091,
          "time_to_last_output_seconds": 228.653474,
          "ttft_seconds": 0.076229,
          "wall_seconds": 228.658468
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 228.653,
        "median": 205.004,
        "minimum": 198.808,
        "p90": 228.653
      },
      "ttft_seconds": {
        "maximum": 0.083,
        "median": 0.076,
        "minimum": 0.074,
        "p90": 0.083
      },
      "wall_seconds": {
        "maximum": 228.658,
        "median": 205.008,
        "minimum": 198.812,
        "p90": 228.658
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 60.249,
        "median": 59.911,
        "minimum": 53.253,
        "p90": 60.249
      },
      "pooled_decode_tokens_per_second": 58.367,
      "runs": [
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 8509,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 141.214607,
          "decode_tokens_per_second": 60.249,
          "finish_reason": "stop",
          "measured_sse_events": 8505,
          "output": "**MEMO \u2014 Internal**\n\n**To:** Platform Engineering; cc Finance\n**From:** R. Okafor, Infrastructure\n**Date:** 12 March 2025\n**Re:** GPU capacity \u2014 buy more, or use what we own better?\n\nWe have two proposals on the table, and I want to settle the question before this quarter's budget conversation. The first is to buy a second inference node \u2014 eight L40S-class cards, roughly $120k all-in with networking and racks. The second is to spend engineering time raising utilization on the hardware we already own. Last quarter our six GPUs averaged 22% compute utilization, yet researchers still hit afternoon queue times, and the retrieval project will add a steady inference load once it ships. Both facts are true at once, which is why this deserves a memo rather than a Slack thread.\n\nStart with cost, because it's the easiest to get wrong. The new node is $120k up front plus around $9k a year in power and cooling, depreciated over four years. At 22% utilization, each hour of useful GPU work costs us roughly four times what the sticker price implies; we own capacity we mostly heat the office with. The utilization path costs engineer-weeks instead \u2014 call it six weeks of one senior engineer, well under a fifth of the capex \u2014 and unlike the hardware, that investment transfers to whatever we own later.\n\nLatency is the reason we own hardware at all. We keep inference local because p99 latency to our own services is predictable and there's no egress tax or provider queue. The tension is that nearly every utilization lever \u2014 dynamic batching, quantization, packing services per card \u2014 trades a little latency or fidelity for throughput. Batching adds tens of milliseconds at the median but can spike the tail under bursty traffic; quantization is usually imperceptible but not always. So any utilization work has to be gated on our latency SLOs, not on a throughput dashboard. If the retrieval service needs single-digit-millisecond responses, some cards should stay underfilled, and that's a legitimate answer.\n\nReliability cuts the other way. Today a dead power supply is an annoyance because we have slack to drain workloads onto healthy cards. Raise utilization and the slack vanishes; the same failure becomes a user-visible outage. Buying more hardware adds components that can fail, but it also buys redundancy we currently lack. Our oldest cards are forty months into service, an age where fan and VRAM failures start clustering. The real question isn't more hardware versus less; it's whether our buffer should live in silicon or in scheduling.\n\nThen there's opportunity cost, and it points in both directions. The engineers best placed to do the utilization work are the same ones the retrieval project needs, and capital spent on GPUs is capital not spent on hiring or tooling. But the reverse cost is real too: every week a researcher spends in a queue is a week of product work we quietly lose, and that cost appears on no invoice.\n\nReversibility is where the asymmetry gets sharp. A GPU purchase is close to irreversible \u2014 depreciation is fast, resale is at a steep discount, and the rack and power commitments don't unwind. Utilization work is software: incremental, reversible, and it produces the telemetry that makes any future purchase decision better. One caveat: measures like per-team quotas or shared scheduling are organizationally sticky even when technically reversible, so we should introduce them as defaults rather than mandates.\n\nMy recommendation is to time-box the utilization push before buying anything. One senior engineer, six weeks, three targets: honest telemetry that distinguishes reserved-but-idle memory from genuinely free capacity; dynamic batching and quantization on the two inference services that can tolerate them; and a simple priority queue so batch jobs yield to interactive traffic. At the end of that window we decide with data: if projected peak utilization clears about 70%, or if the retrieval load lands before the work does, we buy \u2014 and we'll size the purchase correctly instead of guessing. I'll put the go/no-go on the April planning agenda.\n\nI want to be honest about the uncertainty. Our utilization numbers may flatter us, since reserved memory looks idle on dashboards; the retrieval forecast is soft; the engineering work could stall or deliver half of what we hope; and hardware lead times are unpredictable. If two of those break the wrong way, we buy sooner, and the six weeks will have been cheap insurance either way. What I want to avoid is buying a node out of queue-time anxiety and discovering next year that we owned capacity we never used. Happy to walk through the numbers with anyone before we commit.",
          "output_characters": 4684,
          "output_sha256": "c7c9f6e343f0490a68203441cdbdac1a7a3b12c3402f07ac6d37e139171c9574",
          "prompt_tokens": 121,
          "reasoning_characters": 32910,
          "reasoning_sha256": "8a432e66ba79c2405685dc5057ed126f51c18c6d013ef2566cc2f019d9d77c97",
          "response_tail_seconds": 0.003074,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "13bea72eefd316c655b6aedc80b2670cc0a6c17dc357990ed95ab9ea407765a8",
          "time_to_first_visible_seconds": 133.882302,
          "time_to_last_output_seconds": 141.443068,
          "ttft_seconds": 0.228461,
          "wall_seconds": 141.446141
        },
        {
          "cached_prompt_tokens": 64,
          "completion_tokens": 7398,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 123.466337,
          "decode_tokens_per_second": 59.911,
          "finish_reason": "stop",
          "measured_sse_events": 7395,
          "output": "**Internal Memo**\n\n**To:** ML Platform team (cc: Finance)\n**From:** Priya Raman, Engineering Lead\n**Date:** Monday 3 June\n**Subject:** GPU capacity \u2014 buy more, or use what we own better\n\nWe have a decision to make about our GPU fleet, and I want to lay out my thinking ahead of Thursday's planning session. My recommendation, stated up front: we should not buy another training node this quarter. Instead, we should spend six weeks and roughly half an engineer making better use of the hardware we already have, and set a concrete trigger for revisiting a purchase if that effort falls short.\n\nThe facts first. We own one eight-GPU A100 server used for training and batch jobs, and a four-GPU L40S box that serves our inference endpoints. Our dashboards report average GPU utilization around 35 percent, yet researchers complain that training jobs sit in queue for hours and that inference latency degrades badly between ten and two. Both things are true, and the gap between them is the real problem. The A100s are idle in compute terms but their memory is fragmented and hoarded by long-running notebooks; the L40S box has no admission control, so a burst of traffic or one careless batch job degrades everyone's latency at once. We are not short of FLOPs. We are short of discipline.\n\nBuying our way out is the obvious move and not a foolish one. A new node would cost about $180k up front, plus power, cooling, and maybe $600 a month in colocation. It would end the queueing complaints tomorrow and give us headroom for the larger fine-tuning runs on next quarter's roadmap. But it is also the least reversible option on the table. Once bought, the money is spent; resale recovers perhaps a third of the cost; the rack space and the depreciation schedule are committed for three years. There is a soft irreversibility too: teams calibrate their habits to available capacity, and demand ratchets up to meet it. Nobody has ever voluntarily given a GPU back.\n\nImproving utilization, by contrast, is cheap and almost entirely reversible. Concretely: a scheduler with preemption and time limits on the training box, MIG partitioning so a notebook can't reserve a whole device, admission control and real batching on the inference path, and honest telemetry on memory rather than compute. The cost is about half of one engineer for six weeks \u2014 call it $25k of loaded time \u2014 and every change can be rolled back in an afternoon. If it works, we effectively find half a node we already own, at a twentieth of the price of buying one.\n\nThe opportunity cost cuts both ways and deserves honest accounting. Idle GPUs are capital earning nothing; at 35 percent utilization we are paying full price for a third of a machine. But engineer time is also capital, and the person doing this work is not writing product code. I think the trade still favors utilization, because the work is bounded, teaches us things we would need to know before buying anyway, and the alternative spends $180k to avoid learning whether our problem was capacity or configuration.\n\nReliability is the one place where new hardware has a genuine edge, and I want to be precise about it. Our real reliability risk is not the training fleet; a crashed batch job costs an afternoon. It is the inference server, which is a single point of failure for everything customer-facing. If we do spend money this quarter, it should be a second small inference node or a modest cloud burst reservation \u2014 a few thousand dollars a month, cancellable, reversible \u2014 not a training node. That addresses the failure mode that can actually page us at night.\n\nI want to flag the uncertainty in my own recommendation. Utilization numbers can mislead: memory-bound workloads look idle while being unable to share, and some apparent headroom is really latency reserve we shouldn't spend. If next quarter's roadmap includes the large multilingual fine-tune, the calculus shifts and a purchase becomes defensible; I would rather make that call with six weeks of clean telemetry than now, with guesswork. And if the utilization sprint fails \u2014 if we cannot get training queue waits under two hours at p90 \u2014 then we will have earned the evidence to buy, and I will champion it.\n\nProposal for Thursday: approve the six-week effort, authorize the burst reservation as insurance, and put a purchase decision on the calendar for mid-July with agreed criteria. Comments welcome before then.\n\nPriya",
          "output_characters": 4426,
          "output_sha256": "f1ae5e5b67319909822ccf66d25aaaa036d7397c6feed4439d20e002104b61d5",
          "prompt_tokens": 121,
          "reasoning_characters": 27623,
          "reasoning_sha256": "9f9187bbeb5d6ff0d2ea76289247b5f4a680818eafa66d6d3d571a4a572739e6",
          "response_tail_seconds": 0.00273,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "c2178175131b800e2e9727b2f7184424dc89e8ba306e7c84524066034e913bde",
          "time_to_first_visible_seconds": 115.203431,
          "time_to_last_output_seconds": 123.694009,
          "ttft_seconds": 0.227672,
          "wall_seconds": 123.696739
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 4753,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 89.234097,
          "decode_tokens_per_second": 53.253,
          "finish_reason": "stop",
          "measured_sse_events": 4751,
          "output": "**Internal Memo**\n\n**To:** Engineering leadership\n**From:** Inference infrastructure\n**Re:** GPU capacity \u2014 buy more, or use what we own better\n\nWe have a decision to make about our inference capacity, and I want to lay out how I think about it before we commit money or time. We currently operate eight L40S cards across two servers. Weekly average utilization sits around thirty-five percent, which sounds comfortable, but demand is spiky: mid-morning peaks push the cluster to saturation, our p95 latency for interactive inference breaches the 800-millisecond target several times a week, and last month two batch jobs slipped a full day. The two options on the table are buying two additional nodes, roughly $95,000 installed, or spending about six engineer-weeks on scheduling, batching, and model right-sizing to get more from what we have.\n\nOn cost: the sticker price understates the true number. Power, cooling, support contracts, and depreciation over a realistic three-year horizon push the effective figure meaningfully higher, and GPU resale values fall sharply whenever a new generation ships, so the capital is worth less each quarter. The utilization path costs something too, but it costs engineer time, which is our scarcest resource; six weeks is a visible dent in the roadmap. Still, idle capacity is not free. We are paying full price for machines we use a third of the time, and that is a cost we have simply been absorbing.\n\nThe utilization figure misleads in both directions, and this is where latency matters. Our problem is not total compute but concurrency at peak. Most peak load comes from latency-sensitive interactive traffic \u2014 the internal copilot and the search reranker \u2014 while batch embedding jobs can run at any hour. Continuous batching, better bin-packing of requests, and moving the reranker to fp8 could plausibly raise effective peak capacity by half again or more without new metal. But there is a floor. If peak demand turns out to be memory-bandwidth-bound, no amount of scheduling cleverness fixes it, and only more cards will. We do not yet know which regime we are in, and that uncertainty should shape how we sequence the work.\n\nReliability cuts the other way, and I want to be honest about it. Headroom is reliability. Today, losing a node at peak degrades service without breaching our targets; at eighty-five or ninety percent sustained utilization, the same failure becomes an incident. Maintenance windows get riskier as we approach saturation. Software improvements carry their own reliability risk: a scheduler change that lifts throughput can quietly worsen tail latency, so we would need guardrails, canaries, and a rollback path. Hardware is dull, but it is predictable, and there is real value in that.\n\nThen there is opportunity cost and reversibility, which I think decide this. The utilization investment is largely reversible and compounding: better instrumentation, a smarter queue, and the operational muscle we build stay valuable even if we buy hardware later. The purchase is semi-irreversible. Capital gets locked into cards that depreciate quickly, and if our workload changes shape \u2014 a new model architecture, a team leaving, a shift from serving to fine-tuning \u2014 the hardware may no longer fit demand. The counterweight is that engineer time spent on scheduling is time not spent on the product roadmap, and that opportunity cost is the one we control least, because roadmap priorities shift underneath us.\n\nMy recommendation is to fund the utilization work now, time-boxed to one quarter, with a pre-approved hardware purchase as the explicit fallback. The tripwires are simple: if business-hours utilization sustains above roughly seventy percent while p95 still breaches target, or if a known demand step-change lands \u2014 the data team onboarding their fine-tuning workload next quarter \u2014 we order the nodes without re-litigating the decision. This sequencing buys information cheaply, keeps the reversible option first, and preserves the procurement lead time rather than closing it.\n\nI want to name the uncertainties plainly. Our telemetry probably overstates usable capacity, because memory fragmentation and unpreemptable jobs mean real headroom is smaller than the dashboards suggest. Our demand forecast is a guess dressed up as a chart. The utilization work might deliver half of what we hope. And GPU lead times run eight to twelve weeks, so if we guess wrong we will pay in waiting rather than money. The biggest risk, in my view, is under-investing and spending a quarter firefighting latency instead of building; the tripwires exist to cap exactly that downside. I would like a decision by the end of the month so the procurement option stays open either way.",
          "output_characters": 4742,
          "output_sha256": "dfbbc8f2478e2f65b2413bba55685ba2ed7d48d76b4e3e1b01f41e2db44eb73b",
          "prompt_tokens": 122,
          "reasoning_characters": 17489,
          "reasoning_sha256": "b25533babc1f3de8a34d184b747cbe8fb4092d254eac5b0091c62252d745d634",
          "response_tail_seconds": 0.001475,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "01ccf4736d63dca73213cf060e7ab39c16f9dceec863bf8009c490ddc835ccaf",
          "time_to_first_visible_seconds": 81.491,
          "time_to_last_output_seconds": 89.731413,
          "ttft_seconds": 0.497315,
          "wall_seconds": 89.732888
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 141.443,
        "median": 123.694,
        "minimum": 89.731,
        "p90": 141.443
      },
      "ttft_seconds": {
        "maximum": 0.497,
        "median": 0.228,
        "minimum": 0.228,
        "p90": 0.497
      },
      "wall_seconds": {
        "maximum": 141.446,
        "median": 123.697,
        "minimum": 89.733,
        "p90": 141.446
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 3,
        "total": 3
      },
      "decode_tokens_per_second": {
        "maximum": 71.374,
        "median": 70.962,
        "minimum": 70.806,
        "p90": 71.374
      },
      "pooled_decode_tokens_per_second": 71.047,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 13.674384,
          "decode_tokens_per_second": 71.374,
          "finish_reason": "stop",
          "measured_sse_events": 973,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 112,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.000584,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 5.28803,
          "time_to_last_output_seconds": 14.155016,
          "ttft_seconds": 0.480631,
          "wall_seconds": 14.1556
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 13.753828,
          "decode_tokens_per_second": 70.962,
          "finish_reason": "stop",
          "measured_sse_events": 973,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.000508,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 5.359933,
          "time_to_last_output_seconds": 14.238603,
          "ttft_seconds": 0.484775,
          "wall_seconds": 14.239111
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 977,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 13.784192,
          "decode_tokens_per_second": 70.806,
          "finish_reason": "stop",
          "measured_sse_events": 973,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 573,
          "reasoning_sha256": "824d3b1f110b869c8ee763a55226d658718300ffe6d81b6c2d98c9e8e21da0f2",
          "response_tail_seconds": 0.000796,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "cdd0652fec9be14b5c11395523af674c81413cc82ea7353c0fa30a3ad9ebbbf4",
          "time_to_first_visible_seconds": 5.381494,
          "time_to_last_output_seconds": 14.271259,
          "ttft_seconds": 0.487067,
          "wall_seconds": 14.272055
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 14.271,
        "median": 14.239,
        "minimum": 14.155,
        "p90": 14.271
      },
      "ttft_seconds": {
        "maximum": 0.487,
        "median": 0.485,
        "minimum": 0.481,
        "p90": 0.487
      },
      "wall_seconds": {
        "maximum": 14.272,
        "median": 14.239,
        "minimum": 14.156,
        "p90": 14.272
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2000.761,
          "median": 1972.82,
          "minimum": 1942.664,
          "p90": 2000.761
        },
        "runs": [
          {
            "cache_salt_sha256": "be3b6c666e406e8cd7964267d724b788e52b96eb5dbdd50d563b3f3d43372771",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.057125,
            "decode_tokens_per_second": 122.539,
            "effective_prefill_tokens_per_second": 2000.761,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000452,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 16.377765,
            "time_to_last_output_seconds": 16.43489,
            "ttft_seconds": 16.377765,
            "wall_seconds": 16.435342
          },
          {
            "cache_salt_sha256": "2b1884f0739f2244bb580004f436e9f96c704edef07aa0590a1431f79518a54d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.060232,
            "decode_tokens_per_second": 116.217,
            "effective_prefill_tokens_per_second": 1942.664,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000121,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.867556,
            "time_to_last_output_seconds": 16.927788,
            "ttft_seconds": 16.867556,
            "wall_seconds": 16.92791
          },
          {
            "cache_salt_sha256": "f169b95b8f0aa8cf733642a0b208202e8254078ff4fe26e263b0e452defae48e",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.059965,
            "decode_tokens_per_second": 116.735,
            "effective_prefill_tokens_per_second": 1972.82,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000177,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 16.609722,
            "time_to_last_output_seconds": 16.669687,
            "ttft_seconds": 16.609722,
            "wall_seconds": 16.669864
          }
        ],
        "ttft_seconds": {
          "maximum": 16.868,
          "median": 16.61,
          "minimum": 16.378,
          "p90": 16.868
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 649501.497,
          "median": 641114.437,
          "minimum": 632245.118,
          "p90": 649501.497
        },
        "runs": [
          {
            "cache_salt_sha256": "be3b6c666e406e8cd7964267d724b788e52b96eb5dbdd50d563b3f3d43372771",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.056105,
            "decode_tokens_per_second": 124.766,
            "effective_prefill_tokens_per_second": 649501.497,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": "The repository contains a service, tests,",
            "output_characters": 41,
            "output_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 8.8e-05,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "1b45c203065a3cc5a98fb148892c7ab22a326b9f39b4cd46f9fd57885ce52b49",
            "time_to_first_visible_seconds": 0.050451,
            "time_to_last_output_seconds": 0.106556,
            "ttft_seconds": 0.050451,
            "wall_seconds": 0.106644
          },
          {
            "cache_salt_sha256": "2b1884f0739f2244bb580004f436e9f96c704edef07aa0590a1431f79518a54d",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060048,
            "decode_tokens_per_second": 116.573,
            "effective_prefill_tokens_per_second": 641114.437,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 7.9e-05,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.051111,
            "time_to_last_output_seconds": 0.111159,
            "ttft_seconds": 0.051111,
            "wall_seconds": 0.111238
          },
          {
            "cache_salt_sha256": "f169b95b8f0aa8cf733642a0b208202e8254078ff4fe26e263b0e452defae48e",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.061134,
            "decode_tokens_per_second": 114.502,
            "effective_prefill_tokens_per_second": 632245.118,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00096,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 0.051828,
            "time_to_last_output_seconds": 0.112963,
            "ttft_seconds": 0.051828,
            "wall_seconds": 0.113923
          }
        ],
        "ttft_seconds": {
          "maximum": 0.052,
          "median": 0.051,
          "minimum": 0.05,
          "p90": 0.052
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1966.042,
          "median": 1948.183,
          "minimum": 1886.964,
          "p90": 1966.042
        },
        "runs": [
          {
            "cache_salt_sha256": "d4f0fe1bda5189c07dd06cfdaac34adc22054dd9aa79f0cc6b3d5ad1e08e1abc",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062728,
            "decode_tokens_per_second": 111.593,
            "effective_prefill_tokens_per_second": 1966.042,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001003,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 33.333985,
            "time_to_last_output_seconds": 33.396713,
            "ttft_seconds": 33.333985,
            "wall_seconds": 33.397716
          },
          {
            "cache_salt_sha256": "fe733a4fffa5fe96dd74b0aca14af02876c929fa3e3b5576c0842fe348ff4d21",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.08072,
            "decode_tokens_per_second": 86.72,
            "effective_prefill_tokens_per_second": 1886.964,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000507,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 34.730931,
            "time_to_last_output_seconds": 34.811651,
            "ttft_seconds": 34.730931,
            "wall_seconds": 34.812158
          },
          {
            "cache_salt_sha256": "9f636295a711dd33e206ef7ca4cb55a93c3a7dcd2ee19848d555a661127c80c1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064898,
            "decode_tokens_per_second": 107.861,
            "effective_prefill_tokens_per_second": 1948.183,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 7.6e-05,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 33.639552,
            "time_to_last_output_seconds": 33.70445,
            "ttft_seconds": 33.639552,
            "wall_seconds": 33.704526
          }
        ],
        "ttft_seconds": {
          "maximum": 34.731,
          "median": 33.64,
          "minimum": 33.334,
          "p90": 34.731
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1108693.813,
          "median": 1099136.268,
          "minimum": 976153.239,
          "p90": 1108693.813
        },
        "runs": [
          {
            "cache_salt_sha256": "d4f0fe1bda5189c07dd06cfdaac34adc22054dd9aa79f0cc6b3d5ad1e08e1abc",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063315,
            "decode_tokens_per_second": 110.559,
            "effective_prefill_tokens_per_second": 1099136.268,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000292,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.059625,
            "time_to_last_output_seconds": 0.12294,
            "ttft_seconds": 0.059625,
            "wall_seconds": 0.123232
          },
          {
            "cache_salt_sha256": "fe733a4fffa5fe96dd74b0aca14af02876c929fa3e3b5576c0842fe348ff4d21",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.080558,
            "decode_tokens_per_second": 86.894,
            "effective_prefill_tokens_per_second": 976153.239,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000426,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.067137,
            "time_to_last_output_seconds": 0.147695,
            "ttft_seconds": 0.067137,
            "wall_seconds": 0.148121
          },
          {
            "cache_salt_sha256": "9f636295a711dd33e206ef7ca4cb55a93c3a7dcd2ee19848d555a661127c80c1",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064197,
            "decode_tokens_per_second": 109.04,
            "effective_prefill_tokens_per_second": 1108693.813,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000667,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.059111,
            "time_to_last_output_seconds": 0.123308,
            "ttft_seconds": 0.059111,
            "wall_seconds": 0.123975
          }
        ],
        "ttft_seconds": {
          "maximum": 0.067,
          "median": 0.06,
          "minimum": 0.059,
          "p90": 0.067
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1960.885,
          "median": 1926.008,
          "minimum": 1475.396,
          "p90": 1960.885
        },
        "runs": [
          {
            "cache_salt_sha256": "ddc687b4afde14ae05939c8130a0618f7de35618ac6a89b00e0062cefb1b0ae3",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053111,
            "decode_tokens_per_second": 131.8,
            "effective_prefill_tokens_per_second": 1475.396,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " observation records behaviour, evidence, constraints,",
            "output_characters": 54,
            "output_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000734,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "time_to_first_visible_seconds": 5.552407,
            "time_to_last_output_seconds": 5.605518,
            "ttft_seconds": 5.552407,
            "wall_seconds": 5.606252
          },
          {
            "cache_salt_sha256": "68c2a25706a863e2705f5f3854e28973e0ad8862ada819e8ff5a18ae398e1991",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054556,
            "decode_tokens_per_second": 128.308,
            "effective_prefill_tokens_per_second": 1926.008,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001126,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 4.253358,
            "time_to_last_output_seconds": 4.307914,
            "ttft_seconds": 4.253358,
            "wall_seconds": 4.30904
          },
          {
            "cache_salt_sha256": "2593ecd85966c013869c2728c714b8a31adf47a52b7c6918073ab6fbd9b731de",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.056899,
            "decode_tokens_per_second": 123.025,
            "effective_prefill_tokens_per_second": 1960.885,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000127,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.177706,
            "time_to_last_output_seconds": 4.234605,
            "ttft_seconds": 4.177706,
            "wall_seconds": 4.234731
          }
        ],
        "ttft_seconds": {
          "maximum": 5.552,
          "median": 4.253,
          "minimum": 4.178,
          "p90": 5.552
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 258765.557,
          "median": 250000.0,
          "minimum": 244624.94,
          "p90": 258765.557
        },
        "runs": [
          {
            "cache_salt_sha256": "ddc687b4afde14ae05939c8130a0618f7de35618ac6a89b00e0062cefb1b0ae3",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.052961,
            "decode_tokens_per_second": 132.173,
            "effective_prefill_tokens_per_second": 244624.94,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " observation records behaviour, evidence, constraints,",
            "output_characters": 54,
            "output_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001002,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "64de6b20e549222d7456ec41b7e90d976a3186746092d94e04bb350c64dbcfd2",
            "time_to_first_visible_seconds": 0.033488,
            "time_to_last_output_seconds": 0.086449,
            "ttft_seconds": 0.033488,
            "wall_seconds": 0.087452
          },
          {
            "cache_salt_sha256": "68c2a25706a863e2705f5f3854e28973e0ad8862ada819e8ff5a18ae398e1991",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.055443,
            "decode_tokens_per_second": 126.255,
            "effective_prefill_tokens_per_second": 258765.557,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 6.1e-05,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 0.031658,
            "time_to_last_output_seconds": 0.087101,
            "ttft_seconds": 0.031658,
            "wall_seconds": 0.087162
          },
          {
            "cache_salt_sha256": "2593ecd85966c013869c2728c714b8a31adf47a52b7c6918073ab6fbd9b731de",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.057315,
            "decode_tokens_per_second": 122.132,
            "effective_prefill_tokens_per_second": 250000.0,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.0008,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.032768,
            "time_to_last_output_seconds": 0.090082,
            "ttft_seconds": 0.032768,
            "wall_seconds": 0.090882
          }
        ],
        "ttft_seconds": {
          "maximum": 0.033,
          "median": 0.033,
          "minimum": 0.032,
          "p90": 0.033
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "checkpoint MTP head (Mia v1.10 DRAFTER=mtp)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "Mia-AiLab/GLM-5.3-Flash-EXL3-4bpw-TensorFold",
      "model_revision": "078455ffe6472f9a52fbc1139f58b9db2881b25c",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-mtp-max",
    "finished_at": "2026-10-08T11:20:55.864187+00:00",
    "label": "tp2-stock-mtp-max",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T10:59:16.657244+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 32768,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "max"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 3,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}
