{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 60.047,
        "median": 56.397,
        "minimum": 54.566,
        "p90": 60.047
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 67.042,
        "median": 62.115,
        "minimum": 60.217,
        "p90": 67.042
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.457,
        "median": 0.455,
        "minimum": 0.431,
        "p90": 0.457
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 60.047,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 3.803576,
              "decode_tokens_per_second": 67.042,
              "finish_reason": "length",
              "measured_sse_events": 249,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64   // tokens per second\n\tburst    int       // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter with the given refill rate (tokens per second) and\n// burst capacity. The bucket starts full. If clock is nil, the real clock is used.\nfunc New(rate float64, burst int, clock",
              "output_characters": 984,
              "output_sha256": "1725249cafa7d9fb2cd90be0068a0a7a55da400fa73b5c2d63daf4689eb36156",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001115,
              "stream_done_marker": true,
              "stream_sha256": "1725249cafa7d9fb2cd90be0068a0a7a55da400fa73b5c2d63daf4689eb36156",
              "time_to_first_visible_seconds": 0.457231,
              "time_to_last_output_seconds": 4.260807,
              "ttft_seconds": 0.457231,
              "wall_seconds": 4.261922
            }
          ],
          "wall_seconds": 4.263337
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.566,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.234698,
              "decode_tokens_per_second": 60.217,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter that starts full, refilling at rate tokens per\n// second up to capacity.\nfunc New(capacity int,",
              "output_characters": 1036,
              "output_sha256": "8083b98fcf2fe43d3c8eb63f177401a89ab14506a76fbcb334f8eb7915b4d0cd",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001479,
              "stream_done_marker": true,
              "stream_sha256": "8083b98fcf2fe43d3c8eb63f177401a89ab14506a76fbcb334f8eb7915b4d0cd",
              "time_to_first_visible_seconds": 0.454711,
              "time_to_last_output_seconds": 4.689409,
              "ttft_seconds": 0.454711,
              "wall_seconds": 4.690888
            }
          ],
          "wall_seconds": 4.691595
        },
        {
          "aggregate_end_to_end_tokens_per_second": 56.397,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.105265,
              "decode_tokens_per_second": 62.115,
              "finish_reason": "length",
              "measured_sse_events": 249,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n",
              "output_characters": 1027,
              "output_sha256": "b3b75affd7857a65318b955a2ecfbdbb0afb7017ff2e516eb2b953b3debf18d8",
              "prompt_tokens": 130,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.002124,
              "stream_done_marker": true,
              "stream_sha256": "b3b75affd7857a65318b955a2ecfbdbb0afb7017ff2e516eb2b953b3debf18d8",
              "time_to_first_visible_seconds": 0.43119,
              "time_to_last_output_seconds": 4.536454,
              "ttft_seconds": 0.43119,
              "wall_seconds": 4.538578
            }
          ],
          "wall_seconds": 4.539249
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 81.406,
        "median": 78.335,
        "minimum": 71.862,
        "p90": 81.406
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 48.829,
        "median": 45.495,
        "minimum": 44.141,
        "p90": 46.983
      },
      "per_stream_ttft_seconds": {
        "maximum": 1.399,
        "median": 1.106,
        "minimum": 0.511,
        "p90": 1.315
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 78.335,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.222286,
              "decode_tokens_per_second": 48.829,
              "finish_reason": "length",
              "measured_sse_events": 249,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int           // max tokens\n\trate     float64       // tokens per second\n\ttokens   float64       // current tokens\n\tlast     time.Time     // last refill time\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: capacity must be positive\")\n\t}\n\tif rate <= 0",
              "output_characters": 1021,
              "output_sha256": "921c104fc3bab0423cce94c7a42e80739caffe93c1bf9d6d5993c1e61abd5cae",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000925,
              "stream_done_marker": true,
              "stream_sha256": "921c104fc3bab0423cce94c7a42e80739caffe93c1bf9d6d5993c1e61abd5cae",
              "time_to_first_visible_seconds": 1.105858,
              "time_to_last_output_seconds": 6.328143,
              "ttft_seconds": 1.105858,
              "wall_seconds": 6.329068
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.427483,
              "decode_tokens_per_second": 46.983,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second). rate must be > 0 and capacity must be >= 1.\nfunc New(rate float64, capacity int) (*Limiter, error) {\n\treturn NewWithClock(rate, capacity, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(rate float64, capacity int, c Clock) (*Limiter, error) {\n",
              "output_characters": 988,
              "output_sha256": "b84b263bd048de5a24630e0bc7131fb8cab76be0aac4608053a8c46b7095d66c",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001176,
              "stream_done_marker": true,
              "stream_sha256": "b84b263bd048de5a24630e0bc7131fb8cab76be0aac4608053a8c46b7095d66c",
              "time_to_first_visible_seconds": 1.106024,
              "time_to_last_output_seconds": 6.533507,
              "ttft_seconds": 1.106024,
              "wall_seconds": 6.534684
            }
          ],
          "wall_seconds": 6.536017
        },
        {
          "aggregate_end_to_end_tokens_per_second": 81.406,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.531757,
              "decode_tokens_per_second": 46.097,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tcapacity float64\n\trate     float64 // tokens per second\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. A nil clock defaults to the real clock.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn",
              "output_characters": 958,
              "output_sha256": "b33fdc572de2929bb482c7a153b2b261cec53e354263d5997c0147268721a61e",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000142,
              "stream_done_marker": true,
              "stream_sha256": "b33fdc572de2929bb482c7a153b2b261cec53e354263d5997c0147268721a61e",
              "time_to_first_visible_seconds": 0.510557,
              "time_to_last_output_seconds": 6.042314,
              "ttft_seconds": 0.510557,
              "wall_seconds": 6.042457
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.776982,
              "decode_tokens_per_second": 44.141,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens\n// per second, up to a burst capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\nvar (\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n)\n\n// New returns a Limiter with the given refill rate (tokens per second) and\n// burst capacity, starting full. clock may be nil for the real clock.\nfunc New(rate, capacity float64, clock Clock) (*Limiter",
              "output_characters": 982,
              "output_sha256": "d1a52be910e711bfbe85fa43d569db1534f773da27275eb0d0fb874f5bdd5575",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000962,
              "stream_done_marker": true,
              "stream_sha256": "d1a52be910e711bfbe85fa43d569db1534f773da27275eb0d0fb874f5bdd5575",
              "time_to_first_visible_seconds": 0.510591,
              "time_to_last_output_seconds": 6.287573,
              "ttft_seconds": 0.510591,
              "wall_seconds": 6.288534
            }
          ],
          "wall_seconds": 6.289468
        },
        {
          "aggregate_end_to_end_tokens_per_second": 71.862,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.680123,
              "decode_tokens_per_second": 44.893,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at Rate per\n// second up to Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Lim",
              "output_characters": 982,
              "output_sha256": "11fd17e072e4f4c21557be1c70afd1ffee43260367b6204304c322c0ff9bf770",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000714,
              "stream_done_marker": true,
              "stream_sha256": "11fd17e072e4f4c21557be1c70afd1ffee43260367b6204304c322c0ff9bf770",
              "time_to_first_visible_seconds": 1.315386,
              "time_to_last_output_seconds": 6.995508,
              "ttft_seconds": 1.315386,
              "wall_seconds": 6.996222
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.723276,
              "decode_tokens_per_second": 44.555,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It allows bursts up to Burst tokens, refilling at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second\n\tburst int\n\tclock Clock\n}\n\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with a full bucket. rate is tokens per second.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock",
              "output_characters": 939,
              "output_sha256": "252d69240f2f98c0e65c11162a9a72cebffd544e91665be3fb8a3ebb168fd58c",
              "prompt_tokens": 129,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001197,
              "stream_done_marker": true,
              "stream_sha256": "252d69240f2f98c0e65c11162a9a72cebffd544e91665be3fb8a3ebb168fd58c",
              "time_to_first_visible_seconds": 1.399077,
              "time_to_last_output_seconds": 7.122353,
              "ttft_seconds": 1.399077,
              "wall_seconds": 7.12355
            }
          ],
          "wall_seconds": 7.124755
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 107.829,
        "median": 106.397,
        "minimum": 104.153,
        "p90": 107.829
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 34.067,
        "median": 30.429,
        "minimum": 28.494,
        "p90": 32.679
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.992,
        "median": 0.879,
        "minimum": 0.638,
        "p90": 0.971
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 104.153,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.949336,
              "decode_tokens_per_second": 28.494,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "**ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the system time.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second until the bucket holds capacity tokens.\n// The zero value is not usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter that starts full, refilling at rate tokens per\n// second up to capacity. clock may be nil to use the real clock.\nfunc New(capacity, rate float64,",
              "output_characters": 1038,
              "output_sha256": "1cf8a085593ecd18f2024501c955be155360f1b0f91a3ec6730683e8a3e5fa52",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000764,
              "stream_done_marker": true,
              "stream_sha256": "1cf8a085593ecd18f2024501c955be155360f1b0f91a3ec6730683e8a3e5fa52",
              "time_to_first_visible_seconds": 0.879521,
              "time_to_last_output_seconds": 9.828856,
              "ttft_seconds": 0.879521,
              "wall_seconds": 9.82962
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.308618,
              "decode_tokens_per_second": 30.691,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at Rate per second\n// up to Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity int, rate float64, clock",
              "output_characters": 964,
              "output_sha256": "2761aec6051753b32ba315f5a0bef4d990e8add4e45a11511479b4cb79a2f5d0",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.00035,
              "stream_done_marker": true,
              "stream_sha256": "2761aec6051753b32ba315f5a0bef4d990e8add4e45a11511479b4cb79a2f5d0",
              "time_to_first_visible_seconds": 0.878915,
              "time_to_last_output_seconds": 9.187533,
              "ttft_seconds": 0.878915,
              "wall_seconds": 9.187883
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.953519,
              "decode_tokens_per_second": 32.061,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tcapacity float64\n\trate     float64 // tokens per second\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with a full bucket. capacity and rate must be > 0.\n// If clock is nil, the real clock is used.\nfunc New(capacity float64, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn",
              "output_characters": 932,
              "output_sha256": "b647864705ddd55ecafc964b88ffd8e30e5b160b66150f75c56501cfa34de48a",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000285,
              "stream_done_marker": true,
              "stream_sha256": "b647864705ddd55ecafc964b88ffd8e30e5b160b66150f75c56501cfa34de48a",
              "time_to_first_visible_seconds": 0.992213,
              "time_to_last_output_seconds": 8.945732,
              "ttft_seconds": 0.992213,
              "wall_seconds": 8.946017
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.803284,
              "decode_tokens_per_second": 32.679,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// capacity, starting full. If clock is nil, the real clock is used.\nfunc New(rate float64, capacity int, clock Clock) (*Limiter, error) {\n\tif rate",
              "output_characters": 965,
              "output_sha256": "e0c93ba608277e92d4b608e75239e009c021d7f90a79c18653edc4e6191e3bc5",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000649,
              "stream_done_marker": true,
              "stream_sha256": "e0c93ba608277e92d4b608e75239e009c021d7f90a79c18653edc4e6191e3bc5",
              "time_to_first_visible_seconds": 0.879748,
              "time_to_last_output_seconds": 8.683032,
              "ttft_seconds": 0.879748,
              "wall_seconds": 8.683681
            }
          ],
          "wall_seconds": 9.831695
        },
        {
          "aggregate_end_to_end_tokens_per_second": 107.829,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.338844,
              "decode_tokens_per_second": 30.58,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, errors",
              "output_characters": 1005,
              "output_sha256": "74b9d12a5c58277157e737bb2d36872b25c485ff3ecfefad0a8a934b5fd4fe9f",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000494,
              "stream_done_marker": true,
              "stream_sha256": "74b9d12a5c58277157e737bb2d36872b25c485ff3ecfefad0a8a934b5fd4fe9f",
              "time_to_first_visible_seconds": 0.639349,
              "time_to_last_output_seconds": 8.978193,
              "ttft_seconds": 0.639349,
              "wall_seconds": 8.978687
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.666942,
              "decode_tokens_per_second": 29.422,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It is not safe to copy; use the pointer.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n\tclock    Clock\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock         = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// (bucket capacity), starting full. If clock is nil, the real clock is used.\nfunc New(rate, burst float",
              "output_characters": 957,
              "output_sha256": "7304d60a390330e0de70ac7564533249249c81c143fbd59e23e7505a41baceea",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001055,
              "stream_done_marker": true,
              "stream_sha256": "7304d60a390330e0de70ac7564533249249c81c143fbd59e23e7505a41baceea",
              "time_to_first_visible_seconds": 0.638187,
              "time_to_last_output_seconds": 9.305129,
              "ttft_seconds": 0.638187,
              "wall_seconds": 9.306184
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.603798,
              "decode_tokens_per_second": 29.638,
              "finish_reason": "length",
              "measured_sse_events": 249,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a Limiter that starts full, refilling at rate tokens per\n// second with the given burst capacity.\nfunc New(rate float64, capacity int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: rate must be > 0\")\n\t}\n\tif capacity <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: capacity must be > 0\")\n\t}\n",
              "output_characters": 956,
              "output_sha256": "137787b31f8121f93cdf102e8432919fd32d2b136cda6d1dab9953f20fc7aa51",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001014,
              "stream_done_marker": true,
              "stream_sha256": "137787b31f8121f93cdf102e8432919fd32d2b136cda6d1dab9953f20fc7aa51",
              "time_to_first_visible_seconds": 0.639324,
              "time_to_last_output_seconds": 9.243122,
              "ttft_seconds": 0.639324,
              "wall_seconds": 9.244136
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.854334,
              "decode_tokens_per_second": 28.799,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It allows bursts up to Burst tokens, refilling at Rate tokens per second.\n// The zero value is not usable; construct with NewLimiter.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second\n\tburst int\n\tclock Clock\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock         = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// NewLimiter creates a Limiter with the given refill rate (tokens per\n//",
              "output_characters": 966,
              "output_sha256": "a40e1f9e9ac58f6dd73a53d92c6803f4b2c7be56d0a032aef7d50907f2b432fe",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000719,
              "stream_done_marker": true,
              "stream_sha256": "a40e1f9e9ac58f6dd73a53d92c6803f4b2c7be56d0a032aef7d50907f2b432fe",
              "time_to_first_visible_seconds": 0.639304,
              "time_to_last_output_seconds": 9.493638,
              "ttft_seconds": 0.639304,
              "wall_seconds": 9.494357
            }
          ],
          "wall_seconds": 9.496479
        },
        {
          "aggregate_end_to_end_tokens_per_second": 106.397,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.935235,
              "decode_tokens_per_second": 32.135,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tcapacity float64\n\trate     float64 // tokens per second\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n)\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// capacity, starting full. If clock is nil, the real clock is used.\nfunc New(rate float64, capacity int64, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, ErrNonPositiveRate\n\t}\n\tif capacity <=",
              "output_characters": 953,
              "output_sha256": "1afde07d848e925e368cfa6afedf460935995f4393658e8f88e4c7a98a51b8af",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001014,
              "stream_done_marker": true,
              "stream_sha256": "1afde07d848e925e368cfa6afedf460935995f4393658e8f88e4c7a98a51b8af",
              "time_to_first_visible_seconds": 0.971387,
              "time_to_last_output_seconds": 8.906622,
              "ttft_seconds": 0.971387,
              "wall_seconds": 8.907636
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.485193,
              "decode_tokens_per_second": 34.067,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill at rate tokens per second; capacity bounds burst size.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). rate must be > 0 and capacity must be >= 1.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable clock; clock must not be nil.\nfunc NewWithClock(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"rat",
              "output_characters": 981,
              "output_sha256": "747c20bff0b1a798e3c612aabb70a510bfb45e8a41015348198c595a8c80a54f",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.00071,
              "stream_done_marker": true,
              "stream_sha256": "747c20bff0b1a798e3c612aabb70a510bfb45e8a41015348198c595a8c80a54f",
              "time_to_first_visible_seconds": 0.970477,
              "time_to_last_output_seconds": 8.45567,
              "ttft_seconds": 0.970477,
              "wall_seconds": 8.45638
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.421618,
              "decode_tokens_per_second": 30.279,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the standard library.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be constructed with NewLimiter; the zero value is not usable.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\trate   float64   // tokens per second\n\tburst  float64   // bucket capacity\n\ttokens float64   // current tokens\n\tlast   time.Time // last refill time\n}\n\n// ErrInvalidConfig is returned by NewLimiter for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be > 0\")\n\n// NewLimiter creates a Limiter that refills at rate tokens per second with\n// a maximum burst capacity. It starts full.\nfunc NewLimiter(rate, burst float",
              "output_characters": 1028,
              "output_sha256": "12528aa933e7b1fdbc939b59a94d86201c11781324baf215ca2c5ac7d0b60b82",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 5.3e-05,
              "stream_done_marker": true,
              "stream_sha256": "12528aa933e7b1fdbc939b59a94d86201c11781324baf215ca2c5ac7d0b60b82",
              "time_to_first_visible_seconds": 0.970949,
              "time_to_last_output_seconds": 9.392568,
              "ttft_seconds": 0.970949,
              "wall_seconds": 9.392621
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.758831,
              "decode_tokens_per_second": 29.113,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate RefillPerSecond\n// up to capacity Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for invalid parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be positive\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an inject",
              "output_characters": 1029,
              "output_sha256": "2cdfdfb75dd7a1eb9501e02bd946d01b6107580962d5a53cca25c683ad787da2",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000817,
              "stream_done_marker": true,
              "stream_sha256": "2cdfdfb75dd7a1eb9501e02bd946d01b6107580962d5a53cca25c683ad787da2",
              "time_to_first_visible_seconds": 0.862223,
              "time_to_last_output_seconds": 9.621054,
              "ttft_seconds": 0.862223,
              "wall_seconds": 9.621871
            }
          ],
          "wall_seconds": 9.6243
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 75.196,
        "median": 74.535,
        "minimum": 73.671,
        "p90": 75.196
      },
      "pooled_decode_tokens_per_second": 74.553,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1902,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 25.411089,
          "decode_tokens_per_second": 74.81,
          "finish_reason": "stop",
          "measured_sse_events": 1900,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a Limiter with the given burst capacity and refill rate.\n// rate must be > 0 and capacity must be > 0.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(capacity int, rate float64, c Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: capacity must be positive\")\n\t}\n\tif rate <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: rate must be positive\")\n\t}\n\tif c == nil {\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\t}\n\tnow := c.Now()\n\treturn &Limiter{\n\t\tclock:    c,\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   float64(capacity), // start full\n\t\tlast:     now,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now, consuming\n// them if so. It never blocks.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n <= 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > float64(l.capacity) {\n\t\t\tl.tokens = float64(l.capacity)\n\t\t}\n\t\tl.last = now\n\t}\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname      string\n\t\tcapacity  int\n\t\trate      float64\n\t\twantErr   bool\n\t}{\n\t\t{\"ok\", 10, 1.0, false},\n\t\t{\"zero capacity\", 0, 1.0, true},\n\t\t{\"negative capacity\", -1, 1.0, true},\n\t\t{\"zero rate\", 10, 0, true},\n\t\t{\"negative rate\", 10, -0.5, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.capacity, tc.rate)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%d, %v) err = %v, wantErr %v\", tc.capacity, tc.rate, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:     \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"tokens capped at capacity\",\n\t\t\tcapacity: 1, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{time.Minute, 1, true}, // only 1 token despite long wait\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"large request denied without consuming\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 10, false}, {0, 5, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"fractional refill\",\n\t\t\tcapacity: 1, rate: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{250 * time.Millisecond, 1, false}, // 0.5 tokens\n\t\t\t\t{250 * time.Millisecond, 1, true},  // 1.0 tokens\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(base)\n\t\t\tl, err := NewWithClock(tc.capacity, tc.rate, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v (tokens=%v)\", i, s.n, got, s.want, l.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowNInvalidInput(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, err := NewWithClock(1, 1, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tif l.AllowN(0) || l.AllowN(-1) {\n\t\tt.Error(\"AllowN with n <= 0 should return false\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, _ := NewWithClock(100, 1000, fc)\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed %d tokens, exceeds capacity 100\", allowed)\n\t}\n}\n```",
          "output_characters": 5794,
          "output_sha256": "0573da2c2a5c3c5157b7b405b0fac771b76482d121136dfd2f62ed3205582680",
          "prompt_tokens": 130,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.00087,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "0573da2c2a5c3c5157b7b405b0fac771b76482d121136dfd2f62ed3205582680",
          "time_to_first_visible_seconds": 0.436183,
          "time_to_last_output_seconds": 25.847272,
          "ttft_seconds": 0.436183,
          "wall_seconds": 25.848142
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1984,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 26.604983,
          "decode_tokens_per_second": 74.535,
          "finish_reason": "stop",
          "measured_sse_events": 1982,
          "output": "**ratelimit/ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is\n// not usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter that starts full, refilling at rate tokens per\n// second with the given burst capacity.\nfunc New(rate float64, capacity int, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\tcapacity: float64(capacity),\n\t\trate:     rate,\n\t\ttokens:   float64(capacity),\n\t\tlast:     now,\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(time.Now(), 1) }\n\n// AllowN reports whether n tokens are available at time now, consuming\n// them if so. It is safe for concurrent use.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\t// Refill tokens accrued since the last event.\n\tif elapsed := now.Sub(l.last); elapsed > 0 {\n\t\tl.tokens += elapsed.Seconds() * l.rate\n\t\tif l.tokens > l.capacity {\n\t\t\tl.tokens = l.capacity\n\t\t}\n\t\tl.last = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n**ratelimit/ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\tt.Parallel()\n\tcases := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tcapacity int\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1, 5, true},\n\t\t{\"zero capacity\", 10, 0, true},\n\t\t{\"negative capacity\", 10, -3, true},\n\t}\n\tfor _, tc := range cases {\n\t\ttc := tc\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.rate, tc.capacity, newFakeClock(base))\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %v) err = %v, wantErr %v\", tc.rate, tc.capacity, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt.Parallel()\n\tcases := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tcapacity int\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\trate: 1, capacity: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\trate: 2, capacity: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true}, // 1 token accrued\n\t\t\t\t{0, 1, false},\n\t\t\t\t{time.Second, 2, true}, // 2 tokens accrued\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at capacity\",\n\t\t\trate: 10, capacity: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{time.Hour, 3, false}, // only capacity tokens accrue\n\t\t\t\t{0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"request more than capacity\",\n\t\t\trate: 1, capacity: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{time.Hour, 2, false},\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\trate: 1, capacity: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range cases {\n\t\ttc := tc\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := New(tc.rate, tc.capacity, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(clk.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v (tokens=%f)\", i, s.n, got, s.want, l.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tt.Parallel()\n\tclk := newFakeClock(base)\n\tl, err := New(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow = true, want false\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tt.Parallel()\n\tclk := newFakeClock(base)\n\tl, err := New(100, 50, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tconst goroutines, per = 100, 10\n\tvar mu sync.Mutex\n\tallowed := 0\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < goroutines; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < per; j++ {\n\t\t\t\tif l.AllowN(clk.Now(), 1) {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 50 {\n\t\tt.Fatalf(\"allowed = %d, want <= capacity 50\", allowed)\n\t}\n}\n```",
          "output_characters": 6173,
          "output_sha256": "67d4013e06cda832f0189616cb4f10f57c919682d56edfffc14408a08a042cfb",
          "prompt_tokens": 132,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000994,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "67d4013e06cda832f0189616cb4f10f57c919682d56edfffc14408a08a042cfb",
          "time_to_first_visible_seconds": 0.446052,
          "time_to_last_output_seconds": 27.051035,
          "ttft_seconds": 0.446052,
          "wall_seconds": 27.052029
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2113,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 28.08643,
          "decode_tokens_per_second": 75.196,
          "finish_reason": "stop",
          "measured_sse_events": 2111,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens\n// per second, up to a maximum burst size. The zero value is not usable;\n// construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be > 0\")\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum burst capacity. The bucket starts full.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\trate:     rate,\n\t\tburst:    float64(burst),\n\t\ttokens:   float64(burst),\n\t\tlastTime: now,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now, consuming\n// them if so. It never blocks.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n <= 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.lastTime).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > l.burst {\n\t\t\tl.tokens = l.burst\n\t\t}\n\t\tl.lastTime = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens, rounded down\n// to the nearest whole token.\nfunc (l *Limiter) Tokens() int {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.lastTime).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > l.burst {\n\t\t\tl.tokens = l.burst\n\t\t}\n\t\tl.lastTime = now\n\t}\n\treturn int(l.tokens)\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (c *fakeClock) Now() time.Time {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\treturn c.now\n}\n\nfunc (c *fakeClock) Advance(d time.Duration) {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\tc.now = c.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr error\n\t}{\n\t\t{\"valid\", 10, 5, nil},\n\t\t{\"zero rate\", 0, 5, ErrInvalidConfig},\n\t\t{\"negative rate\", -1, 5, ErrInvalidConfig},\n\t\t{\"zero burst\", 10, 0, ErrInvalidConfig},\n\t\t{\"negative burst\", 10, -3, ErrInvalidConfig},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, newFakeClock(base))\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"New() error = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t\tif tc.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"New() returned nil limiter without error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tburst    int\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst exhausted immediately\",\n\t\t\trate:  1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\trate:  1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens cap at burst\",\n\t\t\trate:  10, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{10 * time.Second, 2, true}, // capped, not 100\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"partial refill insufficient\",\n\t\t\trate:  1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{500 * time.Millisecond, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"multi-token request\",\n\t\t\trate:  2, burst: 5,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false},\n\t\t\t\t{time.Second, 2, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"zero and negative n rejected\",\n\t\t\trate:  1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0, false}, {0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := New(tc.rate, tc.burst, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() unexpected error: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestTokens(t *testing.T) {\n\tclk := newFakeClock(base)\n\tl, err := New(2, 4, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New() unexpected error: %v\", err)\n\t}\n\tl.AllowN(3)\n\tif got := l.Tokens(); got != 1 {\n\t\tt.Errorf(\"Tokens() = %d, want 1\", got)\n\t}\n\tclk.Advance(2 * time.Second)\n\tif got := l.Tokens(); got != 4 { // capped at burst\n\t\tt.Errorf(\"Tokens() = %d, want 4\", got)\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tclk := newFakeClock(base)\n\tl, err := New(1000, 100, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New() unexpected error: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.AllowN(1) {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed %d tokens, exceeds burst 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6483,
          "output_sha256": "c51fe6bce62f73e5f17be4e326c880ea9ebe8d51f4d5f0619db075b0bb31f639",
          "prompt_tokens": 133,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001861,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "c51fe6bce62f73e5f17be4e326c880ea9ebe8d51f4d5f0619db075b0bb31f639",
          "time_to_first_visible_seconds": 0.450669,
          "time_to_last_output_seconds": 28.5371,
          "ttft_seconds": 0.450669,
          "wall_seconds": 28.53896
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1947,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 26.109899,
          "decode_tokens_per_second": 74.531,
          "finish_reason": "stop",
          "measured_sse_events": 1945,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens\n// per second, up to a maximum burst size. The zero value is not usable;\n// construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64 // tokens per second\n\tburst    int     // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter that starts full with burst tokens, refilling at\n// rate tokens per second. clock may be nil, in which case the real clock\n// is used.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\trate:     rate,\n\t\tburst:    burst,\n\t\ttokens:   float64(burst),\n\t\tlastTime: now,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(time.Now(), 1) }\n\n// AllowN reports whether n tokens are available at time now, consuming\n// them if so. now is typically l's clock time; it is accepted for\n// flexibility and must not be before the limiter's last observed time.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n <= 0 {\n\t\treturn true\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now.After(l.lastTime) {\n\t\telapsed := now.Sub(l.lastTime).Seconds()\n\t\tl.tokens = minFloat(float64(l.burst), l.tokens+elapsed*l.rate)\n\t\tl.lastTime = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens, for diagnostics.\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n\nfunc minFloat(a, b float64) float64 {\n\tif a < b {\n\t\treturn a\n\t}\n\treturn b\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a deterministic, manually advanced clock.\ntype fakeClock struct {\n\tcurrent time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{current: start} }\n\nfunc (f *fakeClock) Now() time.Time { return f.current }\n\nfunc (f *fakeClock) Advance(d time.Duration) { f.current = f.current.Add(d) }\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr error\n\t}{\n\t\t{\"valid\", 10, 5, nil},\n\t\t{\"zero rate\", 0, 5, ErrInvalidConfig},\n\t\t{\"negative rate\", -1, 5, ErrInvalidConfig},\n\t\t{\"zero burst\", 10, 0, ErrInvalidConfig},\n\t\t{\"negative burst\", 10, -3, ErrInvalidConfig},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, nil)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"New(%v, %v) error = %v, want %v\", tc.rate, tc.burst, err, tc.wantErr)\n\t\t\t}\n\t\t\tif tc.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"expected non-nil limiter on success\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst exhausted then denied\",\n\t\t\trate:  1,\n\t\t\tburst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{0, 1, true},\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 tokens accrued\n\t\t\t\t{500 * time.Millisecond, 1, true},  // 1.0 token accrued\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens capped at burst\",\n\t\t\trate:  10,\n\t\t\tburst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{1 * time.Hour, 2, true}, // long idle: capped at 2, not 36002\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"multi-token request\",\n\t\t\trate:  2,\n\t\t\tburst: 5,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true},\n\t\t\t\t{0, 3, false}, // only 2 left\n\t\t\t\t{1 * time.Second, 3, true},\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"zero and negative n allowed\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0, true},\n\t\t\t\t{0, -5, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(base)\n\t\t\tl, err := New(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(fc.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v) = %v, want %v (tokens=%f)\", i, s.n, got, s.want, l.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesRealClock(t *testing.T) {\n\tl, err := New(1000, 1, nil)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"expected first Allow to succeed\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, err := New(100, 100, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tconst goroutines = 50\n\tdone := make(chan int, goroutines)\n\tfor i := 0; i < goroutines; i++ {\n\t\tgo func() { done <- boolToInt(l.AllowN(fc.Now(), 1)) }()\n\t}\n\tallowed := 0\n\tfor i := 0; i < goroutines; i++ {\n\t\tallowed += <-done\n\t}\n\tif allowed != 100 {\n\t\tt.Errorf(\"allowed = %d, want 100\", allowed)\n\t}\n}\n\nfunc boolToInt(b bool) int {\n\tif b {\n\t\treturn 1\n\t}\n\treturn 0\n}\n```",
          "output_characters": 5974,
          "output_sha256": "54e21b89f0b18e8aa4d2ca451fdefcf3b316d83c66b4903b6b6495030381792a",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.002887,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "54e21b89f0b18e8aa4d2ca451fdefcf3b316d83c66b4903b6b6495030381792a",
          "time_to_first_visible_seconds": 0.443552,
          "time_to_last_output_seconds": 26.55345,
          "ttft_seconds": 0.443552,
          "wall_seconds": 26.556338
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1967,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 26.686355,
          "decode_tokens_per_second": 73.671,
          "finish_reason": "stop",
          "measured_sse_events": 1965,
          "output": "**ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens per\n// second, up to capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. rate must be > 0 and capacity > 0.\nfunc New(rate float64, capacity int) (*Limiter, error) {\n\treturn NewWithClock(rate, capacity, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, capacity int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\t}\n\tif capacity <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: capacity must be positive\")\n\t}\n\tif math.IsNaN(rate) || rate <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: rate must be positive\")\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\tcapacity: float64(capacity),\n\t\trate:     rate,\n\t\ttokens:   float64(capacity),\n\t\tlast:     now,\n\t}, nil\n}\n\n// AllowN reports whether n tokens are available at time now, consuming them\n// if so. It is safe for concurrent use.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now.After(l.last) {\n\t\telapsed := now.Sub(l.last).Seconds()\n\t\tl.tokens = math.Min(l.capacity, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Allow reports whether one token is available at the current time.\nfunc (l *Limiter) Allow() bool { return l.AllowN(l.clock.Now(), 1) }\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n**ratelimit_test.go**\n```go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tcapacity int\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1, 5, true},\n\t\t{\"NaN rate\", nan(), 5, true},\n\t\t{\"zero capacity\", 10, 0, true},\n\t\t{\"negative capacity\", 10, -1, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.rate, tc.capacity)\n\t\t\tif gotErr := err != nil; gotErr != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %d) err = %v, wantErr %v\", tc.rate, tc.capacity, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc nan() float64 { f := 0.0; return f / f }\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tcapacity int\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\trate: 1, capacity: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\trate: 1, capacity: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true}, // 2 tokens accrued, cap 2\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"partial refill insufficient\",\n\t\t\trate: 1, capacity: 5,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 5, true},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 token\n\t\t\t\t{500 * time.Millisecond, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"capacity caps accumulation\",\n\t\t\trate: 10, capacity: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{time.Hour, 2, false}, // still capped at 1\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\trate: 1, capacity: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := NewWithClock(tc.rate, tc.capacity, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tnow := clk.Now()\n\t\t\t\tif got := l.AllowN(now, s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v, %d) = %v, want %v\", i, now, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesClock(t *testing.T) {\n\tclk := newFakeClock(base)\n\tl, err := NewWithClock(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow = true, want false\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrent(t *testing.T) {\n\tclk := newFakeClock(base)\n\tl, err := NewWithClock(1000, 100, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar mu sync.Mutex\n\tallowed := 0\n\tfor i := 0; i < 200; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(clk.Now(), 1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tallowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6048,
          "output_sha256": "332291a37067214c5a9ce32756daa8d5d9cd8587c0bd16bf00ae27177756c34e",
          "prompt_tokens": 132,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001548,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "332291a37067214c5a9ce32756daa8d5d9cd8587c0bd16bf00ae27177756c34e",
          "time_to_first_visible_seconds": 0.446189,
          "time_to_last_output_seconds": 27.132544,
          "ttft_seconds": 0.446189,
          "wall_seconds": 27.134092
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 28.537,
        "median": 27.051,
        "minimum": 25.847,
        "p90": 28.537
      },
      "ttft_seconds": {
        "maximum": 0.451,
        "median": 0.446,
        "minimum": 0.436,
        "p90": 0.451
      },
      "wall_seconds": {
        "maximum": 28.539,
        "median": 27.052,
        "minimum": 25.848,
        "p90": 28.539
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 44.636,
        "median": 43.101,
        "minimum": 42.505,
        "p90": 44.636
      },
      "pooled_decode_tokens_per_second": 43.324,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1035,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.16513,
          "decode_tokens_per_second": 44.636,
          "finish_reason": "stop",
          "measured_sse_events": 1033,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached \u2014 decision needed by end of month**\n\nOver the past three weeks we've been arguing, in standups and in the shared doc, about whether we should purchase additional GPU hardware or invest in improving utilisation of the fleet we already own. I want to lay out my thinking in one place, because the debate has been circling the same points without converging, and I think we're closer to an answer than the discussion suggests.\n\nStart with the numbers we can actually defend. Our fleet currently runs at roughly 40\u201345% average utilisation, but that average hides a bimodal reality: overnight batch jobs and scheduled evaluation runs leave the cards idle for long stretches, while peak daytime inference traffic pushes our primary serving nodes to the point where queue times become visible to users. The latency problem is real but narrow. Our p50 inference latency is fine; it's the p95 and p99 that degrade, and they degrade specifically during the afternoon window when the product team's new features drive concurrent load. Buying more hardware would fix that window. So would smoothing the load. The question is which one we can do well, quickly, and without regret.\n\nCost cuts both ways, and not symmetrically. New hardware is a capital commitment with a lead time \u2014 eight to twelve weeks for delivery, plus rack space, power, and the operational overhead of expanding a cluster that one person currently owns part-time. The utilisation path, by contrast, is mostly engineering time: better request batching, a proper scheduling layer so batch jobs yield to interactive traffic, and moving our evaluation workloads into the overnight trough where they currently don't run. The engineering time is not free, and I don't want to pretend it is. But it's expense we control week to week, against a purchase that locks in our architecture assumptions for three years. If model sizes or traffic patterns shift \u2014 and they have shifted twice in the last year \u2014 hardware bought for today's shape of work may be the wrong shape tomorrow.\n\nReliability is the argument I find most underrated on the \"buy more\" side, and I want to be fair to it. Right now we have no meaningful spare capacity. If a node fails during peak, we degrade visibly. A larger fleet gives us headroom, and headroom is a form of insurance. But insurance bought with capital is expensive insurance, and there's a cheaper version: if we get utilisation under control, we effectively manufacture headroom out of the capacity we're already wasting. A fleet running at 65% with good scheduling has more usable resilience than a fleet at 45% with none, because resilience is about having capacity available when you need it, not about owning it.\n\nThe opportunity cost question is the one that keeps me up. Every week we spend an engineer on scheduler work is a week not spent on the model-quality improvements the product roadmap actually depends on. That's a real cost. But the reverse is also true: every dollar spent on hardware we run at 40% is a dollar not spent on those same improvements, and hardware has the additional property of demanding ongoing attention \u2014 power, failures, driver churn \u2014 that quietly taxes the team forever. I'd rather pay a bounded, one-time engineering cost than an unbounded operational one.\n\nOn reversibility, the comparison isn't close. Scheduler improvements, batching changes, and workload reshuffling are all reversible. If they don't work, we've lost weeks, not quarters, and we'll have learned precisely where our bottlenecks actually are \u2014 which makes any future purchase better targeted. Hardware, once bought, is a sunk decision. We can resell it, but at a loss and with effort.\n\nSo my recommendation: commit to a six-week utilisation push, with a concrete target \u2014 p95 latency under our SLO at current peak, and fleet utilisation above 60% \u2014 and a pre-agreed trigger. If we miss the target, or if demand grows faster than our projections before we even finish, we place the hardware order immediately and without re-litigating. This isn't a way of avoiding the purchase; it's a way of making it, if needed, with better information.\n\nI want to acknowledge the uncertainty honestly. Our traffic forecast has a wide error band, and utilisation gains of the kind I'm describing are promised more often than they're delivered. If the scheduling work stalls or the gains prove illusory, we will have spent six weeks we could have spent shipping. I think that risk is acceptable because the downside is bounded and the information gained is durable either way. But I'm open to being argued out of the sequencing, if someone can show me the demand curve is steeper than I believe.\n\nLet's discuss at Thursday's planning session. Come with objections, not just vibes.",
          "output_characters": 4978,
          "output_sha256": "bbb82f2203e9875d373ff30a7550706c6ca136f043a49a19bcbca2d52b3f1d69",
          "prompt_tokens": 126,
          "reasoning_characters": 59,
          "reasoning_sha256": "2240e9e7fe5cec082dc7ebe9b2384d3b443959c6fdb62a512062bd696289f917",
          "response_tail_seconds": 0.000701,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "69991932c78f7ee42f340a3e180b03b96dd5e26c9927d218dcc1fc3c1667934e",
          "time_to_first_visible_seconds": 1.177379,
          "time_to_last_output_seconds": 23.640385,
          "ttft_seconds": 0.475255,
          "wall_seconds": 23.641086
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 999,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.479639,
          "decode_tokens_per_second": 42.505,
          "finish_reason": "stop",
          "measured_sse_events": 997,
          "output": "**Internal Memo: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** AI hardware capacity decision for the next two quarters\n\nWe've been running hot on our local inference hardware for about six weeks now. Queue times during peak hours have crept from seconds into minutes, and two of our internal customers have started asking whether they should just route around us and use a hosted API. Before we reflexively ask for budget to buy more GPUs, I want to think carefully about whether that's actually the right move, because I don't think it is \u2014 at least not yet.\n\nThe case for buying is obvious and emotionally compelling. Hardware is idle capacity that nobody has to think about. If we double the fleet, latency problems disappear overnight, on-call pages quiet down, and we stop having awkward conversations with teams whose batch jobs got preempted. The cost is real but not prohibitive: roughly the price of one senior engineer-year for the cards, plus rack power, cooling, and the operational burden of managing more physical inventory. The hidden cost, though, is opportunity cost. That same budget could fund the observability work we've been deferring, or a second person on the reliability rotation, or simply sit unspent. And hardware purchases are the least reversible decision available to us. Once we've bought cards, we own them. If our workload mix shifts \u2014 if a model change cuts our compute needs in half, or a project gets cancelled \u2014 we're stuck depreciating silicon we no longer need. Selling used GPUs recovers maybe half the outlay, and the market moves fast enough that waiting six months to resell often means losing more.\n\nThe case for improving utilisation is less glamorous but, I think, stronger. Our telemetry suggests we're running at roughly 40\u201350% effective utilisation across the fleet. That number is suspiciously low, and the gap between it and the 80%+ we should be able to reach is where the answer lives. Some of the waste is structural: we have models pinned to specific machines for no good reason, batch jobs scheduled naively at peak hours, and no preemption between workloads of different priority classes. Some of it is technical: our batching logic is conservative, our KV-cache management is crude, and we're serving several models that could be quantised harder with negligible quality loss. None of this is novel engineering. It's mostly unglamorous work \u2014 a proper scheduler, better metrics, a week of load-shaping \u2014 that we've skipped because buying hardware felt easier.\n\nOn latency, the utilisation path has a real advantage worth naming. Adding hardware doesn't fix the tail latency problems caused by bad scheduling; it just hides them until the new capacity fills up, which at our growth rate means six to nine months before we're having this conversation again. Fixing the scheduler and batching actually improves p99 latency at *lower* average utilisation, because the gains come from packing work intelligently rather than from raw capacity. Reliability cuts the other way, admittedly. A denser, more contended fleet has less slack to absorb a node failure, and our failure domain grows if we consolidate workloads onto fewer machines. That's a genuine risk, and it's why I'd pair any consolidation work with a hard commitment to keep N+1 redundancy for the priority-0 serving tier. We should also be honest that some of our reliability incidents have been self-inflicted \u2014 config drift, silent thermal throttling on one node we didn't monitor \u2014 and more hardware would not have prevented either.\n\nSo my recommendation: spend the next six weeks on utilisation engineering, with a concrete target of 75% effective utilisation and p99 queue times under ten seconds at peak. Treat this as a real project with an owner and weekly checkpoints, not a background aspiration. If we hit the target, we likely defer the purchase by two quarters or more, and we'll have learned enough about our actual demand shape to buy the *right* hardware if we do buy. If we miss it \u2014 if the telemetry was wrong, or the remaining waste turns out to be irreducible \u2014 then we buy, and we'll do so with evidence rather than anxiety. The purchase option doesn't expire; it's fully reversible as a decision, just not as an expenditure. That asymmetry is exactly why we should exhaust the cheap, reversible improvements first.\n\nThe main uncertainty I want to flag: our utilisation numbers come from instrumentation we wrote ourselves and I don't fully trust. The first week of this effort should be validating those numbers. If they're wrong and we're actually at 70%, the whole calculus changes, and we should know that before we invest anyone's time. Let's decide at the next planning cycle whether the data supports the plan.",
          "output_characters": 4829,
          "output_sha256": "3bf7631193b1070b3dc53f2b717dd406a1bffadfc9c7907b0ae5fff9ffce50bb",
          "prompt_tokens": 123,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.001113,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "91a81e6d8067fcf886feda542ed56b8cc0e913e5c89aba61e49d00a851c94ccf",
          "time_to_first_visible_seconds": 1.103487,
          "time_to_last_output_seconds": 23.957826,
          "ttft_seconds": 0.478188,
          "wall_seconds": 23.958939
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1050,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 24.447815,
          "decode_tokens_per_second": 42.908,
          "finish_reason": "stop",
          "measured_sse_events": 1048,
          "output": "**Internal Memo: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Platform Engineering Team\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\nWe've been running hot on our local inference hardware for about six weeks now. Peak-hour GPU utilisation sits in the high eighties, queue times during the afternoon window have crept from seconds into the tens of seconds, and two of our internal customers have started asking whether they should just route around us to a hosted API. Before we reflexively file a purchase order, I want to think carefully about whether buying more hardware is actually the right answer, or whether we're leaving capacity on the table that we've already paid for.\n\nThe case for buying is straightforward on its face. Our three A100 nodes were sized for a workload that has roughly tripled. Latency is a product feature for the teams we serve: when p95 response time crosses about two seconds, adoption of the internal assistant tools drops measurably, and we're brushing against that threshold daily. New hardware would give us immediate headroom, and the procurement path is well understood. But the cost is not trivial. At current pricing, adding a fourth node is roughly a year of one engineer's fully loaded salary, and that comparison is the honest one, because the alternative use of that money is almost always people. Capital spent on silicon is capital not spent on the roadmap work that would reduce our costs structurally \u2014 better batching, model distillation, smarter caching.\n\nThat brings me to the case for improving utilisation, which I think is stronger than our dashboards suggest. Right now our serving stack runs models at batch sizes that were tuned eighteen months ago, we keep a large model resident for a workload that is mostly small requests, and we have no mechanism for preemption between the training jobs that occasionally squat on the cluster and the latency-sensitive inference traffic. Each of these is a known inefficiency, and collectively I'd estimate we could recover twenty to thirty percent of effective capacity without spending a dollar on hardware. Continuous batching alone, which we've prototyped but never shipped, looked like a fifteen percent throughput gain in a limited test. There's also the scheduling question: our utilisation numbers are averages, and averages hide the fact that capacity exists at night and on weekends that nobody is using. Some of our batch-eligible workloads \u2014 embedding refreshes, evaluation runs \u2014 could be shifted off-peak with modest engineering effort, flattening the peak that's driving this whole conversation.\n\nThe reliability dimension cuts both ways, and I want to be honest about it. Adding a fourth node improves our redundancy posture: today, losing a single machine during peak hours means degraded service for everyone, and that's a real operational risk that utilisation improvements do nothing to fix. Better scheduling, by contrast, can actually make reliability worse if we over-optimize \u2014 a cluster running at ninety-five percent utilisation has no slack to absorb a node failure or a traffic spike. If we go the utilisation route, we should deliberately target a ceiling, perhaps eighty-five percent effective utilisation, and treat the remaining headroom as insurance rather than waste.\n\nOn reversibility, the asymmetry favours the software work. Hardware is a commitment with a three-to-five-year depreciation horizon and real exit costs; if our workload mix changes \u2014 if distillation lets us serve most traffic on smaller models, or if hosted inference prices keep falling \u2014 we own machines we can't easily repurpose. Process improvements, by contrast, are almost entirely reversible. If we invest six weeks in better batching and scheduling and the gains disappoint, we've lost time and some engineering focus, but we can still buy hardware afterward, having learned exactly how much we actually need. The reverse is not true: hardware bought in haste is hard to unbuy, and it's remarkably easy to let new capacity absorb new inefficiency rather than solve the underlying problem.\n\nMy recommendation, then: defer the purchase for one quarter. In that time, ship continuous batching, implement priority-based preemption between training and inference, move the batch-eligible workloads off-peak, and re-measure. I'd budget roughly half of one engineer's time on this, which keeps the opportunity cost low. If, at the end of the quarter, p95 latency still breaches our threshold at eighty-five percent utilisation, we buy the fourth node with confidence and a precise understanding of why \u2014 and I'll write the purchase order myself.\n\nI want to acknowledge the uncertainty here honestly. My twenty-to-thirty percent recovery estimate is informed, not measured; the prototyping data is thin; and there's a chance our workload keeps growing fast enough that no amount of tuning keeps pace, in which case every week of delay is a week of degraded experience for our users. If the growth curve steepens, we should revisit this decision in weeks, not months. But given what we know today, I believe the disciplined path is to exhaust the capacity we've already paid for before buying more of it.\n\nComments welcome by end of week.",
          "output_characters": 5300,
          "output_sha256": "48caff3aab979c433015ebd5c3345cfd21e99c2fd4bc423ed64c825119fae6d7",
          "prompt_tokens": 125,
          "reasoning_characters": 59,
          "reasoning_sha256": "2240e9e7fe5cec082dc7ebe9b2384d3b443959c6fdb62a512062bd696289f917",
          "response_tail_seconds": 0.001682,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "9e2961dbb5b25aebd3e7f5724b439cc7f5d56776a51525bd20232e776486f3e6",
          "time_to_first_visible_seconds": 1.234569,
          "time_to_last_output_seconds": 24.929072,
          "ttft_seconds": 0.481258,
          "wall_seconds": 24.930754
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 992,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.775329,
          "decode_tokens_per_second": 43.512,
          "finish_reason": "stop",
          "measured_sse_events": 990,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached \u2014 decision needed by end of month**\n\nOver the past six weeks we've hit GPU saturation on our inference fleet three times, each incident producing p99 latency spikes well past our 800ms SLO and, in one case, a cascade of client retries that made the problem worse. The obvious question on the table is whether we should purchase additional hardware or invest in raising utilisation of the fleet we already own. I want to lay out my thinking before we meet Thursday.\n\nThe case for buying is straightforward on its face. Our peak-hour utilisation sits around 87 percent, which sounds healthy until you remember that GPU clusters degrade badly in the last stretch. Queueing delay grows non-linearly as you approach saturation, so those last few points of utilisation are exactly where our latency problems live. New hardware would give us immediate headroom, and the capital cost \u2014 roughly \u00a3140k for two A100-class nodes \u2014 is not catastrophic relative to the revenue the inference path supports. Procurement also tells me lead times have shortened to about five weeks, which is materially better than last year.\n\nBut the case for buying is weaker than it first appears, and the reason is opportunity cost. That \u00a3140k is not just money; it's the budget we've been earmarking for the fine-tuning workstream, which I'd argue has higher expected value than raw serving capacity. More importantly, buying hardware now would lock in a decision about our workload shape that we're not yet in a position to make. Our traffic mix is shifting \u2014 batch jobs are growing faster than interactive traffic, and if that trend continues, the bottleneck moves from GPU compute to memory bandwidth and storage throughput. Hardware bought to solve today's problem may be the wrong shape for next quarter's.\n\nThe alternative \u2014 improving utilisation of what we own \u2014 is less glamorous but more instructive. Right now we're leaving capacity on the table in three ways. First, our autoscaling floor is conservative; we keep nodes warm that could be drained during trough hours, and our own telemetry suggests we could reclaim 8\u201310 percent of capacity there. Second, we haven't implemented request batching properly. Our serving stack processes requests individually when dynamic batching would let us pack several small requests into a single forward pass, and the literature plus our own load tests suggest this alone could double effective throughput for our traffic profile. Third, we have no meaningful priority queueing, so a burst of low-value batch jobs can starve interactive traffic \u2014 which is precisely what happened in the second incident.\n\nThese fixes are cheap in cash terms and expensive in engineering time: I estimate four to six weeks of focused work from two engineers. That's real cost, and I don't want to pretend otherwise. But it's reversible cost in a way capital expenditure is not. If we invest in batching and scheduling and it turns out the demand curve keeps climbing anyway, nothing is wasted \u2014 the software improvements compound with whatever hardware we eventually buy. If we buy hardware first and the demand curve flattens, we're sitting on depreciating silicon and a fine-tuning project we can't fund.\n\nOn reliability, the utilisation path has a second advantage worth naming. Right now we have no slack in the fleet, which means a single node failure during peak hours produces an outage, not a degradation. Ironically, running hot is itself a reliability risk. Squeezing more throughput from the same hardware, and then deliberately running it at a *lower* headline utilisation with better packing, gives us failure headroom we currently lack. Buying more hardware without fixing the scheduling would just mean more nodes failing simultaneously.\n\nMy recommendation, then: defer the purchase for one quarter. Fund the batching, autoscaling, and priority-queueing work now, with a hard checkpoint in eight weeks. If utilisation-adjusted latency is back inside SLO with headroom, we revisit hardware in Q3 with better data about our true demand curve. If the engineering work stalls or the demand growth outruns it, we buy \u2014 and we'll do so knowing exactly what shape of hardware we need, which will likely save us money on the purchase itself.\n\nI want to be honest about the uncertainty here. Demand forecasting in this business has a poor track record, including mine. If we land a large customer in the next six weeks, this whole analysis changes, and we should treat the checkpoint as a genuine decision point rather than a formality. But the asymmetry favours patience: the cost of waiting one quarter is bounded and small; the cost of buying the wrong hardware is neither.\n\nHappy to argue about this Thursday.",
          "output_characters": 4938,
          "output_sha256": "604ee3f75089a0e5be6cb9f10a7d02e15efad6af324d159e5e02cde83aea5300",
          "prompt_tokens": 127,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.001785,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "b936e63fcaf605bfd20c2635c9c8d95ea9a109d182a2855dadf43ac73f2ed83b",
          "time_to_first_visible_seconds": 1.084343,
          "time_to_last_output_seconds": 23.261474,
          "ttft_seconds": 0.486145,
          "wall_seconds": 23.263259
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1026,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.78128,
          "decode_tokens_per_second": 43.101,
          "finish_reason": "stop",
          "measured_sse_events": 1024,
          "output": "**Internal Memo: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Platform Engineering Team\n**From:** Infrastructure Lead\n**Re:** AI hardware capacity decision for the next two quarters\n\nWe've been running hot on our local inference hardware for about six weeks now. Queue times during peak hours have crept from seconds into the low minutes, and two of our internal customers have started asking whether they should just route their workloads to a hosted API instead. That question \u2014 whether our on-premise fleet is worth defending at all \u2014 is really what this memo is about. The immediate decision in front of us is narrower: do we buy more hardware, or do we invest in squeezing more out of what we already own?\n\nLet me start with the cost picture, because it's less one-sided than it first appears. A new node at the spec we'd want runs roughly the price of two senior engineers for a quarter, before power, cooling, and the rack space we'd need to negotiate for. But the cost of *not* buying is also real: idle engineers waiting on batch jobs are an expense we don't usually book to infrastructure, and they should be. If our utilisation problem is costing the team even ten percent of its throughput in lost focus and context-switching, that's a genuine line item. The honest framing is not \"hardware versus nothing\" but \"hardware versus the opportunity cost of everything else we could do with that capital and those engineer-hours.\"\n\nOn utilisation, the picture is more encouraging than our dashboards suggest. Our average GPU utilisation sits around forty percent, but that number hides the real story: we have a classic peak-trough pattern, with mornings and late afternoons saturated and nights nearly idle. We've never seriously attempted workload shaping \u2014 deferrable jobs like embedding refreshes and evaluation runs could move to off-peak windows with almost no downside. We also haven't implemented dynamic batching on our two highest-traffic model endpoints, and our reading of the literature and a quick spike last week suggests that alone could cut peak-hour queue times by a third to a half. There's a catch, though: these are engineering projects, not configuration changes. Dynamic batching in particular touches our serving layer in ways that could introduce new latency variance for interactive requests, and we'd need careful load testing before we trust it. Budget two to three engineer-months, and accept that during that period the people doing the work aren't shipping features.\n\nOn latency, the two options pull in different directions. Buying hardware is the blunt instrument: it lowers queue times immediately and predictably, and it keeps our interactive latencies comfortably inside the SLA we've promised. Utilisation work is subtler. Workload shaping improves things for everyone by flattening the peak, but dynamic batching trades a small amount of per-request latency for higher aggregate throughput \u2014 a trade that's fine for batch consumers and potentially annoying for the interactive ones. We should be honest that some of our latency complaints are really about *reliability of latency*: people can tolerate a two-second response, but not a two-second response that's sometimes forty seconds. That variance is a utilisation problem more than a capacity problem, which tilts me toward fixing it first.\n\nReliability is the argument I take most seriously for buying. Right now we have no N+1 redundancy; if a node dies, we degrade to a queue-time regime that our customers will not accept, and our current utilisation spike is partly masking a single point of failure. A second node of our existing spec would buy genuine fault tolerance, not just capacity. That said, better scheduling and checkpointing of long-running jobs would also shrink the blast radius of a failure, and it's cheaper.\n\nSo here is my recommendation, with reasoning. I propose we do the utilisation work first \u2014 workload shaping, dynamic batching on the two hot endpoints, and proper queue observability \u2014 with a hard checkpoint at the end of the quarter. If peak queue times drop below our target and the latency variance story holds, we defer the purchase and revisit in the next planning cycle. If they don't, we buy, and we'll do so with much better data about what we actually need, which may well be a smaller and cheaper node than we'd have spec'd today. The utilisation path is also the more reversible one: if it fails, we've spent engineer-months and learned something; if we buy and it turns out unnecessary, the capital is gone and the depreciation clock is running.\n\nThe uncertainty I want to name plainly: our traffic is growing, and if growth accelerates, no amount of scheduling cleverness will save us. The checkpoint exists precisely to catch that. If the trend line bends the wrong way mid-quarter, we pull the trigger early rather than heroically squeezing a fleet that's already past its limit.\n\nI'd like a decision by Friday so we can staff the work. Happy to argue about it in the usual channel.",
          "output_characters": 5043,
          "output_sha256": "efba38122add44532059abb525474875261918957421cee256fc74df2f8d384c",
          "prompt_tokens": 125,
          "reasoning_characters": 62,
          "reasoning_sha256": "de67786d423b4f7b993b812759bdd5ac76d372c9ec74fec8fe6bd9fce0cde02c",
          "response_tail_seconds": 0.000397,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "5bb14d7fb2ff32dbd6968b5d7fee62dde045537a9a0e6051f25febaa6f9929f9",
          "time_to_first_visible_seconds": 1.161741,
          "time_to_last_output_seconds": 24.260561,
          "ttft_seconds": 0.479281,
          "wall_seconds": 24.260957
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 24.929,
        "median": 23.958,
        "minimum": 23.261,
        "p90": 24.929
      },
      "ttft_seconds": {
        "maximum": 0.486,
        "median": 0.479,
        "minimum": 0.475,
        "p90": 0.486
      },
      "wall_seconds": {
        "maximum": 24.931,
        "median": 23.959,
        "minimum": 23.263,
        "p90": 24.931
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 105.107,
        "median": 102.839,
        "minimum": 101.481,
        "p90": 105.107
      },
      "pooled_decode_tokens_per_second": 103.639,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 4.240268,
          "decode_tokens_per_second": 102.824,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 107,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001558,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.517438,
          "time_to_last_output_seconds": 4.757706,
          "ttft_seconds": 0.517438,
          "wall_seconds": 4.759264
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 4.239647,
          "decode_tokens_per_second": 102.839,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001394,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.51101,
          "time_to_last_output_seconds": 4.750658,
          "ttft_seconds": 0.51101,
          "wall_seconds": 4.752052
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 687,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.526694,
          "decode_tokens_per_second": 105.107,
          "finish_reason": "stop",
          "measured_sse_events": 685,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 108,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.00172,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "time_to_first_visible_seconds": 0.521056,
          "time_to_last_output_seconds": 7.047749,
          "ttft_seconds": 0.521056,
          "wall_seconds": 7.049469
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 4.296361,
          "decode_tokens_per_second": 101.481,
          "finish_reason": "stop",
          "measured_sse_events": 429,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.00126,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.517498,
          "time_to_last_output_seconds": 4.813859,
          "ttft_seconds": 0.517498,
          "wall_seconds": 4.81512
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 691,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.594709,
          "decode_tokens_per_second": 104.629,
          "finish_reason": "stop",
          "measured_sse_events": 689,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 17,
          "reasoning_sha256": "0d600f77fc578738aa0a0846f4e4a238c03e8eb8bb8442a6922a08f15a17ace4",
          "response_tail_seconds": 0.001767,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "c40c19b2a873122679d3f796a81d3c0e75f4d504007ea2f412068de8bffd1180",
          "time_to_first_visible_seconds": 0.709353,
          "time_to_last_output_seconds": 7.058021,
          "ttft_seconds": 0.463313,
          "wall_seconds": 7.059788
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 7.058,
        "median": 4.814,
        "minimum": 4.751,
        "p90": 7.058
      },
      "ttft_seconds": {
        "maximum": 0.521,
        "median": 0.517,
        "minimum": 0.463,
        "p90": 0.521
      },
      "wall_seconds": {
        "maximum": 7.06,
        "median": 4.815,
        "minimum": 4.752,
        "p90": 7.06
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2034.144,
          "median": 2032.272,
          "minimum": 2031.231,
          "p90": 2034.144
        },
        "runs": [
          {
            "cache_salt_sha256": "70658ac979822a67b7092fc1b5b6ea7b8039b44f4365ad9ab305a1c83ab233db",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061019,
            "decode_tokens_per_second": 114.718,
            "effective_prefill_tokens_per_second": 2034.144,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001606,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 16.108988,
            "time_to_last_output_seconds": 16.170007,
            "ttft_seconds": 16.108988,
            "wall_seconds": 16.171613
          },
          {
            "cache_salt_sha256": "37ab56f3e54709f8dab883523e5562461386db188c80773ef611848a1fde465d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054498,
            "decode_tokens_per_second": 128.444,
            "effective_prefill_tokens_per_second": 2032.272,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001673,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 16.123823,
            "time_to_last_output_seconds": 16.178321,
            "ttft_seconds": 16.123823,
            "wall_seconds": 16.179995
          },
          {
            "cache_salt_sha256": "ea308a14fc0761a74fb1c6b0e1d8ef64d6bb325d135aacec3401d765fc36b5af",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061889,
            "decode_tokens_per_second": 113.106,
            "effective_prefill_tokens_per_second": 2031.231,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00146,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 16.132092,
            "time_to_last_output_seconds": 16.193981,
            "ttft_seconds": 16.132092,
            "wall_seconds": 16.195441
          }
        ],
        "ttft_seconds": {
          "maximum": 16.132,
          "median": 16.124,
          "minimum": 16.109,
          "p90": 16.132
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 695135.663,
          "median": 599071.264,
          "minimum": 595110.966,
          "p90": 695135.663
        },
        "runs": [
          {
            "cache_salt_sha256": "70658ac979822a67b7092fc1b5b6ea7b8039b44f4365ad9ab305a1c83ab233db",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060131,
            "decode_tokens_per_second": 116.413,
            "effective_prefill_tokens_per_second": 595110.966,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00156,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.055062,
            "time_to_last_output_seconds": 0.115193,
            "ttft_seconds": 0.055062,
            "wall_seconds": 0.116753
          },
          {
            "cache_salt_sha256": "37ab56f3e54709f8dab883523e5562461386db188c80773ef611848a1fde465d",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.053195,
            "decode_tokens_per_second": 131.592,
            "effective_prefill_tokens_per_second": 695135.663,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001978,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 0.047139,
            "time_to_last_output_seconds": 0.100334,
            "ttft_seconds": 0.047139,
            "wall_seconds": 0.102312
          },
          {
            "cache_salt_sha256": "ea308a14fc0761a74fb1c6b0e1d8ef64d6bb325d135aacec3401d765fc36b5af",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.062079,
            "decode_tokens_per_second": 112.76,
            "effective_prefill_tokens_per_second": 599071.264,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002067,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.054698,
            "time_to_last_output_seconds": 0.116776,
            "ttft_seconds": 0.054698,
            "wall_seconds": 0.118844
          }
        ],
        "ttft_seconds": {
          "maximum": 0.055,
          "median": 0.055,
          "minimum": 0.047,
          "p90": 0.055
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2007.542,
          "median": 2006.778,
          "minimum": 2005.824,
          "p90": 2007.542
        },
        "runs": [
          {
            "cache_salt_sha256": "b65593392df7fa509c5138d737debb01a865dd3b5188f88a11ebb425643982ef",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.063332,
            "decode_tokens_per_second": 110.529,
            "effective_prefill_tokens_per_second": 2005.824,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.003081,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 32.672852,
            "time_to_last_output_seconds": 32.736184,
            "ttft_seconds": 32.672852,
            "wall_seconds": 32.739264
          },
          {
            "cache_salt_sha256": "1a344dd31bfbba9e7606f9322bde20ad0946f7d0a839e352818e32540c0d865d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062902,
            "decode_tokens_per_second": 111.285,
            "effective_prefill_tokens_per_second": 2006.778,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001996,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 32.657332,
            "time_to_last_output_seconds": 32.720234,
            "ttft_seconds": 32.657332,
            "wall_seconds": 32.722231
          },
          {
            "cache_salt_sha256": "bf13ae82435c659b602dee0e5b72a432cf72196031a0c058467a4a17085ac1f1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062487,
            "decode_tokens_per_second": 112.023,
            "effective_prefill_tokens_per_second": 2007.542,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001769,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 32.64489,
            "time_to_last_output_seconds": 32.707377,
            "ttft_seconds": 32.64489,
            "wall_seconds": 32.709146
          }
        ],
        "ttft_seconds": {
          "maximum": 32.673,
          "median": 32.657,
          "minimum": 32.645,
          "p90": 32.673
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1080044.167,
          "median": 996063.531,
          "minimum": 984630.178,
          "p90": 1080044.167
        },
        "runs": [
          {
            "cache_salt_sha256": "b65593392df7fa509c5138d737debb01a865dd3b5188f88a11ebb425643982ef",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064517,
            "decode_tokens_per_second": 108.498,
            "effective_prefill_tokens_per_second": 984630.178,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002369,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.066559,
            "time_to_last_output_seconds": 0.131076,
            "ttft_seconds": 0.066559,
            "wall_seconds": 0.133445
          },
          {
            "cache_salt_sha256": "1a344dd31bfbba9e7606f9322bde20ad0946f7d0a839e352818e32540c0d865d",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063824,
            "decode_tokens_per_second": 109.677,
            "effective_prefill_tokens_per_second": 1080044.167,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002438,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.060679,
            "time_to_last_output_seconds": 0.124503,
            "ttft_seconds": 0.060679,
            "wall_seconds": 0.126941
          },
          {
            "cache_salt_sha256": "bf13ae82435c659b602dee0e5b72a432cf72196031a0c058467a4a17085ac1f1",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063385,
            "decode_tokens_per_second": 110.436,
            "effective_prefill_tokens_per_second": 996063.531,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00159,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.065795,
            "time_to_last_output_seconds": 0.12918,
            "ttft_seconds": 0.065795,
            "wall_seconds": 0.13077
          }
        ],
        "ttft_seconds": {
          "maximum": 0.067,
          "median": 0.066,
          "minimum": 0.061,
          "p90": 0.067
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2003.528,
          "median": 2003.413,
          "minimum": 1996.248,
          "p90": 2003.528
        },
        "runs": [
          {
            "cache_salt_sha256": "189de2be02a8a87c120c040eca0de43bbf1c202fadff3a9f3638dab82d659a5b",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053087,
            "decode_tokens_per_second": 131.859,
            "effective_prefill_tokens_per_second": 1996.248,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001553,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.103698,
            "time_to_last_output_seconds": 4.156785,
            "ttft_seconds": 4.103698,
            "wall_seconds": 4.158338
          },
          {
            "cache_salt_sha256": "57d4e9be83295ada084695c43db9f594c76e354b41b86aea09a0edbec91f8f0c",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055341,
            "decode_tokens_per_second": 126.488,
            "effective_prefill_tokens_per_second": 2003.528,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.002334,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 4.088787,
            "time_to_last_output_seconds": 4.144129,
            "ttft_seconds": 4.088787,
            "wall_seconds": 4.146463
          },
          {
            "cache_salt_sha256": "0e2a375e616370c47eb457b064722c55612a5b5ce60ed9b172a1a64f28f8fed0",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053674,
            "decode_tokens_per_second": 130.417,
            "effective_prefill_tokens_per_second": 2003.413,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.002017,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.089022,
            "time_to_last_output_seconds": 4.142696,
            "ttft_seconds": 4.089022,
            "wall_seconds": 4.144713
          }
        ],
        "ttft_seconds": {
          "maximum": 4.104,
          "median": 4.089,
          "minimum": 4.089,
          "p90": 4.104
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 218290.343,
          "median": 211291.945,
          "minimum": 181810.111,
          "p90": 218290.343
        },
        "runs": [
          {
            "cache_salt_sha256": "189de2be02a8a87c120c040eca0de43bbf1c202fadff3a9f3638dab82d659a5b",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.05359,
            "decode_tokens_per_second": 130.621,
            "effective_prefill_tokens_per_second": 181810.111,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001386,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.045058,
            "time_to_last_output_seconds": 0.098648,
            "ttft_seconds": 0.045058,
            "wall_seconds": 0.100034
          },
          {
            "cache_salt_sha256": "57d4e9be83295ada084695c43db9f594c76e354b41b86aea09a0edbec91f8f0c",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.05694,
            "decode_tokens_per_second": 122.936,
            "effective_prefill_tokens_per_second": 218290.343,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001528,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 0.037528,
            "time_to_last_output_seconds": 0.094468,
            "ttft_seconds": 0.037528,
            "wall_seconds": 0.095995
          },
          {
            "cache_salt_sha256": "0e2a375e616370c47eb457b064722c55612a5b5ce60ed9b172a1a64f28f8fed0",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.054962,
            "decode_tokens_per_second": 127.36,
            "effective_prefill_tokens_per_second": 211291.945,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001188,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.038771,
            "time_to_last_output_seconds": 0.093733,
            "ttft_seconds": 0.038771,
            "wall_seconds": 0.094921
          }
        ],
        "ttft_seconds": {
          "maximum": 0.045,
          "median": 0.039,
          "minimum": 0.038,
          "p90": 0.045
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "incoai/GLM-5.3-Flash-DFlash2 @ bf582e4e (Mia v1.10 default settings)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3",
      "model_revision": "13ba3ef94d88bc30b27691efa29a6012f66e964d (AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3 on Hugging Face; the same files, served from a local copy)",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-dflash2-low",
    "finished_at": "2026-10-08T11:19:27.522303+00:00",
    "label": "tp2-ours-dflash2-low",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T11:11:01.240050+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 8192,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "low"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 5,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}