{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 56.837,
        "median": 56.431,
        "minimum": 54.418,
        "p90": 56.837
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 63.016,
        "median": 62.515,
        "minimum": 59.746,
        "p90": 63.016
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.456,
        "median": 0.456,
        "minimum": 0.435,
        "p90": 0.456
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 56.431,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.078994,
              "decode_tokens_per_second": 62.515,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter with a full bucket. rate is tokens per second;\n// burst is the maximum bucket size. Both must be positive.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc",
              "output_characters": 965,
              "output_sha256": "fa260d46e1a08c2f8b9dc3b39806250aec7b86d804f52276fa40fa3d82bd503c",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001369,
              "stream_done_marker": true,
              "stream_sha256": "fa260d46e1a08c2f8b9dc3b39806250aec7b86d804f52276fa40fa3d82bd503c",
              "time_to_first_visible_seconds": 0.455615,
              "time_to_last_output_seconds": 4.534609,
              "ttft_seconds": 0.455615,
              "wall_seconds": 4.535979
            }
          ],
          "wall_seconds": 4.536494
        },
        {
          "aggregate_end_to_end_tokens_per_second": 56.837,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.046608,
              "decode_tokens_per_second": 63.016,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock provides the current time; injectable for testing.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate      = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock             = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the",
              "output_characters": 1013,
              "output_sha256": "7c185e640f0c266b5ef465503e8018712405e4a250d69062296a388b6033152e",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001048,
              "stream_done_marker": true,
              "stream_sha256": "7c185e640f0c266b5ef465503e8018712405e4a250d69062296a388b6033152e",
              "time_to_first_visible_seconds": 0.455859,
              "time_to_last_output_seconds": 4.502467,
              "ttft_seconds": 0.455859,
              "wall_seconds": 4.503515
            }
          ],
          "wall_seconds": 4.504146
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.418,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.268065,
              "decode_tokens_per_second": 59.746,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"fmt\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time so tests can inject a fake.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\tcapacity float64\n\trate     float64 // tokens per second\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with a full bucket. capacity is the maximum burst\n// size; rate is the steady-state refill in tokens per second.\nfunc New(capacity int, rate float64) (*L",
              "output_characters": 1019,
              "output_sha256": "1bb9ce32ead29265431f8fe3eac6f40c7576944a97d21cadc2333acc15b960ca",
              "prompt_tokens": 130,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000542,
              "stream_done_marker": true,
              "stream_sha256": "1bb9ce32ead29265431f8fe3eac6f40c7576944a97d21cadc2333acc15b960ca",
              "time_to_first_visible_seconds": 0.435262,
              "time_to_last_output_seconds": 4.703327,
              "ttft_seconds": 0.435262,
              "wall_seconds": 4.703869
            }
          ],
          "wall_seconds": 4.704352
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 77.656,
        "median": 74.22,
        "minimum": 71.812,
        "p90": 77.656
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 47.183,
        "median": 44.716,
        "minimum": 40.312,
        "p90": 45.118
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.867,
        "median": 0.802,
        "minimum": 0.795,
        "p90": 0.866
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 71.812,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 6.325591,
              "decode_tokens_per_second": 40.312,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n\tmaxDelay time.Duration\n}\n\nvar ErrInvalidParams = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. The bucket starts full. maxDelay bounds the wait returned by\n// AllowN when the caller requests a delay; zero means no bound.\nfunc New(rate, burst float64, clock Clock, maxDelay time.Duration) (*Limiter",
              "output_characters": 1037,
              "output_sha256": "2feafd9f0f9b2c17a8ff85cd530354985d973811f043c632112551f96f619ed9",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000846,
              "stream_done_marker": true,
              "stream_sha256": "2feafd9f0f9b2c17a8ff85cd530354985d973811f043c632112551f96f619ed9",
              "time_to_first_visible_seconds": 0.802087,
              "time_to_last_output_seconds": 7.127678,
              "ttft_seconds": 0.802087,
              "wall_seconds": 7.128523
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.681401,
              "decode_tokens_per_second": 44.883,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second, > 0\n\tburst int     // capacity, > 0\n\tclock Clock\n}\n\nvar (\n\tErrInvalidRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrInvalidBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock     = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given refill rate (tokens/sec), burst\n// capacity, and clock. If",
              "output_characters": 939,
              "output_sha256": "68922af0e2da68e9390accc0f760052028ccfb21d0f4975cc60f57aa8ee71155",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000965,
              "stream_done_marker": true,
              "stream_sha256": "68922af0e2da68e9390accc0f760052028ccfb21d0f4975cc60f57aa8ee71155",
              "time_to_first_visible_seconds": 0.802494,
              "time_to_last_output_seconds": 6.483894,
              "ttft_seconds": 0.802494,
              "wall_seconds": 6.484859
            }
          ],
          "wall_seconds": 7.129685
        },
        {
          "aggregate_end_to_end_tokens_per_second": 74.22,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 6.100006,
              "decode_tokens_per_second": 41.803,
              "finish_reason": "length",
              "measured_sse_events": 250,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It starts full and refills at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(cap",
              "output_characters": 1005,
              "output_sha256": "7e5817253a8aa6435a557462aa69a439cbdc18221beb32f4a6facaa87de7a876",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.002302,
              "stream_done_marker": true,
              "stream_sha256": "7e5817253a8aa6435a557462aa69a439cbdc18221beb32f4a6facaa87de7a876",
              "time_to_first_visible_seconds": 0.794931,
              "time_to_last_output_seconds": 6.894937,
              "ttft_seconds": 0.794931,
              "wall_seconds": 6.897239
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.651841,
              "decode_tokens_per_second": 45.118,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with a full bucket. rate > 0 and burst > 0 are required.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"",
              "output_characters": 976,
              "output_sha256": "819f1573a98b8d798567a1645dede99147e9d07453a7b77d76c8c6bc2d6cab13",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001209,
              "stream_done_marker": true,
              "stream_sha256": "819f1573a98b8d798567a1645dede99147e9d07453a7b77d76c8c6bc2d6cab13",
              "time_to_first_visible_seconds": 0.795665,
              "time_to_last_output_seconds": 6.447506,
              "ttft_seconds": 0.795665,
              "wall_seconds": 6.448715
            }
          ],
          "wall_seconds": 6.898453
        },
        {
          "aggregate_end_to_end_tokens_per_second": 77.656,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.724141,
              "decode_tokens_per_second": 44.548,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds at most Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n}\n\nvar ErrInvalidParams = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with a full bucket. rate > 0 and burst > 0 are required.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"",
              "output_characters": 978,
              "output_sha256": "d3f5933c8ff308d4d65c3ad51d4831e1f756f7bcf0eb4553a2dd62fdc9ca7708",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.002304,
              "stream_done_marker": true,
              "stream_sha256": "d3f5933c8ff308d4d65c3ad51d4831e1f756f7bcf0eb4553a2dd62fdc9ca7708",
              "time_to_first_visible_seconds": 0.866012,
              "time_to_last_output_seconds": 6.590153,
              "ttft_seconds": 0.866012,
              "wall_seconds": 6.592457
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 5.404469,
              "decode_tokens_per_second": 47.183,
              "finish_reason": "length",
              "measured_sse_events": 248,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 || rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock",
              "output_characters": 965,
              "output_sha256": "02e920f262606ce03aaa52a2d6dc396009c9fae4d9800ab4c45337356c600b9a",
              "prompt_tokens": 129,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000834,
              "stream_done_marker": true,
              "stream_sha256": "02e920f262606ce03aaa52a2d6dc396009c9fae4d9800ab4c45337356c600b9a",
              "time_to_first_visible_seconds": 0.866721,
              "time_to_last_output_seconds": 6.27119,
              "ttft_seconds": 0.866721,
              "wall_seconds": 6.272024
            }
          ],
          "wall_seconds": 6.593175
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 109.04,
        "median": 105.551,
        "minimum": 104.553,
        "p90": 109.04
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 33.311,
        "median": 30.002,
        "minimum": 27.834,
        "p90": 30.895
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.634,
        "median": 0.63,
        "minimum": 0.618,
        "p90": 0.633
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 109.04,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.756113,
              "decode_tokens_per_second": 29.123,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\n// A Limiter must be constructed via NewLimiter; the zero value is invalid.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\tburst int\n\trate  float64\n\tclock Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: burst must be > 0 and rate must be > 0\")\n\n// NewLimiter creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc NewLimiter(burst int, rate float64) (*Limiter, error) {\n\treturn NewLimiterWithClock(burst, rate",
              "output_characters": 986,
              "output_sha256": "6dc3fd82ee90871e5fc319b6a8569b0ff3d8a7a0f1b446ce9e5f836d2649dce7",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000889,
              "stream_done_marker": true,
              "stream_sha256": "6dc3fd82ee90871e5fc319b6a8569b0ff3d8a7a0f1b446ce9e5f836d2649dce7",
              "time_to_first_visible_seconds": 0.632511,
              "time_to_last_output_seconds": 9.388624,
              "ttft_seconds": 0.632511,
              "wall_seconds": 9.389513
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.640108,
              "decode_tokens_per_second": 29.514,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// capacity, starting full. A non-nil Clock may be injected for testing.\nfunc New(rate, burst float64, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\tnow := clock.Now()\n\treturn &",
              "output_characters": 987,
              "output_sha256": "7e829d600f954d6bfe24495be82b203665d39a154cfe77e031afe8a39e08cb9d",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000424,
              "stream_done_marker": true,
              "stream_sha256": "7e829d600f954d6bfe24495be82b203665d39a154cfe77e031afe8a39e08cb9d",
              "time_to_first_visible_seconds": 0.632634,
              "time_to_last_output_seconds": 9.272742,
              "ttft_seconds": 0.632634,
              "wall_seconds": 9.273167
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.253691,
              "decode_tokens_per_second": 30.895,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It permits bursts up to Burst tokens, refilling at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second\n\tburst  int\n\tclock  Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate must be > 0 and burst must be >= 1\")\n\n// New creates a Limiter with the given refill rate (tokens per second),\n// burst capacity, and clock. If clock is nil, RealClock is used.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst < 1 {\n\t\treturn nil, ErrInvalid",
              "output_characters": 967,
              "output_sha256": "4f983a7e125325e0cef95b0d21e4282fcf2b7d313ad3484d62bbba7cda0af745",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000624,
              "stream_done_marker": true,
              "stream_sha256": "4f983a7e125325e0cef95b0d21e4282fcf2b7d313ad3484d62bbba7cda0af745",
              "time_to_first_visible_seconds": 0.63254,
              "time_to_last_output_seconds": 8.88623,
              "ttft_seconds": 0.63254,
              "wall_seconds": 8.886854
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.640038,
              "decode_tokens_per_second": 29.514,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock provides the current time; injectable for testing.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// ErrInvalidConfig describes invalid limiter configuration.\nvar ErrInvalidConfig = errors.New(\"ratelimit: invalid configuration\")\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It starts full and refills at Rate tokens per second, up to Capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\tcapacity float64\n\trate     float64 // tokens per second\n\tclock    Clock\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). Both must be finite and > 0.\nfunc New(capacity, rate float64, clock Clock) (*",
              "output_characters": 1050,
              "output_sha256": "b107af7ddf1233c789d77aace9678d805c1670caf4447adc4204dffaebb7cbbe",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000534,
              "stream_done_marker": true,
              "stream_sha256": "b107af7ddf1233c789d77aace9678d805c1670caf4447adc4204dffaebb7cbbe",
              "time_to_first_visible_seconds": 0.633713,
              "time_to_last_output_seconds": 9.273751,
              "ttft_seconds": 0.633713,
              "wall_seconds": 9.274285
            }
          ],
          "wall_seconds": 9.391054
        },
        {
          "aggregate_end_to_end_tokens_per_second": 104.553,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.507927,
              "decode_tokens_per_second": 29.972,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// RealClock uses the system time.\ntype RealClock struct{}\n\n// Now returns the current wall-clock time.\nfunc (RealClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It starts full and refills at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tcapacity float64\n\trate     float64 // tokens per second\n\tlast     time.Time\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity, rate float64, clock Clock) (*Limiter, error",
              "output_characters": 1036,
              "output_sha256": "71867ec3d7623e0a13d73684816977ebc223cfa9367b87069584c1d747adcaef",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000676,
              "stream_done_marker": true,
              "stream_sha256": "71867ec3d7623e0a13d73684816977ebc223cfa9367b87069584c1d747adcaef",
              "time_to_first_visible_seconds": 0.630666,
              "time_to_last_output_seconds": 9.138593,
              "ttft_seconds": 0.630666,
              "wall_seconds": 9.139268
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 9.161463,
              "decode_tokens_per_second": 27.834,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```ratelimit.go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. Implementations must return monotonically\n// non-decreasing values for correct limiter behavior.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n//\n// The bucket holds up to Burst tokens and is refilled at Rate tokens per\n// second. Tokens are fractional internally; AllowN consumes whole tokens.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\tburst int\n\trate  float64\n\tclock Clock\n}\n\n// ErrInvalidConfig is returned by New when arguments are out of range.\nvar ErrInvalidConfig = errors.New(\"ratelimit: burst must be > 0 and rate must be > 0 and finite\")\n\n// New returns a Limiter with a full bucket of burst tokens refilled at rate\n// tokens per second. clock may be nil, in which case the real clock is used.\nfunc New(burst int, rate float64, clock Clock) (*L",
              "output_characters": 1040,
              "output_sha256": "19fc5310364aa3a6fe08f15b8a129a854b32e33b66453f0078ab552dee69e55d",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001115,
              "stream_done_marker": true,
              "stream_sha256": "19fc5310364aa3a6fe08f15b8a129a854b32e33b66453f0078ab552dee69e55d",
              "time_to_first_visible_seconds": 0.629258,
              "time_to_last_output_seconds": 9.790721,
              "ttft_seconds": 0.629258,
              "wall_seconds": 9.791836
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.366499,
              "decode_tokens_per_second": 30.479,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\nvar (\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n)\n\n// New returns a Limiter that starts full, refilling at rate tokens per\n//",
              "output_characters": 996,
              "output_sha256": "28337ae560e0f143db63fbf703fb1eae901defc7d9a51503a8b67ea5ae5db2bc",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001405,
              "stream_done_marker": true,
              "stream_sha256": "28337ae560e0f143db63fbf703fb1eae901defc7d9a51503a8b67ea5ae5db2bc",
              "time_to_first_visible_seconds": 0.629574,
              "time_to_last_output_seconds": 8.996073,
              "ttft_seconds": 0.629574,
              "wall_seconds": 8.997478
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.262867,
              "decode_tokens_per_second": 30.861,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. Implementations must return monotonically\n// non-decreasing values.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system wall clock.\nfunc RealClock() Clock { return realClock{} }\n\nvar (\n\tErrZeroRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrZeroCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n)\n\n// Limiter is a token-bucket rate limiter. The zero value is not usable; use\n// New. All methods are safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tcapacity float64\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// New returns a Limiter that refills at rate tokens per second",
              "output_characters": 993,
              "output_sha256": "0bafed93def9463abc13dff90941a92ecbe25d733aa9f706ca857242d31a2fa9",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000566,
              "stream_done_marker": true,
              "stream_sha256": "0bafed93def9463abc13dff90941a92ecbe25d733aa9f706ca857242d31a2fa9",
              "time_to_first_visible_seconds": 0.629758,
              "time_to_last_output_seconds": 8.892625,
              "ttft_seconds": 0.629758,
              "wall_seconds": 8.893191
            }
          ],
          "wall_seconds": 9.7941
        },
        {
          "aggregate_end_to_end_tokens_per_second": 105.551,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.491132,
              "decode_tokens_per_second": 30.031,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second, > 0\n\tburst int     // bucket capacity, > 0\n\tclock Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate/burst or nil clock.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive and clock must be non-nil\")\n\n// New creates a Limiter with a full bucket.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= 0 || clock",
              "output_characters": 986,
              "output_sha256": "ae8c24d5e4986258d70008a3bfb6b683d142d68cba7143cee0725dd6825881bc",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000255,
              "stream_done_marker": true,
              "stream_sha256": "ae8c24d5e4986258d70008a3bfb6b683d142d68cba7143cee0725dd6825881bc",
              "time_to_first_visible_seconds": 0.617798,
              "time_to_last_output_seconds": 9.10893,
              "ttft_seconds": 0.617798,
              "wall_seconds": 9.109185
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 7.655066,
              "decode_tokens_per_second": 33.311,
              "finish_reason": "length",
              "measured_sse_events": 251,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter with a full bucket, using the real clock.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock returns a Limiter with a full bucket and an injectable clock.\nfunc NewWithClock(rate float64, burst",
              "output_characters": 962,
              "output_sha256": "e7275172f433e0a018ec0a2b85f0de05191f7053c15c51c7f650ae1d7f9e3263",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000599,
              "stream_done_marker": true,
              "stream_sha256": "e7275172f433e0a018ec0a2b85f0de05191f7053c15c51c7f650ae1d7f9e3263",
              "time_to_first_visible_seconds": 0.618697,
              "time_to_last_output_seconds": 8.273763,
              "ttft_seconds": 0.618697,
              "wall_seconds": 8.274361
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 8.489041,
              "decode_tokens_per_second": 30.039,
              "finish_reason": "length",
              "measured_sse_events": 249,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"fmt\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. It is an interface so tests can inject a fake.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n//\n// The bucket starts full with `burst` tokens and is refilled at `rate` tokens\n// per second. All methods are safe for concurrent use.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second\n\tburst  float64 // bucket capacity\n\tclock  Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratel",
              "output_characters": 985,
              "output_sha256": "57caa001985971f9c1633c8ab4fd8fb4e4e89d783cd6e54a7696f08a174135a9",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000525,
              "stream_done_marker": true,
              "stream_sha256": "57caa001985971f9c1633c8ab4fd8fb4e4e89d783cd6e54a7696f08a174135a9",
              "time_to_first_visible_seconds": 0.619497,
              "time_to_last_output_seconds": 9.108538,
              "ttft_seconds": 0.619497,
              "wall_seconds": 9.109063
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 9.079063,
              "decode_tokens_per_second": 28.087,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. It is an abstraction over time.Now to\n// allow deterministic testing.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter. Tokens refill\n// continuously at rate tokensPerSecond, up to the bucket capacity.\n// The zero value is not usable; construct with New.\ntype Limiter struct {\n\tmu   sync.Mutex\n\ttoks float64 // current tokens\n\tcap  float64 // bucket capacity\n\trate float64 // tokens per second\n\tlast time.Time\n\tclk  Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCap  = errors.New(\"ratelimit: capacity",
              "output_characters": 1016,
              "output_sha256": "3e062257c546aa1ce6ad4a50438a83baa7d92508c6095dcfa1a8be8a3218b3a4",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.001561,
              "stream_done_marker": true,
              "stream_sha256": "3e062257c546aa1ce6ad4a50438a83baa7d92508c6095dcfa1a8be8a3218b3a4",
              "time_to_first_visible_seconds": 0.619512,
              "time_to_last_output_seconds": 9.698575,
              "ttft_seconds": 0.619512,
              "wall_seconds": 9.700137
            }
          ],
          "wall_seconds": 9.70144
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 77.468,
        "median": 75.588,
        "minimum": 71.571,
        "p90": 77.468
      },
      "pooled_decode_tokens_per_second": 74.944,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2074,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 28.016902,
          "decode_tokens_per_second": 73.991,
          "finish_reason": "stop",
          "measured_sse_events": 2072,
          "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tcapacity float64\n\trate     float64 // tokens per second\n\tlast     time.Time\n\tclock    Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given burst capacity, refill rate\n// (tokens per second), and clock. It starts full of tokens.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, ErrNonPositiveCapacity\n\t}\n\tif rate <= 0 || math.IsNaN(rate) || math.IsInf(rate, 0) {\n\t\treturn nil, ErrNonPositiveRate\n\t}\n\tif clock == nil {\n\t\treturn nil, ErrNilClock\n\t}\n\treturn &Limiter{\n\t\ttokens:   float64(capacity),\n\t\tcapacity: float64(capacity),\n\t\trate:     rate,\n\t\tlast:     clock.Now(),\n\t\tclock:    clock,\n\t}, nil\n}\n\n// refill advances the bucket to time now, adding accrued tokens.\n// Caller must hold l.mu.\nfunc (l *Limiter) refill(now time.Time) {\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = math.Min(l.capacity, l.tokens+elapsed*l.rate)\n\t}\n\tl.last = now\n}\n\n// AllowN reports whether n tokens may be consumed at time now.\n// If allowed, the tokens are deducted.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(now)\n\tneed := float64(n)\n\tif l.tokens >= need {\n\t\tl.tokens -= need\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Allow reports whether one token may be consumed now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.clock.Now(), 1)\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(l.clock.Now())\n\treturn l.tokens\n}\n```\n\n**ratelimit/ratelimit_test.go**\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n}\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tclock    Clock\n\t\twantErr  error\n\t}{\n\t\t{\"ok\", 10, 5, newFakeClock(), nil},\n\t\t{\"zero capacity\", 0, 5, newFakeClock(), ErrNonPositiveCapacity},\n\t\t{\"negative capacity\", -1, 5, newFakeClock(), ErrNonPositiveCapacity},\n\t\t{\"zero rate\", 10, 0, newFakeClock(), ErrNonPositiveRate},\n\t\t{\"negative rate\", 10, -1, newFakeClock(), ErrNonPositiveRate},\n\t\t{\"NaN rate\", 10, float64(\"nan\" == \"x\" && 1) * 0 / 0, newFakeClock(), ErrNonPositiveRate},\n\t\t{\"Inf rate\", 10, 1e308 * 10, newFakeClock(), ErrNonPositiveRate},\n\t\t{\"nil clock\", 10, 5, nil, ErrNilClock},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.capacity, tc.rate, tc.clock)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"New() err = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t\tif tc.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"New() returned nil limiter without error\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := newFakeClock()\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"capacity caps accumulation\",\n\t\t\tcapacity: 2, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{time.Hour, 3, false},\n\t\t\t\t{0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"large request denied atomically\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 4, true}, {0, 2, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 5, true},\n\t\t\t},\n\t\t},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclock := newFakeClock()\n\t\t\tl, err := New(tc.capacity, tc.rate, clock)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() err = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclock.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(clock.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v, %d) = %v, want %v\", i, s.advance, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestRefillIsContinuous(t *testing.T) {\n\tclock := newFakeClock()\n\tl, _ := New(1, 2, clock) // 1 token capacity, 2 tokens/sec\n\tl.AllowN(clock.Now(), 1) // empty\n\tclock.Advance(250 * time.Millisecond)\n\tif got := l.AllowN(clock.Now(), 1); !got {\n\t\tt.Error(\"expected token after 250ms at 2 tokens/sec\")\n\t}\n}\n\nfunc TestConcurrentUse(t *testing.T) {\n\tclock := newFakeClock()\n\tl, _ := New(100, 1000, clock)\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.AllowN(clock.Now(), 1) {\n\t\t\t\t\tallowed++\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6445,
          "output_sha256": "d0606b6f440bfc44154cb53d6dea6bdeee2e8b9faf30d9dcaf8a3808580e8f92",
          "prompt_tokens": 130,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.003043,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "d0606b6f440bfc44154cb53d6dea6bdeee2e8b9faf30d9dcaf8a3808580e8f92",
          "time_to_first_visible_seconds": 0.433751,
          "time_to_last_output_seconds": 28.450654,
          "ttft_seconds": 0.433751,
          "wall_seconds": 28.453697
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2011,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 28.083912,
          "decode_tokens_per_second": 71.571,
          "finish_reason": "stop",
          "measured_sse_events": 2009,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with a full bucket. rate is tokens per second;\n// burst is the maximum bucket size. Both must be > 0.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\t}\n\tif rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\treturn &Limiter{\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\trate:   rate,\n\t\tburst:  float64(burst),\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now,\n// consuming them if so. It is safe for concurrent use.\nfunc (l *Limiter) AllowN(n int) bool {\n\treturn l.AllowNAt(l.clock.Now(), n)\n}\n\n// AllowNAt is AllowN at an explicit time; useful for tests.\nfunc (l *Limiter) AllowNAt(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now.After(l.last) {\n\t\telapsed := now.Sub(l.last).Seconds()\n\t\tl.tokens = min(l.burst, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (fractional).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n\nfunc min(a, b float64) float64 {\n\tif a < b {\n\t\treturn a\n\t}\n\treturn b\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tnow time.Time\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.now }\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr error\n\t}{\n\t\t{\"ok\", 10, 5, nil},\n\t\t{\"zero rate\", 0, 5, ErrInvalidConfig},\n\t\t{\"negative rate\", -1, 5, ErrInvalidConfig},\n\t\t{\"zero burst\", 10, 0, ErrInvalidConfig},\n\t\t{\"negative burst\", 10, -1, ErrInvalidConfig},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.rate, tc.burst)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"New(%v, %v) error = %v, want %v\", tc.rate, tc.burst, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n\n\tif _, err := NewWithClock(1, 1, nil); err == nil {\n\t\tt.Error(\"NewWithClock(nil clock): want error, got nil\")\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst up to capacity then deny\",\n\t\t\trate:  1, burst: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\trate:  1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens capped at burst\",\n\t\t\trate:  10, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{10 * time.Second, 2, true}, // capped, not 100\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"multi-token request\",\n\t\t\trate:  1, burst: 5,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false}, {3 * time.Second, 3, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"negative request denied\",\n\t\t\trate:  1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := &fakeClock{now: base}\n\t\t\tl, err := NewWithClock(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowNAt(fc.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowNConcurrent(t *testing.T) {\n\tfc := &fakeClock{now: time.Now()}\n\tl, err := NewWithClock(1000, 100, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tallowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed != 100 {\n\t\tt.Errorf(\"allowed = %d, want 100 (burst 100, rate high)\", allowed)\n\t}\n\tif l.AllowN(1) {\n\t\tt.Error(\"expected denial after bucket drained\")\n\t}\n}\n\nfunc TestTokens(t *testing.T) {\n\tfc := &fakeClock{now: time.Now()}\n\tl, err := NewWithClock(2, 4, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tif got := l.Tokens(); got != 4 {\n\t\tt.Errorf(\"initial tokens = %v, want 4\", got)\n\t}\n\tl.AllowN(3)\n\tif got := l.Tokens(); got != 1 {\n\t\tt.Errorf(\"tokens after AllowN(3) = %v, want 1\", got)\n\t}\n\tfc.Advance(time.Second)\n\tif got := l.Tokens(); got != 3 {\n\t\tt.Errorf(\"tokens after 1s = %v, want 3\", got)\n\t}\n}\n```",
          "output_characters": 5986,
          "output_sha256": "40008d6cdf912841852ea6ad2d2b347d3b823adc5583f4cb9023814982d183ab",
          "prompt_tokens": 132,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.002313,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "40008d6cdf912841852ea6ad2d2b347d3b823adc5583f4cb9023814982d183ab",
          "time_to_first_visible_seconds": 0.444735,
          "time_to_last_output_seconds": 28.528646,
          "ttft_seconds": 0.444735,
          "wall_seconds": 28.53096
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2014,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 26.631291,
          "decode_tokens_per_second": 75.588,
          "finish_reason": "stop",
          "measured_sse_events": 2012,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\n// A Limiter must be constructed via NewLimiter; the zero value is invalid.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\tburst  int\n\trate   float64 // tokens per second\n\tclock  Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: burst must be > 0 and rate must be > 0\")\n\n// NewLimiter creates a full bucket with the given burst capacity and\n// refill rate (tokens per second). Both must be positive.\nfunc NewLimiter(burst int, rate float64, clock Clock) (*Limiter, error) {\n\tif burst <= 0 || rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\treturn &Limiter{\n\t\ttokens: float64(burst),\n\t\tlast:   clock.Now(),\n\t\tburst:  burst,\n\t\trate:   rate,\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now,\n// consuming them if so. It never blocks.\nfunc (l *Limiter) AllowN(n int) bool {\n\treturn l.allowAt(l.clock.Now(), n)\n}\n\nfunc (l *Limiter) allowAt(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif !now.Before(l.last) {\n\t\telapsed := now.Sub(l.last).Seconds()\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > float64(l.burst) {\n\t\t\tl.tokens = float64(l.burst)\n\t\t}\n\t\tl.last = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewLimiterValidation(t *testing.T) {\n\tt.Parallel()\n\tcases := []struct {\n\t\tname    string\n\t\tburst   int\n\t\trate    float64\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 1, 1.0, false},\n\t\t{\"zero burst\", 0, 1.0, true},\n\t\t{\"negative burst\", -1, 1.0, true},\n\t\t{\"zero rate\", 1, 0, true},\n\t\t{\"negative rate\", 1, -0.5, true},\n\t}\n\tfor _, tc := range cases {\n\t\ttc := tc\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := NewLimiter(tc.burst, tc.rate, newFakeClock(time.Unix(0, 0)))\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"NewLimiter(%d, %f) err = %v, wantErr %v\", tc.burst, tc.rate, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tt.Parallel()\n\tbase := time.Unix(0, 0)\n\n\tcases := []struct {\n\t\tname  string\n\t\tburst int\n\t\trate  float64\n\t\tsteps []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst then deny\",\n\t\t\tburst: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\tburst: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {time.Second / 2, 1, false},\n\t\t\t\t{time.Second / 2, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens capped at burst\",\n\t\t\tburst: 2, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{time.Minute, 2, true}, // would refill 600, capped at 2\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"multi-token request\",\n\t\t\tburst: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false}, {3 * time.Second, 3, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"zero tokens requested\",\n\t\t\tburst: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0, true}, {0, 0, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"negative request denied\",\n\t\t\tburst: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range cases {\n\t\ttc := tc\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := NewLimiter(tc.burst, tc.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewLimiter: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tt.Parallel()\n\tclk := newFakeClock(time.Unix(0, 0))\n\tl, err := NewLimiter(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewLimiter: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow() = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow() = true, want false\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrentUse(t *testing.T) {\n\tt.Parallel()\n\tclk := newFakeClock(time.Unix(0, 0))\n\tl, err := NewLimiter(100, 1000, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewLimiter: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.AllowN(1) {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= 100 (burst)\", allowed)\n\t}\n}\n```",
          "output_characters": 5998,
          "output_sha256": "e555e13ed7b3d8c85a9cdbb8709de2a3ee8c300699e68d9167ddb1dddb53d7d5",
          "prompt_tokens": 133,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.002232,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "e555e13ed7b3d8c85a9cdbb8709de2a3ee8c300699e68d9167ddb1dddb53d7d5",
          "time_to_first_visible_seconds": 0.450339,
          "time_to_last_output_seconds": 27.08163,
          "ttft_seconds": 0.450339,
          "wall_seconds": 27.083862
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2082,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 26.862715,
          "decode_tokens_per_second": 77.468,
          "finish_reason": "stop",
          "measured_sse_events": 2080,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\n// A Limiter must be constructed with NewLimiter; the zero value is invalid.\ntype Limiter struct {\n\tmu     sync.Mutex\n\trate   float64 // tokens per second\n\tburst  float64 // bucket capacity\n\ttokens float64 // current tokens\n\tlast   time.Time\n\tclock  Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// NewLimiter creates a Limiter with the given refill rate (tokens/sec),\n// burst capacity, and clock. If clock is nil, the real clock is used.\nfunc NewLimiter(rate float64, burst float64, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\treturn &Limiter{\n\t\trate:   rate,\n\t\tburst:  burst,\n\t\ttokens: burst,\n\t\tclock:  clock,\n\t\tlast:   clock.Now(),\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// Tokens accrue lazily based on elapsed time since the last event.\nfunc (l *Limiter) AllowN(n float64) bool {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = min(l.burst, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif n < 0 || n > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= n\n\treturn true\n}\n\n// RetryAfter returns the duration until n tokens would be available,\n// or zero if they are available now. It does not consume tokens.\nfunc (l *Limiter) RetryAfter(n float64) time.Duration {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = min(l.burst, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif n <= l.tokens {\n\t\treturn 0\n\t}\n\tneed := n - l.tokens\n\treturn time.Duration(need / l.rate * float64(time.Second))\n}\n\nfunc min(a, b float64) float64 {\n\tif a < b {\n\t\treturn a\n\t}\n\treturn b\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewLimiterValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   float64\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1, 5, true},\n\t\t{\"zero burst\", 10, 0, true},\n\t\t{\"negative burst\", 10, -3, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := NewLimiter(tc.rate, tc.burst, nil)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"NewLimiter(%v, %v) err = %v, wantErr %v\", tc.rate, tc.burst, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tstart := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\ttests := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tburst    float64\n\t\tactions  []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       float64\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst up to capacity then denied\",\n\t\t\trate:  1, burst: 3,\n\t\t\tactions: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\trate:  2, burst: 2,\n\t\t\tactions: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true}, // 1 token accrued\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens capped at burst\",\n\t\t\trate:  10, burst: 2,\n\t\t\tactions: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{10 * time.Second, 2, true}, // capped, not 100\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"fractional request\",\n\t\t\trate:  1, burst: 1,\n\t\t\tactions: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0.5, true}, {0, 0.5, true}, {0, 0.5, false},\n\t\t\t\t{500 * time.Millisecond, 0.5, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"negative request denied\",\n\t\t\trate:  1, burst: 1,\n\t\t\tactions: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(start)\n\t\t\tl, err := NewLimiter(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewLimiter: %v\", err)\n\t\t\t}\n\t\t\tfor i, a := range tc.actions {\n\t\t\t\tfc.Advance(a.advance)\n\t\t\t\tif got := l.AllowN(a.n); got != a.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v) = %v, want %v\", i, a.n, got, a.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestRetryAfter(t *testing.T) {\n\tfc := newFakeClock(time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC))\n\tl, _ := NewLimiter(2, 2, fc)\n\n\tif d := l.RetryAfter(2); d != 0 {\n\t\tt.Errorf(\"RetryAfter(2) = %v, want 0\", d)\n\t}\n\tl.AllowN(2) // drain\n\n\tif d := l.RetryAfter(1); d != 500*time.Millisecond {\n\t\tt.Errorf(\"RetryAfter(1) = %v, want 500ms\", d)\n\t}\n\t// RetryAfter must not consume tokens.\n\tif d := l.RetryAfter(1); d != 500*time.Millisecond {\n\t\tt.Errorf(\"RetryAfter(1) again = %v, want 500ms\", d)\n\t}\n\tfc.Advance(500 * time.Millisecond)\n\tif !l.AllowN(1) {\n\t\tt.Error(\"AllowN(1) after waiting should succeed\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tl, _ := NewLimiter(1000, 100, RealClock())\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tallowed++\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, exceeds burst 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6291,
          "output_sha256": "7a1513be00674fdaa2f41e628df96f3d88d1f32cf38f88067384e5113d8e7720",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.002044,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "7a1513be00674fdaa2f41e628df96f3d88d1f32cf38f88067384e5113d8e7720",
          "time_to_first_visible_seconds": 0.440728,
          "time_to_last_output_seconds": 27.303443,
          "ttft_seconds": 0.440728,
          "wall_seconds": 27.305488
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2128,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 27.894117,
          "decode_tokens_per_second": 76.253,
          "finish_reason": "stop",
          "measured_sse_events": 2126,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second, > 0\n\tburst int     // capacity, > 0\n\tclock Clock\n}\n\nvar (\n\tErrRate      = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst     = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock  = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. If clock is nil, the real clock is used.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, ErrRate\n\t}\n\tif burst <= 0 {\n\t\treturn nil, ErrBurst\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\ttokens: float64(burst),\n\t\tlast:   now,\n\t\trate:   rate,\n\t\tburst:  burst,\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now, consuming them\n// if so. n must be non-negative; AllowN returns false for n > burst.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > float64(l.burst) {\n\t\t\tl.tokens = float64(l.burst)\n\t\t}\n\t\tl.last = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (for observability).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (c *fakeClock) Now() time.Time {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\treturn c.now\n}\n\nfunc (c *fakeClock) Advance(d time.Duration) {\n\tc.mu.Lock()\n\tdefer c.mu.Unlock()\n\tc.now = c.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 1, ErrRate},\n\t\t{\"negative rate\", -1.5, 1, ErrRate},\n\t\t{\"zero burst\", 1, 0, ErrBurst},\n\t\t{\"negative burst\", 1, -3, ErrBurst},\n\t\t{\"valid\", 2, 5, nil},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, nil)\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"unexpected error: %v\", err)\n\t\t\t\t}\n\t\t\t\tif l == nil {\n\t\t\t\t\tt.Fatal(\"expected non-nil limiter\")\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif err != tc.wantErr {\n\t\t\t\tt.Fatalf(\"got error %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tburst    int\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then denied\",\n\t\t\trate: 1, burst: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"tokens capped at burst\",\n\t\t\trate: 10, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{10 * time.Second, 3, false}, // capped at 2, not 100\n\t\t\t\t{0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"request larger than burst denied\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{time.Hour, 3, false},\n\t\t\t\t{0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"zero request allowed if any tokens\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0, true}, {0, 1, true}, {0, 0, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"fractional refill\",\n\t\t\trate: 2, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{400 * time.Millisecond, 1, false}, // 0.8 tokens\n\t\t\t\t{100 * time.Millisecond, 1, true},  // 1.0 tokens\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(base)\n\t\t\tl, err := New(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v (tokens=%f)\",\n\t\t\t\t\t\ti, s.n, got, s.want, l.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, err := New(1, 1, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow should succeed\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow should fail\")\n\t}\n\tfc.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill should succeed\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tl, err := New(1000, 100, nil)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 100; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed %d requests, want <= 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6351,
          "output_sha256": "fd9fccd6ade63aa6adb36982d6e10742b36c8c88dd88e25d8201c20146cee879",
          "prompt_tokens": 132,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.002461,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "fd9fccd6ade63aa6adb36982d6e10742b36c8c88dd88e25d8201c20146cee879",
          "time_to_first_visible_seconds": 0.443655,
          "time_to_last_output_seconds": 28.337772,
          "ttft_seconds": 0.443655,
          "wall_seconds": 28.340234
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 28.529,
        "median": 28.338,
        "minimum": 27.082,
        "p90": 28.529
      },
      "ttft_seconds": {
        "maximum": 0.45,
        "median": 0.444,
        "minimum": 0.434,
        "p90": 0.45
      },
      "wall_seconds": {
        "maximum": 28.531,
        "median": 28.34,
        "minimum": 27.084,
        "p90": 28.531
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 45.325,
        "median": 44.381,
        "minimum": 41.508,
        "p90": 45.325
      },
      "pooled_decode_tokens_per_second": 43.91,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1014,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.08661,
          "decode_tokens_per_second": 43.878,
          "finish_reason": "stop",
          "measured_sse_events": 1011,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached**\n\nOver the past month we've been asked, informally but repeatedly, whether we should purchase additional local GPU hardware or invest in improving utilisation of the eight A100s we already own. I want to lay out our thinking in one place, because the answer is more nuanced than the question suggests, and because the decision has a different shape depending on which time horizon you care about.\n\nLet me start with the case for buying. Our p95 inference latency has crept from 340ms to just over 900ms during peak hours, and that degradation is not evenly distributed \u2014 it lands almost entirely on the interactive product surface, where users notice. The obvious fix is more hardware. At current spot pricing, a two-node expansion would cost roughly \u00a3180k including networking, and it would solve the latency problem immediately and unambiguously. There is real appeal in that. Latency is not a soft metric for us; the product team has data showing session abandonment rises measurably once response times cross about 700ms, and we are now on the wrong side of that line for several hours each day.\n\nBut the utilisation numbers complicate the story. Our fleet averages 38% GPU utilisation over a full week, and the diurnal pattern is extreme: we saturate the cluster for perhaps four hours daily and leave it nearly idle overnight and on weekends. That is not a capacity problem in the classical sense \u2014 it is a scheduling and batching problem. Our current serving stack was built for a single model with predictable traffic, and we have since layered on three additional workloads, each with its own container, its own memory reservation, and its own conservative headroom assumptions. We are, in effect, running four half-empty aeroplanes on the same routes. Early experiments with dynamic batching and consolidated scheduling suggest we could recover 25\u201330 percentage points of utilisation without touching hardware, which would absorb our peak growth for at least two quarters.\n\nThe cost comparison, then, is not \u00a3180k versus nothing. It is \u00a3180k versus roughly six to eight weeks of engineering time from two people who are currently assigned to the retrieval-quality project. That is the opportunity cost, and it is not trivial \u2014 the retrieval work is the thing the product organisation most wants this quarter. If we redirect those engineers, something visible slips. If we buy hardware, the retrieval work proceeds and the latency problem disappears, but we also lock in a fleet whose running costs (power, colocation, and the eventual refresh cycle) grow by about \u00a340k annually, and we remove the pressure to fix the architectural debt that got us here.\n\nReliability cuts both ways, and I want to be honest about that. More hardware means more redundancy: we can survive a node failure during peak without degraded service, which we cannot do today. That is a genuine argument for buying, and it is the strongest one. On the other hand, the consolidation work required to improve utilisation would also simplify our serving topology \u2014 fewer memory-fragmented containers, fewer failure modes \u2014 and would likely reduce the incident rate we currently see from OOM kills and scheduler thrash. We would still be exposed to a single-node failure at peak until the utilisation gains free up headroom, and we should be candid that the transition period carries some risk.\n\nThe reversibility question is where I land most firmly. Hardware is a one-way door in practice. Once purchased, the capital is spent, the depreciation clock starts, and the organisational habit of \"just add a node\" becomes entrenched. The utilisation work is a two-way door: if it fails, we have spent engineering time, learned a great deal about our own traffic, and can then buy hardware with much better information about how much we actually need. There is also a decent chance the work pays for itself in deferred purchases even if it only partially succeeds.\n\nMy recommendation, then: commit to the utilisation project now, with a hard six-week timebox and a pre-agreed trigger \u2014 if p95 latency at peak has not fallen below 600ms by the end of it, or if the retrieval project slips more than two weeks as a result, we purchase the two-node expansion without further debate. This is not a decision to avoid spending; it is sequencing the spending so it happens once, informed, rather than twice.\n\nThe main uncertainty I want to flag is our growth forecast. If the new enterprise pilot signs, our traffic projections double and none of this analysis holds \u2014 we should buy immediately and in larger quantity. I'd suggest we revisit this memo within a fortnight of hearing on the pilot.\n\nHappy to walk through the utilisation data with anyone who wants the detail.",
          "output_characters": 4963,
          "output_sha256": "52b1669e377f47ff6141d21482ddb853194d53ae41bf8dbc960c1e334335b3d5",
          "prompt_tokens": 126,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.001965,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "e618d0c88e6e335938ac024f23f06fe4dd2624a086943b1e154d960da9203021",
          "time_to_first_visible_seconds": 1.091618,
          "time_to_last_output_seconds": 23.566987,
          "ttft_seconds": 0.480377,
          "wall_seconds": 23.568952
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 962,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.152072,
          "decode_tokens_per_second": 41.508,
          "finish_reason": "stop",
          "measured_sse_events": 960,
          "output": "# Memo: GPU Capacity \u2014 Buy More or Use What We Have Better?\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Local AI hardware \u2014 expansion vs. utilisation\n**Status:** Recommendation for discussion at Thursday's planning session\n\nWe've been asked whether we should purchase additional local AI hardware or invest in getting more out of the machines we already own. After two weeks of measurement and modelling, our view is that we should spend the next quarter improving utilisation first, and only revisit procurement once we've exhausted the cheap wins. Here's the reasoning.\n\nThe starting point is that our current fleet is running at roughly 38% average GPU utilisation, with a diurnal pattern that peaks around model evaluation runs and falls off sharply overnight and on weekends. That number matters because the marginal cost of the next GPU is not just its purchase price. A new high-memory accelerator runs to six figures once you include the server chassis, networking, power provisioning, and the engineering time to rack, configure, and integrate it. Against that, the utilisation work \u2014 a job scheduler with queueing, batch-aware inference routing, and better checkpointing so long-running jobs can be preempted \u2014 is mostly engineering time we can absorb, plus perhaps a modest spend on storage to make checkpointing fast enough to be practical.\n\nLatency is the strongest argument for buying more hardware, and we take it seriously. Our interactive inference p99 sits at 340ms, and under contention it degrades badly; a few batch jobs landing on the wrong node can push it past a second. If our product roadmap makes sub-200ms a hard requirement, more dedicated capacity becomes attractive quickly. But the measurements suggest the contention is largely self-inflicted: workloads are pinned to nodes by habit rather than scheduled by need. Isolating latency-sensitive inference onto two dedicated nodes and letting batch work float across the rest would, on our modelled numbers, hold p99 under 250ms without any new silicon. That's not quite the sub-200 target, but it's close enough that we should test it in practice before spending capital on a guess.\n\nReliability cuts both ways. More hardware means more redundancy, and right now a single node failure during a training run costs us days. But the deeper reliability problem is that we have no graceful preemption: when a node dies, jobs fail rather than migrate. Fixing that improves reliability on the fleet we have and is a prerequisite for using any future fleet well. Buying more machines without fixing this just multiplies the blast radius.\n\nThe opportunity cost consideration is the one that gives us the most pause. The engineering time for the utilisation project is real \u2014 we estimate six to eight weeks of two engineers' time \u2014 and those engineers could instead be shipping product features. If the team's highest-value work is product, not infrastructure, then buying hardware and accepting some inefficiency may genuinely be the better trade. That's a judgement about roadmap priorities that we can't make alone, and it's the main question we'd like the group to weigh in on. Our instinct is that the utilisation work is not throwaway: scheduler and checkpointing investment pays off regardless of fleet size, and every quarter of growth makes it more valuable.\n\nReversibility tips the balance. Hardware purchases are close to irreversible at our scale; a GPU we buy is a GPU we own, and if demand shifts \u2014 say a model change halves our training needs, or a hosted API becomes good enough for a workload \u2014 we're stuck with depreciating kit and a rack bill. Utilisation improvements are largely reversible and, more importantly, they buy us information. If after a quarter of scheduling discipline we're still saturating the fleet and latency targets are unmet, we'll have precise data on exactly what to buy, rather than ordering capacity shaped by our current inefficiencies. There's a real risk we over-provision today and regret it in six months; there's very little risk we regret learning to run what we own efficiently.\n\nTo be clear about uncertainty: our utilisation figures come from a month of telemetry and may not capture seasonal peaks; our latency model is simulated, not measured under the proposed isolation scheme; and we may simply be wrong that demand is flat enough for a quarter of deferral to be safe. If the roadmap contains a launch that will triple inference load in Q3, this analysis changes, and we'd want to start procurement lead times now.\n\n**Recommendation:** Commit to the utilisation quarter. Assign two engineers, set explicit success gates \u2014 p99 under 250ms, utilisation above 60% \u2014 and pre-approve procurement contingent on missing those gates. We'll return with measured results and a concrete purchase proposal either way.",
          "output_characters": 4861,
          "output_sha256": "635cb747028502f4b4cc4c83b46203d51735438e0aa4d21af33ae4b2131a9784",
          "prompt_tokens": 123,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.00177,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "635cb747028502f4b4cc4c83b46203d51735438e0aa4d21af33ae4b2131a9784",
          "time_to_first_visible_seconds": 0.527894,
          "time_to_last_output_seconds": 23.679966,
          "ttft_seconds": 0.527894,
          "wall_seconds": 23.681736
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 973,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 21.831979,
          "decode_tokens_per_second": 44.522,
          "finish_reason": "stop",
          "measured_sse_events": 971,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached**\n\nOver the past month we've been asked, informally but repeatedly, whether we should purchase additional local hardware to serve our growing inference workloads, or whether we should invest in squeezing more out of the machines we already own. This memo is an attempt to settle that question with actual numbers rather than vibes, and to be honest about where the numbers get fuzzy.\n\nLet me start with the case that's easiest to quantify. Our current fleet of four A100-class nodes runs at roughly 38% average utilisation, but that average hides a lot. During business hours on weekdays we routinely peak above 80%, and twice last quarter we queued jobs for 20 minutes or more. Overnight and on weekends, utilisation drops into the teens. The pattern is clear: we don't have a capacity problem so much as a *shape* problem. We own enough aggregate compute to handle our load, but it's concentrated in the wrong hours, and our batch scheduler is too conservative about packing smaller jobs onto partially occupied cards. A serious effort at better bin-packing, dynamic batching for the latency-tolerant workloads, and moving our internal evaluation jobs to an off-peak queue would, by our modelling, raise effective utilisation to the 60\u201370% range. That is the equivalent of adding two nodes to the fleet, at a cost of perhaps three to four engineer-weeks \u2014 a fraction of the capital cost of even one new machine, which currently runs us about $180k all-in with networking and storage.\n\nThe case for buying is not frivolous, though. Latency is where utilisation games get dangerous. Our customer-facing inference path has a p95 latency budget of 400ms, and the moment we start packing jobs tightly, we introduce contention. Dynamic batching helps throughput but adds queueing delay; aggressive bin-packing risks noisy-neighbour effects on the latency-sensitive jobs. We can mitigate this with strict partitioning \u2014 reserving dedicated capacity for the interactive path and packing only the batch workloads into the remainder \u2014 but that caps the achievable utilisation gain. My honest estimate is that we can get to 60% effective utilisation without touching p95 latency, and anything beyond that starts to trade against the user experience. If our growth forecast is right, we hit that ceiling in roughly five to seven months.\n\nSo the real question is not \"buy or optimise\" but \"optimise now, and buy when?\" And here the opportunity cost and reversibility considerations matter more than the raw arithmetic. Optimisation work is cheap, fast, and almost entirely reversible: if a scheduling change degrades latency, we roll it back in an afternoon. Buying hardware is the opposite. It's a six-figure, non-refundable commitment with a three-month lead time, and it locks us into a power and cooling footprint we can't easily shed if our workload shifts \u2014 say, if we move more inference to a hosted API for cost reasons, or if the model mix changes to favour smaller, cheaper deployments. Every dollar spent on hardware we underuse is a dollar not spent on the product work that actually differentiates us.\n\nThere is also a reliability argument that cuts both ways, and I want to be careful here. More nodes means more redundancy: right now, a single node failure during peak hours degrades service noticeably. But more nodes also means more failure surface to operate, and we are a small team. Our on-call burden is already the thing people complain about most. I'd rather operate four well-instrumented machines than six half-watched ones. The optimisation path, done properly, includes better monitoring and automated failover, which improves reliability *without* expanding the fleet.\n\nMy recommendation, then: fund the utilisation project now, with a hard scope of three engineer-weeks and a target of 60% effective utilisation with no p95 regression. Simultaneously, place a conditional purchase order for two additional nodes with a delivery date about four months out, structured so we can cancel or defer at minimal cost. That gives us the option value of hardware arriving just as we hit the optimisation ceiling, without paying for it today. If growth comes in slower than forecast, we defer; if faster, we're covered.\n\nThe main uncertainty is the growth forecast itself, which is based on two enterprise conversations that could go either way. I'd rather hold a cancellable option than bet the budget on them.\n\nHappy to walk through the utilisation model in detail at Thursday's sync.",
          "output_characters": 4714,
          "output_sha256": "2839056ed64f494028f4280cf593cc6c0c46001d11826a2f2d8c773464b0d4d0",
          "prompt_tokens": 125,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.002391,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "f85a65543c0be57ba8273d736085f5475c0a02de1c46a8af0520e0d52d0da851",
          "time_to_first_visible_seconds": 1.08563,
          "time_to_last_output_seconds": 22.312599,
          "ttft_seconds": 0.48062,
          "wall_seconds": 22.31499
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1036,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.834877,
          "decode_tokens_per_second": 45.325,
          "finish_reason": "stop",
          "measured_sse_events": 1033,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached**\n\nOver the past month we've been asked repeatedly whether we should purchase additional local hardware to serve our growing inference workloads, or whether we should invest in raising utilisation of the machines we already own. This memo lays out our thinking. The short version: we should spend the next six to eight weeks improving utilisation first, and only then commit to a purchase \u2014 and even then, likely a smaller one than the current request implies.\n\nLet's start with the economics, because they frame everything else. Our current fleet of eight GPUs is running at roughly 34% average utilisation, with a diurnal pattern that peaks near 70% for a few hours each afternoon and drops below 15% overnight. We are, in effect, paying for capacity we use a third of the time. The capital request on the table \u2014 four additional accelerators plus supporting networking and power \u2014 comes to a figure that would represent the largest single line item in our budget this year. Before we spend that, we owe ourselves a serious attempt to flatten the peaks and fill the troughs on hardware we've already paid for.\n\nThe obvious counterargument is latency. Our p99 inference latency is already near the edge of what product considers acceptable during afternoon peaks, and adding queueing or batch delays to improve utilisation could push it over. This is a real risk and we take it seriously. But the current peak is driven largely by a small number of chatty interactive workloads competing with long-running batch jobs for the same cards. If we partition the fleet \u2014 reserving two cards exclusively for latency-sensitive traffic and consolidating batch work onto the remainder \u2014 we can raise aggregate utilisation substantially without touching the tail latency of interactive requests. Early modelling suggests we can get aggregate utilisation into the 55\u201360% range this way. That is the equivalent of adding three cards, at a fraction of the cost.\n\nThere is also a reliability argument for consolidation that cuts both ways, and we want to be honest about it. Right now, workloads are spread thinly across all eight machines, which means a single hardware failure degrades almost everything a little. Concentrating batch work onto fewer machines means a failure takes out a bigger chunk of that batch capacity at once. On the other hand, a smaller active footprint means more spare cards, faster failover, and a simpler failure domain to reason about. Net, we think consolidation improves our recovery story, but it does concentrate risk in a way we'll need to monitor carefully, and we should build the failover runbook before, not after, we reorganise.\n\nThe opportunity cost cuts the other way, and it's the strongest argument for buying now. Every engineer-week we spend on scheduler work, batching improvements, and capacity partitioning is a week not spent on the product features that are driving this demand in the first place. If demand doubles again next quarter, no amount of utilisation tuning will save us, and we'll have delayed the inevitable purchase while burning engineering time. We've tried to price this: we estimate four to six engineer-weeks for the utilisation work, which is real money, but it is roughly an order of magnitude less than the hardware request, and the scheduling and partitioning infrastructure we build will be valuable regardless of what we eventually buy.\n\nFinally, reversibility. This is, to us, the decisive consideration. Software and process changes can be unwound in an afternoon; a hardware purchase cannot. If we buy four cards and demand shifts \u2014 a customer churns, a workload is optimised, a model change halves our compute per request \u2014 we are stuck with depreciating silicon and a higher power bill. If we invest in utilisation and it proves insufficient, we will have lost some weeks and gained a much sharper picture of exactly what we need to buy: how many cards, what interconnect, what power envelope. The purchase decision we'd make in eight weeks will be a better-informed one than the purchase decision we could make today.\n\nSo our recommendation is: approve the utilisation work now, with a hard checkpoint in eight weeks. If projected peak demand at that point exceeds 80% of consolidated capacity, or if latency SLOs are at risk, we escalate the hardware request with data behind it. We should also pre-qualify a vendor and leasing option now so that, if we do need to buy, the lead time doesn't become its own bottleneck.\n\nWe're confident in the direction and genuinely uncertain about the magnitude \u2014 demand forecasting at our scale has a wide error band, and we've said so in the model. But buying hardware to solve a scheduling problem is the kind of mistake that's expensive to notice and more expensive to undo. Let's spend the smaller money first.\n\nHappy to walk through the utilisation data with anyone who wants the detail.",
          "output_characters": 5118,
          "output_sha256": "57efc650c9b72814daf4f386252162f7566425296ff2b9b581e71bdb28e58aa0",
          "prompt_tokens": 127,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.002763,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "39711178931b3ae07f49cbdac5dfa821bd9ef04c8b310768a952309d09e08244",
          "time_to_first_visible_seconds": 1.074179,
          "time_to_last_output_seconds": 23.312282,
          "ttft_seconds": 0.477405,
          "wall_seconds": 23.315045
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1001,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.532319,
          "decode_tokens_per_second": 44.381,
          "finish_reason": "stop",
          "measured_sse_events": 999,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status:** For discussion ahead of Thursday's planning session\n\n---\n\nWe've been asked to settle a question that has been simmering for a few weeks now: do we spend roughly $180K on two additional GPU nodes, or do we invest a quarter of engineering effort in raising utilisation on the hardware we already own? I want to lay out how I'm thinking about it, because the right answer is less obvious than either camp has suggested in our Slack threads.\n\nFirst, the case for buying. Our p99 inference latency has crept from 400ms to just over a second during peak hours, and that degradation is not evenly distributed \u2014 it lands hardest on the interactive features our users actually notice. When the queue backs up, requests time out, retries pile on, and the whole thing gets worse before it gets better. From a pure reliability standpoint, more capacity is the blunt but dependable fix. It also buys us headroom for the two product launches scheduled for next quarter, which we currently expect to add 30\u201340% to inference load. If those launches land and we haven't added capacity, we will be rationing features by throttling, which is a worse conversation to have with product than the one we're having now about money.\n\nBut the utilisation numbers give me real pause. Our fleet averages 41% GPU utilisation over a 24-hour cycle, and the pattern is stark: we're near saturation for maybe five hours a day and nearly idle overnight. That is a scheduling problem as much as a capacity problem. Batch workloads \u2014 embeddings refreshes, fine-tuning runs, offline evaluation \u2014 currently run during business hours alongside interactive traffic because nobody has prioritised moving them. Shifting that work to the overnight window, plus introducing a proper priority queue and dynamic batching, could plausibly get us to 65\u201370% utilisation. That is the equivalent of adding roughly half a node's worth of capacity for the cost of maybe six engineer-weeks, versus $180K plus the ongoing power, cooling, and colocation costs of new hardware, which our finance contact estimates at $2,400/month per node.\n\nThere's also the opportunity cost to weigh honestly. The engineers best positioned to do the scheduling and batching work are the same ones product teams keep trying to borrow for launch support. Six engineer-weeks from that group is not free; it's six weeks of something else they won't do. And there's a risk that the utilisation work, like all infrastructure projects of its kind, expands to fill more time than estimated. Meanwhile, hardware lead times are currently quoted at 8\u201310 weeks, which means if we decide in March that we need the nodes for a June launch, we may already be too late. That asymmetry matters: the cost of buying too late is a visible product failure, while the cost of buying too early is money and rack space we didn't need yet.\n\nOn reversibility, the picture is genuinely mixed. Hardware is a semi-permanent commitment \u2014 we can resell nodes, but at a discount, and the colocation contract has a twelve-month minimum. Utilisation work, by contrast, is almost entirely reversible: if the batching changes cause problems, we revert and we've lost some engineering time. But that cuts both ways. Utilisation gains are also fragile in a way hardware isn't. They depend on workload mix staying roughly stable, and a new product that shifts traffic patterns can quietly erode them. Steel and silicon don't get optimised away by a product pivot.\n\nMy recommendation is a staged one, though I want to be clear it's a recommendation to sequence, not to hedge indefinitely. Start the utilisation work now \u2014 it's cheap, reversible, and the overnight batching change alone should relieve a meaningful share of the peak-hour pressure. Simultaneously, place a conditional purchase order for one node rather than two, with delivery timed for early next quarter. That splits the difference: we capture the cheap capacity first, we keep a hardware backstop against the launch-driven load spike, and we preserve the option to cancel or defer the order if utilisation work over-delivers. The $90K we don't spend on the second node stays available if the data tells us we need it.\n\nThe honest uncertainty here is that our load forecast is a guess dressed in a spreadsheet. If the launches underperform, we'll have bought capacity we didn't need; if they overperform, one node may not be enough. I'd rather make that correction in June with real traffic data than in February with none.\n\nHappy to walk through the utilisation dashboards with anyone before Thursday.",
          "output_characters": 4755,
          "output_sha256": "afc769c5e9b6c3506138267b19796ba347e369103fbf451970f136a65074a694",
          "prompt_tokens": 125,
          "reasoning_characters": 62,
          "reasoning_sha256": "7eddaec104351da3aba6059600be80b14d17920bd2df8ba3f1ac3d20ffb662d1",
          "response_tail_seconds": 0.001462,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "369d7efa574d1981daf7ee34224207d1d1200293e5a5fe74554b8b7ff03cfe96",
          "time_to_first_visible_seconds": 1.151561,
          "time_to_last_output_seconds": 23.011824,
          "ttft_seconds": 0.479505,
          "wall_seconds": 23.013286
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 23.68,
        "median": 23.312,
        "minimum": 22.313,
        "p90": 23.68
      },
      "ttft_seconds": {
        "maximum": 0.528,
        "median": 0.48,
        "minimum": 0.477,
        "p90": 0.528
      },
      "wall_seconds": {
        "maximum": 23.682,
        "median": 23.315,
        "minimum": 22.315,
        "p90": 23.682
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 103.466,
        "median": 102.443,
        "minimum": 99.255,
        "p90": 103.466
      },
      "pooled_decode_tokens_per_second": 102.208,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 640,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.258908,
          "decode_tokens_per_second": 102.094,
          "finish_reason": "stop",
          "measured_sse_events": 634,
          "output": "[{\"index\": 1, \"square\": 1}, {\"index\": 2, \"square\": 4}, {\"index\": 3, \"square\": 9}, {\"index\": 4, \"square\": 16}, {\"index\": 5, \"square\": 25}, {\"index\": 6, \"square\": 36}, {\"index\": 7, \"square\": 49}, {\"index\": 8, \"square\": 64}, {\"index\": 9, \"square\": 81}, {\"index\": 10, \"square\": 100}, {\"index\": 11, \"square\": 121}, {\"index\": 12, \"square\": 144}, {\"index\": 13, \"square\": 169}, {\"index\": 14, \"square\": 196}, {\"index\": 15, \"square\": 225}, {\"index\": 16, \"square\": 256}, {\"index\": 17, \"square\": 289}, {\"index\": 18, \"square\": 324}, {\"index\": 19, \"square\": 361}, {\"index\": 20, \"square\": 400}, {\"index\": 21, \"square\": 441}, {\"index\": 22, \"square\": 484}, {\"index\": 23, \"square\": 529}, {\"index\": 24, \"square\": 576}, {\"index\": 25, \"square\": 625}, {\"index\": 26, \"square\": 676}, {\"index\": 27, \"square\": 729}, {\"index\": 28, \"square\": 784}, {\"index\": 29, \"square\": 841}, {\"index\": 30, \"square\": 900}, {\"index\": 31, \"square\": 961}, {\"index\": 32, \"square\": 1024}, {\"index\": 33, \"square\": 1089}, {\"index\": 34, \"square\": 1156}, {\"index\": 35, \"square\": 1225}, {\"index\": 36, \"square\": 1296}, {\"index\": 37, \"square\": 1369}, {\"index\": 38, \"square\": 1444}, {\"index\": 39, \"square\": 1521}, {\"index\": 40, \"square\": 1600}, {\"index\": 41, \"square\": 1681}, {\"index\": 42, \"square\": 1764}, {\"index\": 43, \"square\": 1849}, {\"index\": 44, \"square\": 1936}, {\"index\": 45, \"square\": 2025}, {\"index\": 46, \"square\": 2116}, {\"index\": 47, \"square\": 2209}, {\"index\": 48, \"square\": 2304}, {\"index\": 49, \"square\": 2401}, {\"index\": 50, \"square\": 2500}]",
          "output_characters": 1498,
          "output_sha256": "3ca4127620a6fff97fe702e51a0c5ce9ff47da0f11e0ea4fe73ee6837b91f3db",
          "prompt_tokens": 107,
          "reasoning_characters": 18,
          "reasoning_sha256": "abca0f51ad8341cf7ac2e952f743c89fd86589479036d30a16b825f2aeb68568",
          "response_tail_seconds": 0.002278,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "180cb279b79391ae7a8bec45f36868345076e64e6bd87b01154e0e9aaa242d15",
          "time_to_first_visible_seconds": 0.743749,
          "time_to_last_output_seconds": 6.723962,
          "ttft_seconds": 0.465054,
          "wall_seconds": 6.72624
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 691,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.706485,
          "decode_tokens_per_second": 102.885,
          "finish_reason": "stop",
          "measured_sse_events": 689,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 18,
          "reasoning_sha256": "abca0f51ad8341cf7ac2e952f743c89fd86589479036d30a16b825f2aeb68568",
          "response_tail_seconds": 0.002269,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "807d5cb51b38b6d0f6507a903e7b9a9ade0d4d0d1a953ccc597b2f5613f6927e",
          "time_to_first_visible_seconds": 0.717844,
          "time_to_last_output_seconds": 7.173655,
          "ttft_seconds": 0.46717,
          "wall_seconds": 7.175924
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 691,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.668829,
          "decode_tokens_per_second": 103.466,
          "finish_reason": "stop",
          "measured_sse_events": 688,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 108,
          "reasoning_characters": 18,
          "reasoning_sha256": "abca0f51ad8341cf7ac2e952f743c89fd86589479036d30a16b825f2aeb68568",
          "response_tail_seconds": 0.000529,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "807d5cb51b38b6d0f6507a903e7b9a9ade0d4d0d1a953ccc597b2f5613f6927e",
          "time_to_first_visible_seconds": 0.712745,
          "time_to_last_output_seconds": 7.131857,
          "ttft_seconds": 0.463028,
          "wall_seconds": 7.132386
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 446,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 4.483391,
          "decode_tokens_per_second": 99.255,
          "finish_reason": "stop",
          "measured_sse_events": 444,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 32,
          "reasoning_sha256": "81ec3fb7ffc4a0a574a5e949ec47ccaa3594ad0d41ab7cb61639eed44bdaa735",
          "response_tail_seconds": 0.001584,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "30fce3ebe2c6b55d30afe8fa089ba8e6275ec86d9a2f9da811e2268ce12db084",
          "time_to_first_visible_seconds": 0.847048,
          "time_to_last_output_seconds": 4.947183,
          "ttft_seconds": 0.463791,
          "wall_seconds": 4.948767
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 4.256032,
          "decode_tokens_per_second": 102.443,
          "finish_reason": "stop",
          "measured_sse_events": 430,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001772,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.522357,
          "time_to_last_output_seconds": 4.778388,
          "ttft_seconds": 0.522357,
          "wall_seconds": 4.78016
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 7.174,
        "median": 6.724,
        "minimum": 4.778,
        "p90": 7.174
      },
      "ttft_seconds": {
        "maximum": 0.522,
        "median": 0.465,
        "minimum": 0.463,
        "p90": 0.522
      },
      "wall_seconds": {
        "maximum": 7.176,
        "median": 6.726,
        "minimum": 4.78,
        "p90": 7.176
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 2012.577,
          "median": 2012.363,
          "minimum": 2012.242,
          "p90": 2012.577
        },
        "runs": [
          {
            "cache_salt_sha256": "5cda11bec7ecf58c1c048441a333cbe62bbde540a2281753cc25c5e72981a936",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062456,
            "decode_tokens_per_second": 112.078,
            "effective_prefill_tokens_per_second": 2012.363,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001696,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 16.283345,
            "time_to_last_output_seconds": 16.345801,
            "ttft_seconds": 16.283345,
            "wall_seconds": 16.347497
          },
          {
            "cache_salt_sha256": "044c86752e013a8f8e744394bb3d5ca3aa08ba17655b321443d6ccd66d2504e1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054655,
            "decode_tokens_per_second": 128.077,
            "effective_prefill_tokens_per_second": 2012.577,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000843,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 16.281615,
            "time_to_last_output_seconds": 16.33627,
            "ttft_seconds": 16.281615,
            "wall_seconds": 16.337112
          },
          {
            "cache_salt_sha256": "57e38298936a868b563ee962fdcf9098c520a280996c4b266566b7db4bd7575d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.060138,
            "decode_tokens_per_second": 116.399,
            "effective_prefill_tokens_per_second": 2012.242,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001621,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 16.284322,
            "time_to_last_output_seconds": 16.34446,
            "ttft_seconds": 16.284322,
            "wall_seconds": 16.346081
          }
        ],
        "ttft_seconds": {
          "maximum": 16.284,
          "median": 16.283,
          "minimum": 16.282,
          "p90": 16.284
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 755318.904,
          "median": 750045.779,
          "minimum": 690274.062,
          "p90": 755318.904
        },
        "runs": [
          {
            "cache_salt_sha256": "5cda11bec7ecf58c1c048441a333cbe62bbde540a2281753cc25c5e72981a936",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.062128,
            "decode_tokens_per_second": 112.67,
            "effective_prefill_tokens_per_second": 755318.904,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002101,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.043383,
            "time_to_last_output_seconds": 0.105512,
            "ttft_seconds": 0.043383,
            "wall_seconds": 0.107613
          },
          {
            "cache_salt_sha256": "044c86752e013a8f8e744394bb3d5ca3aa08ba17655b321443d6ccd66d2504e1",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.054965,
            "decode_tokens_per_second": 127.353,
            "effective_prefill_tokens_per_second": 690274.062,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.001353,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 0.047471,
            "time_to_last_output_seconds": 0.102437,
            "ttft_seconds": 0.047471,
            "wall_seconds": 0.10379
          },
          {
            "cache_salt_sha256": "57e38298936a868b563ee962fdcf9098c520a280996c4b266566b7db4bd7575d",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060059,
            "decode_tokens_per_second": 116.552,
            "effective_prefill_tokens_per_second": 750045.779,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.002642,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.043688,
            "time_to_last_output_seconds": 0.103747,
            "ttft_seconds": 0.043688,
            "wall_seconds": 0.106388
          }
        ],
        "ttft_seconds": {
          "maximum": 0.047,
          "median": 0.044,
          "minimum": 0.043,
          "p90": 0.047
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1989.118,
          "median": 1987.435,
          "minimum": 1987.06,
          "p90": 1989.118
        },
        "runs": [
          {
            "cache_salt_sha256": "3455f3d3b64f817a5e99ffa56cfdf1c00ae615da67fc058b179a71e05e967547",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061728,
            "decode_tokens_per_second": 113.401,
            "effective_prefill_tokens_per_second": 1987.06,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002275,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 32.981395,
            "time_to_last_output_seconds": 33.043123,
            "ttft_seconds": 32.981395,
            "wall_seconds": 33.045398
          },
          {
            "cache_salt_sha256": "e97ecdd0029a3e3f2288e55146753ab97d1ed67b57aca3d46c2a70954cb3c623",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.063618,
            "decode_tokens_per_second": 110.031,
            "effective_prefill_tokens_per_second": 1987.435,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002113,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 32.97516,
            "time_to_last_output_seconds": 33.038778,
            "ttft_seconds": 32.97516,
            "wall_seconds": 33.040891
          },
          {
            "cache_salt_sha256": "317385332a304b9c60ba924341b0e9a7cc0d3390d5bb22ac1e53ea7cc0ab3474",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064441,
            "decode_tokens_per_second": 108.627,
            "effective_prefill_tokens_per_second": 1989.118,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002311,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 32.947265,
            "time_to_last_output_seconds": 33.011706,
            "ttft_seconds": 32.947265,
            "wall_seconds": 33.014017
          }
        ],
        "ttft_seconds": {
          "maximum": 32.981,
          "median": 32.975,
          "minimum": 32.947,
          "p90": 32.981
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1156633.31,
          "median": 1155511.672,
          "minimum": 1140291.963,
          "p90": 1156633.31
        },
        "runs": [
          {
            "cache_salt_sha256": "3455f3d3b64f817a5e99ffa56cfdf1c00ae615da67fc058b179a71e05e967547",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.061543,
            "decode_tokens_per_second": 113.742,
            "effective_prefill_tokens_per_second": 1155511.672,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.002107,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.056716,
            "time_to_last_output_seconds": 0.118259,
            "ttft_seconds": 0.056716,
            "wall_seconds": 0.120367
          },
          {
            "cache_salt_sha256": "e97ecdd0029a3e3f2288e55146753ab97d1ed67b57aca3d46c2a70954cb3c623",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063042,
            "decode_tokens_per_second": 111.037,
            "effective_prefill_tokens_per_second": 1156633.31,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001427,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.056661,
            "time_to_last_output_seconds": 0.119703,
            "ttft_seconds": 0.056661,
            "wall_seconds": 0.121131
          },
          {
            "cache_salt_sha256": "317385332a304b9c60ba924341b0e9a7cc0d3390d5bb22ac1e53ea7cc0ab3474",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.064991,
            "decode_tokens_per_second": 107.708,
            "effective_prefill_tokens_per_second": 1140291.963,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " action. The repository contains a service",
            "output_characters": 42,
            "output_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001197,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "b5a7611b4f1762cd19a22471acbe46b2c959f51e351ddcc2332e20186a3f3dd8",
            "time_to_first_visible_seconds": 0.057473,
            "time_to_last_output_seconds": 0.122463,
            "ttft_seconds": 0.057473,
            "wall_seconds": 0.12366
          }
        ],
        "ttft_seconds": {
          "maximum": 0.057,
          "median": 0.057,
          "minimum": 0.057,
          "p90": 0.057
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1972.009,
          "median": 1971.015,
          "minimum": 1963.233,
          "p90": 1972.009
        },
        "runs": [
          {
            "cache_salt_sha256": "3f0a127234831b1748033cbec7e52e7d09a3fd0383d04092d980fa5cc32180aa",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.060046,
            "decode_tokens_per_second": 116.578,
            "effective_prefill_tokens_per_second": 1963.233,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001155,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.172709,
            "time_to_last_output_seconds": 4.232755,
            "ttft_seconds": 4.172709,
            "wall_seconds": 4.233909
          },
          {
            "cache_salt_sha256": "0a6b2ee9c7761afe1452de55a5d2b9403311ac834141569d256199a5c4d41bb1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061308,
            "decode_tokens_per_second": 114.178,
            "effective_prefill_tokens_per_second": 1972.009,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001472,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 4.15414,
            "time_to_last_output_seconds": 4.215448,
            "ttft_seconds": 4.15414,
            "wall_seconds": 4.216919
          },
          {
            "cache_salt_sha256": "0c65435509932b479b8c733ad82f2a215c013f695b03a8a5d635b9d4f3fa412f",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055347,
            "decode_tokens_per_second": 126.476,
            "effective_prefill_tokens_per_second": 1971.015,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001818,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.156235,
            "time_to_last_output_seconds": 4.211581,
            "ttft_seconds": 4.156235,
            "wall_seconds": 4.2134
          }
        ],
        "ttft_seconds": {
          "maximum": 4.173,
          "median": 4.156,
          "minimum": 4.154,
          "p90": 4.173
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 241267.597,
          "median": 216834.304,
          "minimum": 171542.247,
          "p90": 241267.597
        },
        "runs": [
          {
            "cache_salt_sha256": "3f0a127234831b1748033cbec7e52e7d09a3fd0383d04092d980fa5cc32180aa",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.059706,
            "decode_tokens_per_second": 117.242,
            "effective_prefill_tokens_per_second": 171542.247,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001243,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.047755,
            "time_to_last_output_seconds": 0.107461,
            "ttft_seconds": 0.047755,
            "wall_seconds": 0.108704
          },
          {
            "cache_salt_sha256": "0a6b2ee9c7761afe1452de55a5d2b9403311ac834141569d256199a5c4d41bb1",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.060968,
            "decode_tokens_per_second": 114.814,
            "effective_prefill_tokens_per_second": 216834.304,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " notes. Each observation records behaviour, evidence",
            "output_characters": 52,
            "output_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.00231,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "16c1fa2c111d6c60c8745728c935a28674dcef12ee7ef184c3b4ce33f486c298",
            "time_to_first_visible_seconds": 0.03778,
            "time_to_last_output_seconds": 0.098749,
            "ttft_seconds": 0.03778,
            "wall_seconds": 0.101059
          },
          {
            "cache_salt_sha256": "0c65435509932b479b8c733ad82f2a215c013f695b03a8a5d635b9d4f3fa412f",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.055453,
            "decode_tokens_per_second": 126.233,
            "effective_prefill_tokens_per_second": 241267.597,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001278,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.033954,
            "time_to_last_output_seconds": 0.089407,
            "ttft_seconds": 0.033954,
            "wall_seconds": 0.090686
          }
        ],
        "ttft_seconds": {
          "maximum": 0.048,
          "median": 0.038,
          "minimum": 0.034,
          "p90": 0.048
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "incoai/GLM-5.3-Flash-DFlash2 @ bf582e4e (Mia v1.10 default settings)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "Mia-AiLab/GLM-5.3-Flash-EXL3-4bpw-TensorFold",
      "model_revision": "078455ffe6472f9a52fbc1139f58b9db2881b25c",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-dflash2-low",
    "finished_at": "2026-10-08T11:31:17.905159+00:00",
    "label": "tp2-stock-dflash2-low",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T11:22:46.445851+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 8192,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "low"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 5,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}
