{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 53.397,
        "median": 52.411,
        "minimum": 52.203,
        "p90": 53.397
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 58.947,
        "median": 57.801,
        "minimum": 57.774,
        "p90": 58.947
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.491,
        "median": 0.47,
        "minimum": 0.468,
        "p90": 0.491
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 52.203,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.411713,
              "decode_tokens_per_second": 57.801,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the standard library.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be constructed with New; the zero value is not usable.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\trate   float64   // tokens per second\n\tburst  float64   // bucket capacity\n\ttokens float64   // current tokens\n\tlast   time.Time // last refill time\n}\n\n// ErrInvalidConfig is returned by New for invalid parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate must be > 0 and burst must be >= 1\")\n\n// New returns a Limiter that refills at rate tokens per second with the\n// given burst capacity, starting full. It returns an error if rate <= 0\n",
              "output_characters": 1021,
              "output_sha256": "855bddba678c193b41b40cc0a776ba9da191cc97562350423c3603549c8def3b",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000117,
              "stream_done_marker": true,
              "stream_sha256": "855bddba678c193b41b40cc0a776ba9da191cc97562350423c3603549c8def3b",
              "time_to_first_visible_seconds": 0.491218,
              "time_to_last_output_seconds": 4.90293,
              "ttft_seconds": 0.491218,
              "wall_seconds": 4.903047
            }
          ],
          "wall_seconds": 4.903892
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.397,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.325944,
              "decode_tokens_per_second": 58.947,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. clock may be nil for the real clock.\nfunc New(capacity, rate float64,",
              "output_characters": 1042,
              "output_sha256": "a1c8fe218a60561ccd7dcb2e64c3cf31b8ead8e8130135ecfca1a6e5008c46f5",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000186,
              "stream_done_marker": true,
              "stream_sha256": "a1c8fe218a60561ccd7dcb2e64c3cf31b8ead8e8130135ecfca1a6e5008c46f5",
              "time_to_first_visible_seconds": 0.46754,
              "time_to_last_output_seconds": 4.793484,
              "ttft_seconds": 0.46754,
              "wall_seconds": 4.79367
            }
          ],
          "wall_seconds": 4.794317
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.411,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.41375,
              "decode_tokens_per_second": 57.774,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(capacity int, rate float64, c Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t",
              "output_characters": 985,
              "output_sha256": "d4015346040fd1ee3392f9ecdadb63b29aa4dbffc09dcb4b5491a93f4a76a4d7",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.00022,
              "stream_done_marker": true,
              "stream_sha256": "d4015346040fd1ee3392f9ecdadb63b29aa4dbffc09dcb4b5491a93f4a76a4d7",
              "time_to_first_visible_seconds": 0.469772,
              "time_to_last_output_seconds": 4.883522,
              "ttft_seconds": 0.469772,
              "wall_seconds": 4.883742
            }
          ],
          "wall_seconds": 4.884426
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 53.438,
        "median": 53.429,
        "minimum": 53.158,
        "p90": 53.438
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 61.206,
        "median": 59.103,
        "minimum": 57.045,
        "p90": 59.713
      },
      "per_stream_ttft_seconds": {
        "maximum": 5.234,
        "median": 2.792,
        "minimum": 0.469,
        "p90": 5.224
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 53.429,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.357011,
              "decode_tokens_per_second": 58.526,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens\n// per second; the bucket holds at most burst tokens. The zero value is\n// not usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be > 0\")\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum capacity of burst tokens, starting full.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate",
              "output_characters": 1027,
              "output_sha256": "dc041517fe76568948a31824e7477a966aed4fe80b836749519b8e1ad7dbccd7",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9.3e-05,
              "stream_done_marker": true,
              "stream_sha256": "dc041517fe76568948a31824e7477a966aed4fe80b836749519b8e1ad7dbccd7",
              "time_to_first_visible_seconds": 5.224311,
              "time_to_last_output_seconds": 9.581322,
              "ttft_seconds": 5.224311,
              "wall_seconds": 9.581415
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.270458,
              "decode_tokens_per_second": 59.713,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec,\n// with capacity tokens held at most.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given burst capacity and refill rate.\n// If clock is nil, the real clock is used.\nfunc New(capacity int, rate float64, clock Clock)",
              "output_characters": 983,
              "output_sha256": "dd89da92c9acd0eaae3edcaf1fc49426294e4e907b8a5557565519e7db422eaf",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000179,
              "stream_done_marker": true,
              "stream_sha256": "dd89da92c9acd0eaae3edcaf1fc49426294e4e907b8a5557565519e7db422eaf",
              "time_to_first_visible_seconds": 0.475272,
              "time_to_last_output_seconds": 4.74573,
              "ttft_seconds": 0.475272,
              "wall_seconds": 4.745909
            }
          ],
          "wall_seconds": 9.582745
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.158,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.395995,
              "decode_tokens_per_second": 58.007,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the system time.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens/sec up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with a full bucket.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock allows injecting a custom Clock (for tests).\nfunc NewWithClock(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tif clock == nil",
              "output_characters": 1010,
              "output_sha256": "ebd7ea8778b1b8b467f789fe710782cd633dddac6cfff22e18b7a3883712c85b",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000412,
              "stream_done_marker": true,
              "stream_sha256": "ebd7ea8778b1b8b467f789fe710782cd633dddac6cfff22e18b7a3883712c85b",
              "time_to_first_visible_seconds": 5.233963,
              "time_to_last_output_seconds": 9.629958,
              "ttft_seconds": 5.233963,
              "wall_seconds": 9.63037
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.272732,
              "decode_tokens_per_second": 59.681,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "**ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity Burst. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tburst    int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlastRefill time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive burst or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: burst and rate must be positive\")\n\n// New returns a Limiter that starts full with burst tokens, refilling at\n// rate tokens per second. clock may be nil for the real clock.\nfunc New(burst int, rate float64, clock Clock) (*Limiter, error) {\n\tif burst <= 0 || rate <= ",
              "output_characters": 1001,
              "output_sha256": "29ce6487bd8b5a495c6c99b15f77f412f1491017d28107998ff8874babd86079",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000126,
              "stream_done_marker": true,
              "stream_sha256": "29ce6487bd8b5a495c6c99b15f77f412f1491017d28107998ff8874babd86079",
              "time_to_first_visible_seconds": 0.476327,
              "time_to_last_output_seconds": 4.749059,
              "ttft_seconds": 0.476327,
              "wall_seconds": 4.749185
            }
          ],
          "wall_seconds": 9.631743
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.438,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.470179,
              "decode_tokens_per_second": 57.045,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the standard library.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be constructed with NewLimiter; the zero value is not usable.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\trate   float64 // tokens per second\n\tburst  float64 // bucket capacity\n\ttokens float64 // current tokens\n\tlast   time.Time\n}\n\n// ErrInvalidConfig is returned by NewLimiter for invalid parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be > 0\")\n\n// NewLimiter creates a Limiter that refills at rate tokens per second with\n// a maximum burst capacity. It starts full.\nfunc NewLimiter(rate float64, burst int) (*Limiter,",
              "output_characters": 1022,
              "output_sha256": "dbeb0ddfc948694e72a2fcc85e92de085bd2f1f6a4ee6f7865308d5bfdf53e96",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.00078,
              "stream_done_marker": true,
              "stream_sha256": "dbeb0ddfc948694e72a2fcc85e92de085bd2f1f6a4ee6f7865308d5bfdf53e96",
              "time_to_first_visible_seconds": 5.108363,
              "time_to_last_output_seconds": 9.578543,
              "ttft_seconds": 5.108363,
              "wall_seconds": 9.579323
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.166262,
              "decode_tokens_per_second": 61.206,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It is not safe to copy; use the pointer.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n\tnowFn    func() time.Time\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter that refills at rate tokens per second with\n// capacity burst. The bucket starts full.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is",
              "output_characters": 953,
              "output_sha256": "9133dc0cc9b7fca3071726d830ecf8fadcf133d1f17e97d797ac5610ea81b893",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000251,
              "stream_done_marker": true,
              "stream_sha256": "9133dc0cc9b7fca3071726d830ecf8fadcf133d1f17e97d797ac5610ea81b893",
              "time_to_first_visible_seconds": 0.469349,
              "time_to_last_output_seconds": 4.635611,
              "ttft_seconds": 0.469349,
              "wall_seconds": 4.635862
            }
          ],
          "wall_seconds": 9.58118
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 53.865,
        "median": 53.547,
        "minimum": 52.355,
        "p90": 53.865
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 62.003,
        "median": 58.886,
        "minimum": 55.744,
        "p90": 60.568
      },
      "per_stream_ttft_seconds": {
        "maximum": 15.1,
        "median": 7.74,
        "minimum": 0.488,
        "p90": 14.895
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 53.547,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.210141,
              "decode_tokens_per_second": 60.568,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at Rate per second\n// up to Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int\n\trate     float64 // tokens per second\n\n\ttokens float64\n\tlast   time.Time\n}\n\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given capacity and refill rate (tokens per\n// second), starting full. clock may be nil for the real clock.\nfunc New(cap",
              "output_characters": 983,
              "output_sha256": "eb34cbf08919697a8cca07926ba7b28d1abf70ec7ee458e121e81b6a7e0e6865",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000143,
              "stream_done_marker": true,
              "stream_sha256": "eb34cbf08919697a8cca07926ba7b28d1abf70ec7ee458e121e81b6a7e0e6865",
              "time_to_first_visible_seconds": 0.487526,
              "time_to_last_output_seconds": 4.697667,
              "ttft_seconds": 0.487526,
              "wall_seconds": 4.697811
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.354534,
              "decode_tokens_per_second": 58.56,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second; the bucket holds at most Burst tokens.\n// The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlast     time.Time\n\tnowFn    Clock\n}\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum capacity of burst tokens, starting full.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, errors.New(\"ratel",
              "output_characters": 1020,
              "output_sha256": "4ee0542779a7c413c5aef35c8a86f14069a440f81c199c316945a9390b14af5b",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000166,
              "stream_done_marker": true,
              "stream_sha256": "4ee0542779a7c413c5aef35c8a86f14069a440f81c199c316945a9390b14af5b",
              "time_to_first_visible_seconds": 5.167492,
              "time_to_last_output_seconds": 9.522026,
              "ttft_seconds": 5.167492,
              "wall_seconds": 9.522193
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.345659,
              "decode_tokens_per_second": 58.679,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the standard library.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error",
              "output_characters": 1030,
              "output_sha256": "aa1481b54af783f038e5ef4848b9be1ae584c1424f27c5f5f5beb6c07e6950a1",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000146,
              "stream_done_marker": true,
              "stream_sha256": "aa1481b54af783f038e5ef4848b9be1ae584c1424f27c5f5f5beb6c07e6950a1",
              "time_to_first_visible_seconds": 14.775569,
              "time_to_last_output_seconds": 19.121228,
              "ttft_seconds": 14.775569,
              "wall_seconds": 19.121374
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.298766,
              "decode_tokens_per_second": 59.319,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity,",
              "output_characters": 1044,
              "output_sha256": "697e2b077faefd9d453c6d9b2b1d84168bc6841ecb3562566ae973b0d17148bb",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9.1e-05,
              "stream_done_marker": true,
              "stream_sha256": "697e2b077faefd9d453c6d9b2b1d84168bc6841ecb3562566ae973b0d17148bb",
              "time_to_first_visible_seconds": 10.002849,
              "time_to_last_output_seconds": 14.301615,
              "ttft_seconds": 10.002849,
              "wall_seconds": 14.301706
            }
          ],
          "wall_seconds": 19.123537
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.865,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.310747,
              "decode_tokens_per_second": 59.154,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to the bucket capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is",
              "output_characters": 1013,
              "output_sha256": "b173883abd80e8d45226e9a367f7401d66f304582aebdbb77e1cfce001a4d3b2",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.00033,
              "stream_done_marker": true,
              "stream_sha256": "b173883abd80e8d45226e9a367f7401d66f304582aebdbb77e1cfce001a4d3b2",
              "time_to_first_visible_seconds": 0.489604,
              "time_to_last_output_seconds": 4.800352,
              "ttft_seconds": 0.489604,
              "wall_seconds": 4.800682
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.112726,
              "decode_tokens_per_second": 62.003,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n\tnowFn    func() time.Time\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum capacity of burst. The bucket starts full.\nfunc New(rate float64, burst int) (*Limiter, error)",
              "output_characters": 972,
              "output_sha256": "1165146cb9d3864ed86ab74b2b8ef5b9cbf2bac81a12f62410d99342f226131b",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000698,
              "stream_done_marker": true,
              "stream_sha256": "1165146cb9d3864ed86ab74b2b8ef5b9cbf2bac81a12f62410d99342f226131b",
              "time_to_first_visible_seconds": 14.89489,
              "time_to_last_output_seconds": 19.007617,
              "ttft_seconds": 14.89489,
              "wall_seconds": 19.008314
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.325989,
              "decode_tokens_per_second": 58.946,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\n",
              "output_characters": 1026,
              "output_sha256": "b9e6714bde039870276a9f6035b6d369cfed1b64741e6d8f3fb5143878fc5f78",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000121,
              "stream_done_marker": true,
              "stream_sha256": "b9e6714bde039870276a9f6035b6d369cfed1b64741e6d8f3fb5143878fc5f78",
              "time_to_first_visible_seconds": 10.101504,
              "time_to_last_output_seconds": 14.427493,
              "ttft_seconds": 10.101504,
              "wall_seconds": 14.427614
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.334836,
              "decode_tokens_per_second": 58.826,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int           // max tokens\n\trate     float64       // tokens per second\n\ttokens   float64       // current tokens\n\tlast     time.Time     // last refill time\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(capacity int, rate float64, c Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn",
              "output_characters": 1010,
              "output_sha256": "4b9f44f04bf00b51f8450b4882658530ff467c3db13c6f524a5acca638b69768",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000567,
              "stream_done_marker": true,
              "stream_sha256": "4b9f44f04bf00b51f8450b4882658530ff467c3db13c6f524a5acca638b69768",
              "time_to_first_visible_seconds": 5.27338,
              "time_to_last_output_seconds": 9.608215,
              "ttft_seconds": 5.27338,
              "wall_seconds": 9.608782
            }
          ],
          "wall_seconds": 19.01064
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.355,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.456251,
              "decode_tokens_per_second": 57.223,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n",
              "output_characters": 1026,
              "output_sha256": "61ce50ea119264635b2b3879d0cf7def36c284d20740f7bfb8d8bfd8b1d479cd",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000155,
              "stream_done_marker": true,
              "stream_sha256": "61ce50ea119264635b2b3879d0cf7def36c284d20740f7bfb8d8bfd8b1d479cd",
              "time_to_first_visible_seconds": 15.100001,
              "time_to_last_output_seconds": 19.556252,
              "ttft_seconds": 15.100001,
              "wall_seconds": 19.556407
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.448744,
              "decode_tokens_per_second": 57.32,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate tokens/sec\n// up to capacity Burst. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu     sync.Mutex\n\tclock  Clock\n\tburst  int           // bucket capacity\n\trate   float64       // tokens per second\n\ttokens float64       // current tokens\n\tlast   time.Time     // last refill time\n}\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second). rate must be > 0 and burst must be >= 1.\nfunc New(burst int, rate float64) (*Limiter, error) {\n\treturn NewWithClock(burst, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(burst",
              "output_characters": 1000,
              "output_sha256": "6c20ab6ef91db77c4c49c3969412b5e33e1caefe63c9ce21f2a8e72d23ddccd5",
              "prompt_tokens": 128,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000381,
              "stream_done_marker": true,
              "stream_sha256": "6c20ab6ef91db77c4c49c3969412b5e33e1caefe63c9ce21f2a8e72d23ddccd5",
              "time_to_first_visible_seconds": 5.477134,
              "time_to_last_output_seconds": 9.925879,
              "ttft_seconds": 5.477134,
              "wall_seconds": 9.926259
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.239157,
              "decode_tokens_per_second": 60.153,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```ratelimit.go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n\tnowFn    func() time.Time\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum burst capacity. The bucket starts full.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWith",
              "output_characters": 979,
              "output_sha256": "1bd20c02aedbfaf4c5140753e373162350221614499e759ad5c7f7ce4db19343",
              "prompt_tokens": 130,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000128,
              "stream_done_marker": true,
              "stream_sha256": "1bd20c02aedbfaf4c5140753e373162350221614499e759ad5c7f7ce4db19343",
              "time_to_first_visible_seconds": 10.392355,
              "time_to_last_output_seconds": 14.631512,
              "ttft_seconds": 10.392355,
              "wall_seconds": 14.631639
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.574505,
              "decode_tokens_per_second": 55.744,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at rate RefillPerSecond\n// up to capacity Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity int\n\t// refillPerToken is the duration to accrue per single token.\n\trefillPerToken time.Duration\n\ttokens         float64\n\tlast           time.Time\n\tclock          Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and refill rate must be positive\")\n\n// New returns a full Limiter with the given burst capacity and refill rate\n// (tokens added per second). If clock is nil, the real clock is used.\nfunc New(capacity int, refillPerSecond float64, clock Clock)",
              "output_characters": 1063,
              "output_sha256": "ac25c0300a6858959b3339ac962d4b498224c5b9827874cf73b248fa47f5edd2",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9.9e-05,
              "stream_done_marker": true,
              "stream_sha256": "ac25c0300a6858959b3339ac962d4b498224c5b9827874cf73b248fa47f5edd2",
              "time_to_first_visible_seconds": 0.491377,
              "time_to_last_output_seconds": 5.065882,
              "ttft_seconds": 0.491377,
              "wall_seconds": 5.065981
            }
          ],
          "wall_seconds": 19.558896
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 64.493,
        "median": 64.237,
        "minimum": 64.028,
        "p90": 64.493
      },
      "pooled_decode_tokens_per_second": 64.246,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1802,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 27.925641,
          "decode_tokens_per_second": 64.493,
          "finish_reason": "stop",
          "measured_sse_events": 1800,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the wall clock.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens per second, up to Burst.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64 // current tokens\n\tlast     time.Time\n\tlastSet  bool\n\tnowFn    func() time.Time\n}\n\n// New creates a Limiter allowing bursts of burst events, refilling at\n// rate tokens per second. The bucket starts full.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: rate must be > 0\")\n\t}\n\tif burst <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: burst must be > 0\")\n\t}\n\tif clock == nil {\n\t\treturn nil, errors.New(\"ratelimit: clock must not be nil\")\n\t}\n\treturn &Limiter{\n\t\trate:  rate,\n\t\tburst: float64(burst),\n\t\ttokens: float64(burst),\n\t\tnowFn: clock.Now,\n\t}, nil\n}\n\n// AllowN reports whether n events are permitted at time now.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif l.lastSet {\n\t\telapsed := now.Sub(l.last).Seconds()\n\t\tif elapsed > 0 {\n\t\t\tl.tokens += elapsed * l.rate\n\t\t\tif l.tokens > l.burst {\n\t\t\t\tl.tokens = l.burst\n\t\t\t}\n\t\t}\n\t}\n\tl.last = now\n\tl.lastSet = true\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Allow reports whether a single event is permitted now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.nowFn(), 1)\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct{ t time.Time }\n\nfunc (f *fakeClock) Now() time.Time { return f.t }\n\nfunc (f *fakeClock) advance(d time.Duration) { f.t = f.t.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\tcases := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tclock Clock\n\t\twant  bool\n\t}{\n\t\t{\"valid\", 10, 5, RealClock(), true},\n\t\t{\"zero rate\", 0, 5, RealClock(), false},\n\t\t{\"negative rate\", -1, 5, RealClock(), false},\n\t\t{\"zero burst\", 10, 0, RealClock(), false},\n\t\t{\"negative burst\", 10, -1, RealClock(), false},\n\t\t{\"nil clock\", 10, 5, nil, false},\n\t}\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.rate, tc.burst, tc.clock)\n\t\t\tif (err == nil) != tc.want {\n\t\t\t\tt.Fatalf(\"New() err = %v, want valid=%v\", err, tc.want)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\tcases := []struct {\n\t\tname     string\n\t\trate     float64\n\t\tburst    int\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst exhausted then denied\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{0, 1, false},\n\t\t\t\t{2 * time.Second, 1, true},\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"tokens capped at burst\",\n\t\t\trate: 10, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{10 * time.Second, 2, false}, // only 1 token despite long wait\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"zero request always allowed\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{0, 0, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\trate: 1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"fractional refill\",\n\t\t\trate: 2, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{400 * time.Millisecond, 1, false}, // 0.8 tokens\n\t\t\t\t{100 * time.Millisecond, 1, true},  // 1.0 tokens\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := &fakeClock{t: base}\n\t\t\tl, err := New(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.advance(s.advance)\n\t\t\t\tif got := l.AllowN(fc.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tfc := &fakeClock{t: time.Now()}\n\tl, err := New(100, 100, fc)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar mu sync.Mutex\n\tallowed := 0\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(fc.Now(), 1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tallowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed != 100 {\n\t\tt.Errorf(\"allowed = %d, want 100 (full burst)\", allowed)\n\t}\n\tif !l.AllowN(fc.Now(), 1) {\n\t\t// refill of 0 elapsed time; should be denied\n\t} else {\n\t\tt.Error(\"expected denial after burst exhausted\")\n\t}\n}\n```\n\n*(Add `\"sync\"` to the test file's imports.)*",
          "output_characters": 5358,
          "output_sha256": "e41443c5f95b5b1b90b623377a3c6ff1b1c514345de6623e6140885a63fd024b",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000444,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "e41443c5f95b5b1b90b623377a3c6ff1b1c514345de6623e6140885a63fd024b",
          "time_to_first_visible_seconds": 0.477687,
          "time_to_last_output_seconds": 28.403328,
          "ttft_seconds": 0.477687,
          "wall_seconds": 28.403772
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2057,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 32.110805,
          "decode_tokens_per_second": 64.028,
          "finish_reason": "stop",
          "measured_sse_events": 2054,
          "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill continuously at\n// rate tokens per second, up to the bucket capacity. The zero value is not\n// usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(capacity, rate float64, c Clock) (*Limiter, error) {\n\tif c == nil {\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\t}\n\tif capacity <= 0 || rate <= 0 || capacity != capacity || rate != rate {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\treturn &Limiter{\n\t\tclock:    c,\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   capacity,\n\t\tlast:     c.Now(),\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now, consuming\n// them if so. It never blocks.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\tif elapsed := now.Sub(l.last); elapsed > 0 {\n\t\tl.tokens += elapsed.Seconds() * l.rate\n\t\tif l.tokens > l.capacity {\n\t\t\tl.tokens = l.capacity\n\t\t}\n\t\tl.last = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (approximate under\n// concurrency; primarily for tests and monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tnow := l.clock.Now()\n\tif elapsed := now.Sub(l.last); elapsed > 0 {\n\t\tt := l.tokens + elapsed.Seconds()*l.rate\n\t\tif t > l.capacity {\n\t\t\tt = l.capacity\n\t\t}\n\t\treturn t\n\t}\n\treturn l.tokens\n}\n```\n\n**ratelimit/ratelimit_test.go**\n```go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(t time.Time) *fakeClock { return &fakeClock{now: t} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero capacity\", 0, 5, true},\n\t\t{\"negative capacity\", -1, 5, true},\n\t\t{\"zero rate\", 10, 0, true},\n\t\t{\"negative rate\", 10, -2, true},\n\t\t{\"NaN capacity\", nan(), 5, true},\n\t\t{\"NaN rate\", 10, nan(), true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.capacity, tc.rate)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %v) err = %v, wantErr %v\", tc.capacity, tc.rate, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc nan() float64 { n := 0.0; return n / n }\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 tokens accrued\n\t\t\t\t{500 * time.Millisecond, 1, true},  // 1.0 token now\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"capacity caps refill\",\n\t\t\tcapacity: 1, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{time.Hour, 2, false}, // capped at 1 token\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"multi-token request\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false}, {2 * time.Second, 3, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock(base)\n\t\t\tl, err := NewWithClock(tc.capacity, tc.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestTokensRefill(t *testing.T) {\n\tclk := newFakeClock(time.Now())\n\tl, err := NewWithClock(10, 2, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tl.AllowN(10) // drain\n\tif got := l.Tokens(); got > 0.001 {\n\t\tt.Errorf(\"Tokens after drain = %v, want ~0\", got)\n\t}\n\tclk.Advance(time.Second)\n\tif got := l.Tokens(); got < 1.9 || got > 2.1 {\n\t\tt.Errorf(\"Tokens after 1s = %v, want ~2\", got)\n\t}\n}\n\nfunc TestConcurrent(t *testing.T) {\n\tclk := newFakeClock(time.Now())\n\tl, err := NewWithClock(100, 100, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed %d requests, want <= 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6299,
          "output_sha256": "915ece09d0f34dd767debb01d78fc992cd623967a7b692d911c1c0afcefe36f6",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000974,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "915ece09d0f34dd767debb01d78fc992cd623967a7b692d911c1c0afcefe36f6",
          "time_to_first_visible_seconds": 0.469956,
          "time_to_last_output_seconds": 32.580761,
          "ttft_seconds": 0.469956,
          "wall_seconds": 32.581734
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1829,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 28.510695,
          "decode_tokens_per_second": 64.116,
          "finish_reason": "stop",
          "measured_sse_events": 1825,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlast     time.Time // last refill time\n\tnowFn    func() time.Time\n}\n\n// New creates a Limiter that refills at rate tokens per second, with a\n// maximum burst capacity. The bucket starts full.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, burst int, c Clock) (*Limiter, error) {\n\tif c == nil {\n\t\treturn nil, errors.New(\"ratelimit: nil clock\")\n\t}\n\tif rate <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: rate must be positive\")\n\t}\n\tif burst <= 0 {\n\t\treturn nil, errors.New(\"ratelimit: burst must be positive\")\n\t}\n\tnow := c.Now()\n\treturn &Limiter{\n\t\trate:  rate,\n\t\tburst: float64(burst),\n\t\ttokens: float64(burst),\n\t\tlast:  now,\n\t\tnowFn: c.Now,\n\t}, nil\n}\n\n// AllowN reports whether n tokens may be consumed at time now.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif now.Before(l.last) {\n\t\tnow = l.last // guard against clock going backwards\n\t}\n\telapsed := now.Sub(l.last).Seconds()\n\tl.tokens += elapsed * l.rate\n\tif l.tokens > l.burst {\n\t\tl.tokens = l.burst\n\t}\n\tl.last = now\n\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Allow reports whether one token may be consumed now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.nowFn(), 1)\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tt time.Time\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.t }\n\nfunc (f *fakeClock) Advance(d time.Duration) { f.t = f.t.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\twantErr bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero rate\", 0, 5, true},\n\t\t{\"negative rate\", -1, 5, true},\n\t\t{\"zero burst\", 10, 0, true},\n\t\t{\"negative burst\", 10, -1, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.rate, tc.burst)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %v) err = %v, wantErr %v\", tc.rate, tc.burst, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\tsteps []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then denied\",\n\t\t\trate: 1, burst: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"tokens refill over time\",\n\t\t\trate: 2, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true}, // 1 token accrued\n\t\t\t\t{0, 1, false},\n\t\t\t\t{time.Second, 2, true}, // 2 tokens accrued\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill capped at burst\",\n\t\t\trate: 10, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{time.Hour, 1, true}, // only 1 token despite long wait\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"large request denied but state preserved\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 5, false},\n\t\t\t\t{0, 2, true}, // tokens untouched by failed request\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\trate: 1, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := &fakeClock{t: base}\n\t\t\tl, err := NewWithClock(tc.rate, tc.burst, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tgot := l.AllowN(clk.Now(), s.n)\n\t\t\t\tif got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesClock(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, err := NewWithClock(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow = true, want false\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, err := NewWithClock(1000, 100, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"NewWithClock: %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tallowed := make(chan struct{}, 1000)\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.Allow() {\n\t\t\t\tallowed <- struct{}{}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tclose(allowed)\n\tif len(allowed) != 100 {\n\t\tt.Fatalf(\"allowed %d requests, want 100\", len(allowed))\n\t}\n}\n```\n\n*(Add `\"sync\"` to the test file's imports.)*",
          "output_characters": 5525,
          "output_sha256": "2a0fee0002df76988738e162366842b4142e746bd843f4b7af274cd23501b2e9",
          "prompt_tokens": 133,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000599,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "2a0fee0002df76988738e162366842b4142e746bd843f4b7af274cd23501b2e9",
          "time_to_first_visible_seconds": 0.487767,
          "time_to_last_output_seconds": 28.998462,
          "ttft_seconds": 0.487767,
          "wall_seconds": 28.999062
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1883,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 29.229419,
          "decode_tokens_per_second": 64.387,
          "finish_reason": "stop",
          "measured_sse_events": 1881,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. It is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int           // max tokens\n\trate     float64       // tokens per second\n\ttokens   float64       // current tokens\n\tlast     time.Time     // last refill time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a full Limiter with the given capacity and refill rate\n// (tokens per second). A nil clock uses the real time.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 || rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   float64(capacity),\n\t\tlast:     clock.Now(),\n\t}, nil\n}\n\n// AllowN reports whether n tokens may be consumed at time now.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = min(float64(l.capacity), l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif n < 0 || float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Allow reports whether one token may be consumed now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.clock.Now(), 1)\n}\n\nfunc min(a, b float64) float64 {\n\tif a < b {\n\t\treturn a\n\t}\n\treturn b\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct{ t time.Time }\n\nfunc (f *fakeClock) Now() time.Time { return f.t }\n\nfunc (f *fakeClock) advance(d time.Duration) { f.t = f.t.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\tcases := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\twantErr  error\n\t}{\n\t\t{\"valid\", 10, 1.0, nil},\n\t\t{\"zero capacity\", 0, 1.0, ErrInvalidConfig},\n\t\t{\"negative capacity\", -1, 1.0, ErrInvalidConfig},\n\t\t{\"zero rate\", 10, 0, ErrInvalidConfig},\n\t\t{\"negative rate\", 10, -0.5, ErrInvalidConfig},\n\t}\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.capacity, tc.rate, nil)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"err = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t\tif tc.wantErr == nil && l == nil {\n\t\t\t\tt.Fatal(\"expected non-nil limiter\")\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\tcases := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:     \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"tokens capped at capacity\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{10 * time.Second, 2, true}, // only 2 refilled, not 10\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"fractional refill accumulates\",\n\t\t\tcapacity: 1, rate: 0.5,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, false}, // 0.5 tokens\n\t\t\t\t{time.Second, 1, true},  // 1.0 tokens\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"negative request denied\",\n\t\t\tcapacity: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"large request denied when insufficient\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 4, true}, {0, 2, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := &fakeClock{t: base}\n\t\t\tl, err := New(tc.capacity, tc.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.advance(s.advance)\n\t\t\t\tif got := l.AllowN(clk.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, err := New(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow should succeed\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow should fail\")\n\t}\n\tclk.advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill should succeed\")\n\t}\n}\n\nfunc TestConcurrentUse(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, _ := New(100, 10, clk)\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(clk.Now(), 1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tallowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Fatalf(\"allowed %d > capacity 100\", allowed)\n\t}\n}\n```\n\n*(Add `\"sync\"` to the test file's imports.)*",
          "output_characters": 5664,
          "output_sha256": "f0c1af95bc83ce8557ec0111499667089b65d2a182698fa762a1b42e5b3b224a",
          "prompt_tokens": 134,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.00015,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "f0c1af95bc83ce8557ec0111499667089b65d2a182698fa762a1b42e5b3b224a",
          "time_to_first_visible_seconds": 0.493078,
          "time_to_last_output_seconds": 29.722496,
          "ttft_seconds": 0.493078,
          "wall_seconds": 29.722646
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2156,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 33.54741,
          "decode_tokens_per_second": 64.237,
          "finish_reason": "stop",
          "measured_sse_events": 2154,
          "output": "**ratelimit.go**\n\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a token-bucket rate limiter. Tokens refill at Rate per second\n// up to Capacity. The zero value is not usable; use New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity int\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with the given capacity and refill rate (tokens/sec),\n// starting full. clock may be nil, in which case the real clock is used.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, ErrNonPositiveCapacity\n\t}\n\tif rate <= 0 {\n\t\treturn nil, ErrNonPositiveRate\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   float64(capacity),\n\t\tlast:     clock.Now(),\n\t\tclock:    clock,\n\t}, nil\n}\n\n// refill advances the bucket to time now.\nfunc (l *Limiter) refill(now time.Time) {\n\tif elapsed := now.Sub(l.last); elapsed > 0 {\n\t\tl.tokens += elapsed.Seconds() * l.rate\n\t\tif l.tokens > float64(l.capacity) {\n\t\t\tl.tokens = float64(l.capacity)\n\t\t}\n\t}\n\tl.last = now\n}\n\n// AllowN reports whether n tokens may be consumed at time now.\n// If allowed, the tokens are consumed.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(now)\n\tif float64(n) > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= float64(n)\n\treturn true\n}\n\n// Allow reports whether one token may be consumed now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.clock.Now(), 1)\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.refill(l.clock.Now())\n\treturn l.tokens\n}\n```\n\n**ratelimit_test.go**\n\n```go\npackage ratelimit\n\nimport (\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tt time.Time\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.t }\n\nfunc (f *fakeClock) Advance(d time.Duration) { f.t = f.t.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\twantErr  error\n\t}{\n\t\t{\"valid\", 10, 1.0, nil},\n\t\t{\"zero capacity\", 0, 1.0, ErrNonPositiveCapacity},\n\t\t{\"negative capacity\", -1, 1.0, ErrNonPositiveCapacity},\n\t\t{\"zero rate\", 10, 0, ErrNonPositiveRate},\n\t\t{\"negative rate\", 10, -0.5, ErrNonPositiveRate},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.capacity, tc.rate, nil)\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"unexpected error: %v\", err)\n\t\t\t\t}\n\t\t\t\tif l == nil {\n\t\t\t\t\tt.Fatal(\"expected non-nil limiter\")\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif err != tc.wantErr {\n\t\t\t\tt.Fatalf(\"got error %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:     \"burst up to capacity then deny\",\n\t\t\tcapacity: 3,\n\t\t\trate:     1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"refill over time\",\n\t\t\tcapacity: 2,\n\t\t\trate:     1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 tokens\n\t\t\t\t{500 * time.Millisecond, 1, true},  // 1.0 token now\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"tokens capped at capacity\",\n\t\t\tcapacity: 1,\n\t\t\trate:     10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{1 * time.Hour, 2, false}, // refilled to 1, not 36000\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"multi-token request\",\n\t\t\tcapacity: 5,\n\t\t\trate:     1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false}, {2 * time.Second, 3, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"negative request denied\",\n\t\t\tcapacity: 1,\n\t\t\trate:     1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"backward time does not add tokens\",\n\t\t\tcapacity: 1,\n\t\t\trate:     1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {-10 * time.Second, 1, false},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := &fakeClock{t: base}\n\t\t\tl, err := New(tc.capacity, tc.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(clk.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestTokensRefillAndCap(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, err := New(2, 0.5, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tl.AllowN(clk.Now(), 2) // drain\n\tif got := l.Tokens(); got != 0 {\n\t\tt.Fatalf(\"after drain, tokens = %v, want 0\", got)\n\t}\n\tclk.Advance(2 * time.Second) // 1 token\n\tif got := l.Tokens(); got != 1.0 {\n\t\tt.Fatalf(\"tokens = %v, want 1.0\", got)\n\t}\n\tclk.Advance(1 * time.Hour) // capped at capacity\n\tif got := l.Tokens(); got != 2.0 {\n\t\tt.Fatalf(\"tokens = %v, want 2.0 (capped)\", got)\n\t}\n}\n\nfunc TestConcurrentAllow(t *testing.T) {\n\tclk := &fakeClock{t: time.Now()}\n\tl, err := New(100, 100, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New: %v\", err)\n\t}\n\tvar mu struct {\n\t\tsync.Mutex\n\t\tallowed int\n\t}\n\tvar wg sync.WaitGroup\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(clk.Now(), 1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tmu.allowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif mu.allowed != 100 && mu.allowed > 100 {\n\t\tt.Fatalf(\"allowed %d, must not exceed capacity 100\", mu.allowed)\n\t}\n}\n```",
          "output_characters": 6465,
          "output_sha256": "0fd010565fb124c33152fb6f6a6d38e43940cde86de98e879ab3aa2f4ffaa12d",
          "prompt_tokens": 134,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000654,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "0fd010565fb124c33152fb6f6a6d38e43940cde86de98e879ab3aa2f4ffaa12d",
          "time_to_first_visible_seconds": 0.487268,
          "time_to_last_output_seconds": 34.034678,
          "ttft_seconds": 0.487268,
          "wall_seconds": 34.035332
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 34.035,
        "median": 29.722,
        "minimum": 28.403,
        "p90": 34.035
      },
      "ttft_seconds": {
        "maximum": 0.493,
        "median": 0.487,
        "minimum": 0.47,
        "p90": 0.493
      },
      "wall_seconds": {
        "maximum": 34.035,
        "median": 29.723,
        "minimum": 28.404,
        "p90": 34.035
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 44.371,
        "median": 43.148,
        "minimum": 42.655,
        "p90": 44.371
      },
      "pooled_decode_tokens_per_second": 43.511,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1063,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 24.65597,
          "decode_tokens_per_second": 43.073,
          "finish_reason": "stop",
          "measured_sse_events": 1061,
          "output": "**Internal Memo: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\nWe've been running hot on our local inference hardware for about six weeks now. Peak-hour GPU utilisation regularly exceeds 85 percent, queue times during business hours have crept from milliseconds into seconds, and two teams have started quietly scheduling batch jobs overnight to avoid contention. The obvious question on the table is whether we should buy more hardware. My recommendation, after looking at the numbers and talking to the teams affected, is that we should not \u2014 not yet. We should first spend six weeks improving utilisation of what we own, and only revisit purchasing if that effort demonstrably fails.\n\nThe case for buying is real, and I don't want to dismiss it. Our three A100 nodes were sized for a workload half our current traffic, and hardware lead times mean that if we do need to buy, we should decide sooner rather than later. But the case for buying rests on an assumption I don't think we've tested: that our current hardware is actually being used efficiently. The telemetry suggests it isn't. Average utilisation during off-peak hours sits around 30 percent, and even at peak we see significant fragmentation \u2014 models pinned to specific GPUs that could be co-located, batch sizes left at defaults set a year ago, and at least one legacy service holding a full GPU for traffic that would fit comfortably on a shared instance. Before we spend roughly \u00a3180,000 on a fourth node, plus the ongoing power, cooling, and rack costs, we owe ourselves a serious attempt to reclaim the capacity we've already paid for.\n\nLatency is the strongest argument on the buy side, and it deserves honest treatment. Queueing delay is the worst kind of latency because it's unpredictable; a request that takes 40ms of compute can take 900ms end-to-end if it lands behind a burst. That variance degrades user experience in ways average latency figures hide, and it's already causing friction for the customer-facing search feature. But most of that variance comes from contention, not compute. If we consolidate models, enable dynamic batching, and move the overnight batch workloads to a proper scheduler with preemption, I expect p99 latency to improve substantially without any new silicon. If it doesn't, that failure itself is valuable evidence for the purchase case.\n\nReliability cuts both ways. A fourth node gives us more headroom for failover, which matters \u2014 right now losing a GPU takes us close to degraded service. But adding hardware also adds failure modes: another node to patch, another power feed, another point of configuration drift. The cheaper reliability win is architectural. Right now our redundancy is thin because everything is crammed onto the same three nodes with no slack. If utilisation work frees up even half a node's worth of capacity, we can hold it as a genuine reserve rather than letting work expand to fill it, which is what happened last time. Capacity you don't deliberately reserve gets consumed.\n\nThe opportunity cost question is the one I take most seriously. Six weeks of senior engineering time on utilisation work is not free; that's time not spent on the feature work the product team wants. But I'd frame it this way: the utilisation work is largely a one-time investment that compounds \u2014 better scheduling, better batching, and honest capacity telemetry benefit us whether or not we ever buy more hardware. The purchase, by contrast, is a recurring commitment. Once we buy, we own the maintenance, the power bill, and eventually the disposal problem. And hardware depreciates fast; a node bought today to solve a problem that better scheduling would have solved is money we can't spend on the thing that actually might need it next year.\n\nReversibility favours the same path. Deferring the purchase is fully reversible \u2014 if utilisation work stalls or demand keeps growing past what we've reclaimed, we can place the order in six weeks having lost little except time. Buying is much harder to undo. We can't easily return a rack-mounted node, and the operational commitment it represents shapes our infrastructure decisions for years. When one option is reversible and the other isn't, and the evidence is genuinely ambiguous, the reversible option should win.\n\nI want to be clear about the uncertainty here. My utilisation estimates come from partial telemetry, and our monitoring has blind spots \u2014 it's possible the fragmentation I've described is less severe than it looks, or that the low-hanging fruit is smaller than I hope. Demand forecasts are also shaky; if the new internal tooling ships on schedule, traffic could grow faster than any efficiency gains. That's why I'm proposing a decision point, not a permanent deferral. If by mid-quarter we haven't reclaimed meaningful capacity, or if the demand curve steepens, we buy, and we'll do so with much better data than we have today.\n\nConcretely: I'd like two engineers on the utilisation effort starting next sprint, with a checkpoint review in six weeks. Please flag concerns before Thursday's planning meeting.\n\n\u2014 Infrastructure Lead",
          "output_characters": 5248,
          "output_sha256": "0856c9fa675a111abbe0a2e3fc64e9fd4aa4671464c193a12ad3695299bc895b",
          "prompt_tokens": 123,
          "reasoning_characters": 59,
          "reasoning_sha256": "2240e9e7fe5cec082dc7ebe9b2384d3b443959c6fdb62a512062bd696289f917",
          "response_tail_seconds": 0.001178,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "6b680b6774d9fbaff5254722e6db4defbebf2bb96f6f1b77118312e5c87d357d",
          "time_to_first_visible_seconds": 1.172023,
          "time_to_last_output_seconds": 25.153584,
          "ttft_seconds": 0.497614,
          "wall_seconds": 25.154762
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1023,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.064071,
          "decode_tokens_per_second": 44.311,
          "finish_reason": "stop",
          "measured_sse_events": 1021,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\n**Status:** For discussion at Thursday's planning session\n\nOver the past month we've hit GPU saturation on our inference fleet three times, each incident producing p99 latency spikes that degraded the assistant experience for a meaningful slice of users. The instinctive response \u2014 buy more hardware \u2014 deserves scrutiny before we commit. This memo lays out my thinking and a recommendation, with the caveat that several numbers below are estimates I'd want us to tighten before any purchase order is signed.\n\nLet's start with the case for buying. Our current fleet of eight A100s runs at roughly 78% average utilisation, but that average conceals the real problem: peak-hour utilisation regularly exceeds 95%, while overnight and weekend utilisation drops below 30%. The hardware isn't uniformly scarce; it's scarce at exactly the moments our users are most active. Buying four more cards would cost approximately $140K upfront, plus roughly $2K monthly in colocation and power. That's real money for a team our size, and it's worth asking what else that capital could do \u2014 a dedicated engineer-quarter for inference optimisation, for instance, costs a fraction of that and could plausibly deliver comparable relief.\n\nThe case for improving utilisation rests on the shape of our own telemetry. Our traffic is heavily diurnal, and our current scheduler treats all requests as equal-priority batch jobs. Three concrete opportunities stand out. First, dynamic batching: we currently batch with a fixed window of 10ms, and early experiments suggest an adaptive window could raise effective throughput 15\u201320% at identical latency. Second, workload shaping: our internal evaluation and fine-tuning jobs run on the same fleet as production inference, and they're scheduled naively. Moving batch workloads to off-peak windows, or pre-empting them under load, would free meaningful capacity during exactly the peaks that hurt us. Third, quantisation: moving our largest model from FP16 to INT8 would cut memory footprint nearly in half, letting us consolidate workloads across fewer cards. None of these are exotic; all are well-trodden paths we simply haven't walked.\n\nOn latency, the optimisation path is actually safer than buying. Adding hardware without changing scheduling would lower queueing delay somewhat, but our worst latency incidents are driven by contention \u2014 batch jobs colliding with interactive traffic \u2014 which more hardware only dilutes rather than fixes. Optimisation attacks the root cause. On reliability, there's a counterintuitive point in favour of better utilisation: a fleet running at 95% peak has no headroom to absorb a node failure. If one of our eight cards dies during peak, we're not degraded, we're down. Improving scheduling efficiency effectively buys us failure headroom for free, which four additional cards at naive scheduling would not.\n\nOn opportunity cost and reversibility, the asymmetry is decisive. Capital spent on GPUs is largely sunk: hardware depreciates, and if our traffic mix shifts \u2014 say, toward smaller distilled models \u2014 those cards may be poorly matched to future needs. Optimisation work, by contrast, is mostly engineering time, and its outputs (better schedulers, quantised models, adaptive batching) are portable across whatever hardware we eventually run. If optimisation succeeds, we've saved $140K. If it partially succeeds, we can still buy hardware later, with better information about what we actually need. The option value of waiting is genuinely high here, and buying now forfeits it.\n\nMy recommendation, then: commit the next six weeks to a utilisation push \u2014 adaptive batching, workload separation, and a quantisation spike \u2014 with a concrete success threshold. If, after that work, peak-hour p99 latency still exceeds 400ms or peak utilisation still exceeds 90%, we proceed with the hardware purchase, and I'll sponsor it enthusiastically. This is a staged decision, not a refusal to spend.\n\nI want to be honest about the uncertainties. My throughput estimates for adaptive batching come from a two-day experiment, not a production trial; quantisation may degrade output quality in ways our evals don't yet catch; and if product launches next quarter drive a step-change in traffic, no amount of scheduling cleverness will save us. There's also an engineering cost to optimisation work that's easy to undercount \u2014 this is senior-engineer time pulled from other priorities, and I've estimated it at roughly 0.75 FTE for the six weeks.\n\nI'd like alignment on the success thresholds by Thursday, and volunteers for the quantisation spike. If we can't reach consensus, I'll default to the hardware purchase \u2014 the cost of prolonged latency incidents to user trust may exceed the cost of the cards, and I don't want analysis paralysis to make this decision for us.",
          "output_characters": 5006,
          "output_sha256": "39360d6028f821f760ba57e9976e81bf3b42e1dba2011283ea90b788ffa82e87",
          "prompt_tokens": 121,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.00116,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "6113fbdcfed758184a37f44f42c7a605a06606266bde309825c504ed1a391d68",
          "time_to_first_visible_seconds": 1.06793,
          "time_to_last_output_seconds": 23.562194,
          "ttft_seconds": 0.498123,
          "wall_seconds": 23.563354
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 982,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.735923,
          "decode_tokens_per_second": 43.148,
          "finish_reason": "stop",
          "measured_sse_events": 980,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Platform Engineering\n**From:** Infrastructure Planning\n**Re:** Inference capacity for the next two quarters\n\n**Status: Recommendation attached; decision needed by end of month**\n\n---\n\nWe've been running hot on our local inference hardware for about six weeks now. Peak-hour GPU utilisation sits in the high eighties, queue times during the morning and late-afternoon spikes have crept from milliseconds into seconds, and two of our internal customers have started asking whether they should budget for a hosted API fallback. The question on the table is whether we buy more hardware or invest in squeezing more out of what we already own. I want to lay out how I'm thinking about it, because the answer is less obvious than the utilisation graph suggests.\n\nThe case for buying is straightforward on its face. Our current fleet was sized for a workload that has roughly tripled, and hardware is a one-time decision that removes the constraint entirely. But the cost side deserves honest arithmetic. Capital purchase is only the visible part; the real number includes power, cooling, rack space, the engineering time to integrate and monitor new machines, and the depreciation clock that starts ticking the moment the boxes arrive. At our current utilisation, a new node would spend most of its life idle \u2014 we'd be buying capacity for the ninety-ninth percentile spike and paying for it across every quiet hour in between. That's an expensive way to solve a problem that is, so far, concentrated in a few predictable windows.\n\nThe case for improving utilisation rests on the observation that our peaks are shaped, not random. A meaningful fraction of our traffic is batch-shaped \u2014 embedding jobs, evaluation runs, report generation \u2014 and none of it actually needs sub-second latency. Right now it competes with interactive traffic for the same GPUs because we never built the scheduling to separate them. A modest investment in a priority queue and time-shifting batch work into off-peak hours would, by my rough model, cut our effective peak load by twenty to thirty percent. Add dynamic batching for the interactive path, which our current serving stack supports but we never tuned, and we may buy ourselves two quarters of headroom for the cost of a few engineering weeks rather than a capital line item.\n\nThere's also a reliability argument that cuts both ways, and I want to be careful here. More hardware means more redundancy: a single failed node today degrades us noticeably, whereas a larger fleet absorbs failure gracefully. That's a real benefit and I don't want to dismiss it. But our reliability incidents this year have mostly been software-side \u2014 a bad model rollout, a memory leak in the router \u2014 and a bigger fleet would not have prevented either. Meanwhile, higher utilisation itself carries risk: less slack means a traffic surprise or a stuck job turns into visible degradation faster. If we go the utilisation route, we should pair it with a hard rule that we never plan above roughly eighty-five percent sustained peak, so the buffer we're reclaiming is deliberate rather than accidental.\n\nThe opportunity-cost question is the one I keep returning to. The engineering weeks spent on scheduling and batching are weeks not spent on the model-quality work several teams are waiting on. But the capital we'd spend on hardware is similarly not spent on anything else, and unlike engineering time, it's committed. That asymmetry matters. Money spent on GPUs is largely sunk; engineering effort on scheduling is reversible, and the tooling we'd build \u2014 better queueing, better observability, load-shedding logic \u2014 pays off regardless of what we eventually decide about fleet size. Reversibility is worth something, and right now it's worth more to us than certainty.\n\nSo my recommendation is to defer the purchase for one quarter and spend that quarter on utilisation: batch/interactive separation, dynamic batching, and a proper capacity dashboard so we're arguing from data rather than anecdotes. I want to be clear about the uncertainty in this position. My twenty-to-thirty-percent figure is modelled, not measured, and if the real number is half that, we'll be back at this decision in eight weeks with less margin than I'd like. To hedge, I propose we pre-qualify a vendor and a configuration now \u2014 get quotes, validate compatibility, agree the spec \u2014 so that if utilisation work underdelivers, we can execute a purchase in days rather than restarting the analysis. If demand growth continues at its current trajectory, we will likely need hardware eventually regardless; the question is only whether we buy it now, at high cost and low utilisation, or later, with better information and a serving stack that uses it well.\n\nHappy to walk through the numbers with anyone who wants to pressure-test them.",
          "output_characters": 4892,
          "output_sha256": "05dbfcb2e3553a3ba1fc8adb72c4fea4f9c43ec8e7ca427de7808fad423614f1",
          "prompt_tokens": 126,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000725,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "c0e22b00777f3a7d6739d1a503806bfe5be459de41977ac3cf29294956b03327",
          "time_to_first_visible_seconds": 1.079022,
          "time_to_last_output_seconds": 23.234868,
          "ttft_seconds": 0.498945,
          "wall_seconds": 23.235593
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1032,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.235915,
          "decode_tokens_per_second": 44.371,
          "finish_reason": "stop",
          "measured_sse_events": 1029,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for the next two quarters\n\n**Status:** For discussion at Thursday's planning session\n\nWe've been running hot on our inference fleet for about six weeks now. Peak-hour GPU utilisation has been sitting above 85 percent, queue times during the morning and late-afternoon spikes have crept from milliseconds into seconds, and we've had two incidents where latency-sensitive endpoints degraded enough to generate user complaints. The obvious question on the table is whether we should buy more hardware or get smarter about what we already own. I want to lay out how I'm thinking about this before we meet, because I don't think the answer is as simple as either option suggests.\n\nThe case for buying is straightforward on the surface. Our current fleet cost us roughly what we'd pay again for a modest expansion, and lead times on the hardware we'd want are now measured in weeks rather than months. Adding capacity would immediately relieve the queue pressure, and it would give us headroom for the two new internal services scheduled to launch next quarter. But the cost isn't just the invoice. There's power, cooling, rack space, and \u2014 more importantly \u2014 the ongoing operational burden. Every additional node is something we patch, monitor, and eventually decommission. We're a small team, and our scarcest resource isn't budget; it's engineering attention. I'd estimate that a new cluster would consume somewhere between a quarter and a third of one engineer's ongoing time, which is time not spent on the product work that actually differentiates us.\n\nThe case for improving utilisation rests on the shape of our load. Our traffic is heavily diurnal: we peak hard for a few hours and idle at 20 to 30 percent utilisation overnight. That pattern is a classic signal that scheduling, not raw capacity, is the constraint. We currently pin models to specific nodes, which means a latency-insensitive batch job can't borrow capacity from a node reserved for a model nobody is querying at 2 a.m. If we implemented proper dynamic scheduling with priority classes \u2014 letting batch and offline evaluation workloads yield to interactive traffic \u2014 I believe we could effectively recover 20 to 30 percent of our fleet without spending anything on hardware. We could also right-size some of our model deployments; at least two services are running quantisation-unfriendly configurations that a modest engineering effort could shrink by half.\n\nThe risk with the utilisation path is that it's engineering work with an uncertain payoff and a real opportunity cost. Scheduling work is fiddly, it touches the hot path for every request, and a botched implementation could cause exactly the reliability incidents we're trying to avoid. There's also a hard limit: if our baseline load grows rather than our peak, no amount of scheduling cleverness will save us. Buying hardware, by contrast, is nearly risk-free in execution terms \u2014 it either arrives or it doesn't \u2014 but it's the less reversible choice. Hardware we buy is capital we've committed; if our workload mix shifts, if a hosted API becomes good enough for one of our services, or if the team's priorities change, that spend doesn't come back. Scheduling improvements, by contrast, are almost entirely reversible: if the dynamic scheduler turns out to be a mistake, we revert and we've lost a few engineer-weeks.\n\nOn latency specifically, I want to be honest that scheduling changes carry a tail risk that buying does not. Preemption and dynamic batching can introduce p99 latency spikes even when average latency improves. For our most latency-sensitive endpoint \u2014 the one embedded in the editor experience \u2014 I would not want to bet its reliability on a new scheduler without a long soak period. That endpoint is also small enough, in traffic terms, that we could carve out dedicated capacity for it and still get most of the utilisation win elsewhere.\n\nMy recommendation: do both, but sequence them. Fund the utilisation work now \u2014 it's cheap, reversible, and the diurnal shape of our load says there's real headroom hiding in our scheduling. Simultaneously, place the hardware order but with a delivery date six to eight weeks out, which gives us a natural decision point. If the scheduling work lands and peak utilisation drops below 70 percent, we can defer or reduce the order. If it stalls, or if baseline load grows faster than expected, the hardware arrives right as we need it. The deposit on the order is a small price for keeping both options open.\n\nThe main uncertainty I'd flag is our own growth forecast. If the new services launching next quarter drive more traffic than the current estimates suggest, the utilisation play buys us time but not a solution, and we should treat the hardware as inevitable rather than optional. I'd like us to revisit the traffic model with two weeks of real data before finalising the order quantity.\n\nHappy to walk through the numbers on Thursday.",
          "output_characters": 5092,
          "output_sha256": "1475a557635c05fb91d0f7dbf1c9fa4f5f1fc77aaee2acd76090f40c29f90b47",
          "prompt_tokens": 124,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.001124,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "828559fb3849c9bb48abfe05c1704464edeb4c099ae2070dd6dc668e4cd5320e",
          "time_to_first_visible_seconds": 1.078962,
          "time_to_last_output_seconds": 23.740635,
          "ttft_seconds": 0.504719,
          "wall_seconds": 23.741759
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 963,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.553262,
          "decode_tokens_per_second": 42.655,
          "finish_reason": "stop",
          "measured_sse_events": 961,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Platform Engineering\n**From:** Infrastructure Lead\n**Re:** Inference capacity planning for Q3\n\n**Status:** For discussion at Thursday's planning session\n\n---\n\nWe've been running hot on our inference fleet for about six weeks now. Peak-hour GPU utilisation has been sitting above 85%, queue times during the morning and late-afternoon spikes have crept from milliseconds into seconds, and two of our internal customers have started asking whether they should plan around our constraints rather than our capabilities. That last part worries me more than the numbers themselves. The question on the table is whether we buy more hardware or invest in squeezing more out of what we own. I want to lay out how I'm thinking about it, and then make a recommendation, because I think we've been circling this without committing.\n\nThe cost picture is more nuanced than the sticker price. Buying four more A100-class nodes is a capital hit of roughly $180K upfront, plus maybe $2K a month in power, cooling, and rack space, plus the less visible cost of someone's time to rack, image, and maintain them. Improving utilisation is not free either, though. The batching and scheduling work we'd need is realistically six to eight weeks of our strongest engineer's time, and that's time not spent on the retrieval-quality project the search team has been waiting on. That opportunity cost is real and I don't want to pretend it away. But it's also a cost we pay once, in labour we've already budgeted, rather than an ongoing commitment that compounds.\n\nLatency is where the two options genuinely diverge. Right now our tail latencies degrade badly at peak, and that's a utilisation problem, not a capacity problem \u2014 we're queuing requests behind badly packed batches. Better continuous batching and smarter request routing could plausibly cut our p99 in half without adding a single GPU. Buying hardware would also help, but it would help by making our inefficiency cheaper to tolerate. That's the trap I want us to avoid: hardware is a way of paying money instead of paying attention. If we buy first and optimise never, we'll be back here in a year with the same conversation and a bigger bill.\n\nReliability cuts the other way, and I want to be honest about that. Our fleet has no meaningful headroom. When node 3 dropped its NVLink last month, we degraded gracefully only because demand happened to be low that week. A fleet running at 85% utilisation has no room to lose a machine, and machines do lose themselves at inconvenient times. More hardware buys us failure tolerance that optimisation cannot. There's also a resilience argument for capacity in the abstract: if an internal team ships something that goes viral internally, we want to absorb the spike, not apologise for it. Utilisation work makes us efficient; headroom makes us robust. These are different virtues and we shouldn't confuse them.\n\nReversibility is the tiebreaker for me. The scheduling and batching improvements are almost entirely reversible and portable \u2014 the engineering effort teaches us things about our own workload that we keep regardless of what we decide next. Buying hardware is much harder to walk back. Depreciated servers are a sunk cost, and the temptation to keep running them because \"we already own them\" is exactly how fleets rot. If we optimise first and still find ourselves saturated, the purchase decision will be better informed: we'll know our true demand curve, not the demand curve implied by our current inefficiency. There's a real chance the answer changes \u2014 maybe we need capacity for a different workload shape than we assumed \u2014 and we can only see that clearly from the other side of the optimisation work.\n\nSo my recommendation: do the utilisation work now, and pre-approve a hardware purchase contingent on its outcome. Specifically, spend the next six weeks on batching, routing, and killing the worst of the peak-hour contention, with a hard checkpoint at the end. If utilisation is still breaching 80% at peak after that, we buy \u2014 and we'll buy knowing exactly what we need. I'm genuinely uncertain here. It's possible the optimisation gains are half what the vendor benchmarks suggest, and we'll have spent six weeks learning something expensive. But six weeks of one engineer is a small, recoverable bet compared to $180K of silicon that we might not need, or that might be the wrong shape. The risk asymmetry favours looking before buying.\n\nI'd like a decision by Thursday so we can start the scheduling work the following sprint.",
          "output_characters": 4605,
          "output_sha256": "bd42eaa5a415dc874463dfa44aa3b6cfafc9cffd5e4ab4352f4c34663bf83ef1",
          "prompt_tokens": 122,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000732,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "dd377f64e47b4171aed683c92ac1493fa4d2731b5bea45d9cf56edb2a7fb9877",
          "time_to_first_visible_seconds": 1.073797,
          "time_to_last_output_seconds": 23.051776,
          "ttft_seconds": 0.498514,
          "wall_seconds": 23.052508
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 25.154,
        "median": 23.562,
        "minimum": 23.052,
        "p90": 25.154
      },
      "ttft_seconds": {
        "maximum": 0.505,
        "median": 0.499,
        "minimum": 0.498,
        "p90": 0.505
      },
      "wall_seconds": {
        "maximum": 25.155,
        "median": 23.563,
        "minimum": 23.053,
        "p90": 25.155
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 72.189,
        "median": 69.993,
        "minimum": 67.815,
        "p90": 72.189
      },
      "pooled_decode_tokens_per_second": 70.269,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.159485,
          "decode_tokens_per_second": 70.785,
          "finish_reason": "stop",
          "measured_sse_events": 433,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 111,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000145,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.544518,
          "time_to_last_output_seconds": 6.704002,
          "ttft_seconds": 0.544518,
          "wall_seconds": 6.704147
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 447,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.576691,
          "decode_tokens_per_second": 67.815,
          "finish_reason": "stop",
          "measured_sse_events": 445,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 108,
          "reasoning_characters": 40,
          "reasoning_sha256": "1fc5cfc87b905f49ac82467b9ebaa49ed6b407021db4c1a3132459522ca867e0",
          "response_tail_seconds": 0.00053,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "342ef8f4e6e61494f52a121e011d937dff1aa19e23c118e4072aed44d2372d34",
          "time_to_first_visible_seconds": 0.951593,
          "time_to_last_output_seconds": 7.056883,
          "ttft_seconds": 0.480192,
          "wall_seconds": 7.057413
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.229235,
          "decode_tokens_per_second": 69.993,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000136,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.53019,
          "time_to_last_output_seconds": 6.759425,
          "ttft_seconds": 0.53019,
          "wall_seconds": 6.759562
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 694,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 9.599821,
          "decode_tokens_per_second": 72.189,
          "finish_reason": "stop",
          "measured_sse_events": 692,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 110,
          "reasoning_characters": 22,
          "reasoning_sha256": "8eb7b7a7427b61499264af851bab5b130e33df9c8496347175d3c7c26512b253",
          "response_tail_seconds": 0.000608,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "0593955c7022186ee66d7f8d90effa4aaf77e159994adbbaf5740bf8aefd14d4",
          "time_to_first_visible_seconds": 0.856195,
          "time_to_last_output_seconds": 10.085896,
          "ttft_seconds": 0.486075,
          "wall_seconds": 10.086503
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.257928,
          "decode_tokens_per_second": 69.672,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000441,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.529029,
          "time_to_last_output_seconds": 6.786958,
          "ttft_seconds": 0.529029,
          "wall_seconds": 6.787399
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 10.086,
        "median": 6.787,
        "minimum": 6.704,
        "p90": 10.086
      },
      "ttft_seconds": {
        "maximum": 0.545,
        "median": 0.529,
        "minimum": 0.48,
        "p90": 0.545
      },
      "wall_seconds": {
        "maximum": 10.087,
        "median": 6.787,
        "minimum": 6.704,
        "p90": 10.087
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1985.386,
          "median": 1985.328,
          "minimum": 1984.657,
          "p90": 1985.386
        },
        "runs": [
          {
            "cache_salt_sha256": "5fd8fcf2ba97b1fc2cbc2d09a75af0922ab8e2d32073b0774ef062bec7a91b50",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062996,
            "decode_tokens_per_second": 111.118,
            "effective_prefill_tokens_per_second": 1985.328,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000741,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.505084,
            "time_to_last_output_seconds": 16.56808,
            "ttft_seconds": 16.505084,
            "wall_seconds": 16.568821
          },
          {
            "cache_salt_sha256": "128ddd0f6c1ff3da49e35684c32d9b75ebc0f37dbe5da7431469908ed55f4512",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.055427,
            "decode_tokens_per_second": 126.293,
            "effective_prefill_tokens_per_second": 1985.386,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00017,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 16.5046,
            "time_to_last_output_seconds": 16.560027,
            "ttft_seconds": 16.5046,
            "wall_seconds": 16.560197
          },
          {
            "cache_salt_sha256": "2310c4f5435fc1886f3499e440083c11779487dedc961e24ace121875ee06e12",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.06232,
            "decode_tokens_per_second": 112.324,
            "effective_prefill_tokens_per_second": 1984.657,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.00054,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 16.510661,
            "time_to_last_output_seconds": 16.572981,
            "ttft_seconds": 16.510661,
            "wall_seconds": 16.573521
          }
        ],
        "ttft_seconds": {
          "maximum": 16.511,
          "median": 16.505,
          "minimum": 16.505,
          "p90": 16.511
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 782650.234,
          "median": 764375.19,
          "minimum": 739300.138,
          "p90": 782650.234
        },
        "runs": [
          {
            "cache_salt_sha256": "5fd8fcf2ba97b1fc2cbc2d09a75af0922ab8e2d32073b0774ef062bec7a91b50",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.061977,
            "decode_tokens_per_second": 112.945,
            "effective_prefill_tokens_per_second": 739300.138,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000621,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.044323,
            "time_to_last_output_seconds": 0.1063,
            "ttft_seconds": 0.044323,
            "wall_seconds": 0.106921
          },
          {
            "cache_salt_sha256": "128ddd0f6c1ff3da49e35684c32d9b75ebc0f37dbe5da7431469908ed55f4512",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.05482,
            "decode_tokens_per_second": 127.691,
            "effective_prefill_tokens_per_second": 782650.234,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000962,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 0.041868,
            "time_to_last_output_seconds": 0.096688,
            "ttft_seconds": 0.041868,
            "wall_seconds": 0.097649
          },
          {
            "cache_salt_sha256": "2310c4f5435fc1886f3499e440083c11779487dedc961e24ace121875ee06e12",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.063372,
            "decode_tokens_per_second": 110.458,
            "effective_prefill_tokens_per_second": 764375.19,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000895,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.042869,
            "time_to_last_output_seconds": 0.106242,
            "ttft_seconds": 0.042869,
            "wall_seconds": 0.107137
          }
        ],
        "ttft_seconds": {
          "maximum": 0.044,
          "median": 0.043,
          "minimum": 0.042,
          "p90": 0.044
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1957.407,
          "median": 1956.876,
          "minimum": 1956.606,
          "p90": 1957.407
        },
        "runs": [
          {
            "cache_salt_sha256": "49eb69aee41b6f69c293592475fb4fe781a4754ba1341bf7914662d1481b10b5",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.06213,
            "decode_tokens_per_second": 112.667,
            "effective_prefill_tokens_per_second": 1956.876,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000667,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 33.490119,
            "time_to_last_output_seconds": 33.552249,
            "ttft_seconds": 33.490119,
            "wall_seconds": 33.552917
          },
          {
            "cache_salt_sha256": "23ae45dc62c54cabf87a162d45c77bd2f02c40df9dba107fdca3c38c5c7b7b7a",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.064427,
            "decode_tokens_per_second": 108.65,
            "effective_prefill_tokens_per_second": 1956.606,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00068,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 33.49473,
            "time_to_last_output_seconds": 33.559157,
            "ttft_seconds": 33.49473,
            "wall_seconds": 33.559837
          },
          {
            "cache_salt_sha256": "0615a7fb7e85fb5c7b88530329d5a83b2be8b893ac4c1e577fb8e882df300caa",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.059349,
            "decode_tokens_per_second": 117.947,
            "effective_prefill_tokens_per_second": 1957.407,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000357,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 33.481021,
            "time_to_last_output_seconds": 33.540369,
            "ttft_seconds": 33.481021,
            "wall_seconds": 33.540727
          }
        ],
        "ttft_seconds": {
          "maximum": 33.495,
          "median": 33.49,
          "minimum": 33.481,
          "p90": 33.495
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1516721.053,
          "median": 1516405.202,
          "minimum": 1446838.573,
          "p90": 1516721.053
        },
        "runs": [
          {
            "cache_salt_sha256": "49eb69aee41b6f69c293592475fb4fe781a4754ba1341bf7914662d1481b10b5",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.061295,
            "decode_tokens_per_second": 114.201,
            "effective_prefill_tokens_per_second": 1516405.202,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000401,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.043218,
            "time_to_last_output_seconds": 0.104514,
            "ttft_seconds": 0.043218,
            "wall_seconds": 0.104915
          },
          {
            "cache_salt_sha256": "23ae45dc62c54cabf87a162d45c77bd2f02c40df9dba107fdca3c38c5c7b7b7a",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063452,
            "decode_tokens_per_second": 110.319,
            "effective_prefill_tokens_per_second": 1446838.573,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.001167,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.045296,
            "time_to_last_output_seconds": 0.108748,
            "ttft_seconds": 0.045296,
            "wall_seconds": 0.109915
          },
          {
            "cache_salt_sha256": "0615a7fb7e85fb5c7b88530329d5a83b2be8b893ac4c1e577fb8e882df300caa",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.060973,
            "decode_tokens_per_second": 114.806,
            "effective_prefill_tokens_per_second": 1516721.053,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.00085,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 0.043209,
            "time_to_last_output_seconds": 0.104181,
            "ttft_seconds": 0.043209,
            "wall_seconds": 0.105031
          }
        ],
        "ttft_seconds": {
          "maximum": 0.045,
          "median": 0.043,
          "minimum": 0.043,
          "p90": 0.045
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1959.995,
          "median": 1959.427,
          "minimum": 1957.116,
          "p90": 1959.995
        },
        "runs": [
          {
            "cache_salt_sha256": "9b561a5af4489be7f55477121ceb9d015ba940455c0db765c4e6849d72006937",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.052989,
            "decode_tokens_per_second": 132.104,
            "effective_prefill_tokens_per_second": 1959.995,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000131,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 4.179603,
            "time_to_last_output_seconds": 4.232592,
            "ttft_seconds": 4.179603,
            "wall_seconds": 4.232723
          },
          {
            "cache_salt_sha256": "34511194b33f34cc79014126ab6f35253b13e8e54ccfa3d2d4f1aa4675ec8253",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054541,
            "decode_tokens_per_second": 128.344,
            "effective_prefill_tokens_per_second": 1959.427,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000129,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.180814,
            "time_to_last_output_seconds": 4.235355,
            "ttft_seconds": 4.180814,
            "wall_seconds": 4.235484
          },
          {
            "cache_salt_sha256": "0232a9cb2226336d1e88519ae4802fb77a5badb5a3c349e042154c5e4688991d",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.054531,
            "decode_tokens_per_second": 128.367,
            "effective_prefill_tokens_per_second": 1957.116,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 6.2e-05,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 4.18575,
            "time_to_last_output_seconds": 4.240282,
            "ttft_seconds": 4.18575,
            "wall_seconds": 4.240344
          }
        ],
        "ttft_seconds": {
          "maximum": 4.186,
          "median": 4.181,
          "minimum": 4.18,
          "p90": 4.186
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 260460.384,
          "median": 257585.762,
          "minimum": 198051.399,
          "p90": 260460.384
        },
        "runs": [
          {
            "cache_salt_sha256": "9b561a5af4489be7f55477121ceb9d015ba940455c0db765c4e6849d72006937",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.053249,
            "decode_tokens_per_second": 131.457,
            "effective_prefill_tokens_per_second": 257585.762,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000378,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 0.031803,
            "time_to_last_output_seconds": 0.085052,
            "ttft_seconds": 0.031803,
            "wall_seconds": 0.08543
          },
          {
            "cache_salt_sha256": "34511194b33f34cc79014126ab6f35253b13e8e54ccfa3d2d4f1aa4675ec8253",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.054624,
            "decode_tokens_per_second": 128.149,
            "effective_prefill_tokens_per_second": 260460.384,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000392,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.031452,
            "time_to_last_output_seconds": 0.086076,
            "ttft_seconds": 0.031452,
            "wall_seconds": 0.086468
          },
          {
            "cache_salt_sha256": "0232a9cb2226336d1e88519ae4802fb77a5badb5a3c349e042154c5e4688991d",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.051489,
            "decode_tokens_per_second": 135.952,
            "effective_prefill_tokens_per_second": 198051.399,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.00077,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 0.041363,
            "time_to_last_output_seconds": 0.092852,
            "ttft_seconds": 0.041363,
            "wall_seconds": 0.093622
          }
        ],
        "ttft_seconds": {
          "maximum": 0.041,
          "median": 0.032,
          "minimum": 0.031,
          "p90": 0.041
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "checkpoint MTP head (Mia v1.10 DRAFTER=mtp)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3",
      "model_revision": "13ba3ef94d88bc30b27691efa29a6012f66e964d (AxonZeta/GLM-5.3-Flash-AZ-Focus-EXL3 on Hugging Face; the same files, served from a local copy)",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-mtp-low",
    "finished_at": "2026-10-08T11:26:49.992230+00:00",
    "label": "tp2-ours-mtp-low",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T11:17:15.011994+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 8192,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "low"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 5,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}