{
  "concurrency": {
    "1": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 54.489,
        "median": 53.328,
        "minimum": 52.852,
        "p90": 54.489
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 60.371,
        "median": 58.859,
        "minimum": 58.598,
        "p90": 60.371
      },
      "per_stream_ttft_seconds": {
        "maximum": 0.491,
        "median": 0.473,
        "minimum": 0.467,
        "p90": 0.491
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 52.852,
          "completion_tokens": 256,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.351676,
              "decode_tokens_per_second": 58.598,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second\n\tburst  int\n\tclock  Clock\n}\n\nvar ErrInvalidParams = errors.New(\"ratelimit: rate must be > 0 and burst must be >= 1\")\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. A nil clock uses the real time.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst < 1 {\n\t\treturn nil, ErrInvalidParams\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\ttokens:",
              "output_characters": 933,
              "output_sha256": "c40d65aba0a7bac11c9a16ea4cc3987dd03e914a3b4f26069456d118ec222c4e",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9e-05,
              "stream_done_marker": true,
              "stream_sha256": "c40d65aba0a7bac11c9a16ea4cc3987dd03e914a3b4f26069456d118ec222c4e",
              "time_to_first_visible_seconds": 0.491178,
              "time_to_last_output_seconds": 4.842854,
              "ttft_seconds": 0.491178,
              "wall_seconds": 4.842944
            }
          ],
          "wall_seconds": 4.84368
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.328,
          "completion_tokens": 256,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.332403,
              "decode_tokens_per_second": 58.859,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second, > 0\n\tburst int     // capacity, > 0\n\tclock Clock\n}\n\nvar (\n\tErrInvalidRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrInvalidBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock     = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. If clock is",
              "output_characters": 940,
              "output_sha256": "7733a62b96fa0d540be6fec936fa9a4e8e2eaa45ec19452b748b7f7c12219d80",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 8.3e-05,
              "stream_done_marker": true,
              "stream_sha256": "7733a62b96fa0d540be6fec936fa9a4e8e2eaa45ec19452b748b7f7c12219d80",
              "time_to_first_visible_seconds": 0.46725,
              "time_to_last_output_seconds": 4.799653,
              "ttft_seconds": 0.46725,
              "wall_seconds": 4.799736
            }
          ],
          "wall_seconds": 4.800467
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.489,
          "completion_tokens": 256,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.22391,
              "decode_tokens_per_second": 60.371,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds at most burst tokens and refills at rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\tclock    Clock\n}\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. rate must be > 0 and burst must be >= 1.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"ratelimit:",
              "output_characters": 973,
              "output_sha256": "bc467f0f02aca01117a928a3f3e4ba33651f275f9541579d42cb713d70a396da",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9.8e-05,
              "stream_done_marker": true,
              "stream_sha256": "bc467f0f02aca01117a928a3f3e4ba33651f275f9541579d42cb713d70a396da",
              "time_to_first_visible_seconds": 0.473402,
              "time_to_last_output_seconds": 4.697312,
              "ttft_seconds": 0.473402,
              "wall_seconds": 4.69741
            }
          ],
          "wall_seconds": 4.698216
        }
      ]
    },
    "2": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 54.025,
        "median": 53.579,
        "minimum": 52.958,
        "p90": 54.025
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 60.497,
        "median": 59.581,
        "minimum": 57.204,
        "p90": 60.412
      },
      "per_stream_ttft_seconds": {
        "maximum": 5.41,
        "median": 2.83,
        "minimum": 0.472,
        "p90": 5.334
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 52.958,
          "completion_tokens": 512,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.457761,
              "decode_tokens_per_second": 57.204,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second, > 0\n\tburst  float64 // max tokens, > 0\n\tclock  Clock\n}\n\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrClock = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New returns a Limiter with a full bucket. rate is tokens per second;\n// burst is the maximum bucket size. A nil clock defaults to the real clock.\nfunc New(rate float64, burst int, clock Clock) (*",
              "output_characters": 926,
              "output_sha256": "bbc81713750a3bc3f039b470e8f95b5b72bf3ddc787e497de8554edd6c82aec1",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000138,
              "stream_done_marker": true,
              "stream_sha256": "bbc81713750a3bc3f039b470e8f95b5b72bf3ddc787e497de8554edd6c82aec1",
              "time_to_first_visible_seconds": 0.481157,
              "time_to_last_output_seconds": 4.938918,
              "ttft_seconds": 0.481157,
              "wall_seconds": 4.939056
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.256571,
              "decode_tokens_per_second": 59.907,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a",
              "output_characters": 999,
              "output_sha256": "8401ad1e118749ad0502e8f866112694a232ca03f251e66104677c8cbbe07e61",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 7.8e-05,
              "stream_done_marker": true,
              "stream_sha256": "8401ad1e118749ad0502e8f866112694a232ca03f251e66104677c8cbbe07e61",
              "time_to_first_visible_seconds": 5.410032,
              "time_to_last_output_seconds": 9.666603,
              "ttft_seconds": 5.410032,
              "wall_seconds": 9.666681
            }
          ],
          "wall_seconds": 9.668044
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.025,
          "completion_tokens": 512,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.215076,
              "decode_tokens_per_second": 60.497,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It starts full and refills at Rate tokens per second, up to Capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// Errors returned by New.\nvar (\n\tErrNonPositiveCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n\tErrNonPositiveRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNilClock            = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n//",
              "output_characters": 985,
              "output_sha256": "04e183b6eb7288e3334fcf39b270638f12639250974c3319988b8d37fc74e559",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000463,
              "stream_done_marker": true,
              "stream_sha256": "04e183b6eb7288e3334fcf39b270638f12639250974c3319988b8d37fc74e559",
              "time_to_first_visible_seconds": 0.48736,
              "time_to_last_output_seconds": 4.702436,
              "ttft_seconds": 0.48736,
              "wall_seconds": 4.702898
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.303406,
              "decode_tokens_per_second": 59.255,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. It is an abstraction over time.Now to\n// allow deterministic testing.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// ErrInvalidConfig is returned by New when its arguments are invalid.\nvar ErrInvalidConfig = errors.New(\"ratelimit: invalid configuration\")\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64   // max tokens\n\ttokens   float64   // current tokens\n\trate     float64   // tokens per second\n\tlast     time.Time // last refill time\n}\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second). The bucket starts full.\nfunc New(capacity int, rate",
              "output_characters": 1071,
              "output_sha256": "e2fb29387fd7d5416f616e80d17e75897030bed4addc35d5f6d90600b0eb9782",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 8.7e-05,
              "stream_done_marker": true,
              "stream_sha256": "e2fb29387fd7d5416f616e80d17e75897030bed4addc35d5f6d90600b0eb9782",
              "time_to_first_visible_seconds": 5.172658,
              "time_to_last_output_seconds": 9.476064,
              "ttft_seconds": 5.172658,
              "wall_seconds": 9.476151
            }
          ],
          "wall_seconds": 9.477051
        },
        {
          "aggregate_end_to_end_tokens_per_second": 53.579,
          "completion_tokens": 512,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.220983,
              "decode_tokens_per_second": 60.412,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\nvar (\n\tErrZeroRate     = errors.New(\"ratelimit: rate must be > 0\")\n\tErrZeroCapacity = errors.New(\"ratelimit: capacity must be > 0\")\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// It starts full and refills at Rate tokens per second, up to Capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tcapacity float64\n\tclock    Clock\n}\n\n// New returns a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. The bucket starts full. clock may",
              "output_characters": 990,
              "output_sha256": "cf907d67e09b626be2657df24a373a72a277754ee8a50210d55ae360bf6d27da",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000126,
              "stream_done_marker": true,
              "stream_sha256": "cf907d67e09b626be2657df24a373a72a277754ee8a50210d55ae360bf6d27da",
              "time_to_first_visible_seconds": 5.333692,
              "time_to_last_output_seconds": 9.554675,
              "ttft_seconds": 5.333692,
              "wall_seconds": 9.554801
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.388215,
              "decode_tokens_per_second": 58.11,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. It is injectable for testing.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// ClockFunc adapts a function to a Clock.\ntype ClockFunc func() time.Time\n\n// Now implements Clock.\nfunc (f ClockFunc) Now() time.Time { return f() }\n\n// SystemClock uses the real wall clock.\nvar SystemClock Clock = ClockFunc(time.Now)\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter starts full and refills at Rate tokens per second, up to Capacity.\n// The zero value is not usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\tcapacity float64\n\trate     float64 // tokens per second\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n",
              "output_characters": 1025,
              "output_sha256": "c60539e7e81a31b8f18108a7bb2dfdfab8fcb6a8134124fadc8547b9dea991e0",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000171,
              "stream_done_marker": true,
              "stream_sha256": "c60539e7e81a31b8f18108a7bb2dfdfab8fcb6a8134124fadc8547b9dea991e0",
              "time_to_first_visible_seconds": 0.471836,
              "time_to_last_output_seconds": 4.860051,
              "ttft_seconds": 0.471836,
              "wall_seconds": 4.860222
            }
          ],
          "wall_seconds": 9.555904
        }
      ]
    },
    "4": {
      "aggregate_end_to_end_tokens_per_second": {
        "maximum": 54.105,
        "median": 53.041,
        "minimum": 52.966,
        "p90": 54.105
      },
      "per_stream_decode_tokens_per_second": {
        "maximum": 61.128,
        "median": 59.032,
        "minimum": 56.375,
        "p90": 59.918
      },
      "per_stream_ttft_seconds": {
        "maximum": 15.047,
        "median": 7.735,
        "minimum": 0.412,
        "p90": 14.976
      },
      "rounds": [
        {
          "aggregate_end_to_end_tokens_per_second": 53.041,
          "completion_tokens": 1024,
          "round": 1,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.335577,
              "decode_tokens_per_second": 58.816,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with a full bucket. rate is tokens per second;\n// burst is the maximum bucket size. Both must be > 0.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock ==",
              "output_characters": 988,
              "output_sha256": "63baef418a70990ac5ef252e5b82c9aea036b46552de070354949a5b52de63ed",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000209,
              "stream_done_marker": true,
              "stream_sha256": "63baef418a70990ac5ef252e5b82c9aea036b46552de070354949a5b52de63ed",
              "time_to_first_visible_seconds": 5.479475,
              "time_to_last_output_seconds": 9.815052,
              "ttft_seconds": 5.479475,
              "wall_seconds": 9.815261
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.255826,
              "decode_tokens_per_second": 59.918,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```ratelimit.go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. time.Now satisfies this interface.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be constructed with NewLimiter; the zero value is not usable.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\tcapacity float64\n\trate     float64 // tokens per second\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by NewLimiter for invalid parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// NewLimiter creates a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. If clock is nil, the real clock is used.\nfunc NewLimiter(capacity, rate float64, clock Clock) (*Limiter, error) {\n",
              "output_characters": 1030,
              "output_sha256": "3a51b97ada0068d4bdcaa6e32994f2b51dcd87e525aab40a6651c5f2a2ed39d7",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000133,
              "stream_done_marker": true,
              "stream_sha256": "3a51b97ada0068d4bdcaa6e32994f2b51dcd87e525aab40a6651c5f2a2ed39d7",
              "time_to_first_visible_seconds": 15.047464,
              "time_to_last_output_seconds": 19.30329,
              "ttft_seconds": 15.047464,
              "wall_seconds": 19.303423
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.295408,
              "decode_tokens_per_second": 59.366,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be created with New; the zero value is not usable.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n}\n\n// ErrInvalidConfig is returned by New for invalid parameters.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be > 0\")\n\n// New returns a Limiter that refills at rate tokens per second with a\n// maximum capacity of burst tokens. The bucket starts full.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n",
              "output_characters": 1021,
              "output_sha256": "072c4001aeeb4f56714e7c13ef1978f048e28efd66dc3bcce4582efe2ccb6649",
              "prompt_tokens": 132,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000102,
              "stream_done_marker": true,
              "stream_sha256": "072c4001aeeb4f56714e7c13ef1978f048e28efd66dc3bcce4582efe2ccb6649",
              "time_to_first_visible_seconds": 10.282395,
              "time_to_last_output_seconds": 14.577803,
              "ttft_seconds": 10.282395,
              "wall_seconds": 14.577905
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.523259,
              "decode_tokens_per_second": 56.375,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock reports the current time. time.Time's zero method set is satisfied by *time.Time wrappers in tests.\ntype Clock interface {\n\tNow() time.Time\n}\n\n// realClock uses the wall clock.\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter starts full and refills at Rate tokens per second, up to Capacity.\n// The zero value is not usable; construct with New.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\tcapacity float64\n\trate     float64 // tokens per second\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second),",
              "output_characters": 1040,
              "output_sha256": "f444da12b7eb420bdfd475d7ae7bf2e1cf1c09eae94eeb6534d7fc63af6dee4e",
              "prompt_tokens": 133,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000123,
              "stream_done_marker": true,
              "stream_sha256": "f444da12b7eb420bdfd475d7ae7bf2e1cf1c09eae94eeb6534d7fc63af6dee4e",
              "time_to_first_visible_seconds": 0.483517,
              "time_to_last_output_seconds": 5.006776,
              "ttft_seconds": 0.483517,
              "wall_seconds": 5.006898
            }
          ],
          "wall_seconds": 19.305683
        },
        {
          "aggregate_end_to_end_tokens_per_second": 52.966,
          "completion_tokens": 1024,
          "round": 2,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.374352,
              "decode_tokens_per_second": 58.294,
              "finish_reason": "length",
              "measured_sse_events": 252,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second\n\tburst  int\n\tclock  Clock\n}\n\nvar ErrInvalidParams = errors.New(\"ratelimit: rate must be > 0 and burst must be >= 1\")\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. The bucket starts full. A nil clock uses the real time.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, nil)\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter,",
              "output_characters": 967,
              "output_sha256": "806d9cc2ba40b1d1237047d4d6cbf74a9ca92a645e0620edb709914a8451809d",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000109,
              "stream_done_marker": true,
              "stream_sha256": "806d9cc2ba40b1d1237047d4d6cbf74a9ca92a645e0620edb709914a8451809d",
              "time_to_first_visible_seconds": 5.239964,
              "time_to_last_output_seconds": 9.614315,
              "ttft_seconds": 5.239964,
              "wall_seconds": 9.614424
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.406579,
              "decode_tokens_per_second": 57.868,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\trate   float64 // tokens per second\n\tburst  int\n\tclock  Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New creates a Limiter with the given refill rate (tokens per second),\n// burst capacity, and clock. A nil clock uses the real system clock.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= ",
              "output_characters": 1014,
              "output_sha256": "dba029323a12e21961ffd0da04dd97bfc26ef6a20b505fa193b71ec833490676",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 7.4e-05,
              "stream_done_marker": true,
              "stream_sha256": "dba029323a12e21961ffd0da04dd97bfc26ef6a20b505fa193b71ec833490676",
              "time_to_first_visible_seconds": 10.082678,
              "time_to_last_output_seconds": 14.489257,
              "ttft_seconds": 10.082678,
              "wall_seconds": 14.489331
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.353999,
              "decode_tokens_per_second": 58.567,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds at most Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64\n\tclock    Clock\n}\n\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New returns a Limiter with a full bucket. rate > 0 and burst > 0 are required.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable Clock.\nfunc NewWithClock(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif clock == nil {\n\t\treturn nil, errors.New(\"",
              "output_characters": 978,
              "output_sha256": "6cf9e601962deb75cedcb190e0b8cc0b14da667ebe090377ed8177fd2f607916",
              "prompt_tokens": 135,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000127,
              "stream_done_marker": true,
              "stream_sha256": "6cf9e601962deb75cedcb190e0b8cc0b14da667ebe090377ed8177fd2f607916",
              "time_to_first_visible_seconds": 14.976174,
              "time_to_last_output_seconds": 19.330173,
              "ttft_seconds": 14.976174,
              "wall_seconds": 19.3303
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.287326,
              "decode_tokens_per_second": 59.478,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// Tokens refill continuously at rate tokens per second, up to capacity.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. If clock is nil, RealClock is used.\nfunc New(capacity",
              "output_characters": 1043,
              "output_sha256": "551435757f6e706829c378015028fbef8a70d41537a5e8ad9b48e6f19962b97e",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9.4e-05,
              "stream_done_marker": true,
              "stream_sha256": "551435757f6e706829c378015028fbef8a70d41537a5e8ad9b48e6f19962b97e",
              "time_to_first_visible_seconds": 0.469609,
              "time_to_last_output_seconds": 4.756936,
              "ttft_seconds": 0.469609,
              "wall_seconds": 4.75703
            }
          ],
          "wall_seconds": 19.333253
        },
        {
          "aggregate_end_to_end_tokens_per_second": 54.105,
          "completion_tokens": 1024,
          "round": 3,
          "streams": [
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.289562,
              "decode_tokens_per_second": 59.447,
              "finish_reason": "length",
              "measured_sse_events": 253,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds at most Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\ttokens   float64\n\tlast     time.Time\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive rate or burst.\nvar ErrInvalidConfig = errors.New(\"ratelimit: rate and burst must be positive\")\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity, starting full. A nil clock uses the real time.\nfunc New(rate, burst float64, clock Clock) (*Limiter, error) {\n\tif rate <= 0 || burst <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock",
              "output_characters": 1000,
              "output_sha256": "cd613a485431e0ebb898f89411c4896c9e20bf5c493054f2a9599a7b00ecf568",
              "prompt_tokens": 131,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000106,
              "stream_done_marker": true,
              "stream_sha256": "cd613a485431e0ebb898f89411c4896c9e20bf5c493054f2a9599a7b00ecf568",
              "time_to_first_visible_seconds": 5.216927,
              "time_to_last_output_seconds": 9.506489,
              "ttft_seconds": 5.216927,
              "wall_seconds": 9.506594
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.341938,
              "decode_tokens_per_second": 58.73,
              "finish_reason": "length",
              "measured_sse_events": 254,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n\tclock    Clock\n}\n\nvar (\n\tErrRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrBurst = errors.New(\"ratelimit: burst must be > 0\")\n)\n\n// New returns a Limiter with a full bucket. rate is tokens per second;\n// burst is the maximum bucket size. Both must be positive.\nfunc New(rate float64, burst int) (*Limiter, error) {\n\treturn NewWithClock(rate, burst, realClock{})\n}\n\n// NewWithClock is New with an injectable clock, for",
              "output_characters": 964,
              "output_sha256": "c841bb7341d8567f98a6a9e503d410d3b0ab8bd93ca031a70c4e106b03e37d25",
              "prompt_tokens": 128,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 9e-05,
              "stream_done_marker": true,
              "stream_sha256": "c841bb7341d8567f98a6a9e503d410d3b0ab8bd93ca031a70c4e106b03e37d25",
              "time_to_first_visible_seconds": 0.411514,
              "time_to_last_output_seconds": 4.753453,
              "ttft_seconds": 0.411514,
              "wall_seconds": 4.753543
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.171563,
              "decode_tokens_per_second": 61.128,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\tburst int\n\trate  float64 // tokens per second\n\tclock Clock\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock         = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given refill rate (tokens/sec) and burst\n// capacity. The bucket starts full. If clock is nil, the real",
              "output_characters": 964,
              "output_sha256": "5962230d191a7ca49c70a6df597063dd7a3abd74201fabaef46f5a09f67ce3bd",
              "prompt_tokens": 130,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000148,
              "stream_done_marker": true,
              "stream_sha256": "5962230d191a7ca49c70a6df597063dd7a3abd74201fabaef46f5a09f67ce3bd",
              "time_to_first_visible_seconds": 14.751023,
              "time_to_last_output_seconds": 18.922587,
              "ttft_seconds": 14.751023,
              "wall_seconds": 18.922735
            },
            {
              "cached_prompt_tokens": 0,
              "completion_tokens": 256,
              "decode_seconds": 4.303993,
              "decode_tokens_per_second": 59.247,
              "finish_reason": "length",
              "measured_sse_events": 255,
              "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be created with NewLimiter and is safe for concurrent use.\ntype Limiter struct {\n\tmu       sync.Mutex\n\trate     float64   // tokens per second\n\tburst    float64   // bucket capacity\n\ttokens   float64   // current tokens\n\tlastTime time.Time // last refill time\n\tclock    Clock\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock         = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// NewLimiter returns a Limiter that allows rate events per second with a\n// maximum burst size. The bucket starts full. If clock is nil, the real\n// clock",
              "output_characters": 966,
              "output_sha256": "78c4ae2a8fc77d66f6a417577910d4c748cb5ba89d41332a3d5e625dd1ea638e",
              "prompt_tokens": 134,
              "reasoning_characters": 0,
              "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
              "response_tail_seconds": 0.000217,
              "stream_done_marker": true,
              "stream_sha256": "78c4ae2a8fc77d66f6a417577910d4c748cb5ba89d41332a3d5e625dd1ea638e",
              "time_to_first_visible_seconds": 9.990357,
              "time_to_last_output_seconds": 14.294351,
              "ttft_seconds": 9.990357,
              "wall_seconds": 14.294567
            }
          ],
          "wall_seconds": 18.92632
        }
      ]
    }
  },
  "decode": {
    "code": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 65.147,
        "median": 64.415,
        "minimum": 63.685,
        "p90": 65.147
      },
      "pooled_decode_tokens_per_second": 64.457,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2014,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 31.250604,
          "decode_tokens_per_second": 64.415,
          "finish_reason": "stop",
          "measured_sse_events": 2011,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds capacity tokens, refilling at rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\nvar ErrInvalidParams = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with a full bucket. capacity is the max burst size,\n// rate is the refill rate in tokens per second.\nfunc New(capacity, rate float64) (*Limiter, error) {\n\treturn NewWithClock(capacity, rate, realClock{})\n}\n\n// NewWithClock is New with an injectable clock.\nfunc NewWithClock(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 || rate <= 0 {\n\t\treturn nil, ErrInvalidParams\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   capacity,\n\t\tclock:    clock,\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\nfunc (l *Limiter) AllowN(n float64) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = min(l.capacity, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif l.tokens >= n {\n\t\tl.tokens -= n\n\t\treturn true\n\t}\n\treturn false\n}\n\n// RetryAfter returns the duration until n tokens would be available.\n// It does not consume tokens.\nfunc (l *Limiter) RetryAfter(n float64) time.Duration {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tif n <= 0 || n > l.capacity {\n\t\treturn -1\n\t}\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens = min(l.capacity, l.tokens+elapsed*l.rate)\n\t\tl.last = now\n\t}\n\tif l.tokens >= n {\n\t\treturn 0\n\t}\n\treturn time.Duration((n - l.tokens) / l.rate * float64(time.Second))\n}\n\nfunc min(a, b float64) float64 {\n\tif a < b {\n\t\treturn a\n\t}\n\treturn b\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n}\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 10, 1, false},\n\t\t{\"zero capacity\", 0, 1, true},\n\t\t{\"negative capacity\", -1, 1, true},\n\t\t{\"zero rate\", 10, 0, true},\n\t\t{\"negative rate\", 10, -2, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := NewWithClock(tc.capacity, tc.rate, newFakeClock())\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %v) err = %v, wantErr %v\", tc.capacity, tc.rate, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       float64\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{time.Second, 1, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"capacity caps refill\",\n\t\t\tcapacity: 2, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {time.Hour, 2, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"large request denied, small allowed\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 5, true}, {0, 6, false}, {0, 0.5, false},\n\t\t\t\t{time.Second, 0.5, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\tcapacity: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := NewWithClock(tc.capacity, tc.rate, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"unexpected error: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestRetryAfter(t *testing.T) {\n\tfc := newFakeClock()\n\tl, _ := NewWithClock(2, 2, fc) // 2 tokens/sec\n\n\tif d := l.RetryAfter(2); d != 0 {\n\t\tt.Errorf(\"RetryAfter(2) with full bucket = %v, want 0\", d)\n\t}\n\tl.AllowN(2)\n\tif d := l.RetryAfter(1); d != 500*time.Millisecond {\n\t\tt.Errorf(\"RetryAfter(1) = %v, want 500ms\", d)\n\t}\n\tif d := l.RetryAfter(3); d != -1 {\n\t\tt.Errorf(\"RetryAfter(3) > capacity = %v, want -1\", d)\n\t}\n\t// RetryAfter must not consume tokens.\n\tif !l.AllowN(0.5) {\n\t\tt.Error(\"AllowN(0.5) after RetryAfter should still succeed\")\n\t}\n}\n\nfunc TestConcurrentAccess(t *testing.T) {\n\tl, _ := New(100, 1000)\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.Allow() {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6065,
          "output_sha256": "687ab12314951ff2e239847cf6d7c5e91adabf0532606eddb640e27a3d9a4fdb",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000798,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "687ab12314951ff2e239847cf6d7c5e91adabf0532606eddb640e27a3d9a4fdb",
          "time_to_first_visible_seconds": 0.472,
          "time_to_last_output_seconds": 31.722604,
          "ttft_seconds": 0.472,
          "wall_seconds": 31.723402
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2241,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 35.1729,
          "decode_tokens_per_second": 63.685,
          "finish_reason": "stop",
          "measured_sse_events": 2239,
          "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"math\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be created with New; the zero value is not usable.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\trate     float64 // tokens per second\n\tburst    float64 // bucket capacity\n\ttokens   float64\n\tlastTime time.Time\n}\n\nvar (\n\tErrNonPositiveRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrNonPositiveBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock         = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// NewOption configures New.\ntype NewOption func(*options) error\n\ntype options struct {\n\tclock Clock\n}\n\n// WithClock supplies a custom Clock (e.g., a test fake).\nfunc WithClock(c Clock) NewOption {\n\treturn func(o *options) error {\n\t\tif c == nil {\n\t\t\treturn ErrNilClock\n\t\t}\n\t\to.clock = c\n\t\treturn nil\n\t}\n}\n\n// New returns a Limiter that refills tokens at rate tokens per second,\n// with a maximum capacity of burst tokens. The bucket starts full.\nfunc New(rate, burst float64, opts ...NewOption) (*Limiter, error) {\n\tif rate <= 0 || math.IsNaN(rate) || math.IsInf(rate, 0) {\n\t\treturn nil, ErrNonPositiveRate\n\t}\n\tif burst <= 0 || math.IsNaN(burst) || math.IsInf(burst, 0) {\n\t\treturn nil, ErrNonPositiveBurst\n\t}\n\to := options{clock: realClock{}}\n\tfor _, opt := range opts {\n\t\tif opt == nil {\n\t\t\tcontinue\n\t\t}\n\t\tif err := opt(&o); err != nil {\n\t\t\treturn nil, err\n\t\t}\n\t}\n\treturn &Limiter{\n\t\tclock:    o.clock,\n\t\trate:     rate,\n\t\tburst:    burst,\n\t\ttokens:   burst,\n\t\tlastTime: o.clock.Now(),\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at time now,\n// consuming them if so. It never blocks.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.lastTime).Seconds()\n\tif elapsed < 0 {\n\t\telapsed = 0 // guard against clock going backwards\n\t}\n\tl.tokens = math.Min(l.burst, l.tokens+elapsed*l.rate)\n\tl.lastTime = now\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the number of tokens currently available.\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\telapsed := l.clock.Now().Sub(l.lastTime).Seconds()\n\tif elapsed < 0 {\n\t\telapsed = 0\n\t}\n\treturn math.Min(l.burst, l.tokens+elapsed*l.rate)\n}\n```\n\n**ratelimit/ratelimit_test.go**\n```go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n}\n\nfunc newFakeClock() *fakeClock { return &fakeClock{now: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   float64\n\t\topts    []NewOption\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 10, nil, ErrNonPositiveRate},\n\t\t{\"negative rate\", -1, 10, nil, ErrNonPositiveRate},\n\t\t{\"NaN rate\", float64(\"x\")[0] * 0, 10, nil, ErrNonPositiveRate},\n\t\t{\"zero burst\", 1, 0, nil, ErrNonPositiveBurst},\n\t\t{\"negative burst\", 1, -5, nil, ErrNonPositiveBurst},\n\t\t{\"nil clock\", 1, 1, []NewOption{WithClock(nil)}, ErrNilClock},\n\t\t{\"valid\", 10, 5, nil, nil},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, tc.opts...)\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"New() error = %v, want nil\", err)\n\t\t\t\t}\n\t\t\t\tif l == nil {\n\t\t\t\t\tt.Fatal(\"New() = nil, want non-nil\")\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif err != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New() error = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname      string\n\t\trate      float64\n\t\tburst     float64\n\t\tsteps     []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst exhausted then refills\",\n\t\t\trate:  1, // 1 token per second\n\t\t\tburst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},  // consume full burst\n\t\t\t\t{0, 1, false}, // empty\n\t\t\t\t{time.Second, 1, true},\n\t\t\t\t{0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 tokens\n\t\t\t\t{500 * time.Millisecond, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"tokens cap at burst\",\n\t\t\trate:  10,\n\t\t\tburst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{time.Minute, 1, true}, // refilled, capped at 1\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"zero tokens request\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 0, true},\n\t\t\t\t{0, 0, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"negative request denied\",\n\t\t\trate:  1,\n\t\t\tburst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"large request exceeds burst\",\n\t\t\trate:  1,\n\t\t\tburst: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{time.Hour, 4, false}, // even full bucket can't serve 4\n\t\t\t\t{0, 3, true},\n\t\t\t},\n\t\t},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock()\n\t\t\tl, err := New(tc.rate, tc.burst, WithClock(fc))\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tfc := newFakeClock()\n\tl, err := New(2, 1, WithClock(fc))\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Error(\"first Allow() = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Error(\"second Allow() = true, want false\")\n\t}\n\tfc.Advance(500 * time.Millisecond)\n\tif !l.Allow() {\n\t\tt.Error(\"Allow() after refill = false, want true\")\n\t}\n}\n\nfunc TestConcurrent(t *testing.T) {\n\tfc := newFakeClock()\n\tl, err := New(1000, 100, WithClock(fc))\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tfor i := 0; i < 100; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\twg.Add(1)\n\t\t\t\tgo func() { defer wg.Done(); allowed++ }()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed = %d, want <= burst 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6855,
          "output_sha256": "cf102a970e087c8addafe2f14c9575a807761f5fd9ed4f42e51b1b30f3cb568b",
          "prompt_tokens": 131,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001124,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "cf102a970e087c8addafe2f14c9575a807761f5fd9ed4f42e51b1b30f3cb568b",
          "time_to_first_visible_seconds": 0.469278,
          "time_to_last_output_seconds": 35.642179,
          "ttft_seconds": 0.469278,
          "wall_seconds": 35.643303
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1952,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 30.32948,
          "decode_tokens_per_second": 64.327,
          "finish_reason": "stop",
          "measured_sse_events": 1950,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by the system time.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds up to Burst tokens and refills at Rate tokens per second.\ntype Limiter struct {\n\tmu     sync.Mutex\n\ttokens float64\n\tlast   time.Time\n\n\trate  float64 // tokens per second, > 0\n\tburst int     // max tokens, > 0\n\tclock Clock\n}\n\nvar (\n\tErrInvalidRate  = errors.New(\"ratelimit: rate must be > 0\")\n\tErrInvalidBurst = errors.New(\"ratelimit: burst must be > 0\")\n\tErrNilClock     = errors.New(\"ratelimit: clock must not be nil\")\n)\n\n// New creates a Limiter with the given refill rate (tokens/sec), burst\n// capacity, and clock. If clock is nil, the real clock is used.\nfunc New(rate float64, burst int, clock Clock) (*Limiter, error) {\n\tif rate <= 0 {\n\t\treturn nil, ErrInvalidRate\n\t}\n\tif burst <= 0 {\n\t\treturn nil, ErrInvalidBurst\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\tnow := clock.Now()\n\treturn &Limiter{\n\t\ttokens: float64(burst),\n\t\tlast:   now,\n\t\trate:   rate,\n\t\tburst:  burst,\n\t\tclock:  clock,\n\t}, nil\n}\n\n// Allow reports whether a single token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available at the current time,\n// consuming them if so. It is safe for concurrent use.\nfunc (l *Limiter) AllowN(n int) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > float64(l.burst) {\n\t\t\tl.tokens = float64(l.burst)\n\t\t}\n\t\tl.last = now\n\t}\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Tokens returns the current number of available tokens (for tests/monitoring).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\treturn l.tokens\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tmu   sync.Mutex\n\tnow  time.Time\n\tstep time.Duration\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname    string\n\t\trate    float64\n\t\tburst   int\n\t\tclock   Clock\n\t\twantErr error\n\t}{\n\t\t{\"zero rate\", 0, 1, RealClock(), ErrInvalidRate},\n\t\t{\"negative rate\", -1.5, 1, RealClock(), ErrInvalidRate},\n\t\t{\"zero burst\", 1, 0, RealClock(), ErrInvalidBurst},\n\t\t{\"negative burst\", 1, -3, RealClock(), ErrInvalidBurst},\n\t\t{\"nil clock ok\", 1, 1, nil, nil},\n\t\t{\"valid\", 2.5, 10, RealClock(), nil},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.rate, tc.burst, tc.clock)\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"unexpected error: %v\", err)\n\t\t\t\t}\n\t\t\t\tif l == nil {\n\t\t\t\t\tt.Fatal(\"expected non-nil limiter\")\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif err != tc.wantErr {\n\t\t\t\tt.Fatalf(\"got error %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tstart := time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\n\ttests := []struct {\n\t\tname  string\n\t\trate  float64\n\t\tburst int\n\t\t// sequence of (advance duration, n, want)\n\t\tsteps []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:  \"burst within capacity\",\n\t\t\trate:  1, burst: 3,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"refill over time\",\n\t\t\trate:  2, burst: 2,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, true}, // 1 token refilled\n\t\t\t\t{0, 1, false},\n\t\t\t\t{1 * time.Second, 2, true},        // capped at burst\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"fractional refill\",\n\t\t\trate:  1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true},\n\t\t\t\t{250 * time.Millisecond, 1, false}, // only 0.25 tokens\n\t\t\t\t{750 * time.Millisecond, 1, true},  // now 1.0\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:  \"negative n rejected\",\n\t\t\trate:  1, burst: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false},\n\t\t\t\t{0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(start)\n\t\t\tl, err := New(tc.rate, tc.burst, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%d) = %v, want %v (tokens=%f)\",\n\t\t\t\t\t\ti, s.n, got, s.want, l.Tokens())\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tfc := newFakeClock(time.Time{})\n\tl, err := New(10, 1, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tif !l.Allow() {\n\t\tt.Error(\"first Allow should succeed\")\n\t}\n\tif l.Allow() {\n\t\tt.Error(\"second Allow without refill should fail\")\n\t}\n\tfc.Advance(100 * time.Millisecond)\n\tif !l.Allow() {\n\t\tt.Error(\"Allow after refill should succeed\")\n\t}\n}\n\nfunc TestConcurrentUse(t *testing.T) {\n\tfc := newFakeClock(time.Time{})\n\tl, err := New(1000, 100, fc)\n\tif err != nil {\n\t\tt.Fatal(err)\n\t}\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 50; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tfor j := 0; j < 10; j++ {\n\t\t\t\tif l.AllowN(1) {\n\t\t\t\t\tmu.Lock()\n\t\t\t\t\tallowed++\n\t\t\t\t\tmu.Unlock()\n\t\t\t\t}\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed > 100 {\n\t\tt.Errorf(\"allowed %d tokens, exceeds burst 100\", allowed)\n\t}\n}\n```",
          "output_characters": 5971,
          "output_sha256": "95033681009e6adeb25da6a514dce1c8396baaddeef1342d9dbfdef313becd70",
          "prompt_tokens": 133,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001226,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "95033681009e6adeb25da6a514dce1c8396baaddeef1342d9dbfdef313becd70",
          "time_to_first_visible_seconds": 0.47975,
          "time_to_last_output_seconds": 30.809231,
          "ttft_seconds": 0.47975,
          "wall_seconds": 30.810457
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2043,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 31.508273,
          "decode_tokens_per_second": 64.808,
          "finish_reason": "stop",
          "measured_sse_events": 2041,
          "output": "**ratelimit/ratelimit.go**\n```go\n// Package ratelimit implements a concurrency-safe token-bucket rate limiter.\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"fmt\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// A Limiter must be created with New; the zero value is not usable.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tclock    Clock\n\tcapacity int      // max tokens\n\ttokens   float64  // current tokens\n\trate     float64  // tokens per second\n\tlast     time.Time\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New returns a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. A nil clock uses the real time.\nfunc New(capacity int, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 {\n\t\treturn nil, fmt.Errorf(\"%w: capacity=%d\", ErrInvalidConfig, capacity)\n\t}\n\tif rate <= 0 {\n\t\treturn nil, fmt.Errorf(\"%w: rate=%v\", ErrInvalidConfig, rate)\n\t}\n\tif clock == nil {\n\t\tclock = realClock{}\n\t}\n\treturn &Limiter{\n\t\tclock:    clock,\n\t\tcapacity: capacity,\n\t\ttokens:   float64(capacity),\n\t\trate:     rate,\n\t\tlast:     clock.Now(),\n\t}, nil\n}\n\n// AllowN reports whether n tokens may be consumed at time now.\n// If allowed, the tokens are consumed.\nfunc (l *Limiter) AllowN(now time.Time, n int) bool {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tif n < 0 || n > l.capacity {\n\t\treturn false\n\t}\n\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed < 0 {\n\t\telapsed = 0 // guard against clock going backwards\n\t}\n\tl.tokens += elapsed * l.rate\n\tif l.tokens > float64(l.capacity) {\n\t\tl.tokens = float64(l.capacity)\n\t}\n\tl.last = now\n\n\tif float64(n) <= l.tokens {\n\t\tl.tokens -= float64(n)\n\t\treturn true\n\t}\n\treturn false\n}\n\n// Allow reports whether one token may be consumed now.\nfunc (l *Limiter) Allow() bool {\n\treturn l.AllowN(l.clock.Now(), 1)\n}\n\n// Tokens returns the current number of available tokens (refilled lazily).\nfunc (l *Limiter) Tokens() float64 {\n\tnow := l.clock.Now()\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed < 0 {\n\t\telapsed = 0\n\t}\n\tt := l.tokens + elapsed*l.rate\n\tif t > float64(l.capacity) {\n\t\tt = float64(l.capacity)\n\t}\n\treturn t\n}\n```\n\n**ratelimit/ratelimit_test.go**\n```go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"testing\"\n\t\"time\"\n)\n\ntype fakeClock struct {\n\tt time.Time\n}\n\nfunc newFakeClock() *fakeClock {\n\treturn &fakeClock{t: time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)}\n}\n\nfunc (f *fakeClock) Now() time.Time { return f.t }\n\nfunc (f *fakeClock) Advance(d time.Duration) { f.t = f.t.Add(d) }\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\twantErr  error\n\t}{\n\t\t{\"ok\", 10, 5, nil},\n\t\t{\"zero capacity\", 0, 5, ErrInvalidConfig},\n\t\t{\"negative capacity\", -1, 5, ErrInvalidConfig},\n\t\t{\"zero rate\", 10, 0, ErrInvalidConfig},\n\t\t{\"negative rate\", 10, -0.5, ErrInvalidConfig},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tl, err := New(tc.capacity, tc.rate, nil)\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"New() error = %v, want nil\", err)\n\t\t\t\t}\n\t\t\t\tif l == nil {\n\t\t\t\t\tt.Fatal(\"New() = nil limiter, want non-nil\")\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"New() error = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\tbase := newFakeClock()\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity int\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       int\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname:     \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{2 * time.Second, 2, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"partial refill insufficient\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {1 * time.Second, 2, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"capacity caps refill\",\n\t\t\tcapacity: 2, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {10 * time.Second, 3, false}, {0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"request larger than capacity denied\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{100 * time.Second, 3, false}, {0, 2, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname:     \"negative request denied\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       int\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tclk := newFakeClock()\n\t\t\tl, err := New(tc.capacity, tc.rate, clk)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New() error = %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tclk.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(clk.Now(), s.n); got != s.want {\n\t\t\t\t\tt.Errorf(\"step %d: AllowN(%v) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowUsesInjectedClock(t *testing.T) {\n\tclk := newFakeClock()\n\tl, err := New(1, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow() = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow() = true, want false\")\n\t}\n\tclk.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow() after refill = false, want true\")\n\t}\n}\n\nfunc TestTokens(t *testing.T) {\n\tclk := newFakeClock()\n\tl, err := New(2, 1, clk)\n\tif err != nil {\n\t\tt.Fatalf(\"New() error = %v\", err)\n\t}\n\tl.AllowN(clk.Now(), 2)\n\tif got := l.Tokens(); got != 0 {\n\t\tt.Errorf(\"Tokens() = %v, want 0\", got)\n\t}\n\tclk.Advance(500 * time.Millisecond)\n\tif got := l.Tokens(); got != 0.5 {\n\t\tt.Errorf(\"Tokens() = %v, want 0.5\", got)\n\t}\n}\n```",
          "output_characters": 6276,
          "output_sha256": "e800454c97c8ceab0e18808231001c136c41e6d6ff5f24ce006a153c3bd21660",
          "prompt_tokens": 134,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000162,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "e800454c97c8ceab0e18808231001c136c41e6d6ff5f24ce006a153c3bd21660",
          "time_to_first_visible_seconds": 0.486653,
          "time_to_last_output_seconds": 31.994926,
          "ttft_seconds": 0.486653,
          "wall_seconds": 31.995088
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 2021,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 31.006633,
          "decode_tokens_per_second": 65.147,
          "finish_reason": "stop",
          "measured_sse_events": 2019,
          "output": "```go\n// ratelimit.go\npackage ratelimit\n\nimport (\n\t\"errors\"\n\t\"sync\"\n\t\"time\"\n)\n\n// Clock abstracts time for testability.\ntype Clock interface {\n\tNow() time.Time\n}\n\ntype realClock struct{}\n\nfunc (realClock) Now() time.Time { return time.Now() }\n\n// RealClock returns a Clock backed by time.Now.\nfunc RealClock() Clock { return realClock{} }\n\n// Limiter is a concurrency-safe token-bucket rate limiter.\n// The bucket holds capacity tokens, refilling at rate tokens per second.\ntype Limiter struct {\n\tmu       sync.Mutex\n\tcapacity float64\n\trate     float64 // tokens per second\n\ttokens   float64\n\tlast     time.Time\n\tclock    Clock\n}\n\n// ErrInvalidConfig is returned by New for non-positive capacity or rate.\nvar ErrInvalidConfig = errors.New(\"ratelimit: capacity and rate must be > 0\")\n\n// New creates a Limiter with the given burst capacity and refill rate\n// (tokens per second), starting full. clock may be nil for the real clock.\nfunc New(capacity, rate float64, clock Clock) (*Limiter, error) {\n\tif capacity <= 0 || rate <= 0 {\n\t\treturn nil, ErrInvalidConfig\n\t}\n\tif clock == nil {\n\t\tclock = RealClock()\n\t}\n\treturn &Limiter{\n\t\tcapacity: capacity,\n\t\trate:     rate,\n\t\ttokens:   capacity,\n\t\tlast:     clock.Now(),\n\t\tclock:    clock,\n\t}, nil\n}\n\n// Allow reports whether one token is available.\nfunc (l *Limiter) Allow() bool { return l.AllowN(1) }\n\n// AllowN reports whether n tokens are available, consuming them if so.\n// It is safe for concurrent use.\nfunc (l *Limiter) AllowN(n float64) bool {\n\tif n < 0 {\n\t\treturn false\n\t}\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\n\tnow := l.clock.Now()\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > l.capacity {\n\t\t\tl.tokens = l.capacity\n\t\t}\n\t\tl.last = now\n\t}\n\tif n > l.tokens {\n\t\treturn false\n\t}\n\tl.tokens -= n\n\treturn true\n}\n\n// Tokens returns the current number of available tokens (advanced to now).\nfunc (l *Limiter) Tokens() float64 {\n\tl.mu.Lock()\n\tdefer l.mu.Unlock()\n\tl.advance(l.clock.Now())\n\treturn l.tokens\n}\n\n// advance refills tokens up to now; caller must hold l.mu.\nfunc (l *Limiter) advance(now time.Time) {\n\telapsed := now.Sub(l.last).Seconds()\n\tif elapsed > 0 {\n\t\tl.tokens += elapsed * l.rate\n\t\tif l.tokens > l.capacity {\n\t\t\tl.tokens = l.capacity\n\t\t}\n\t\tl.last = now\n\t}\n}\n```\n\n```go\n// ratelimit_test.go\npackage ratelimit\n\nimport (\n\t\"sync\"\n\t\"testing\"\n\t\"time\"\n)\n\n// fakeClock is a manually advanced clock for deterministic tests.\ntype fakeClock struct {\n\tmu  sync.Mutex\n\tnow time.Time\n}\n\nfunc newFakeClock(start time.Time) *fakeClock { return &fakeClock{now: start} }\n\nfunc (f *fakeClock) Now() time.Time {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\treturn f.now\n}\n\nfunc (f *fakeClock) Advance(d time.Duration) {\n\tf.mu.Lock()\n\tdefer f.mu.Unlock()\n\tf.now = f.now.Add(d)\n}\n\nvar base = time.Date(2024, 1, 1, 0, 0, 0, 0, time.UTC)\n\nfunc TestNewValidation(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\twantErr  bool\n\t}{\n\t\t{\"valid\", 10, 5, false},\n\t\t{\"zero capacity\", 0, 5, true},\n\t\t{\"negative capacity\", -1, 5, true},\n\t\t{\"zero rate\", 10, 0, true},\n\t\t{\"negative rate\", 10, -2, true},\n\t}\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\t_, err := New(tc.capacity, tc.rate, nil)\n\t\t\tif (err != nil) != tc.wantErr {\n\t\t\t\tt.Fatalf(\"New(%v, %v) err = %v, wantErr %v\", tc.capacity, tc.rate, err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllowN(t *testing.T) {\n\ttests := []struct {\n\t\tname     string\n\t\tcapacity float64\n\t\trate     float64\n\t\tsteps    []struct {\n\t\t\tadvance time.Duration\n\t\t\tn       float64\n\t\t\twant    bool\n\t\t}\n\t}{\n\t\t{\n\t\t\tname: \"burst up to capacity then deny\",\n\t\t\tcapacity: 3, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 1, true}, {0, 1, true}, {0, 1, true}, {0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"refill over time\",\n\t\t\tcapacity: 2, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true}, {0, 1, false},\n\t\t\t\t{500 * time.Millisecond, 1, false}, // only 0.5 tokens\n\t\t\t\t{500 * time.Millisecond, 1, true},  // 1.0 token now\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"capacity caps refill\",\n\t\t\tcapacity: 2, rate: 10,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 2, true},\n\t\t\t\t{time.Hour, 2, true}, // capped at capacity, not 36000\n\t\t\t\t{0, 1, false},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"large request denied atomically\",\n\t\t\tcapacity: 5, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, 3, true}, {0, 3, false}, {2 * time.Second, 3, true},\n\t\t\t},\n\t\t},\n\t\t{\n\t\t\tname: \"negative request denied\",\n\t\t\tcapacity: 1, rate: 1,\n\t\t\tsteps: []struct {\n\t\t\t\tadvance time.Duration\n\t\t\t\tn       float64\n\t\t\t\twant    bool\n\t\t\t}{\n\t\t\t\t{0, -1, false}, {0, 1, true},\n\t\t\t},\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tfc := newFakeClock(base)\n\t\t\tl, err := New(tc.capacity, tc.rate, fc)\n\t\t\tif err != nil {\n\t\t\t\tt.Fatalf(\"New: %v\", err)\n\t\t\t}\n\t\t\tfor i, s := range tc.steps {\n\t\t\t\tfc.Advance(s.advance)\n\t\t\t\tif got := l.AllowN(s.n); got != s.want {\n\t\t\t\t\tt.Fatalf(\"step %d: AllowN(%v) = %v, want %v\", i, s.n, got, s.want)\n\t\t\t\t}\n\t\t\t}\n\t\t})\n\t}\n}\n\nfunc TestAllow(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, _ := New(1, 1, fc)\n\tif !l.Allow() {\n\t\tt.Fatal(\"first Allow = false, want true\")\n\t}\n\tif l.Allow() {\n\t\tt.Fatal(\"second Allow = true, want false\")\n\t}\n\tfc.Advance(time.Second)\n\tif !l.Allow() {\n\t\tt.Fatal(\"Allow after refill = false, want true\")\n\t}\n}\n\nfunc TestTokens(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, _ := New(2, 1, fc)\n\tif got := l.Tokens(); got != 2 {\n\t\tt.Fatalf(\"Tokens = %v, want 2\", got)\n\t}\n\tl.AllowN(2)\n\tfc.Advance(500 * time.Millisecond)\n\tif got := l.Tokens(); got != 0.5 {\n\t\tt.Fatalf(\"Tokens = %v, want 0.5\", got)\n\t}\n}\n\nfunc TestConcurrentUse(t *testing.T) {\n\tfc := newFakeClock(base)\n\tl, _ := New(100, 0, fc) // no refill: exactly 100 tokens total\n\tvar wg sync.WaitGroup\n\tvar allowed int64\n\tvar mu sync.Mutex\n\tfor i := 0; i < 200; i++ {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tif l.AllowN(1) {\n\t\t\t\tmu.Lock()\n\t\t\t\tallowed++\n\t\t\t\tmu.Unlock()\n\t\t\t}\n\t\t}()\n\t}\n\twg.Wait()\n\tif allowed != 100 {\n\t\tt.Fatalf(\"allowed = %d, want 100\", allowed)\n\t}\n}\n```",
          "output_characters": 6223,
          "output_sha256": "3e9b1921e6298ec82d5c1038a1cc409db6f0a9e8e803e6bc8ae7ef05a7dde512",
          "prompt_tokens": 134,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.001102,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "3e9b1921e6298ec82d5c1038a1cc409db6f0a9e8e803e6bc8ae7ef05a7dde512",
          "time_to_first_visible_seconds": 0.484097,
          "time_to_last_output_seconds": 31.49073,
          "ttft_seconds": 0.484097,
          "wall_seconds": 31.491832
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 35.642,
        "median": 31.723,
        "minimum": 30.809,
        "p90": 35.642
      },
      "ttft_seconds": {
        "maximum": 0.487,
        "median": 0.48,
        "minimum": 0.469,
        "p90": 0.487
      },
      "wall_seconds": {
        "maximum": 35.643,
        "median": 31.723,
        "minimum": 30.81,
        "p90": 35.643
      }
    },
    "prose": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 44.987,
        "median": 44.771,
        "minimum": 44.035,
        "p90": 44.987
      },
      "pooled_decode_tokens_per_second": 44.555,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1057,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 23.980739,
          "decode_tokens_per_second": 44.035,
          "finish_reason": "stop",
          "measured_sse_events": 1055,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation attached**\n\nOver the past month we've been asked, informally but repeatedly, whether we should purchase additional local GPU hardware or invest in raising the utilisation of the four A100-class nodes we already own. The question sounds like a simple procurement decision, but the more we've dug into it, the more it has turned into a question about what kind of engineering organisation we want to be: one that solves capacity problems with capital, or one that solves them with attention. I want to lay out our thinking honestly, including the places where we're genuinely unsure, and end with a recommendation.\n\nLet's start with the uncomfortable number. Our fleet currently averages 31% utilisation, measured across a full week including nights and weekends. During business hours we peak around 60%, and twice last quarter we queued jobs for more than forty minutes. That spread \u2014 idle most of the time, occasionally saturated \u2014 is the classic signature of a fleet that is sized for its worst hour rather than its average one. Buying more hardware would fix the peak. It would do nothing for the trough, and the trough is where most of our money currently sits depreciating quietly on a rack.\n\nThe cost picture is straightforward enough. A new node at current pricing runs roughly $180k all-in, plus perhaps $8k a year in power, cooling, and maintenance contracts. Against that, we estimate six to eight engineer-weeks to implement the changes we believe would lift utilisation meaningfully: a proper job queue with preemption, batch-size tuning on our two highest-volume services, and consolidating the three small models that each currently reserve a full GPU but use less than a quarter of its memory. Even at fully loaded consulting-grade cost assumptions, the engineering work is an order of magnitude cheaper than the hardware. The payback period on a purchase, given our actual demand curve, is somewhere north of four years \u2014 longer than the useful competitive life of the cards themselves.\n\nLatency is the argument for buying, and it deserves to be taken seriously. Right now, our interactive workloads share the fleet with batch jobs, and when the two collide, p95 latency degrades badly. The clean fix is dedicated hardware: give the interactive services their own silicon and stop worrying. But that clean fix is exactly what a well-built scheduler gives you more cheaply. Preemption with checkpointing, or even crude partitioning of the fleet by workload class, would deliver most of the latency isolation without the capital. We should be honest that we haven't proven this yet \u2014 our confidence that software can deliver acceptable latency isolation is maybe 75%, not 95% \u2014 but the downside if it fails is that we buy the hardware a quarter later, having spent a modest amount of engineering time learning something durable about our own workloads.\n\nReliability cuts both ways. More nodes means more redundancy, and our current fleet has a single point of failure in the sense that two simultaneous node failures would force us to drop batch work entirely. That is a real risk. But it's also a risk we can mitigate for a few thousand dollars by improving our failover handling and making our batch jobs resumable, rather than for $180k. Meanwhile, more hardware means more failure surface, more driver-version drift, and more of the low-grade operational toil that a five-person team feels acutely.\n\nThe opportunity cost may be the strongest argument of all. Eight engineer-weeks spent on scheduling and consolidation is eight weeks not spent on the model-quality work the product team has been asking for. That's a genuine cost, and I don't want to wave it away. But hardware spend is not free of opportunity cost either \u2014 it's $180k that doesn't go toward headcount, tooling, or the reserve we'll want when the next platform migration arrives. The difference is that the engineering investment is reversible in its *lessons* even where it's sunk in its hours: the scheduler we build, the profiling we do, the batching improvements we land all retain value regardless of what hardware we own later. A depreciating GPU purchase is the least reversible option on the table.\n\nSo the recommendation: do not buy hardware this quarter. Fund a focused six-week effort to raise utilisation, with an explicit success threshold \u2014 p95 interactive latency under 800ms and fleet utilisation above 55% \u2014 and a pre-agreed trigger: if we miss it, we proceed with the purchase in the following quarter, no re-litigation required. That structure lets us buy the option to be wrong cheaply.\n\nThe honest caveat is that demand forecasts are where all of this could break. If the new product launch lands the way the PMs hope, our utilisation curve shifts under our feet and the calculus changes within a quarter. We should re-run this analysis in ninety days, and sooner if launch traffic arrives early.\n\nHappy to walk through the numbers in detail at Thursday's sync.",
          "output_characters": 5178,
          "output_sha256": "911afe98a23c3b9b85fbb0e52e6f648e1e1506910919d62f2390eac673210881",
          "prompt_tokens": 123,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000161,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "564dded25e2aba8e2c2adfbe5b20c8d5f89c9a51a373d40b25e457a321c512f4",
          "time_to_first_visible_seconds": 1.065644,
          "time_to_last_output_seconds": 24.474808,
          "ttft_seconds": 0.494069,
          "wall_seconds": 24.474969
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 932,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 20.794691,
          "decode_tokens_per_second": 44.771,
          "finish_reason": "stop",
          "measured_sse_events": 930,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering Leadership\n**From:** Infrastructure Team\n**Re:** AI hardware capacity planning for the next two quarters\n\n**Summary and recommendation**\n\nWe recommend that we invest the next quarter in improving utilisation of our existing GPU fleet before committing to any new hardware purchases. We believe the case for buying more capacity is real but not yet urgent, and that a focused effort on scheduling, batching, and workload consolidation will either defer the purchase by six to twelve months or, at minimum, tell us precisely what we actually need to buy and when. We also recommend setting a concrete trigger point now \u2014 sustained utilisation above roughly 75 percent with queue times breaching our latency SLOs \u2014 so that the decision to buy becomes mechanical rather than political.\n\n**The current situation**\n\nOur fleet of eight A100-class nodes currently runs at an average utilisation of around 40 percent, but that average hides a great deal. Peak hours push several nodes to saturation while nights and weekends sit nearly idle. Interactive inference traffic and batch training jobs compete for the same resources, and our current scheduler treats them identically, which means we provision for the peak of both simultaneously. In practice, we are paying for capacity we own but cannot use when we need it, and leaving capacity idle when we don't.\n\n**Cost**\n\nThe purchase option is straightforward to price: roughly $180K for two additional nodes, plus perhaps $15K annually in incremental power, cooling, and rack space, plus the engineering time to integrate and monitor them. The utilisation option costs less in capital but is not free. It will consume perhaps six engineer-weeks of work on scheduling, dynamic batching, and separating inference from training workloads. That is real opportunity cost \u2014 those engineers would otherwise be shipping product features. But it is reversible, incremental, and it builds capability we will want regardless of what we eventually buy. Hardware, by contrast, depreciates quickly, locks us into a power and maintenance commitment, and is difficult to resell at anything close to book value.\n\n**Latency and reliability**\n\nLatency is where the utilisation argument gets complicated. Consolidating workloads and raising average utilisation necessarily reduces headroom, and headroom is what absorbs traffic spikes and node failures. If we push utilisation too aggressively without proper workload isolation, p99 inference latency will degrade, and we will have traded a capital problem for an on-call problem. This is a genuine risk and the main reason we are not recommending an aggressive target. Our proposal is to separate interactive inference onto a reserved pool with strict latency guarantees, and to pack batch training into the remaining capacity with preemption. That gets us most of the efficiency gains while protecting the user-facing path. Reliability also favours a modest amount of slack: when a node fails today, we have somewhere to put its work. If we run hot, every failure becomes an incident with user-visible impact.\n\n**Opportunity cost and reversibility**\n\nThere is a counterargument we take seriously: every week engineers spend on scheduler work is a week not spent on the product, and if demand doubles in the next quarter anyway, we will end up buying hardware regardless, having spent effort we could have skipped. That scenario is plausible. But even in it, the utilisation work is not wasted \u2014 better scheduling means we buy two nodes instead of four, and the batching infrastructure improves inference cost per request permanently. Buying hardware first, on the other hand, is largely irreversible in the short term. Once capital is committed and racks are filled, the pressure to justify them distorts future decisions.\n\n**Uncertainty**\n\nWe want to be honest about what we don't know. Our demand forecast has a wide error band; the product team's roadmap could shift inference traffic substantially in either direction. Our utilisation numbers come from sampling that may undercount idle-but-allocated memory, which is a distinct bottleneck from compute. And vendor pricing may move; if accelerator prices drop meaningfully, the buy case strengthens. None of these uncertainties change our recommendation for this quarter, but they do argue for keeping the purchase decision live rather than closing it.\n\n**Next steps**\n\nWe propose a six-week utilisation push with a checkpoint review, clear SLO guardrails, and a pre-agreed purchase trigger. If we hit the trigger, we buy quickly and without relitigating the analysis. If we don't, we will have bought ourselves time, information, and a cheaper eventual expansion.",
          "output_characters": 4776,
          "output_sha256": "0f49ae44005189bad89b99134e7d0147849366cca19bea1af3190371cabcef78",
          "prompt_tokens": 121,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000744,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "a50957c427674a3fdb0c0196ffa037fbd536e3e957d30f238d2689438d98113e",
          "time_to_first_visible_seconds": 1.064669,
          "time_to_last_output_seconds": 21.291156,
          "ttft_seconds": 0.496465,
          "wall_seconds": 21.2919
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 985,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.309544,
          "decode_tokens_per_second": 44.107,
          "finish_reason": "stop",
          "measured_sse_events": 982,
          "output": "**Internal Memo: GPU Capacity \u2014 Buy More or Use What We Have Better**\n\n**To:** Engineering Leadership\n**From:** Infrastructure Team\n**Date:** 2025\n**Re:** Capacity strategy for our local inference fleet\n\n**Summary and Recommendation**\n\nWe recommend that we invest the next quarter in improving utilisation of our existing hardware before purchasing additional machines. The case for buying is real, but it is weaker than it first appears, and the option to buy later does not expire. The option to learn how to run what we own more efficiently does not compound unless we start now.\n\n**The Situation**\n\nOur four A100 nodes currently run at roughly 35% average utilisation, but that average conceals a painful pattern: utilisation spikes past 90% during working hours and collapses overnight and on weekends. Several teams have reported queue times of twenty minutes or more for batch jobs, and at least two have quietly moved workloads to a cloud provider, which is both a cost signal and a vote of no confidence. The instinctive response \u2014 buy two more nodes \u2014 would cost approximately \u00a3180,000 in capital plus ongoing power, cooling, and maintenance. That is real money for a team our size, and it deserves scrutiny rather than reflex.\n\n**Cost**\n\nThe purchase price is only the beginning. Power draw at our scale is nontrivial, our rack space is nearly exhausted, and we would need to budget for the operational burden of managing a larger fleet with the same number of people. By contrast, the utilisation work \u2014 a proper scheduling layer, time-slicing for small jobs, consolidating batch work into off-peak windows, and a shared queue with honest priorities \u2014 costs mostly engineering time we would spend anyway. If we can lift average utilisation from 35% to 60%, we have effectively added capacity equal to most of one node for a fraction of the capital cost. Even if we fall short of that, we will know precisely how much hardware we actually need, which is information we currently lack.\n\n**Latency**\n\nLatency is where the buy-more argument is strongest. Interactive workloads \u2014 the notebook sessions, the evaluation runs people run while debugging \u2014 genuinely suffer when the queue is long, and no amount of clever scheduling makes a busy GPU instant. If our researchers lose an hour a day waiting, that is a real cost that does not show up on any invoice. However, much of our current latency pain comes from jobs that could run at any time occupying the same pool as latency-sensitive ones. Separating interactive from batch traffic is exactly the kind of improvement that costs little and could remove most of the pain without new metal. We should measure this properly before concluding that hardware is the bottleneck rather than policy.\n\n**Reliability**\n\nBuying more nodes does improve reliability in one narrow sense: more redundancy means a single failure hurts less. But our reliability problems to date have been mostly software \u2014 a flaky scheduler, jobs that leak memory and require node reboots, no automated health checks. Adding hardware to a system with these problems gives us more hardware to reboot. Hardening what we have improves reliability for every workload, including the ones we would buy capacity for anyway.\n\n**Opportunity Cost and Reversibility**\n\nThis is the crux. The \u00a3180,000 spent on hardware is largely irreversible: GPUs depreciate quickly, and if our workload mix changes \u2014 say, we shift toward smaller models or a hosted provider improves \u2014 we cannot easily recover that capital. Engineering time spent on scheduling and observability, by contrast, produces durable assets: better tooling, better metrics, better habits. It is also highly reversible in the sense that if we do the work and still find ourselves short, nothing is wasted; we simply buy the hardware then, better informed and possibly with a clearer sense of what specification we need. The reverse is not true. If we buy first and the utilisation problem persists \u2014 as it likely would, since our current pattern suggests policy failure more than capacity failure \u2014 we will have spent the money and still face the same queues.\n\n**Uncertainty**\n\nI want to be honest about what we do not know. Our utilisation figures come from partial instrumentation and may understate true demand. If the team grows or a major new project lands next quarter, the calculus shifts meaningfully toward buying. There is also a risk that utilisation work stalls \u2014 scheduling projects have a way of expanding \u2014 in which case we should set a hard checkpoint: if we have not materially reduced queue times in eight weeks, we revisit the purchase with data in hand.\n\n**Next Steps**\n\nWe propose a two-month utilisation push with explicit success metrics, a parallel audit of true demand, and a pre-agreed decision point on hardware purchase at the end of it. Comments welcome before Friday.",
          "output_characters": 4873,
          "output_sha256": "cc2195bf8613b5d20c1928b8ad4ebcce0bc82d8adfecedbff07fb817e44da7b5",
          "prompt_tokens": 126,
          "reasoning_characters": 62,
          "reasoning_sha256": "7eddaec104351da3aba6059600be80b14d17920bd2df8ba3f1ac3d20ffb662d1",
          "response_tail_seconds": 0.000256,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "a15e86eee1d994b7761a6cb73b55003479a148c90c77e182a43d024a2894dc94",
          "time_to_first_visible_seconds": 1.122627,
          "time_to_last_output_seconds": 22.811134,
          "ttft_seconds": 0.50159,
          "wall_seconds": 22.81139
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 1005,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 22.317734,
          "decode_tokens_per_second": 44.987,
          "finish_reason": "stop",
          "measured_sse_events": 1003,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Inference capacity planning for the next two quarters\n\n**Status: Recommendation to defer hardware purchase and invest in utilisation work first**\n\n---\n\nWe've been asked to decide whether to spend roughly $180K on two additional GPU nodes or to invest the next quarter in raising utilisation on the hardware we already own. I want to lay out how I'm thinking about this, because the right answer is less obvious than either the \"we're out of capacity\" or \"we're wasting money\" camps suggest.\n\nStart with what the data actually shows. Our fleet averages 41% utilisation by GPU-hours, but that number hides a lot. Peak-hour utilisation on our primary inference cluster runs above 85%, and we've had three incidents this quarter where queue depth pushed p95 latency past our 800ms SLO. So the constraint is real \u2014 but it's a peak-load constraint, not a total-capacity constraint. We own machines that sit mostly idle overnight and on weekends, and our batch workloads (embeddings refresh, fine-tuning runs) are scheduled at fixed times rather than shaped around inference demand. That's a scheduling problem before it's a hardware problem.\n\nThe cost picture reinforces this. The $180K purchase price is only the visible part. Add colocation power, networking, and roughly 15 hours a month of our own operational attention, and the true annual cost of ownership lands closer to $260K. Against that, the utilisation work is mostly engineering time we already control: better request batching, dynamic scheduling of batch jobs into off-peak windows, and a queueing policy that lets latency-tolerant traffic yield to latency-sensitive traffic. My rough estimate is that these changes could lift effective capacity 30\u201340% at peak, which covers our projected growth for at least two quarters. Even if I'm wrong by half, we buy ourselves a quarter of breathing room for a fraction of the cost.\n\nLatency deserves its own paragraph, because it's where the trade-off bites. Squeezing more out of existing hardware means more contention, and contention shows up as tail latency. If we implement a proper priority queue, the latency-sensitive requests should be largely insulated \u2014 they're a small fraction of volume \u2014 but there's real risk that our p99 degrades for the batch and internal-tool traffic that we've been treating as best-effort. Buying hardware, by contrast, is latency-free in the sense that nothing gets worse. That asymmetry matters and I don't want to pretend it away. But I'd note that our current p99 is already worse than we'd like during peaks, so the status quo isn't a clean baseline either.\n\nOn reliability: adding nodes doesn't just add capacity, it adds failure surface, and our experience says a new node takes two to three weeks of shakedown before we trust it in production. The utilisation path has its own reliability risk \u2014 a bad scheduler change can take down everything at once \u2014 but it's reversible in minutes, which is the key difference. This is really the crux of the whole decision. Hardware is a one-way door: once the purchase order is signed and the machines are racked, the money is spent, and if demand shifts or our traffic mix changes, we're carrying sunk cost with weak resale options. Scheduler and batching improvements are a two-way door. We can ship them behind a flag, measure, and roll back. When two options have similar expected value, the reversible one is worth a lot.\n\nThe opportunity cost cuts both ways, and I want to be honest about that. Every week the team spends on scheduler work is a week not spent on the model-quality roadmap, which is arguably where our competitive edge lives. If the utilisation project drags past six weeks without clear wins, we'll have paid in the currency of our most valuable engineers. That's why I'm proposing a hard timebox: four weeks, one engineer primary, with a pre-agreed measurement plan. If effective peak capacity hasn't improved by at least 20% by then, we place the hardware order immediately and without further debate.\n\nWhat I'm not claiming is certainty about demand. If the new customer pipeline closes the way sales hopes, our growth curve steepens and the utilisation gains buy weeks, not quarters. That scenario is exactly why the timebox matters \u2014 it keeps the hardware option open rather than foreclosing it.\n\n**Recommendation:** fund the four-week utilisation sprint now, with a pre-committed hardware purchase as the fallback. We keep the option, we learn something real about our own workload, and we only spend the $180K if the data says we must.\n\nHappy to walk through the utilisation data in Thursday's sync.",
          "output_characters": 4746,
          "output_sha256": "9618f694fe4342c37337fe7801fabfd042972a82a098cdb5aab56dc0e4805ecd",
          "prompt_tokens": 124,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000887,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "3650aea304e8d5e33f3a5d44b137a9907cc72bb75fe22ad0f2db456c16c5eec5",
          "time_to_first_visible_seconds": 1.076045,
          "time_to_last_output_seconds": 22.818567,
          "ttft_seconds": 0.500833,
          "wall_seconds": 22.819454
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 945,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 21.00094,
          "decode_tokens_per_second": 44.95,
          "finish_reason": "stop",
          "measured_sse_events": 943,
          "output": "**MEMO: GPU Capacity \u2014 Buy More or Use What We Have Better?**\n\n**To:** Engineering leadership\n**From:** Infrastructure team\n**Re:** Local AI hardware investment decision\n\n**Summary and recommendation**\n\nWe recommend that we spend the next quarter improving utilisation of our existing hardware before committing to any new purchases. We believe the case for buying is real but premature: our own telemetry suggests we are leaving meaningful capacity on the table, and the cost of finding out whether that's true is small compared to the cost of being wrong in the other direction.\n\n**Where we stand**\n\nOur cluster currently consists of eight accelerators across two nodes, serving batch inference jobs overnight and interactive workloads during business hours. Utilisation data tells an uncomfortable story. Average GPU utilisation during business hours sits around 40 percent, and overnight batch jobs, while heavier, still leave gaps measured in tens of minutes between stages. Peak demand \u2014 the number that actually drives the \"we need more hardware\" instinct \u2014 occurs in short bursts, typically when several engineers run evaluation suites simultaneously or when a scheduled fine-tuning job collides with interactive traffic.\n\nThe instinct to buy is understandable. Waiting on a queue for a GPU is demoralising, and it interrupts the flow of exactly the people we most want in flow. But queueing pain and capacity shortage are not the same thing. Right now they are conflated, and we should separate them before spending capital.\n\n**The case for better utilisation first**\n\nThe cheapest capacity is the capacity we already own. A serious scheduling effort \u2014 a proper job queue, priority tiers, and time-shifting batch work into the overnight window \u2014 is perhaps three to four weeks of engineering time. Against that, a single new accelerator costs roughly the equivalent of two engineers for a quarter, before power, cooling, and the operational burden of expanding a small cluster we already struggle to keep well-patched. The opportunity cost cuts both ways, of course: those three weeks of scheduling work are three weeks not spent on product features. But the scheduling work is a one-time investment with compounding returns, whereas hardware is a recurring commitment that arrives with a maintenance tail.\n\nLatency is the strongest argument for buying. Interactive inference latency is genuinely sensitive to contention, and no amount of clever queueing fixes a model that needs to respond in under a second while a training job hogs the memory. Our answer here is not to buy more hardware but to partition more deliberately: reserve capacity for latency-sensitive workloads and confine batch jobs to what remains. If reserved capacity turns out to be insufficient after partitioning \u2014 if interactive jobs are genuinely queueing behind other interactive jobs \u2014 that is a clean, measurable signal that we have a real shortage, and we should buy then, with data in hand.\n\nReliability cuts in favour of caution. Our cluster has no meaningful redundancy; a single node failure takes half our capacity. Adding one more accelerator to the same node improves throughput but does nothing for resilience, and arguably makes us more dependent on hardware we cannot quickly replace. If reliability is the concern, the money is better spent on a failover plan, better checkpointing, and a documented arrangement for burst capacity elsewhere \u2014 cloud or a partner lab \u2014 than on marginal local capacity.\n\n**Reversibility**\n\nThis is the crux. Hardware purchases are largely irreversible at our scale: accelerators depreciate quickly, resale is messy, and the commitment shapes our power and cooling footprint for years. Scheduling and partitioning work is almost perfectly reversible \u2014 we can undo a queue configuration in an afternoon. When one option is reversible and the other is not, and the evidence is ambiguous, the reversible option should go first. It is also worth noting that the accelerator market is not standing still; deferring a purchase six months likely buys us better hardware per dollar, which partially offsets the cost of waiting.\n\n**What we acknowledge**\n\nWe could be wrong. If our utilisation numbers are misleading \u2014 if the 40 percent figure hides memory fragmentation that makes the GPUs effectively unusable \u2014 then scheduling work will not deliver, and we will have spent a quarter learning what we could have learned in a week. We propose a two-week measurement sprint first to validate the telemetry before committing to the full effort. And we should pre-commit to a trigger: if, after the utilisation work, p95 wait times for interactive jobs still exceed our target, we buy, without relitigating the whole question.\n\n**Decision requested**\n\nApprove the measurement sprint and the utilisation quarter, with a pre-agreed purchase trigger for the following quarter if targets are missed.",
          "output_characters": 4911,
          "output_sha256": "8ec3c002c0703e9ac2842bd879c71f070b96271997dbb9e67360c86a63c02ce0",
          "prompt_tokens": 122,
          "reasoning_characters": 60,
          "reasoning_sha256": "95d808b5317c8f2f1c0036a73a2ded3b440c088919c9f6da9732c29145461789",
          "response_tail_seconds": 0.000714,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "37d8e13774910437cadf766a5edb84d57de7de95bdb7681973a785592817909f",
          "time_to_first_visible_seconds": 1.068043,
          "time_to_last_output_seconds": 21.500383,
          "ttft_seconds": 0.499443,
          "wall_seconds": 21.501097
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 24.475,
        "median": 22.811,
        "minimum": 21.291,
        "p90": 24.475
      },
      "ttft_seconds": {
        "maximum": 0.502,
        "median": 0.499,
        "minimum": 0.494,
        "p90": 0.502
      },
      "wall_seconds": {
        "maximum": 24.475,
        "median": 22.811,
        "minimum": 21.292,
        "p90": 24.475
      }
    },
    "structured": {
      "completion_gate": {
        "passed": 5,
        "total": 5
      },
      "decode_tokens_per_second": {
        "maximum": 72.247,
        "median": 70.611,
        "minimum": 69.532,
        "p90": 72.247
      },
      "pooled_decode_tokens_per_second": 71.218,
      "runs": [
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.270526,
          "decode_tokens_per_second": 69.532,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 111,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000449,
          "run": 1,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.542105,
          "time_to_last_output_seconds": 6.81263,
          "ttft_seconds": 0.542105,
          "wall_seconds": 6.813079
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 691,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 9.558082,
          "decode_tokens_per_second": 72.19,
          "finish_reason": "stop",
          "measured_sse_events": 689,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 108,
          "reasoning_characters": 18,
          "reasoning_sha256": "abca0f51ad8341cf7ac2e952f743c89fd86589479036d30a16b825f2aeb68568",
          "response_tail_seconds": 0.000503,
          "run": 2,
          "stream_done_marker": true,
          "stream_sha256": "807d5cb51b38b6d0f6507a903e7b9a9ade0d4d0d1a953ccc597b2f5613f6927e",
          "time_to_first_visible_seconds": 0.751709,
          "time_to_last_output_seconds": 10.037162,
          "ttft_seconds": 0.479081,
          "wall_seconds": 10.037665
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 691,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 9.550516,
          "decode_tokens_per_second": 72.247,
          "finish_reason": "stop",
          "measured_sse_events": 689,
          "output": "[\n  {\"index\": 1, \"square\": 1},\n  {\"index\": 2, \"square\": 4},\n  {\"index\": 3, \"square\": 9},\n  {\"index\": 4, \"square\": 16},\n  {\"index\": 5, \"square\": 25},\n  {\"index\": 6, \"square\": 36},\n  {\"index\": 7, \"square\": 49},\n  {\"index\": 8, \"square\": 64},\n  {\"index\": 9, \"square\": 81},\n  {\"index\": 10, \"square\": 100},\n  {\"index\": 11, \"square\": 121},\n  {\"index\": 12, \"square\": 144},\n  {\"index\": 13, \"square\": 169},\n  {\"index\": 14, \"square\": 196},\n  {\"index\": 15, \"square\": 225},\n  {\"index\": 16, \"square\": 256},\n  {\"index\": 17, \"square\": 289},\n  {\"index\": 18, \"square\": 324},\n  {\"index\": 19, \"square\": 361},\n  {\"index\": 20, \"square\": 400},\n  {\"index\": 21, \"square\": 441},\n  {\"index\": 22, \"square\": 484},\n  {\"index\": 23, \"square\": 529},\n  {\"index\": 24, \"square\": 576},\n  {\"index\": 25, \"square\": 625},\n  {\"index\": 26, \"square\": 676},\n  {\"index\": 27, \"square\": 729},\n  {\"index\": 28, \"square\": 784},\n  {\"index\": 29, \"square\": 841},\n  {\"index\": 30, \"square\": 900},\n  {\"index\": 31, \"square\": 961},\n  {\"index\": 32, \"square\": 1024},\n  {\"index\": 33, \"square\": 1089},\n  {\"index\": 34, \"square\": 1156},\n  {\"index\": 35, \"square\": 1225},\n  {\"index\": 36, \"square\": 1296},\n  {\"index\": 37, \"square\": 1369},\n  {\"index\": 38, \"square\": 1444},\n  {\"index\": 39, \"square\": 1521},\n  {\"index\": 40, \"square\": 1600},\n  {\"index\": 41, \"square\": 1681},\n  {\"index\": 42, \"square\": 1764},\n  {\"index\": 43, \"square\": 1849},\n  {\"index\": 44, \"square\": 1936},\n  {\"index\": 45, \"square\": 2025},\n  {\"index\": 46, \"square\": 2116},\n  {\"index\": 47, \"square\": 2209},\n  {\"index\": 48, \"square\": 2304},\n  {\"index\": 49, \"square\": 2401},\n  {\"index\": 50, \"square\": 2500}\n]",
          "output_characters": 1600,
          "output_sha256": "b357eaecb1f80ddb0c0131e67d4bd0edc32f8df925960270694b2f896b1c8d04",
          "prompt_tokens": 109,
          "reasoning_characters": 18,
          "reasoning_sha256": "abca0f51ad8341cf7ac2e952f743c89fd86589479036d30a16b825f2aeb68568",
          "response_tail_seconds": 0.000104,
          "run": 3,
          "stream_done_marker": true,
          "stream_sha256": "807d5cb51b38b6d0f6507a903e7b9a9ade0d4d0d1a953ccc597b2f5613f6927e",
          "time_to_first_visible_seconds": 0.754668,
          "time_to_last_output_seconds": 10.030201,
          "ttft_seconds": 0.479685,
          "wall_seconds": 10.030305
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.189209,
          "decode_tokens_per_second": 70.445,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 110,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000201,
          "run": 4,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.526338,
          "time_to_last_output_seconds": 6.715547,
          "ttft_seconds": 0.526338,
          "wall_seconds": 6.715748
        },
        {
          "cached_prompt_tokens": 0,
          "completion_tokens": 437,
          "completion_validation": {
            "valid": true
          },
          "decode_seconds": 6.174672,
          "decode_tokens_per_second": 70.611,
          "finish_reason": "stop",
          "measured_sse_events": 435,
          "output": "[{\"index\":1,\"square\":1},{\"index\":2,\"square\":4},{\"index\":3,\"square\":9},{\"index\":4,\"square\":16},{\"index\":5,\"square\":25},{\"index\":6,\"square\":36},{\"index\":7,\"square\":49},{\"index\":8,\"square\":64},{\"index\":9,\"square\":81},{\"index\":10,\"square\":100},{\"index\":11,\"square\":121},{\"index\":12,\"square\":144},{\"index\":13,\"square\":169},{\"index\":14,\"square\":196},{\"index\":15,\"square\":225},{\"index\":16,\"square\":256},{\"index\":17,\"square\":289},{\"index\":18,\"square\":324},{\"index\":19,\"square\":361},{\"index\":20,\"square\":400},{\"index\":21,\"square\":441},{\"index\":22,\"square\":484},{\"index\":23,\"square\":529},{\"index\":24,\"square\":576},{\"index\":25,\"square\":625},{\"index\":26,\"square\":676},{\"index\":27,\"square\":729},{\"index\":28,\"square\":784},{\"index\":29,\"square\":841},{\"index\":30,\"square\":900},{\"index\":31,\"square\":961},{\"index\":32,\"square\":1024},{\"index\":33,\"square\":1089},{\"index\":34,\"square\":1156},{\"index\":35,\"square\":1225},{\"index\":36,\"square\":1296},{\"index\":37,\"square\":1369},{\"index\":38,\"square\":1444},{\"index\":39,\"square\":1521},{\"index\":40,\"square\":1600},{\"index\":41,\"square\":1681},{\"index\":42,\"square\":1764},{\"index\":43,\"square\":1849},{\"index\":44,\"square\":1936},{\"index\":45,\"square\":2025},{\"index\":46,\"square\":2116},{\"index\":47,\"square\":2209},{\"index\":48,\"square\":2304},{\"index\":49,\"square\":2401},{\"index\":50,\"square\":2500}]",
          "output_characters": 1299,
          "output_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "prompt_tokens": 109,
          "reasoning_characters": 0,
          "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
          "response_tail_seconds": 0.000787,
          "run": 5,
          "stream_done_marker": true,
          "stream_sha256": "654a5359e59bad57071f454922f88d55d393a17d6eec293ae79ea185c3cd7453",
          "time_to_first_visible_seconds": 0.528686,
          "time_to_last_output_seconds": 6.703358,
          "ttft_seconds": 0.528686,
          "wall_seconds": 6.704146
        }
      ],
      "time_to_last_output_seconds": {
        "maximum": 10.037,
        "median": 6.813,
        "minimum": 6.703,
        "p90": 10.037
      },
      "ttft_seconds": {
        "maximum": 0.542,
        "median": 0.526,
        "minimum": 0.479,
        "p90": 0.542
      },
      "wall_seconds": {
        "maximum": 10.038,
        "median": 6.813,
        "minimum": 6.704,
        "p90": 10.038
      }
    }
  },
  "prefill": {
    "32768": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1996.605,
          "median": 1991.243,
          "minimum": 1990.047,
          "p90": 1996.605
        },
        "runs": [
          {
            "cache_salt_sha256": "3320503e3eaacc1d21f8b628c813549018e684b9f8155aa685e8a84319a9925a",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.058469,
            "decode_tokens_per_second": 119.721,
            "effective_prefill_tokens_per_second": 1996.605,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000525,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 16.411859,
            "time_to_last_output_seconds": 16.470328,
            "ttft_seconds": 16.411859,
            "wall_seconds": 16.470854
          },
          {
            "cache_salt_sha256": "ffb0569e0d6ebf6089843299f0bcc16454397dedadf6d4bcccdbae69c69bd7be",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053783,
            "decode_tokens_per_second": 130.154,
            "effective_prefill_tokens_per_second": 1990.047,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000105,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 16.465942,
            "time_to_last_output_seconds": 16.519724,
            "ttft_seconds": 16.465942,
            "wall_seconds": 16.51983
          },
          {
            "cache_salt_sha256": "7310e58a6d1cf2241bc18d982d3690ae1af9d1cb9cfe10c9834f3d3e344f8f27",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061356,
            "decode_tokens_per_second": 114.089,
            "effective_prefill_tokens_per_second": 1991.243,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000117,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 16.456053,
            "time_to_last_output_seconds": 16.517409,
            "ttft_seconds": 16.456053,
            "wall_seconds": 16.517526
          }
        ],
        "ttft_seconds": {
          "maximum": 16.466,
          "median": 16.456,
          "minimum": 16.412,
          "p90": 16.466
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 812698.413,
          "median": 724042.69,
          "minimum": 707609.916,
          "p90": 812698.413
        },
        "runs": [
          {
            "cache_salt_sha256": "3320503e3eaacc1d21f8b628c813549018e684b9f8155aa685e8a84319a9925a",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060215,
            "decode_tokens_per_second": 116.249,
            "effective_prefill_tokens_per_second": 707609.916,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 8.4e-05,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.046308,
            "time_to_last_output_seconds": 0.106523,
            "ttft_seconds": 0.046308,
            "wall_seconds": 0.106607
          },
          {
            "cache_salt_sha256": "ffb0569e0d6ebf6089843299f0bcc16454397dedadf6d4bcccdbae69c69bd7be",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.053037,
            "decode_tokens_per_second": 131.983,
            "effective_prefill_tokens_per_second": 812698.413,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " repository contains a service, tests, documentation",
            "output_characters": 52,
            "output_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000105,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "a8872e509e80818cffa1a80a96832f9841d99cdc2262bfe7a1b9fb45a547e12c",
            "time_to_first_visible_seconds": 0.04032,
            "time_to_last_output_seconds": 0.093357,
            "ttft_seconds": 0.04032,
            "wall_seconds": 0.093462
          },
          {
            "cache_salt_sha256": "7310e58a6d1cf2241bc18d982d3690ae1af9d1cb9cfe10c9834f3d3e344f8f27",
            "cached_prompt_tokens": 32768,
            "completion_tokens": 8,
            "decode_seconds": 0.060115,
            "decode_tokens_per_second": 116.443,
            "effective_prefill_tokens_per_second": 724042.69,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " next action. The repository contains a",
            "output_characters": 39,
            "output_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "prompt_tokens": 32768,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 32768,
            "response_tail_seconds": 0.000342,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "dc5bcdac54c41a22c11643f735434b20c9079448f0b90a92b5234f70a0094140",
            "time_to_first_visible_seconds": 0.045257,
            "time_to_last_output_seconds": 0.105372,
            "ttft_seconds": 0.045257,
            "wall_seconds": 0.105714
          }
        ],
        "ttft_seconds": {
          "maximum": 0.046,
          "median": 0.045,
          "minimum": 0.04,
          "p90": 0.046
        }
      }
    },
    "65536": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1966.271,
          "median": 1963.934,
          "minimum": 1963.044,
          "p90": 1966.271
        },
        "runs": [
          {
            "cache_salt_sha256": "2b1453f4db6615a8faa7b21d44861f694807d0776f969f0dc234075c8bef0405",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061963,
            "decode_tokens_per_second": 112.971,
            "effective_prefill_tokens_per_second": 1966.271,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 7.5e-05,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 33.330093,
            "time_to_last_output_seconds": 33.392056,
            "ttft_seconds": 33.330093,
            "wall_seconds": 33.392132
          },
          {
            "cache_salt_sha256": "4a8b4b959337df8fa4f1f047e0fa46ce731c8746cbd816748c5fccbd74566ce1",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.062732,
            "decode_tokens_per_second": 111.585,
            "effective_prefill_tokens_per_second": 1963.934,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.0008,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 33.369749,
            "time_to_last_output_seconds": 33.432482,
            "ttft_seconds": 33.369749,
            "wall_seconds": 33.433282
          },
          {
            "cache_salt_sha256": "3b8da899be9d4d1304d45b08065e935a4eecfa14544a2db899f1cade18603d7b",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.061826,
            "decode_tokens_per_second": 113.222,
            "effective_prefill_tokens_per_second": 1963.044,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000678,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 33.384889,
            "time_to_last_output_seconds": 33.446715,
            "ttft_seconds": 33.384889,
            "wall_seconds": 33.447393
          }
        ],
        "ttft_seconds": {
          "maximum": 33.385,
          "median": 33.37,
          "minimum": 33.33,
          "p90": 33.385
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1430167.598,
          "median": 1394264.318,
          "minimum": 1186794.879,
          "p90": 1430167.598
        },
        "runs": [
          {
            "cache_salt_sha256": "2b1453f4db6615a8faa7b21d44861f694807d0776f969f0dc234075c8bef0405",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.063599,
            "decode_tokens_per_second": 110.064,
            "effective_prefill_tokens_per_second": 1394264.318,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": ". The repository contains a service,",
            "output_characters": 36,
            "output_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000761,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "6b7a31399cbcdf5c725277ca09708289bb776a5e506696b6d065d920946f0841",
            "time_to_first_visible_seconds": 0.047004,
            "time_to_last_output_seconds": 0.110603,
            "ttft_seconds": 0.047004,
            "wall_seconds": 0.111364
          },
          {
            "cache_salt_sha256": "4a8b4b959337df8fa4f1f047e0fa46ce731c8746cbd816748c5fccbd74566ce1",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062678,
            "decode_tokens_per_second": 111.681,
            "effective_prefill_tokens_per_second": 1430167.598,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " the next action. The repository contains",
            "output_characters": 41,
            "output_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000703,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "d0ecfe654cea47153ea77ef98b0550d792072bcb3a007c5d9fe0f911301fcbb7",
            "time_to_first_visible_seconds": 0.045824,
            "time_to_last_output_seconds": 0.108502,
            "ttft_seconds": 0.045824,
            "wall_seconds": 0.109206
          },
          {
            "cache_salt_sha256": "3b8da899be9d4d1304d45b08065e935a4eecfa14544a2db899f1cade18603d7b",
            "cached_prompt_tokens": 65536,
            "completion_tokens": 8,
            "decode_seconds": 0.062046,
            "decode_tokens_per_second": 112.819,
            "effective_prefill_tokens_per_second": 1186794.879,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " and the next action. The repository",
            "output_characters": 36,
            "output_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "prompt_tokens": 65536,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 65536,
            "response_tail_seconds": 0.000384,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "57be7c5160a81419095c8a5e6cb16166b7f1930a0c7450fdaefb448bcb8925c4",
            "time_to_first_visible_seconds": 0.055221,
            "time_to_last_output_seconds": 0.117267,
            "ttft_seconds": 0.055221,
            "wall_seconds": 0.117651
          }
        ],
        "ttft_seconds": {
          "maximum": 0.055,
          "median": 0.047,
          "minimum": 0.046,
          "p90": 0.055
        }
      }
    },
    "8192": {
      "cold": {
        "effective_prefill_tokens_per_second": {
          "maximum": 1961.899,
          "median": 1961.366,
          "minimum": 1957.679,
          "p90": 1961.899
        },
        "runs": [
          {
            "cache_salt_sha256": "a9ae1e7cac18e1f5dbedd5d2c41fda0e805bbc393a4083fd8dffd8bce39645d3",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.052282,
            "decode_tokens_per_second": 133.889,
            "effective_prefill_tokens_per_second": 1961.366,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000643,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 4.176681,
            "time_to_last_output_seconds": 4.228963,
            "ttft_seconds": 4.176681,
            "wall_seconds": 4.229606
          },
          {
            "cache_salt_sha256": "d8a6b47951d74b57bdeaa48b3280837283307fdd6acb5fd7c8352906067ad33e",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.052975,
            "decode_tokens_per_second": 132.138,
            "effective_prefill_tokens_per_second": 1961.899,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000473,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 4.175546,
            "time_to_last_output_seconds": 4.228521,
            "ttft_seconds": 4.175546,
            "wall_seconds": 4.228994
          },
          {
            "cache_salt_sha256": "7d2ef213c9e5989f5ce99a3397f7919f84ab559374da62d49da8775b8f81e1b5",
            "cached_prompt_tokens": 0,
            "completion_tokens": 8,
            "decode_seconds": 0.053341,
            "decode_tokens_per_second": 131.231,
            "effective_prefill_tokens_per_second": 1957.679,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000726,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 4.184548,
            "time_to_last_output_seconds": 4.237889,
            "ttft_seconds": 4.184548,
            "wall_seconds": 4.238615
          }
        ],
        "ttft_seconds": {
          "maximum": 4.185,
          "median": 4.177,
          "minimum": 4.176,
          "p90": 4.185
        }
      },
      "warm_replay": {
        "effective_prefill_tokens_per_second": {
          "maximum": 239875.846,
          "median": 222790.318,
          "minimum": 221699.007,
          "p90": 239875.846
        },
        "runs": [
          {
            "cache_salt_sha256": "a9ae1e7cac18e1f5dbedd5d2c41fda0e805bbc393a4083fd8dffd8bce39645d3",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.051377,
            "decode_tokens_per_second": 136.248,
            "effective_prefill_tokens_per_second": 221699.007,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " behaviour, evidence, constraints, and the",
            "output_characters": 42,
            "output_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.001136,
            "run": 1,
            "stream_done_marker": true,
            "stream_sha256": "128002ef37a31be1bbdb67225d75ef3c0c801e9fa83861d8e04741bdff280850",
            "time_to_first_visible_seconds": 0.036951,
            "time_to_last_output_seconds": 0.088329,
            "ttft_seconds": 0.036951,
            "wall_seconds": 0.089465
          },
          {
            "cache_salt_sha256": "d8a6b47951d74b57bdeaa48b3280837283307fdd6acb5fd7c8352906067ad33e",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.052908,
            "decode_tokens_per_second": 132.305,
            "effective_prefill_tokens_per_second": 222790.318,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " Each observation records behaviour, evidence, constraints",
            "output_characters": 58,
            "output_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.000723,
            "run": 2,
            "stream_done_marker": true,
            "stream_sha256": "3f91e56380da40aa484278ca6ad46278f1415f32fb168c9a22b4e5fb220e39c0",
            "time_to_first_visible_seconds": 0.03677,
            "time_to_last_output_seconds": 0.089678,
            "ttft_seconds": 0.03677,
            "wall_seconds": 0.090401
          },
          {
            "cache_salt_sha256": "7d2ef213c9e5989f5ce99a3397f7919f84ab559374da62d49da8775b8f81e1b5",
            "cached_prompt_tokens": 8192,
            "completion_tokens": 8,
            "decode_seconds": 0.053325,
            "decode_tokens_per_second": 131.27,
            "effective_prefill_tokens_per_second": 239875.846,
            "finish_reason": "length",
            "measured_sse_events": 2,
            "output": " records behaviour, evidence, constraints, and",
            "output_characters": 46,
            "output_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "prompt_tokens": 8192,
            "reasoning_characters": 0,
            "reasoning_sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
            "requested_prompt_tokens": 8192,
            "response_tail_seconds": 0.00099,
            "run": 3,
            "stream_done_marker": true,
            "stream_sha256": "f23b3bec4868cd4bb2a9e53fb2c9efd4e5e04b814a4ed7cf5e709584669551b9",
            "time_to_first_visible_seconds": 0.034151,
            "time_to_last_output_seconds": 0.087476,
            "ttft_seconds": 0.034151,
            "wall_seconds": 0.088466
          }
        ],
        "ttft_seconds": {
          "maximum": 0.037,
          "median": 0.037,
          "minimum": 0.034,
          "p90": 0.037
        }
      }
    }
  },
  "protocol": {
    "prompts_sha256": "0c3ac4015813af6a1b967659bd2289990fde895a5e43ae065b894d1e1f909ecb",
    "prompts_version": "1.0.0",
    "repository_dirty": false,
    "repository_revision": "52c29c3fc1d741b9aef8895b23372aafa5dc6161",
    "repository_source": "git-worktree",
    "repository_source_sha256": "0b5414d24838449a4b70f77959d6e2eee96356206bdaa1b919ed6aa59488e333",
    "version": "1.3.0"
  },
  "run": {
    "appliance": {
      "competing_traffic": "none",
      "context_limit": 1048576,
      "drafter": "checkpoint MTP head (Mia v1.10 DRAFTER=mtp)",
      "hardware": "2x NVIDIA DGX Spark (GB10, 128 GB unified memory each)",
      "kv_cache_dtype": "fp8 (e4m3 DSA latent cache and indexer keys, Mia v1.10 KV=fp8 default); bf16 prompt activations",
      "model": "Mia-AiLab/GLM-5.3-Flash-EXL3-4bpw-TensorFold",
      "model_revision": "078455ffe6472f9a52fbc1139f58b9db2881b25c",
      "negotiated_link_speed": "not measured",
      "quantisation": "EXL3 4 bpw routed experts (mcg codebook), BF16 elsewhere",
      "scheduler": "Mia v1.10 defaults: --parallel 4, shared KV pool",
      "serving_engine": "TensorFold v0.6.0 + Mia GLM-5.3-Flash kit v1.10 (96 patches, commit 549fdc5)",
      "serving_image": "ghcr.io/miaai-lab/glm-5.3-flash-exl3-2x-dgx-sparks-tensorfold@sha256:bc34d7d63f978cf601f42863b284bc95a567c50c10e9adb0866a635be568bf5f",
      "topology": "TP2 across 2 Sparks, RoCE over ConnectX-7 via switch"
    },
    "comparison_id": "glm53flash-tp2-mtp-low",
    "finished_at": "2026-10-08T11:30:34.880627+00:00",
    "label": "tp2-stock-mtp-low",
    "model": "GLM-5.3-Flash-EXL3",
    "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
    "python": "3.12.13",
    "started_at": "2026-10-08T11:20:55.957530+00:00"
  },
  "schema": 1,
  "settings": {
    "concurrency": [
      1,
      2,
      4
    ],
    "concurrency_runs": 3,
    "concurrency_tokens": 256,
    "concurrency_workload": "code",
    "decode_tokens": 8192,
    "extra_body": {
      "chat_template_kwargs": {
        "reasoning_effort": "low"
      }
    },
    "prefill_cache_isolation": "random-salt-per-pair",
    "prefill_depths": [
      8192,
      32768,
      65536
    ],
    "prefill_runs": 3,
    "runs": 5,
    "seed": 20260905,
    "temperature": 0.0,
    "top_p": 1.0
  }
}
