{
  "schema_version": 1,
  "version": "v2.0.0-rc.1",
  "historical": true,
  "evidence_sources": [
    {
      "id": "comparison",
      "sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "description": "Frozen comparison-data.json, September 13 2026; selected fields exported below"
    },
    {
      "id": "results",
      "sha256": "f5280f527341e039a9cba7c765d7b5c32e4167eb565a8ee5a1bede85b08fd4d2"
    }
  ],
  "metrics": [
    {
      "id": "ttft_cold_64k",
      "label": "64K cold TTFT",
      "value": 42.623,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "uncached salted prompt, warmed kernels",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; uncached salted prompt, warmed kernels.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/65536-cold",
      "raw_values": [
        42.62296821997734,
        42.768533490016125,
        42.60825550497975
      ],
      "caveats": []
    },
    {
      "id": "ttft_repeat_64k",
      "label": "64K repeat TTFT",
      "value": 0.708,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "same-salt committed prefix; exact repeat",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; same-salt committed prefix; exact repeat.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/65536-warm",
      "raw_values": [
        0.6873520700028166,
        0.7077873190282844,
        0.7220125509775244
      ],
      "caveats": []
    },
    {
      "id": "ttft_suffix_64k",
      "label": "64K suffix TTFT",
      "value": 0.691,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "same-salt committed prefix; changed suffix",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; same-salt committed prefix; changed suffix.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/65536-suffix",
      "raw_values": [
        0.6907970379688777,
        0.7090128429699689,
        0.6672140319715254
      ],
      "caveats": []
    },
    {
      "id": "ttft_cold_76k",
      "label": "76K cold TTFT",
      "value": 49.659,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "uncached salted prompt, warmed kernels",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; uncached salted prompt, warmed kernels.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/76000-cold",
      "raw_values": [
        50.199357914039865,
        49.65898229397135,
        49.430427264014725
      ],
      "caveats": []
    },
    {
      "id": "ttft_repeat_76k",
      "label": "76K repeat TTFT",
      "value": 0.471,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "same-salt committed prefix; exact repeat",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; same-salt committed prefix; exact repeat.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/76000-warm",
      "raw_values": [
        0.4747049320139922,
        0.4705578510183841,
        0.44246281700907275
      ],
      "caveats": []
    },
    {
      "id": "ttft_suffix_76k",
      "label": "76K suffix TTFT",
      "value": 0.475,
      "unit": "s",
      "estimator": "median",
      "samples": 3,
      "cohort": "long-full",
      "cache_state": "same-salt committed prefix; changed suffix",
      "timing_boundary": "client request start to first nonempty SSE output",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Three trials; same-salt committed prefix; changed suffix.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/long/cells/76000-suffix",
      "raw_values": [
        0.47536997497081757,
        0.4516123969806358,
        0.4777440389734693
      ],
      "caveats": []
    },
    {
      "id": "generation_prose",
      "label": "Prose post-first-output rate",
      "value": [
        33.21,
        34.28
      ],
      "unit": "tok/s",
      "estimator": "range across task rates",
      "samples": 3,
      "cohort": "generation-stream-companion",
      "cache_state": "not controlled as uncached; separate streaming pass",
      "timing_boundary": "(completion_tokens - 1) / (HTTP seconds - TTFT)",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "One full streaming answer per task; includes transport/finalization and speculative chunks; not GPU-only decode.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/generation/rows",
      "raw_values": [
        34.28399470928687,
        34.00510857587749,
        33.21436540015559
      ],
      "caveats": []
    },
    {
      "id": "generation_code",
      "label": "Code post-first-output rate",
      "value": [
        67.31,
        75.64
      ],
      "unit": "tok/s",
      "estimator": "range across task rates",
      "samples": 4,
      "cohort": "generation-stream-companion",
      "cache_state": "not controlled as uncached; separate streaming pass",
      "timing_boundary": "(completion_tokens - 1) / (HTTP seconds - TTFT)",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "One full streaming answer per task; includes transport/finalization and speculative chunks; not GPU-only decode.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/generation/rows",
      "raw_values": [
        75.63853950535659,
        68.99434800229787,
        71.5850922236087,
        67.3100295657615
      ],
      "caveats": []
    },
    {
      "id": "short_c1",
      "label": "C1 short-answer aggregate",
      "value": 49.05,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c1",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/0",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_c2",
      "label": "C2 short-answer aggregate",
      "value": 73.55,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c2",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 2,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/1",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_c3",
      "label": "C3 short-answer aggregate",
      "value": 97.21,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c3",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 3,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/2",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_c4",
      "label": "C4 short-answer aggregate",
      "value": 114.16,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c4",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 4,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/3",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_c5",
      "label": "C5 short-answer aggregate",
      "value": 131.46,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c5",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 5,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/4",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_c6",
      "label": "C6 short-answer aggregate",
      "value": 146.12,
      "unit": "tok/s",
      "estimator": "arithmetic mean of 8 category wave rates",
      "samples": 8,
      "cohort": "short-c6",
      "cache_state": "matrix warmup; one pass per category",
      "timing_boundary": "sum output tokens / (last HTTP end - first HTTP start)",
      "concurrency": 6,
      "active_cap": 8,
      "conditions": "Eight categories, 150–256-token caps, one wave per category; transport fixtures, not answer-quality grades.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5ConcurrencyDetails/rows/5",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "work_c3_first",
      "label": "Work C3-FIRST",
      "value": 80.12,
      "unit": "tok/s",
      "estimator": "accepted backend tokens / fixed 180-second window",
      "samples": 1,
      "cohort": "work-v0.1-c3-first",
      "cache_state": "first task round; not guaranteed cold cache",
      "timing_boundary": "backend counter difference over 180 seconds, including unfinished output after drain",
      "concurrency": 3,
      "active_cap": 8,
      "conditions": "One three-minute task round; later tool trajectories diverge; no output-quality score.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/work/recipes/L5/rounds/0",
      "raw_values": [
        80.12222222222222
      ],
      "caveats": [
        "Service-wide counter includes unfinished streams and any health probes; counter snapshots include small launch/drain overhead."
      ]
    },
    {
      "id": "work_c3_repeat",
      "label": "Work C3-REPEAT",
      "value": 81.98,
      "unit": "tok/s",
      "estimator": "accepted backend tokens / fixed 180-second window",
      "samples": 1,
      "cohort": "work-v0.1-c3-repeat",
      "cache_state": "same initial payload repeat",
      "timing_boundary": "backend counter difference over 180 seconds, including unfinished output after drain",
      "concurrency": 3,
      "active_cap": 8,
      "conditions": "One three-minute task round; later tool trajectories diverge; no output-quality score.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/work/recipes/L5/rounds/1",
      "raw_values": [
        81.97777777777777
      ],
      "caveats": [
        "Service-wide counter includes unfinished streams and any health probes; counter snapshots include small launch/drain overhead."
      ]
    },
    {
      "id": "work_c6_first",
      "label": "Work C6-FIRST",
      "value": 102.79,
      "unit": "tok/s",
      "estimator": "accepted backend tokens / fixed 180-second window",
      "samples": 1,
      "cohort": "work-v0.1-c6-first",
      "cache_state": "first task round; not guaranteed cold cache",
      "timing_boundary": "backend counter difference over 180 seconds, including unfinished output after drain",
      "concurrency": 6,
      "active_cap": 8,
      "conditions": "One three-minute task round; later tool trajectories diverge; no output-quality score.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/work/recipes/L5/rounds/2",
      "raw_values": [
        102.78888888888889
      ],
      "caveats": [
        "Service-wide counter includes unfinished streams and any health probes; counter snapshots include small launch/drain overhead."
      ]
    },
    {
      "id": "work_c6_repeat",
      "label": "Work C6-REPEAT",
      "value": 110.49,
      "unit": "tok/s",
      "estimator": "accepted backend tokens / fixed 180-second window",
      "samples": 1,
      "cohort": "work-v0.1-c6-repeat",
      "cache_state": "same initial payload repeat",
      "timing_boundary": "backend counter difference over 180 seconds, including unfinished output after drain",
      "concurrency": 6,
      "active_cap": 8,
      "conditions": "One three-minute task round; later tool trajectories diverge; no output-quality score.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/work/recipes/L5/rounds/3",
      "raw_values": [
        110.4888888888889
      ],
      "caveats": [
        "Service-wide counter includes unfinished streams and any health probes; counter snapshots include small launch/drain overhead."
      ]
    },
    {
      "id": "code_correctness",
      "label": "Functional code correctness",
      "value": 2,
      "unit": "passed / 4",
      "estimator": "count",
      "samples": 4,
      "cohort": "full-answer-quality",
      "cache_state": "single pass",
      "timing_boundary": "independent functional checks after full answer",
      "concurrency": 1,
      "active_cap": 8,
      "conditions": "2/4 small code tasks passed; interval ordering and dependency topology failed.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/l5/quality/functional_code_passes",
      "raw_values": null,
      "caveats": []
    },
    {
      "id": "short_behind_long",
      "label": "Short behind long TTFT",
      "value": 40.458,
      "unit": "s",
      "estimator": "single observation",
      "samples": 1,
      "cohort": "inverse-arrival",
      "cache_state": "short arrives 2 seconds into uncached 64K prefill",
      "timing_boundary": "short request HTTP start to first output",
      "concurrency": 2,
      "active_cap": 8,
      "conditions": "No matched historical inverse-arrival control; unresolved admission weakness.",
      "source_sha256": "c6b886887598c589797600b79863d01a64790ca9ff6aa2ed57c1a5557d4f2ad5",
      "source_pointer": "/followup",
      "raw_values": null,
      "caveats": []
    }
  ],
  "limitations": [
    "Experimental serving recipe; historical one-fleet evidence, fresh portable installation pending.",
    "Functional code passed 2/4 tasks; this is not a broad quality ranking.",
    "A short request arriving two seconds into a 64K prefill waited 40.458 seconds.",
    "300,000 tokens is configured context, not certified usable context or C8 capacity.",
    "Uncached prompt measurements use warmed kernels; they are not cold machine starts.",
    "Work counts unfinished output and does not score quality. Mia active cap 4 differs from Tempo cap 8.",
    "Native vision and optional Pi desktop use are separate from the throughput cohort."
  ]
}
