{
  "extract_sha": "73860e2c00f5dbf3d74270fe5d0264b7f77b6d0f",
  "ledger": "docs/benchmarks/2026-09-11-matrix-runs.ndjson",
  "generated_at": "2026-09-11T17:41:03.446Z",
  "origin": "https://airon-coach-bench.pages.dev/",
  "filter": "last terminal:true success with output and without error, per (model, corpus_id, seed); parse_error and 429s are error rows",
  "scored_n": 1822,
  "error_n": 2,
  "skipped": [
    {
      "model": "google/gemini-3-flash",
      "reason": "not in /api/v1/models"
    },
    {
      "model": "x-ai/grok-4-fast",
      "reason": "not in /api/v1/models"
    },
    {
      "model": "google/gemini-3-flash",
      "reason": "not in /api/v1/models"
    },
    {
      "model": "x-ai/grok-4-fast",
      "reason": "not in /api/v1/models"
    }
  ],
  "ledger_cost_usd": 1.772733,
  "scored_cost_usd": 1.772733,
  "tokens_prompt_sum": 1408203,
  "tokens_completion_sum": 311778,
  "picks": {
    "primary": "google/gemini-2.5-flash",
    "d_band": "google/gemini-3-flash-preview",
    "honesty": "openai/gpt-5.6-luna",
    "honesty_slow": "x-ai/grok-4.5"
  },
  "caption": "OpenRouter bench envelope (temperature 0, seed, json_schema name=prescription_rows, strict:false, usage.include). Production interpretPrescription is Base44 InvokeLLM({prompt, response_json_schema}) with model omitted, then sane() drops declined rows. Same RULES string; different envelope.",
  "paths": {
    "prompt": "prompt.md",
    "schema": "schema.json",
    "request": "request.md",
    "score": "data/score.json",
    "corpus": "data/corpus.json",
    "llms": "llms.txt"
  }
}
