{
  "surface": "openaddict-model-routing",
  "version": "model-routing-v1",
  "url": "https://openaddict.com/models/routing",
  "generatedAt": "2026-09-19",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every pick is read inside one job, one instrument and one run. No pick compares two instruments, and a job measured on two instruments carries a separate side for each.",
  "guide": [
    "Choose a model for each job, not one model for every job.",
    "For each job, take the cheapest model that is too close to call against the best, unless it repeats itself less often or answers more slowly.",
    "Never compare scores from two test methods, except through a model tested both ways, like Claude Haiku 4.5.",
    "Fetch this page or its data file again before you rely on it, because the models on offer and their versions change.",
    "A tier means we made the tasks harder until the models could be told apart.",
    "“Not run” means we never tested that model on that job, not that it scored zero."
  ],
  "runs": [
    {
      "instrument": "api",
      "run": "expansion-cohort",
      "label": "Wider run, three ways"
    },
    {
      "instrument": "api",
      "run": "launch",
      "label": "Main run, tip by tip"
    },
    {
      "instrument": "api",
      "run": "reliability-five-pass",
      "label": "Each tip asked five times"
    },
    {
      "instrument": "api",
      "run": "skill-cohort",
      "label": "First run, with and without the skill"
    },
    {
      "instrument": "api",
      "run": "tier-calibration",
      "label": "Harder tasks, each model at its own tier"
    },
    {
      "instrument": "api",
      "run": "tip-tier2-run",
      "label": "Harder tasks, tier 2"
    },
    {
      "instrument": "claude-code",
      "run": "expansion-cohort",
      "label": "Wider run, three ways"
    },
    {
      "instrument": "claude-code",
      "run": "launch",
      "label": "Main run"
    },
    {
      "instrument": "claude-code",
      "run": "panel-v2",
      "label": "Second run, every task twice"
    },
    {
      "instrument": "claude-code",
      "run": "reliability-five-pass",
      "label": "Each tip asked five times"
    },
    {
      "instrument": "claude-code",
      "run": "tier-calibration",
      "label": "Harder tasks, each model at its own tier"
    },
    {
      "instrument": "claude-code",
      "run": "tip-tier2-run",
      "label": "Harder tasks, tier 2"
    },
    {
      "instrument": "claude-code",
      "run": "workflow-full-run",
      "label": "Workflow run, five times each"
    },
    {
      "instrument": "codex-cli",
      "run": "codex-tips-run",
      "label": "Codex run, tip by tip"
    }
  ],
  "rows": [
    {
      "job": "accessibility-audit",
      "jobKind": "skill-class",
      "jobLabel": "Accessibility audit",
      "measure": "share of the seeded accessibility defects found",
      "scale": "unit",
      "sentence": "In the API, each model was given tasks at its own level of difficulty, so no one model is ranked first. In Claude Code, each model was given tasks at its own level of difficulty, so no one model is ranked first. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "tier-calibration",
            "label": "Harder tasks, each model at its own tier",
            "date": "2026-09-18"
          },
          "tiers": [
            "t1",
            "t2"
          ],
          "best": {
            "measured": false,
            "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
          },
          "cheapest": {
            "measured": false,
            "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In the API, each model was given tasks at its own level of difficulty, so no one model is ranked first."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "tier-calibration",
            "label": "Harder tasks, each model at its own tier",
            "date": "2026-09-18"
          },
          "tiers": [
            "t1",
            "t2",
            "t3"
          ],
          "best": {
            "measured": false,
            "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
          },
          "cheapest": {
            "measured": false,
            "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
          },
          "consistent": {
            "measured": true,
            "run": "panel-v2",
            "runLabel": "Second run, every task twice",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 38,
                "repeated": 40,
                "share": 0.95,
                "shown": "38 of 40"
              }
            ],
            "close": []
          },
          "fastest": {
            "measured": false,
            "reason": "Only one model's calls were timed on this job this way."
          },
          "sentence": "In Claude Code, each model was given tasks at its own level of difficulty, so no one model is ranked first."
        }
      ]
    },
    {
      "job": "code-review",
      "jobKind": "skill-class",
      "jobLabel": "Code review",
      "measure": "share of the seeded code defects found",
      "scale": "unit",
      "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call. In Claude Code, Claude Opus 5 scores highest. Claude Fable 5.1 and Claude Sonnet 5 are too close to call. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "expansion-cohort",
            "label": "Wider run, three ways",
            "date": "2026-09-06"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "expansion-cohort",
            "runLabel": "Wider run, three ways",
            "tier": "t1",
            "leaders": [
              {
                "model": "gpt-5-mini",
                "name": "GPT-5 mini",
                "mean": 0.6995833333333333,
                "shown": "70.0%"
              }
            ],
            "close": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.575,
                "shown": "57.5%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "expansion-cohort",
            "label": "Wider run, three ways",
            "date": "2026-09-06"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "expansion-cohort",
            "runLabel": "Wider run, three ways",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.9349999999999999,
                "shown": "93.5%"
              }
            ],
            "close": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.8899999999999999,
                "shown": "89.0%"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "mean": 0.7493749999999999,
                "shown": "74.9%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "Only one model's calls were timed on this job this way."
          },
          "sentence": "In Claude Code, Claude Opus 5 scores highest. Claude Fable 5.1 and Claude Sonnet 5 are too close to call."
        }
      ]
    },
    {
      "job": "on-page-audit",
      "jobKind": "skill-class",
      "jobLabel": "On-page audit",
      "measure": "share of the seeded on-page issues found",
      "scale": "unit",
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini and Claude Haiku 4.5 are too close to call. In Claude Code, Claude Fable 5.1 scores highest. Claude Opus 5 is too close to call. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "skill-cohort",
            "label": "First run, with and without the skill",
            "date": "2026-08-30"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "tier": "t1",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.488095238095238,
                "shown": "48.8%"
              }
            ],
            "close": [
              {
                "model": "gpt-5-mini",
                "name": "GPT-5 mini",
                "mean": 0.4492857142857143,
                "shown": "44.9%"
              },
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.23928571428571427,
                "shown": "23.9%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "pick": {
              "model": "gemini-3.1-flash-lite",
              "name": "Gemini 3.1 Flash Lite",
              "mean": 0.488095238095238,
              "usd": 0.000228925,
              "shown": "$0.00023",
              "basis": "billed",
              "label": "billed through the API"
            },
            "best": {
              "model": "gemini-3.1-flash-lite",
              "name": "Gemini 3.1 Flash Lite",
              "mean": 0.488095238095238,
              "usd": 0.000228925,
              "shown": "$0.00023",
              "basis": "billed",
              "label": "billed through the API"
            },
            "pickIsBest": true,
            "basesDiffer": false
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini and Claude Haiku 4.5 are too close to call."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "expansion-cohort",
            "label": "Wider run, three ways",
            "date": "2026-09-06"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "expansion-cohort",
            "runLabel": "Wider run, three ways",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.9722222222222223,
                "shown": "97.2%"
              }
            ],
            "close": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.8805555555555555,
                "shown": "88.1%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": true,
            "run": "panel-v2",
            "runLabel": "Second run, every task twice",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 33,
                "repeated": 40,
                "share": 0.825,
                "shown": "33 of 40"
              }
            ],
            "close": []
          },
          "fastest": {
            "measured": false,
            "reason": "Only one model's calls were timed on this job this way."
          },
          "sentence": "In Claude Code, Claude Fable 5.1 scores highest. Claude Opus 5 is too close to call."
        }
      ]
    },
    {
      "job": "skill-authoring",
      "jobKind": "skill-class",
      "jobLabel": "Skill authoring",
      "measure": "share of the specification requirements found",
      "scale": "unit",
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini is too close to call. In Claude Code, Claude Fable 5.1, Claude Opus 5 and Claude Sonnet 5 tie for the top score. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "skill-cohort",
            "label": "First run, with and without the skill",
            "date": "2026-08-30"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "tier": "t1",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.8727272727272727,
                "shown": "87.3%"
              }
            ],
            "close": [
              {
                "model": "gpt-5-mini",
                "name": "GPT-5 mini",
                "mean": 0.7318181818181818,
                "shown": "73.2%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "pick": {
              "model": "gemini-3.1-flash-lite",
              "name": "Gemini 3.1 Flash Lite",
              "mean": 0.8727272727272727,
              "usd": 0.000884075,
              "shown": "$0.00088",
              "basis": "billed",
              "label": "billed through the API"
            },
            "best": {
              "model": "gemini-3.1-flash-lite",
              "name": "Gemini 3.1 Flash Lite",
              "mean": 0.8727272727272727,
              "usd": 0.000884075,
              "shown": "$0.00088",
              "basis": "billed",
              "label": "billed through the API"
            },
            "pickIsBest": true,
            "basesDiffer": false
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini is too close to call."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "expansion-cohort",
            "label": "Wider run, three ways",
            "date": "2026-09-06"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "expansion-cohort",
            "runLabel": "Wider run, three ways",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 1,
                "shown": "100.0%"
              },
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 1,
                "shown": "100.0%"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "mean": 1,
                "shown": "100.0%"
              }
            ],
            "close": [],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": true,
            "run": "panel-v2",
            "runLabel": "Second run, every task twice",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 28,
                "repeated": 29,
                "share": 0.9655172413793104,
                "shown": "28 of 29"
              }
            ],
            "close": []
          },
          "fastest": {
            "measured": false,
            "reason": "Only one model's calls were timed on this job this way."
          },
          "sentence": "In Claude Code, Claude Fable 5.1, Claude Opus 5 and Claude Sonnet 5 tie for the top score."
        }
      ]
    },
    {
      "job": "spec-writing",
      "jobKind": "skill-class",
      "jobLabel": "Spec writing",
      "measure": "share of the required sections and acceptance criteria found",
      "scale": "unit",
      "sentence": "In the API, Claude Haiku 4.5 scores highest. Gemini 3.1 Flash Lite is too close to call. Gemini 3.1 Flash Lite is the cheapest of those. In Claude Code, Claude Haiku 4.5 scores highest. Claude Fable 5.1 is too close to call. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "skill-cohort",
            "label": "First run, with and without the skill",
            "date": "2026-08-30"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.9954545454545455,
                "shown": "99.5%"
              }
            ],
            "close": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.990909090909091,
                "shown": "99.1%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": true,
            "run": "skill-cohort",
            "runLabel": "First run, with and without the skill",
            "pick": {
              "model": "gemini-3.1-flash-lite",
              "name": "Gemini 3.1 Flash Lite",
              "mean": 0.990909090909091,
              "usd": 0.00079535,
              "shown": "$0.00080",
              "basis": "billed",
              "label": "billed through the API"
            },
            "best": {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.9954545454545455,
              "usd": 0.00332185,
              "shown": "$0.00332",
              "basis": "billed",
              "label": "billed through the API"
            },
            "pickIsBest": false,
            "basesDiffer": false
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In the API, Claude Haiku 4.5 scores highest. Gemini 3.1 Flash Lite is too close to call. Gemini 3.1 Flash Lite is the cheapest of those."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "expansion-cohort",
            "label": "Wider run, three ways",
            "date": "2026-09-06"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "expansion-cohort",
            "runLabel": "Wider run, three ways",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.9924242424242424,
                "shown": "99.2%"
              }
            ],
            "close": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.9787878787878788,
                "shown": "97.9%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": true,
            "run": "panel-v2",
            "runLabel": "Second run, every task twice",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 27,
                "repeated": 40,
                "share": 0.675,
                "shown": "27 of 40"
              }
            ],
            "close": []
          },
          "fastest": {
            "measured": false,
            "reason": "Only one model's calls were timed on this job this way."
          },
          "sentence": "In Claude Code, Claude Haiku 4.5 scores highest. Claude Fable 5.1 is too close to call."
        }
      ]
    },
    {
      "job": "link-graph-and-metadata-parity-audit",
      "jobKind": "workflow",
      "jobLabel": "Link graph and metadata parity audit",
      "measure": "share of the planted faults found",
      "scale": "unit",
      "sentence": "In Claude Code, Claude Fable 5.1 scores highest.",
      "sides": [
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "workflow-full-run",
            "label": "Workflow run, five times each",
            "date": "2026-09-10"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "workflow-full-run",
            "runLabel": "Workflow run, five times each",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 1,
                "shown": "100.0%"
              }
            ],
            "close": [],
            "rangeAbsent": true
          },
          "cheapest": {
            "measured": false,
            "reason": "This run does not publish a cost per task."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Claude Code, Claude Fable 5.1 scores highest."
        }
      ]
    },
    {
      "job": "corpus-integrity-and-correction",
      "jobKind": "workflow",
      "jobLabel": "Corpus integrity and correction",
      "measure": "share of the planted faults found",
      "scale": "unit",
      "sentence": "In Claude Code, Claude Opus 5 scores highest.",
      "sides": [
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "workflow-full-run",
            "label": "Workflow run, five times each",
            "date": "2026-09-10"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "workflow-full-run",
            "runLabel": "Workflow run, five times each",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.7583333333333333,
                "shown": "75.8%"
              }
            ],
            "close": [],
            "rangeAbsent": true
          },
          "cheapest": {
            "measured": false,
            "reason": "This run does not publish a cost per task."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Claude Code, Claude Opus 5 scores highest."
        }
      ]
    },
    {
      "job": "traffic-drop-triage",
      "jobKind": "workflow",
      "jobLabel": "Traffic drop triage",
      "measure": "share of runs that named the right cause",
      "scale": "unit",
      "sentence": "In Claude Code, Claude Opus 5 and Claude Fable 5.1 tie for the top score.",
      "sides": [
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "workflow-full-run",
            "label": "Workflow run, five times each",
            "date": "2026-09-10"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "workflow-full-run",
            "runLabel": "Workflow run, five times each",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 1,
                "shown": "100.0%"
              },
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 1,
                "shown": "100.0%"
              }
            ],
            "close": [],
            "rangeAbsent": true
          },
          "cheapest": {
            "measured": false,
            "reason": "This run does not publish a cost per task."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Claude Code, Claude Opus 5 and Claude Fable 5.1 tie for the top score."
        }
      ]
    },
    {
      "job": "C01-json-schema",
      "jobKind": "tip",
      "jobLabel": "Getting clean JSON back",
      "measure": "score with no tip applied",
      "scale": "unit",
      "sentence": "In the API, no model managed this on its own. In Claude Code, no model managed this on its own. In Codex, no model managed this on its own. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "tip-tier2-run",
            "label": "Harder tasks, tier 2",
            "date": "2026-09-19"
          },
          "tiers": [
            "t2"
          ],
          "best": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "cheapest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "consistent": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "fastest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "sentence": "In the API, no model managed this on its own."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "tip-tier2-run",
            "label": "Harder tasks, tier 2",
            "date": "2026-09-19"
          },
          "tiers": [
            "t2"
          ],
          "best": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "cheapest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "consistent": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "fastest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "sentence": "In Claude Code, no model managed this on its own."
        },
        {
          "instrument": "codex-cli",
          "instrumentLabel": "tested in Codex",
          "run": {
            "id": "codex-tips-run",
            "label": "Codex run, tip by tip",
            "date": "2026-09-09"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "cheapest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "consistent": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "fastest": {
            "measured": false,
            "reason": "No model managed this on its own."
          },
          "sentence": "In Codex, no model managed this on its own."
        }
      ]
    },
    {
      "job": "C17-exact-length",
      "jobKind": "tip",
      "jobLabel": "Getting the length right",
      "measure": "score with no tip applied",
      "scale": "unit",
      "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call. In Claude Code, Claude Haiku 4.5 scores highest. Claude Opus 5, Claude Sonnet 5 and Claude Fable 5.1 are too close to call. In Codex, GPT-5.6 Terra scores highest. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "tip-tier2-run",
            "label": "Harder tasks, tier 2",
            "date": "2026-09-19"
          },
          "tiers": [
            "t2"
          ],
          "best": {
            "measured": true,
            "run": "tip-tier2-run",
            "runLabel": "Harder tasks, tier 2",
            "tier": "t2",
            "leaders": [
              {
                "model": "gpt-5-mini",
                "name": "GPT-5 mini",
                "mean": 0.05,
                "shown": "5.0%"
              }
            ],
            "close": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.03666666666666667,
                "shown": "3.7%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "This run does not publish a cost per task."
          },
          "consistent": {
            "measured": false,
            "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "medianMs": 563,
                "shown": "563 ms"
              }
            ],
            "models": 2
          },
          "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "tip-tier2-run",
            "label": "Harder tasks, tier 2",
            "date": "2026-09-19"
          },
          "tiers": [
            "t2"
          ],
          "best": {
            "measured": true,
            "run": "tip-tier2-run",
            "runLabel": "Harder tasks, tier 2",
            "tier": "t2",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.041666666666666664,
                "shown": "4.2%"
              }
            ],
            "close": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.0025000000000000022,
                "shown": "0.3%"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "mean": 0.014166666666666671,
                "shown": "1.4%"
              },
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.0025000000000000022,
                "shown": "0.3%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "This run does not publish a cost per task."
          },
          "consistent": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 8,
                "repeated": 10,
                "share": 0.8,
                "shown": "8 of 10"
              }
            ],
            "close": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "identical": 7,
                "repeated": 10,
                "share": 0.7,
                "shown": "7 of 10"
              },
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "identical": 4,
                "repeated": 10,
                "share": 0.4,
                "shown": "4 of 10"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "identical": 7,
                "repeated": 10,
                "share": 0.7,
                "shown": "7 of 10"
              }
            ]
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "medianMs": 11372,
                "shown": "11.4 s"
              }
            ],
            "models": 4
          },
          "sentence": "In Claude Code, Claude Haiku 4.5 scores highest. Claude Opus 5, Claude Sonnet 5 and Claude Fable 5.1 are too close to call."
        },
        {
          "instrument": "codex-cli",
          "instrumentLabel": "tested in Codex",
          "run": {
            "id": "codex-tips-run",
            "label": "Codex run, tip by tip",
            "date": "2026-09-09"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "codex-tips-run",
            "runLabel": "Codex run, tip by tip",
            "tier": "t1",
            "leaders": [
              {
                "model": "gpt-5.6-terra",
                "name": "GPT-5.6 Terra",
                "mean": 0.6235999999999999,
                "shown": "62.4%"
              }
            ],
            "close": [],
            "rangeAbsent": true
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Codex, GPT-5.6 Terra scores highest."
        }
      ]
    },
    {
      "job": "C16-cutoff-disclosure",
      "jobKind": "tip",
      "jobLabel": "Questions about recent events",
      "measure": "score with no tip applied",
      "scale": "unit",
      "sentence": "In the API, Claude Haiku 4.5 scores highest. In Claude Code, Claude Haiku 4.5 scores highest. Claude Sonnet 5, Claude Fable 5.1, Claude Fable 5 and Claude Opus 5 are too close to call. In Codex, only one model was measured on this run, so there is nothing to rank. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "launch",
            "label": "Main run, tip by tip",
            "date": "2026-08-17"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run, tip by tip",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.8555555555555556,
                "shown": "85.6%"
              }
            ],
            "close": [],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "medianMs": 563,
                "shown": "563 ms"
              }
            ],
            "models": 2
          },
          "sentence": "In the API, Claude Haiku 4.5 scores highest."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "launch",
            "label": "Main run",
            "date": "2026-09-03"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.9666666666666667,
                "shown": "96.7%"
              }
            ],
            "close": [
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "mean": 0.9333333333333333,
                "shown": "93.3%"
              },
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.9222222222222223,
                "shown": "92.2%"
              },
              {
                "model": "claude-fable-5",
                "name": "Claude Fable 5",
                "mean": 0.8777777777777778,
                "shown": "87.8%"
              },
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.8027777777777778,
                "shown": "80.3%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run",
            "pick": {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.9666666666666667,
              "usd": 0.0021981,
              "shown": "$0.00220",
              "basis": "list-price",
              "label": "run on subscription; shown at API list price for comparison"
            },
            "best": {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.9666666666666667,
              "usd": 0.0021981,
              "shown": "$0.00220",
              "basis": "list-price",
              "label": "run on subscription; shown at API list price for comparison"
            },
            "pickIsBest": true,
            "basesDiffer": false
          },
          "consistent": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "identical": 28,
                "repeated": 30,
                "share": 0.9333333333333333,
                "shown": "28 of 30"
              }
            ],
            "close": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "identical": 20,
                "repeated": 30,
                "share": 0.6666666666666666,
                "shown": "20 of 30"
              },
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 23,
                "repeated": 30,
                "share": 0.7666666666666667,
                "shown": "23 of 30"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "identical": 22,
                "repeated": 30,
                "share": 0.7333333333333333,
                "shown": "22 of 30"
              }
            ]
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "medianMs": 11372,
                "shown": "11.4 s"
              }
            ],
            "models": 4
          },
          "sentence": "In Claude Code, Claude Haiku 4.5 scores highest. Claude Sonnet 5, Claude Fable 5.1, Claude Fable 5 and Claude Opus 5 are too close to call."
        },
        {
          "instrument": "codex-cli",
          "instrumentLabel": "tested in Codex",
          "run": {
            "id": "codex-tips-run",
            "label": "Codex run, tip by tip",
            "date": "2026-09-09"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": false,
            "reason": "Only one model was measured on this run, so there is nothing to rank."
          },
          "cheapest": {
            "measured": false,
            "reason": "Only one model was measured on this run, so there is nothing to rank."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Codex, only one model was measured on this run, so there is nothing to rank."
        }
      ]
    },
    {
      "job": "C05-think-step-by-step",
      "jobKind": "tip",
      "jobLabel": "Better step-by-step answers",
      "measure": "score with no tip applied",
      "scale": "unit",
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. Claude Haiku 4.5 is too close to call. In Claude Code, Claude Fable 5, Claude Fable 5.1, Claude Opus 5 and Claude Sonnet 5 tie for the top score. Claude Haiku 4.5 is too close to call. Claude Haiku 4.5 is the cheapest of those. In Codex, GPT-5.4 mini, GPT-5.6 Luna and GPT-5.6 Terra tie for the top score. Those test methods are never compared with each other.",
      "sides": [
        {
          "instrument": "api",
          "instrumentLabel": "tested via API",
          "run": {
            "id": "launch",
            "label": "Main run, tip by tip",
            "date": "2026-08-17"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run, tip by tip",
            "tier": "t1",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "mean": 0.4,
                "shown": "40.0%"
              }
            ],
            "close": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.3,
                "shown": "30.0%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "gemini-3.1-flash-lite",
                "name": "Gemini 3.1 Flash Lite",
                "medianMs": 563,
                "shown": "563 ms"
              }
            ],
            "models": 2
          },
          "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. Claude Haiku 4.5 is too close to call."
        },
        {
          "instrument": "claude-code",
          "instrumentLabel": "tested in Claude Code",
          "run": {
            "id": "launch",
            "label": "Main run",
            "date": "2026-09-03"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run",
            "tier": "t1",
            "leaders": [
              {
                "model": "claude-fable-5",
                "name": "Claude Fable 5",
                "mean": 0.4,
                "shown": "40.0%"
              },
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "mean": 0.4,
                "shown": "40.0%"
              },
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "mean": 0.4,
                "shown": "40.0%"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "mean": 0.4,
                "shown": "40.0%"
              }
            ],
            "close": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "mean": 0.2,
                "shown": "20.0%"
              }
            ],
            "rangeAbsent": false
          },
          "cheapest": {
            "measured": true,
            "run": "launch",
            "runLabel": "Main run",
            "pick": {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.2,
              "usd": 0.00135,
              "shown": "$0.00135",
              "basis": "list-price",
              "label": "run on subscription; shown at API list price for comparison"
            },
            "best": {
              "model": "claude-opus-5",
              "name": "Claude Opus 5",
              "mean": 0.4,
              "usd": 0.00807,
              "shown": "$0.00807",
              "basis": "list-price",
              "label": "run on subscription; shown at API list price for comparison"
            },
            "pickIsBest": false,
            "basesDiffer": false
          },
          "consistent": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "condition": "no-sampling-control",
            "conditionLabel": "no sampling control here",
            "leaders": [
              {
                "model": "claude-fable-5-1",
                "name": "Claude Fable 5.1",
                "identical": 10,
                "repeated": 10,
                "share": 1,
                "shown": "10 of 10"
              },
              {
                "model": "claude-opus-5",
                "name": "Claude Opus 5",
                "identical": 10,
                "repeated": 10,
                "share": 1,
                "shown": "10 of 10"
              },
              {
                "model": "claude-sonnet-5",
                "name": "Claude Sonnet 5",
                "identical": 10,
                "repeated": 10,
                "share": 1,
                "shown": "10 of 10"
              }
            ],
            "close": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "identical": 8,
                "repeated": 10,
                "share": 0.8,
                "shown": "8 of 10"
              }
            ]
          },
          "fastest": {
            "measured": true,
            "run": "reliability-five-pass",
            "runLabel": "Each tip asked five times",
            "scope": "time per call, across all twelve tips asked five times",
            "leaders": [
              {
                "model": "claude-haiku-4-5",
                "name": "Claude Haiku 4.5",
                "medianMs": 11372,
                "shown": "11.4 s"
              }
            ],
            "models": 4
          },
          "sentence": "In Claude Code, Claude Fable 5, Claude Fable 5.1, Claude Opus 5 and Claude Sonnet 5 tie for the top score. Claude Haiku 4.5 is too close to call. Claude Haiku 4.5 is the cheapest of those."
        },
        {
          "instrument": "codex-cli",
          "instrumentLabel": "tested in Codex",
          "run": {
            "id": "codex-tips-run",
            "label": "Codex run, tip by tip",
            "date": "2026-09-09"
          },
          "tiers": [
            "t1"
          ],
          "best": {
            "measured": true,
            "run": "codex-tips-run",
            "runLabel": "Codex run, tip by tip",
            "tier": "t1",
            "leaders": [
              {
                "model": "gpt-5.4-mini",
                "name": "GPT-5.4 mini",
                "mean": 0.3,
                "shown": "30.0%"
              },
              {
                "model": "gpt-5.6-luna",
                "name": "GPT-5.6 Luna",
                "mean": 0.3,
                "shown": "30.0%"
              },
              {
                "model": "gpt-5.6-terra",
                "name": "GPT-5.6 Terra",
                "mean": 0.3,
                "shown": "30.0%"
              }
            ],
            "close": [],
            "rangeAbsent": true
          },
          "cheapest": {
            "measured": false,
            "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
          },
          "consistent": {
            "measured": false,
            "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
          },
          "fastest": {
            "measured": false,
            "reason": "No run of this job timed its calls this way."
          },
          "sentence": "In Codex, GPT-5.4 mini, GPT-5.6 Luna and GPT-5.6 Terra tie for the top score."
        }
      ]
    }
  ]
}
