{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "code-review",
  "jobKind": "skill-class",
  "jobLabel": "Code review",
  "measure": "share of the seeded code defects found",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/code-review",
  "sentence": "A code review means reading a change to some code and finding the mistakes that were put in it. Each model was given the same ten tasks with no help, tested in the API and Claude Code.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "tier": "t1",
        "leaders": [
          {
            "model": "gpt-5-mini",
            "name": "GPT-5 mini",
            "mean": 0.6995833333333333,
            "shown": "70.0%",
            "run": "expansion-cohort"
          }
        ],
        "close": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.575,
            "shown": "57.5%",
            "run": "expansion-cohort"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": false,
        "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "best": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.9349999999999999,
            "shown": "93.5%",
            "run": "expansion-cohort"
          }
        ],
        "close": [
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.8899999999999999,
            "shown": "89.0%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.7493749999999999,
            "shown": "74.9%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.8541666666666666,
            "shown": "85.4%",
            "run": "expansion-opus-5-5"
          },
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "mean": 0.8579166666666665,
            "shown": "85.8%",
            "run": "expansion-fable-5"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "pick": {
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.7493749999999999,
          "usd": 0.004063550000000001,
          "shown": "$0.00406",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "best": {
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.9349999999999999,
          "usd": 0.009484500000000002,
          "shown": "$0.00948",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "pickIsBest": false,
        "basesDiffer": false
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "Only one model's calls were timed on this job this way."
      },
      "sentence": "In Claude Code, Claude Opus 5 scores highest. Claude Fable 5.1, Claude Sonnet 5, Claude Opus 5.5 and Claude Fable 5 are too close to call. Claude Sonnet 5 is the cheapest of those."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "absent",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "word": "not run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.6995833333333333,
          "shown": "70.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.575,
          "shown": "57.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.5825,
          "shown": "58.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 0.8579166666666665,
          "shown": "85.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-fable-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.8899999999999999,
          "shown": "89.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.9349999999999999,
          "shown": "93.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.8541666666666666,
          "shown": "85.4%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-opus-5-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.7493749999999999,
          "shown": "74.9%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/code-review-web",
      "href": "/skills/rampstackco-code-review-web",
      "lift": "+26.7 percentage points (+3.2 to +50.1)",
      "instruction": "could not be told apart from the one-line instruction (−8.9 to +35.5)",
      "runLabel": "code-review-tier-skills",
      "tier": "t3"
    }
  ],
  "nothingLeftToMeasure": [
    {
      "instrument": "claude-code",
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "unaided": "93.5%",
      "tier": "t1"
    }
  ],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On code review, wider run, Claude Fable 5.1 finds 89% of seeded code defects unaided against 86% for Claude Fable 5; no measurable change."
  ]
}
