{
  "surface": "openaddict-comparison",
  "version": "comparison-v1",
  "url": "https://openaddict.com/compare/claude-opus-vs-sonnet-vs-haiku",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Models are compared on a job only where the jobs-by-models grid holds a figure for each of them on one instrument at one tier, read from one run or from runs that asked exactly the same units. A model page exists only where that holds on at least 3 jobs. Who is ahead is the overlap rule over the compared models alone. Nothing crosses instruments.",
  "line": "Eight jobs with the same work, tested inside Claude Code. They could not be told apart on 3 of the 7 that could be ranked.",
  "page": {
    "kind": "models",
    "copy": {
      "slug": "claude-opus-vs-sonnet-vs-haiku",
      "kind": "models",
      "models": [
        "claude-opus-5",
        "claude-sonnet-5",
        "claude-haiku-4-5"
      ]
    },
    "models": [
      {
        "model": "claude-opus-5",
        "name": "Claude Opus 5"
      },
      {
        "model": "claude-sonnet-5",
        "name": "Claude Sonnet 5"
      },
      {
        "model": "claude-haiku-4-5",
        "name": "Claude Haiku 4.5"
      }
    ],
    "shared": [
      {
        "job": {
          "id": "code-review",
          "kind": "skill-class",
          "label": "Code review",
          "href": "/models/compare#ts-skill-class-code-review",
          "scale": "unit",
          "measure": "share of the seeded code defects found"
        },
        "jobPath": "/jobs/code-review",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.9349999999999999,
            "shown": "93.5%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "best",
            "cost": "$0.00948",
            "costUsd": 0.009484500000000002,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.7493749999999999,
            "shown": "74.9%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "close",
            "cost": "$0.00406",
            "costUsd": 0.004063550000000001,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.5825,
            "shown": "58.3%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.00120",
            "costUsd": 0.0012021125000000001,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          }
        ],
        "state": "apart",
        "tied": false,
        "sentence": "On code review, tested inside Claude Code, Claude Opus 5 (93.5%) and Claude Sonnet 5 (74.9%) scored higher than Claude Haiku 4.5 (58.3%)."
      },
      {
        "job": {
          "id": "on-page-audit",
          "kind": "skill-class",
          "label": "On-page audit",
          "href": "/models/compare#ts-skill-class-on-page-audit",
          "scale": "unit",
          "measure": "share of the seeded on-page issues found"
        },
        "jobPath": "/jobs/on-page-audit",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.8805555555555555,
            "shown": "88.1%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "best",
            "cost": "$0.00715",
            "costUsd": 0.0071491666666666665,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.5297619047619048,
            "shown": "53.0%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.00362",
            "costUsd": 0.0036230000000000004,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.23095238095238094,
            "shown": "23.1%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.00105",
            "costUsd": 0.0010481333333333333,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          }
        ],
        "state": "apart",
        "tied": false,
        "sentence": "On on-page audit, tested inside Claude Code, Claude Opus 5 (88.1%) scored higher than Claude Sonnet 5 (53.0%) and Claude Haiku 4.5 (23.1%)."
      },
      {
        "job": {
          "id": "skill-authoring",
          "kind": "skill-class",
          "label": "Skill authoring",
          "href": "/models/compare#ts-skill-class-skill-authoring",
          "scale": "unit",
          "measure": "share of the specification requirements found"
        },
        "jobPath": "/jobs/skill-authoring",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 1,
            "shown": "100.0%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "best",
            "cost": "$0.05962",
            "costUsd": 0.05961924999999999,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 1,
            "shown": "100.0%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "best",
            "cost": "$0.02027",
            "costUsd": 0.020272199999999997,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.6727272727272728,
            "shown": "67.3%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.00504",
            "costUsd": 0.005036275,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          }
        ],
        "state": "apart",
        "tied": false,
        "sentence": "On skill authoring, tested inside Claude Code, Claude Opus 5 (100.0%) and Claude Sonnet 5 (100.0%) scored higher than Claude Haiku 4.5 (67.3%)."
      },
      {
        "job": {
          "id": "spec-writing",
          "kind": "skill-class",
          "label": "Spec writing",
          "href": "/models/compare#ts-skill-class-spec-writing",
          "scale": "unit",
          "measure": "share of the required sections and acceptance criteria found"
        },
        "jobPath": "/jobs/spec-writing",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.9227272727272726,
            "shown": "92.3%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.06098",
            "costUsd": 0.06098191666666666,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.9621212121212122,
            "shown": "96.2%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": null,
            "cost": "$0.01601",
            "costUsd": 0.016010433333333334,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.9924242424242424,
            "shown": "99.2%",
            "tier": "t1",
            "run": "expansion-cohort",
            "mark": "best",
            "cost": "$0.00315",
            "costUsd": 0.003150683333333333,
            "repeat": "one pass, not measured",
            "repeatSetting": null,
            "time": "not timed"
          }
        ],
        "state": "apart",
        "tied": false,
        "sentence": "On spec writing, tested inside Claude Code, Claude Haiku 4.5 (99.2%) scored higher than Claude Opus 5 (92.3%) and Claude Sonnet 5 (96.2%)."
      },
      {
        "job": {
          "id": "C01-json-schema",
          "kind": "tip",
          "label": "Getting clean JSON back",
          "href": "/models/compare#ts-tip-single-turn-C01-json-schema",
          "scale": "unit",
          "measure": "score with no tip applied"
        },
        "jobPath": "/jobs/json-extraction",
        "instrument": "claude-code",
        "tier": "t2",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0,
            "shown": "0.0%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": null,
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "10 of 10",
            "repeatSetting": "no sampling control here",
            "time": "12.0 s a call, across every tip asked five times"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0,
            "shown": "0.0%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": null,
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "10 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.8 s a call, across every tip asked five times"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0,
            "shown": "0.0%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": null,
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "10 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.4 s a call, across every tip asked five times"
          }
        ],
        "state": "none",
        "tied": true,
        "sentence": "On getting clean json back, tested inside Claude Code, none of them managed it with no help."
      },
      {
        "job": {
          "id": "C17-exact-length",
          "kind": "tip",
          "label": "Getting the length right",
          "href": "/models/compare#ts-tip-single-turn-C17-exact-length",
          "scale": "unit",
          "measure": "score with no tip applied"
        },
        "jobPath": "/jobs/length-control",
        "instrument": "claude-code",
        "tier": "t2",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.0025000000000000022,
            "shown": "0.3%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": "close",
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "8 of 10",
            "repeatSetting": "no sampling control here",
            "time": "12.0 s a call, across every tip asked five times"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.014166666666666671,
            "shown": "1.4%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": "close",
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "7 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.8 s a call, across every tip asked five times"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.041666666666666664,
            "shown": "4.2%",
            "tier": "t2",
            "run": "tip-tier2-run",
            "mark": "best",
            "cost": "no cost published",
            "costUsd": null,
            "repeat": "7 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.4 s a call, across every tip asked five times"
          }
        ],
        "state": "tied",
        "tied": true,
        "sentence": "On getting the length right, tested inside Claude Code, they could not be told apart (Claude Opus 5 0.3%, Claude Sonnet 5 1.4%, Claude Haiku 4.5 4.2%)."
      },
      {
        "job": {
          "id": "C16-cutoff-disclosure",
          "kind": "tip",
          "label": "Questions about recent events",
          "href": "/models/compare#ts-tip-single-turn-C16-cutoff-disclosure",
          "scale": "unit",
          "measure": "score with no tip applied"
        },
        "jobPath": "/jobs/recent-events",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.8027777777777778,
            "shown": "80.3%",
            "tier": "t1",
            "run": "launch",
            "mark": "close",
            "cost": "$0.01566",
            "costUsd": 0.015659,
            "repeat": "23 of 30",
            "repeatSetting": "no sampling control here",
            "time": "12.0 s a call, across every tip asked five times"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.9333333333333333,
            "shown": "93.3%",
            "tier": "t1",
            "run": "launch",
            "mark": "close",
            "cost": "$0.00757",
            "costUsd": 0.007570266666666667,
            "repeat": "22 of 30",
            "repeatSetting": "no sampling control here",
            "time": "11.8 s a call, across every tip asked five times"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.9666666666666667,
            "shown": "96.7%",
            "tier": "t1",
            "run": "launch",
            "mark": "best",
            "cost": "$0.00220",
            "costUsd": 0.0021981,
            "repeat": "28 of 30",
            "repeatSetting": "no sampling control here",
            "time": "11.4 s a call, across every tip asked five times"
          }
        ],
        "state": "tied",
        "tied": true,
        "sentence": "On questions about recent events, tested inside Claude Code, they could not be told apart (Claude Opus 5 80.3%, Claude Sonnet 5 93.3%, Claude Haiku 4.5 96.7%)."
      },
      {
        "job": {
          "id": "C05-think-step-by-step",
          "kind": "tip",
          "label": "Better step-by-step answers",
          "href": "/models/compare#ts-tip-single-turn-C05-think-step-by-step",
          "scale": "unit",
          "measure": "score with no tip applied"
        },
        "jobPath": "/jobs/step-by-step",
        "instrument": "claude-code",
        "tier": "t1",
        "figures": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.4,
            "shown": "40.0%",
            "tier": "t1",
            "run": "launch",
            "mark": "best",
            "cost": "$0.00807",
            "costUsd": 0.00807,
            "repeat": "10 of 10",
            "repeatSetting": "no sampling control here",
            "time": "12.0 s a call, across every tip asked five times"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.4,
            "shown": "40.0%",
            "tier": "t1",
            "run": "launch",
            "mark": "best",
            "cost": "$0.00630",
            "costUsd": 0.006297,
            "repeat": "10 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.8 s a call, across every tip asked five times"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.2,
            "shown": "20.0%",
            "tier": "t1",
            "run": "launch",
            "mark": "close",
            "cost": "$0.00135",
            "costUsd": 0.00135,
            "repeat": "8 of 10",
            "repeatSetting": "no sampling control here",
            "time": "11.4 s a call, across every tip asked five times"
          }
        ],
        "state": "tied",
        "tied": true,
        "sentence": "On better step-by-step answers, tested inside Claude Code, they could not be told apart (Claude Opus 5 40.0%, Claude Sonnet 5 40.0%, Claude Haiku 4.5 20.0%)."
      }
    ],
    "jobs": 8,
    "tiedCount": 3,
    "rankable": 7,
    "differ": [
      "On code review, tested inside Claude Code, Claude Opus 5 (93.5%) and Claude Sonnet 5 (74.9%) scored higher than Claude Haiku 4.5 (58.3%).",
      "On on-page audit, tested inside Claude Code, Claude Opus 5 (88.1%) scored higher than Claude Sonnet 5 (53.0%) and Claude Haiku 4.5 (23.1%).",
      "On skill authoring, tested inside Claude Code, Claude Opus 5 (100.0%) and Claude Sonnet 5 (100.0%) scored higher than Claude Haiku 4.5 (67.3%).",
      "On spec writing, tested inside Claude Code, Claude Haiku 4.5 (99.2%) scored higher than Claude Opus 5 (92.3%) and Claude Sonnet 5 (96.2%)."
    ],
    "untested": [],
    "stops": [
      {
        "workflow": "link-graph-and-metadata-parity-audit",
        "workflowName": "Link graph and metadata parity audit",
        "rows": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "notTold": "went ahead 0 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "notTold": "went ahead 5 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "notTold": "went ahead 5 of 5",
            "told": "not run"
          }
        ]
      },
      {
        "workflow": "corpus-integrity-and-correction",
        "workflowName": "Corpus integrity and correction",
        "rows": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "notTold": "went ahead 0 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "notTold": "went ahead 0 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "notTold": "went ahead 2 of 4",
            "told": "not run"
          }
        ]
      },
      {
        "workflow": "traffic-drop-triage",
        "workflowName": "Traffic drop triage",
        "rows": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "notTold": "went ahead 0 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "notTold": "went ahead 0 of 5",
            "told": "went ahead 0 of 5"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "notTold": "went ahead 1 of 4",
            "told": "not run"
          }
        ]
      }
    ],
    "version": null,
    "methods": [
      "claude-code"
    ]
  }
}
