{
  "surface": "openaddict-comparison",
  "version": "comparison-v1",
  "url": "https://openaddict.com/compare/claude-vs-gpt",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Models are compared on a job only where the jobs-by-models grid holds a figure for each of them on one instrument at one tier, read from one run or from runs that asked exactly the same units. A model page exists only where that holds on at least 3 jobs. Who is ahead is the overlap rule over the compared models alone. Nothing crosses instruments.",
  "line": "Five jobs with the same work, tested through the API. They could not be told apart on 3 of the 5 that could be ranked.",
  "page": {
    "kind": "vendor",
    "copy": {
      "slug": "claude-vs-gpt",
      "kind": "vendor",
      "models": [
        "claude-haiku-4-5",
        "gpt-5-mini"
      ],
      "short": "Claude and GPT"
    },
    "pair": {
      "kind": "models",
      "copy": {
        "slug": "claude-vs-gpt",
        "kind": "vendor",
        "models": [
          "claude-haiku-4-5",
          "gpt-5-mini"
        ],
        "short": "Claude and GPT"
      },
      "models": [
        {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5"
        },
        {
          "model": "gpt-5-mini",
          "name": "GPT-5 mini"
        }
      ],
      "shared": [
        {
          "job": {
            "id": "on-page-audit",
            "kind": "skill-class",
            "label": "On-page audit",
            "href": "/models/compare#ts-skill-class-on-page-audit",
            "scale": "unit",
            "measure": "share of the seeded on-page issues found"
          },
          "jobPath": "/jobs/on-page-audit",
          "instrument": "api",
          "tier": "t1",
          "figures": [
            {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.23928571428571427,
              "shown": "23.9%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": "close",
              "cost": "$0.00085",
              "costUsd": 0.0008533000000000001,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            },
            {
              "model": "gpt-5-mini",
              "name": "GPT-5 mini",
              "mean": 0.4492857142857143,
              "shown": "44.9%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": "best",
              "cost": "$0.00029",
              "costUsd": 0.00029170000000000004,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            }
          ],
          "state": "tied",
          "tied": true,
          "sentence": "On on-page audit, tested through the API, they could not be told apart (Claude Haiku 4.5 23.9%, GPT-5 mini 44.9%)."
        },
        {
          "job": {
            "id": "skill-authoring",
            "kind": "skill-class",
            "label": "Skill authoring",
            "href": "/models/compare#ts-skill-class-skill-authoring",
            "scale": "unit",
            "measure": "share of the specification requirements found"
          },
          "jobPath": "/jobs/skill-authoring",
          "instrument": "api",
          "tier": "t1",
          "figures": [
            {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.6181818181818179,
              "shown": "61.8%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": "close",
              "cost": "$0.00618",
              "costUsd": 0.00617665,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            },
            {
              "model": "gpt-5-mini",
              "name": "GPT-5 mini",
              "mean": 0.7318181818181818,
              "shown": "73.2%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": "best",
              "cost": "$0.00297",
              "costUsd": 0.0029716250000000003,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            }
          ],
          "state": "tied",
          "tied": true,
          "sentence": "On skill authoring, tested through the API, they could not be told apart (Claude Haiku 4.5 61.8%, GPT-5 mini 73.2%)."
        },
        {
          "job": {
            "id": "spec-writing",
            "kind": "skill-class",
            "label": "Spec writing",
            "href": "/models/compare#ts-skill-class-spec-writing",
            "scale": "unit",
            "measure": "share of the required sections and acceptance criteria found"
          },
          "jobPath": "/jobs/spec-writing",
          "instrument": "api",
          "tier": "t1",
          "figures": [
            {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.9954545454545455,
              "shown": "99.5%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": "best",
              "cost": "$0.00332",
              "costUsd": 0.00332185,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            },
            {
              "model": "gpt-5-mini",
              "name": "GPT-5 mini",
              "mean": 0.9454545454545455,
              "shown": "94.5%",
              "tier": "t1",
              "run": "skill-cohort",
              "mark": null,
              "cost": "$0.00248",
              "costUsd": 0.00248005,
              "repeat": "one pass, not measured",
              "repeatSetting": null,
              "time": "not timed"
            }
          ],
          "state": "apart",
          "tied": false,
          "sentence": "On spec writing, tested through the API, Claude Haiku 4.5 (99.5%) scored higher than GPT-5 mini (94.5%)."
        },
        {
          "job": {
            "id": "C16-cutoff-disclosure",
            "kind": "tip",
            "label": "Questions about recent events",
            "href": "/models/compare#ts-tip-single-turn-C16-cutoff-disclosure",
            "scale": "unit",
            "measure": "score with no tip applied"
          },
          "jobPath": "/jobs/recent-events",
          "instrument": "api",
          "tier": "t1",
          "figures": [
            {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.8555555555555556,
              "shown": "85.6%",
              "tier": "t1",
              "run": "launch",
              "mark": "best",
              "cost": "Nothing was billed on this run for the version being priced here.",
              "costUsd": null,
              "repeat": "28 of 30",
              "repeatSetting": "temperature 0 asked for",
              "time": "not timed"
            },
            {
              "model": "gpt-5-mini",
              "name": "GPT-5 mini",
              "mean": 0.4861111111111111,
              "shown": "48.6%",
              "tier": "t1",
              "run": "launch",
              "mark": null,
              "cost": "Nothing was billed on this run for the version being priced here.",
              "costUsd": null,
              "repeat": "6 of 30",
              "repeatSetting": "at the vendor default",
              "time": "1.5 s a call, across every tip asked five times"
            }
          ],
          "state": "apart",
          "tied": false,
          "sentence": "On questions about recent events, tested through the API, Claude Haiku 4.5 (85.6%) scored higher than GPT-5 mini (48.6%)."
        },
        {
          "job": {
            "id": "C05-think-step-by-step",
            "kind": "tip",
            "label": "Better step-by-step answers",
            "href": "/models/compare#ts-tip-single-turn-C05-think-step-by-step",
            "scale": "unit",
            "measure": "score with no tip applied"
          },
          "jobPath": "/jobs/step-by-step",
          "instrument": "api",
          "tier": "t1",
          "figures": [
            {
              "model": "claude-haiku-4-5",
              "name": "Claude Haiku 4.5",
              "mean": 0.3,
              "shown": "30.0%",
              "tier": "t1",
              "run": "launch",
              "mark": "best",
              "cost": "Nothing was billed on this run for the version being priced here.",
              "costUsd": null,
              "repeat": "20 of 20",
              "repeatSetting": "at temperature 0",
              "time": "not timed"
            },
            {
              "model": "gpt-5-mini",
              "name": "GPT-5 mini",
              "mean": 0.16,
              "shown": "16.0%",
              "tier": "t1",
              "run": "launch",
              "mark": "close",
              "cost": "Nothing was billed on this run for the version being priced here.",
              "costUsd": null,
              "repeat": "10 of 10",
              "repeatSetting": "at the vendor default",
              "time": "1.5 s a call, across every tip asked five times"
            }
          ],
          "state": "tied",
          "tied": true,
          "sentence": "On better step-by-step answers, tested through the API, they could not be told apart (Claude Haiku 4.5 30.0%, GPT-5 mini 16.0%)."
        }
      ],
      "jobs": 5,
      "tiedCount": 3,
      "rankable": 5,
      "differ": [
        "On spec writing, tested through the API, Claude Haiku 4.5 (99.5%) scored higher than GPT-5 mini (94.5%).",
        "On questions about recent events, tested through the API, Claude Haiku 4.5 (85.6%) scored higher than GPT-5 mini (48.6%)."
      ],
      "untested": [],
      "stops": [],
      "version": null,
      "methods": [
        "api"
      ]
    },
    "split": [
      {
        "instrument": "api",
        "claude": [
          "Claude Haiku 4.5"
        ],
        "gpt": [
          "GPT-5 mini",
          "GPT-5.4 mini"
        ]
      },
      {
        "instrument": "claude-code",
        "claude": [
          "Claude Haiku 4.5",
          "Claude Fable 5",
          "Claude Fable 5.1",
          "Claude Opus 5",
          "Claude Opus 5.5",
          "Claude Sonnet 5"
        ],
        "gpt": []
      },
      {
        "instrument": "codex-cli",
        "claude": [],
        "gpt": [
          "GPT-5.4 mini",
          "GPT-5.6 Luna",
          "GPT-5.6 Terra",
          "GPT-6 Luna",
          "GPT-6 Sol"
        ]
      }
    ],
    "notComparable": {
      "claude": [
        {
          "method": "claude-code",
          "models": [
            "Claude Haiku 4.5",
            "Claude Fable 5",
            "Claude Fable 5.1",
            "Claude Opus 5",
            "Claude Opus 5.5",
            "Claude Sonnet 5"
          ]
        }
      ],
      "gpt": [
        {
          "method": "codex-cli",
          "models": [
            "GPT-5.4 mini",
            "GPT-5.6 Luna",
            "GPT-5.6 Terra",
            "GPT-6 Luna",
            "GPT-6 Sol"
          ]
        }
      ]
    },
    "offsets": [
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Accessibility audit",
        "runs": "First run, with and without the skill against Second run, every task twice",
        "offset": "+5.5 percentage points",
        "range": "−11.7 to +22.8"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Accessibility audit",
        "runs": "First run, with and without the skill against Wider run, three ways",
        "offset": "+6.1 percentage points",
        "range": "−12.1 to +24.3"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "On-page audit",
        "runs": "First run, with and without the skill against Second run, every task twice",
        "offset": "−0.8 percentage points",
        "range": "−28.4 to +26.8"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "On-page audit",
        "runs": "First run, with and without the skill against Wider run, three ways",
        "offset": "−0.8 percentage points",
        "range": "−30.3 to +28.7"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Skill authoring",
        "runs": "First run, with and without the skill against Second run, every task twice",
        "offset": "+8.2 percentage points",
        "range": "−3.4 to +19.7"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Skill authoring",
        "runs": "First run, with and without the skill against Wider run, three ways",
        "offset": "+5.5 percentage points",
        "range": "−6.7 to +17.6"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Spec writing",
        "runs": "First run, with and without the skill against Second run, every task twice",
        "offset": "−0.9 percentage points",
        "range": "−2.2 to +0.4"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Spec writing",
        "runs": "First run, with and without the skill against Wider run, three ways",
        "offset": "−0.3 percentage points",
        "range": "−1.3 to +0.7"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Getting clean JSON back",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "+0.0 to +0.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Instructions before or after",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "+0.0 to +0.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Using tags and formatting",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "+0.0 to +0.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Showing examples",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "−6.0 to +6.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Showing examples",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "−6.0 to +6.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Better step-by-step answers",
        "runs": "Main run, tip by tip against Main run",
        "offset": "−10.0 percentage points",
        "range": "−27.0 to +7.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Stop made-up answers",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+22.0 percentage points",
        "range": "+10.4 to +33.6"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Instructions in long prompts",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+0.0 percentage points",
        "range": "+0.0 to +0.0"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Offering the model money",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+37 words",
        "range": "+3019.2 to +4440.8"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Questions about recent events",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+11.1 percentage points",
        "range": "−2.9 to +25.1"
      },
      {
        "bridge": "Claude Haiku 4.5",
        "from": "api",
        "to": "claude-code",
        "job": "Getting the length right",
        "runs": "Main run, tip by tip against Main run",
        "offset": "+2.8 percentage points",
        "range": "−33.3 to +38.9"
      },
      {
        "bridge": "GPT-5.4 mini",
        "from": "api",
        "to": "codex-cli",
        "job": "Questions about recent events",
        "runs": "API twin for the Codex run against Codex run, replaying two earlier tests",
        "offset": "−12.5 percentage points",
        "range": "no range published"
      },
      {
        "bridge": "GPT-5.4 mini",
        "from": "api",
        "to": "codex-cli",
        "job": "Getting the length right",
        "runs": "API twin for the Codex run against Codex run, replaying two earlier tests",
        "offset": "+2.9 percentage points",
        "range": "no range published"
      },
      {
        "bridge": "GPT-5.4 mini",
        "from": "api",
        "to": "codex-cli",
        "job": "Getting the length right",
        "runs": "API twin for the Codex run against Codex run, tip by tip",
        "offset": "+4.9 percentage points",
        "range": "no range published"
      }
    ]
  }
}
