{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "on-page-audit",
  "jobKind": "skill-class",
  "jobLabel": "On-page audit",
  "measure": "share of the seeded on-page issues found",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/on-page-audit",
  "sentence": "An on-page SEO audit means reading a web page and finding the search problems that were put in it. Each model was given the same ten tasks with no help, tested in the API and Claude Code.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "tier": "t1",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.488095238095238,
            "shown": "48.8%",
            "run": "skill-cohort"
          }
        ],
        "close": [
          {
            "model": "gpt-5-mini",
            "name": "GPT-5 mini",
            "mean": 0.4492857142857143,
            "shown": "44.9%",
            "run": "skill-cohort"
          },
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.23928571428571427,
            "shown": "23.9%",
            "run": "skill-cohort"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "pick": {
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.488095238095238,
          "usd": 0.000228925,
          "shown": "$0.00023",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "best": {
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.488095238095238,
          "usd": 0.000228925,
          "shown": "$0.00023",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "pickIsBest": true,
        "basesDiffer": false
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini and Claude Haiku 4.5 are too close to call."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "best": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.9722222222222223,
            "shown": "97.2%",
            "run": "expansion-cohort"
          }
        ],
        "close": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.8805555555555555,
            "shown": "88.1%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.9333333333333333,
            "shown": "93.3%",
            "run": "expansion-opus-5-5"
          },
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "mean": 0.9321428571428572,
            "shown": "93.2%",
            "run": "expansion-fable-5"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "pick": {
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.8805555555555555,
          "usd": 0.0071491666666666665,
          "shown": "$0.00715",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "best": {
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.9722222222222223,
          "usd": 0.03873,
          "shown": "$0.03873",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "pickIsBest": false,
        "basesDiffer": false
      },
      "consistent": {
        "measured": true,
        "run": "panel-v2",
        "runLabel": "Second run, every task twice",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 33,
            "repeated": 40,
            "share": 0.825,
            "shown": "33 of 40"
          }
        ],
        "close": []
      },
      "fastest": {
        "measured": false,
        "reason": "Only one model's calls were timed on this job this way."
      },
      "sentence": "In Claude Code, Claude Fable 5.1 scores highest. Claude Opus 5, Claude Opus 5.5 and Claude Fable 5 are too close to call. Claude Opus 5 is the cheapest of those."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "tasks": null,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.23928571428571427,
          "shown": "23.9%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.4492857142857143,
          "shown": "44.9%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.488095238095238,
          "shown": "48.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.23095238095238094,
          "shown": "23.1%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 0.9321428571428572,
          "shown": "93.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-fable-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.9722222222222223,
          "shown": "97.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.8805555555555555,
          "shown": "88.1%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.9333333333333333,
          "shown": "93.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-opus-5-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.5297619047619048,
          "shown": "53.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash Lite",
      "what": "skills/seo-onpage",
      "href": "/skills/rampstackco-seo-onpage",
      "lift": "+21.9 percentage points (+4.6 to +39.2)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    }
  ],
  "nothingLeftToMeasure": [
    {
      "instrument": "claude-code",
      "model": "claude-fable-5",
      "name": "Claude Fable 5",
      "unaided": "93.2%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "unaided": "97.2%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "unaided": "93.3%",
      "tier": "t1"
    }
  ],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On on-page audit, Claude Fable 5.1 finds 98% of seeded on-page issues unaided against 92% for Claude Fable 5; scores higher.",
    "On on-page audit, wider run, Claude Fable 5.1 finds 97% of seeded on-page issues unaided against 93% for Claude Fable 5; no measurable change."
  ]
}
