{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "accessibility-audit",
  "jobKind": "skill-class",
  "jobLabel": "Accessibility audit",
  "measure": "share of the seeded accessibility defects found",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/accessibility-audit",
  "sentence": "An accessibility audit means reading a web page and finding the problems that would stop someone using it. Each model was given the same 30 tasks with no help, tested in the API and Claude Code.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tier-calibration",
        "label": "Harder tasks, each model at its own tier",
        "date": "2026-09-18"
      },
      "joined": [],
      "best": {
        "measured": false,
        "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
      },
      "cheapest": {
        "measured": false,
        "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In the API, each model was given tasks at its own level of difficulty, so no one model is ranked first."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tier-calibration",
        "label": "Harder tasks, each model at its own tier",
        "date": "2026-09-18"
      },
      "joined": [
        {
          "id": "opus-5-5-tier-calibration",
          "label": "Harder tasks, Claude Opus 5.5 at its own tier",
          "date": "2026-09-26"
        }
      ],
      "best": {
        "measured": false,
        "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
      },
      "cheapest": {
        "measured": false,
        "reason": "Each model was given tasks at its own level of difficulty, so no one model is ranked first."
      },
      "consistent": {
        "measured": true,
        "run": "panel-v2",
        "runLabel": "Second run, every task twice",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 38,
            "repeated": 40,
            "share": 0.95,
            "shown": "38 of 40"
          }
        ],
        "close": []
      },
      "fastest": {
        "measured": false,
        "reason": "Only one model's calls were timed on this job this way."
      },
      "sentence": "In Claude Code, each model was given tasks at its own level of difficulty, so no one model is ranked first."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tier-calibration",
        "label": "Harder tasks, each model at its own tier",
        "date": "2026-09-18"
      },
      "joined": [],
      "tasks": 30,
      "tiers": [
        "t1",
        "t2"
      ],
      "entries": [
        {
          "kind": "absent",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.5874999999999999,
          "shown": "58.7%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tier-calibration",
        "label": "Harder tasks, each model at its own tier",
        "date": "2026-09-18"
      },
      "joined": [
        {
          "id": "opus-5-5-tier-calibration",
          "label": "Harder tasks, Claude Opus 5.5 at its own tier",
          "date": "2026-09-26"
        }
      ],
      "tasks": 30,
      "tiers": [
        "t1",
        "t2",
        "t3"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.4477777777777778,
          "shown": "44.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.8944444444444445,
          "shown": "89.4%",
          "tier": "t3",
          "tierWord": "tier 3",
          "mark": "close",
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.5611111111111111,
          "shown": "56.1%",
          "tier": "t3",
          "tierWord": "tier 3",
          "mark": null,
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.9277777777777777,
          "shown": "92.8%",
          "tier": "t3",
          "tierWord": "tier 3",
          "mark": "best",
          "run": "opus-5-5-tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.4333333333333333,
          "shown": "43.3%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tier-calibration",
          "repeat": "not asked twice",
          "repeatSetting": null,
          "atCeiling": false
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "skill",
      "instrument": "api",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/accessibility-audit",
      "href": "/skills/rampstackco-accessibility-audit",
      "lift": "+28.9 percentage points (+14.5 to +43.3)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/accessibility-audit",
      "href": "/skills/rampstackco-accessibility-audit",
      "lift": "+23.2 percentage points (+8.8 to +37.7)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "Second run, every task twice",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "engineering-team/a11y-audit/skills/a11y-audit",
      "href": "/skills/alirezarezvani-a11y-audit",
      "lift": "62.4% with it against 48.0% with no help",
      "instruction": "beat the one-line instruction by +20.5 percentage points (+9.7 to +31.3)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "plugins/accessibility-compliance/skills/wcag-audit-patterns",
      "href": "/skills/wshobson-wcag-audit-patterns",
      "lift": "+23.5 percentage points (+5.4 to +41.6)",
      "instruction": "could not be told apart from the one-line instruction (−2.7 to +29.2)",
      "runLabel": "accessibility-tier-skills",
      "tier": "t1"
    }
  ],
  "nothingLeftToMeasure": [
    {
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "unaided": "92.8%",
      "tier": "t3"
    }
  ],
  "honestStop": null,
  "tierNote": "Each model takes its next skills test at its own tier: the one where its score, with no skill loaded, leaves room to improve. A model that scores high everywhere gets the hardest tier. So a skill's result can be compared with other skills on the same model, but not used to rank one model against another, because the models did different work.",
  "versionPairs": [
    "On accessibility audit, wider run, Claude Fable 5.1 finds 88% of seeded accessibility defects unaided against 90% for Claude Fable 5; no measurable change."
  ]
}
