{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "spec-writing",
  "jobKind": "skill-class",
  "jobLabel": "Spec writing",
  "measure": "share of the required sections and acceptance criteria found",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/spec-writing",
  "sentence": "Writing a spec means turning a product idea into a written plan with the parts and the tests it is asked for. Each model was given the same ten tasks with no help, tested in the API and Claude Code.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.9954545454545455,
            "shown": "99.5%",
            "run": "skill-cohort"
          }
        ],
        "close": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.990909090909091,
            "shown": "99.1%",
            "run": "skill-cohort"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "pick": {
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.990909090909091,
          "usd": 0.00079535,
          "shown": "$0.00080",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "best": {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9954545454545455,
          "usd": 0.00332185,
          "shown": "$0.00332",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "pickIsBest": false,
        "basesDiffer": false
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In the API, Claude Haiku 4.5 scores highest. Gemini 3.1 Flash Lite is too close to call. Gemini 3.1 Flash Lite is the cheapest of those."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "best": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.9954545454545455,
            "shown": "99.5%",
            "run": "expansion-opus-5-5"
          }
        ],
        "close": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.9924242424242424,
            "shown": "99.2%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.9787878787878788,
            "shown": "97.9%",
            "run": "expansion-cohort"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "pick": {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9924242424242424,
          "usd": 0.003150683333333333,
          "shown": "$0.00315",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "best": {
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.9954545454545455,
          "usd": 0.04783986666666666,
          "shown": "$0.04784",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-opus-5-5",
          "runLabel": "Opus 5.5 series, wider run, three ways"
        },
        "pickIsBest": false,
        "basesDiffer": false
      },
      "consistent": {
        "measured": true,
        "run": "panel-v2",
        "runLabel": "Second run, every task twice",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 27,
            "repeated": 40,
            "share": 0.675,
            "shown": "27 of 40"
          }
        ],
        "close": []
      },
      "fastest": {
        "measured": false,
        "reason": "Only one model's calls were timed on this job this way."
      },
      "sentence": "In Claude Code, Claude Opus 5.5 scores highest. Claude Haiku 4.5 and Claude Fable 5.1 are too close to call. Claude Haiku 4.5 is the cheapest of those."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "tasks": null,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9954545454545455,
          "shown": "99.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.9454545454545455,
          "shown": "94.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.990909090909091,
          "shown": "99.1%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9924242424242424,
          "shown": "99.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 0.959090909090909,
          "shown": "95.9%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-fable-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.9787878787878788,
          "shown": "97.9%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.9227272727272726,
          "shown": "92.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.9954545454545455,
          "shown": "99.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-opus-5-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.9621212121212122,
          "shown": "96.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        }
      ]
    }
  ],
  "helps": [],
  "nothingLeftToMeasure": [
    {
      "instrument": "api",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "unaided": "99.5%",
      "tier": "t1"
    },
    {
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "unaided": "94.5%",
      "tier": "t1"
    },
    {
      "instrument": "api",
      "model": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash Lite",
      "unaided": "99.1%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "unaided": "99.2%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-fable-5",
      "name": "Claude Fable 5",
      "unaided": "95.9%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "unaided": "97.9%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "unaided": "92.3%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "unaided": "99.5%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "unaided": "96.2%",
      "tier": "t1"
    }
  ],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On spec writing, Claude Fable 5.1 finds 98% of required sections and acceptance criteria unaided against 93% for Claude Fable 5; scores higher.",
    "On spec writing, wider run, Claude Fable 5.1 finds 98% of required sections and acceptance criteria unaided against 96% for Claude Fable 5; scores higher."
  ]
}
