{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "skill-authoring",
  "jobKind": "skill-class",
  "jobLabel": "Skill authoring",
  "measure": "share of the specification requirements found",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/skill-authoring",
  "sentence": "Writing a skill means turning a request into a skill file that meets a written list of needs. Each model was given the same ten tasks with no help, tested in the API and Claude Code.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "tier": "t1",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.8727272727272727,
            "shown": "87.3%",
            "run": "skill-cohort"
          }
        ],
        "close": [
          {
            "model": "gpt-5-mini",
            "name": "GPT-5 mini",
            "mean": 0.7318181818181818,
            "shown": "73.2%",
            "run": "skill-cohort"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "skill-cohort",
        "runLabel": "First run, with and without the skill",
        "pick": {
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.8727272727272727,
          "usd": 0.000884075,
          "shown": "$0.00088",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "best": {
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.8727272727272727,
          "usd": 0.000884075,
          "shown": "$0.00088",
          "basis": "billed",
          "label": "billed through the API",
          "run": "skill-cohort",
          "runLabel": "First run, with and without the skill"
        },
        "pickIsBest": true,
        "basesDiffer": false
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. GPT-5 mini is too close to call."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "best": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 1,
            "shown": "100.0%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 1,
            "shown": "100.0%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 1,
            "shown": "100.0%",
            "run": "expansion-cohort"
          },
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "mean": 1,
            "shown": "100.0%",
            "run": "expansion-fable-5"
          }
        ],
        "close": [
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.9977272727272727,
            "shown": "99.8%",
            "run": "expansion-opus-5-5"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "expansion-cohort",
        "runLabel": "Wider run, three ways",
        "pick": {
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 1,
          "usd": 0.020272199999999997,
          "shown": "$0.02027",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "best": {
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 1,
          "usd": 0.020272199999999997,
          "shown": "$0.02027",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "expansion-cohort",
          "runLabel": "Wider run, three ways"
        },
        "pickIsBest": true,
        "basesDiffer": false
      },
      "consistent": {
        "measured": true,
        "run": "panel-v2",
        "runLabel": "Second run, every task twice",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 28,
            "repeated": 29,
            "share": 0.9655172413793104,
            "shown": "28 of 29"
          }
        ],
        "close": []
      },
      "fastest": {
        "measured": false,
        "reason": "Only one model's calls were timed on this job this way."
      },
      "sentence": "In Claude Code, Claude Fable 5.1, Claude Opus 5, Claude Sonnet 5 and Claude Fable 5 tie for the top score. Claude Opus 5.5 is too close to call."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "skill-cohort",
        "label": "First run, with and without the skill",
        "date": "2026-08-30"
      },
      "joined": [],
      "tasks": null,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.6181818181818179,
          "shown": "61.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.7318181818181818,
          "shown": "73.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.8727272727272727,
          "shown": "87.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "skill-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "expansion-cohort",
        "label": "Wider run, three ways",
        "date": "2026-09-06"
      },
      "joined": [
        {
          "id": "expansion-opus-5-5",
          "label": "Opus 5.5 series, wider run, three ways",
          "date": "2026-09-26"
        },
        {
          "id": "expansion-fable-5",
          "label": "Fable 5, wider run, three ways",
          "date": "2026-09-19"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.6727272727272728,
          "shown": "67.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 1,
          "shown": "100.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-fable-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 1,
          "shown": "100.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 1,
          "shown": "100.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.9977272727272727,
          "shown": "99.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "expansion-opus-5-5",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 1,
          "shown": "100.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "expansion-cohort",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": true
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/skill-creation-walkthrough",
      "href": "/skills/rampstackco-skill-creation-walkthrough",
      "lift": "+34.5 percentage points (+17.6 to +51.5)",
      "instruction": "beat the one-line instruction by +38.2 percentage points (+19.3 to +57.1)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/skill-creator",
      "href": "/skills/anthropics-skill-creator",
      "lift": "+22.7 percentage points (+10.5 to +35.0)",
      "instruction": "beat the one-line instruction by +33.6 percentage points (+19.3 to +48.0)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creation-walkthrough",
      "href": "/skills/rampstackco-skill-creation-walkthrough",
      "lift": "+38.2 percentage points (+28.7 to +47.7)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/skill-creation-walkthrough",
      "href": "/skills/rampstackco-skill-creation-walkthrough",
      "lift": "+28.2 percentage points (+12.8 to +43.5)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "distribution/claude-plugin/skills/skill-builder",
      "href": "/skills/yusufkaraaslan-skill-builder",
      "lift": "+44.5 percentage points (+26.8 to +62.3)",
      "instruction": "beat the one-line instruction by +27.3 percentage points (+14.8 to +39.7)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creator",
      "href": "/skills/anthropics-skill-creator",
      "lift": "+38.2 percentage points (+28.7 to +47.7)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/skill-creator",
      "href": "/skills/anthropics-skill-creator",
      "lift": "+27.3 percentage points (+8.3 to +46.2)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "First run, with and without the skill",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "skills/skill-creator",
      "href": "/skills/openclaw-skill-creator",
      "lift": "+35.6 percentage points (+19.8 to +51.4)",
      "instruction": "could not be told apart from the one-line instruction (+13.7 to +24.8)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creation-walkthrough",
      "href": "/skills/rampstackco-skill-creation-walkthrough",
      "lift": "+30.9 percentage points (+22.0 to +39.8)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "Second run, every task twice",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creator",
      "href": "/skills/anthropics-skill-creator",
      "lift": "+29.1 percentage points (+22.7 to +35.5)",
      "instruction": "the one-line instruction was not asked on this run",
      "runLabel": "Second run, every task twice",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creation-walkthrough",
      "href": "/skills/rampstackco-skill-creation-walkthrough",
      "lift": "100.0% with it against 69.1% with no help",
      "instruction": "beat the one-line instruction by +45.5 percentage points (+45.5 to +45.5)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creator",
      "href": "/skills/anthropics-skill-creator",
      "lift": "100.0% with it against 69.1% with no help",
      "instruction": "beat the one-line instruction by +43.6 percentage points (+35.3 to +52.0)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "skills/skill-creator",
      "href": "/skills/openclaw-skill-creator",
      "lift": "100.0% with it against 65.5% with no help",
      "instruction": "beat the one-line instruction by +34.5 percentage points (+23.7 to +45.4)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    },
    {
      "kind": "skill",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "distribution/claude-plugin/skills/skill-builder",
      "href": "/skills/yusufkaraaslan-skill-builder",
      "lift": "100.0% with it against 65.5% with no help",
      "instruction": "beat the one-line instruction by +40.0 percentage points (+29.3 to +50.7)",
      "runLabel": "Wider run, three ways",
      "tier": "t1"
    }
  ],
  "nothingLeftToMeasure": [
    {
      "instrument": "claude-code",
      "model": "claude-fable-5",
      "name": "Claude Fable 5",
      "unaided": "100.0%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "unaided": "100.0%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "unaided": "100.0%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "unaided": "99.8%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "unaided": "100.0%",
      "tier": "t1"
    }
  ],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On skill authoring, Claude Fable 5.1 finds 100% of specification requirements unaided against 100% for Claude Fable 5; could not be measured.",
    "On skill authoring, wider run, Claude Fable 5.1 finds 100% of specification requirements unaided against 100% for Claude Fable 5; could not be measured."
  ]
}
