{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "C05-think-step-by-step",
  "jobKind": "tip",
  "jobLabel": "Better step-by-step answers",
  "measure": "score with no tip applied",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/step-by-step",
  "sentence": "Step-by-step answers means a problem with one exact answer that takes several steps to reach. Each model was given the same ten tasks with no help, tested in the API, Claude Code and Codex.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "launch",
        "label": "Main run, tip by tip",
        "date": "2026-08-17"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run, tip by tip",
        "tier": "t1",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "launch"
          }
        ],
        "close": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.3,
            "shown": "30.0%",
            "run": "launch"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": false,
        "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
      },
      "consistent": {
        "measured": false,
        "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "medianMs": 563,
            "shown": "563 ms"
          }
        ],
        "models": 2
      },
      "sentence": "In the API, Gemini 3.1 Flash Lite scores highest. Claude Haiku 4.5 is too close to call."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "launch",
        "label": "Main run",
        "date": "2026-09-03"
      },
      "joined": [
        {
          "id": "reliability-opus-5-5",
          "label": "Claude Opus 5.5, asked each task five times",
          "date": "2026-09-28"
        }
      ],
      "best": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "launch"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "launch"
          },
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "launch"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "launch"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "reliability-opus-5-5"
          }
        ],
        "close": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.2,
            "shown": "20.0%",
            "run": "launch"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run",
        "pick": {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.2,
          "usd": 0.00135,
          "shown": "$0.00135",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "launch",
          "runLabel": "Main run"
        },
        "best": {
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.4,
          "usd": 0.006297,
          "shown": "$0.00630",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "launch",
          "runLabel": "Main run"
        },
        "pickIsBest": false,
        "basesDiffer": false
      },
      "consistent": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "identical": 10,
            "repeated": 10,
            "share": 1,
            "shown": "10 of 10"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "identical": 10,
            "repeated": 10,
            "share": 1,
            "shown": "10 of 10"
          },
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 10,
            "repeated": 10,
            "share": 1,
            "shown": "10 of 10"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "identical": 10,
            "repeated": 10,
            "share": 1,
            "shown": "10 of 10"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "identical": 10,
            "repeated": 10,
            "share": 1,
            "shown": "10 of 10"
          }
        ],
        "close": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "identical": 8,
            "repeated": 10,
            "share": 0.8,
            "shown": "8 of 10"
          }
        ]
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "medianMs": 11372,
            "shown": "11.4 s"
          }
        ],
        "models": 6
      },
      "sentence": "In Claude Code, Claude Fable 5, Claude Fable 5.1, Claude Opus 5, Claude Sonnet 5 and Claude Opus 5.5 tie for the top score. Claude Haiku 4.5 is too close to call. Claude Haiku 4.5 is the cheapest of those."
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "codex-tips-run",
        "runLabel": "Codex run, tip by tip",
        "tier": "t1",
        "leaders": [
          {
            "model": "gpt-6-sol",
            "name": "GPT-6 Sol",
            "mean": 0.4,
            "shown": "40.0%",
            "run": "codex-tips-run"
          }
        ],
        "close": [],
        "rangeAbsent": true
      },
      "cheapest": {
        "measured": false,
        "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In Codex, GPT-6 Sol scores highest."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "launch",
        "label": "Main run, tip by tip",
        "date": "2026-08-17"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "20 of 20",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.16,
          "shown": "16.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "at the vendor default",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "launch",
        "label": "Main run",
        "date": "2026-09-03"
      },
      "joined": [
        {
          "id": "reliability-opus-5-5",
          "label": "Claude Opus 5.5, asked each task five times",
          "date": "2026-09-28"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.2,
          "shown": "20.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "8 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "reliability-opus-5-5",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        }
      ]
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-luna",
          "name": "GPT-5.6 Luna",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-terra",
          "name": "GPT-5.6 Terra",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-6-luna",
          "name": "GPT-6 Luna",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-6-sol",
          "name": "GPT-6 Sol",
          "mean": 0.4,
          "shown": "40.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "tip",
      "instrument": "api",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "Better step-by-step answers",
      "href": "/tips/think-step-by-step",
      "lift": "+20.0 percentage points (+1.0 to +39.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Main run",
      "tier": "t1"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "Better step-by-step answers",
      "href": "/tips/think-step-by-step",
      "lift": "+34.0 percentage points (+16.6 to +51.4)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Main run",
      "tier": "t1"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "Better step-by-step answers",
      "href": "/tips/think-step-by-step",
      "lift": "+30.0 percentage points (+12.1 to +47.9)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Main run",
      "tier": "t1"
    }
  ],
  "nothingLeftToMeasure": [],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On better step-by-step answers, Claude Fable 5.1 scores 40% unaided against 40% for Claude Fable 5; no measurable change."
  ]
}
