{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "C01-json-schema",
  "jobKind": "tip",
  "jobLabel": "Getting clean JSON back",
  "measure": "score with no tip applied",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/json-extraction",
  "sentence": "Getting clean JSON back means asking for data in an exact format and getting nothing else in the reply. Each model was given the same ten tasks with no help, tested in the API, Claude Code and Codex.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [],
      "best": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "cheapest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "consistent": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "fastest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "sentence": "In the API, no model managed this on its own."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [
        {
          "id": "opus-5-5-tips",
          "label": "Claude Opus 5.5, tier 2",
          "date": "2026-09-25"
        }
      ],
      "best": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "cheapest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "consistent": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "fastest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "sentence": "In Claude Code, no model managed this on its own."
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "best": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "cheapest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "consistent": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "fastest": {
        "measured": false,
        "reason": "No model managed this on its own."
      },
      "sentence": "In Codex, no model managed this on its own."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t2"
      ],
      "entries": [
        {
          "kind": "absent",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "at the vendor default",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "not run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [
        {
          "id": "opus-5-5-tips",
          "label": "Claude Opus 5.5, tier 2",
          "date": "2026-09-25"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t2"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "opus-5-5-tips",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": null,
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        }
      ]
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-luna",
          "name": "GPT-5.6 Luna",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-terra",
          "name": "GPT-5.6 Terra",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-6-luna",
          "name": "GPT-6 Luna",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-6-sol",
          "name": "GPT-6 Sol",
          "mean": 0,
          "shown": "0.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+80.0 percentage points (+53.9 to +106.1)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash Lite",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-5.6-terra",
      "name": "GPT-5.6 Terra",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-5.6-luna",
      "name": "GPT-5.6 Luna",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-6-sol",
      "name": "GPT-6 Sol",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-6-luna",
      "name": "GPT-6 Luna",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "what": "Getting clean JSON back",
      "href": "/tips/json-schema",
      "lift": "+100.0 percentage points (+100.0 to +100.0)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Claude Opus 5.5, tier 2",
      "tier": "t2"
    }
  ],
  "nothingLeftToMeasure": [],
  "honestStop": null,
  "tierNote": "The first tasks got too easy to tell the models apart, so we built a harder set, tier 2, and the scores here are from it. In Codex the harder set was not run, so those scores are from tier 1.",
  "versionPairs": [
    "On getting clean JSON back, Claude Fable 5.1 scores 0% unaided against 0% for Claude Fable 5; could not be measured."
  ]
}
