{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "C17-exact-length",
  "jobKind": "tip",
  "jobLabel": "Getting the length right",
  "measure": "score with no tip applied",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/length-control",
  "sentence": "Getting the length right means asking for an answer of an exact number of words and getting it. Each model was given the same ten tasks with no help, tested in the API, Claude Code and Codex.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "tip-tier2-run",
        "runLabel": "Harder tasks, tier 2",
        "tier": "t2",
        "leaders": [
          {
            "model": "gpt-5-mini",
            "name": "GPT-5 mini",
            "mean": 0.05,
            "shown": "5.0%",
            "run": "tip-tier2-run"
          }
        ],
        "close": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "mean": 0.03666666666666667,
            "shown": "3.7%",
            "run": "tip-tier2-run"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": false,
        "reason": "This run does not publish a cost per task."
      },
      "consistent": {
        "measured": false,
        "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "medianMs": 563,
            "shown": "563 ms"
          }
        ],
        "models": 2
      },
      "sentence": "In the API, GPT-5 mini scores highest. Gemini 3.1 Flash Lite is too close to call."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [
        {
          "id": "opus-5-5-tips",
          "label": "Claude Opus 5.5, tier 2",
          "date": "2026-09-25"
        }
      ],
      "best": {
        "measured": true,
        "run": "tip-tier2-run",
        "runLabel": "Harder tasks, tier 2",
        "tier": "t2",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.041666666666666664,
            "shown": "4.2%",
            "run": "tip-tier2-run"
          }
        ],
        "close": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.0025000000000000022,
            "shown": "0.3%",
            "run": "tip-tier2-run"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.014166666666666671,
            "shown": "1.4%",
            "run": "tip-tier2-run"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.0025000000000000022,
            "shown": "0.3%",
            "run": "tip-tier2-run"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "mean": 0.0025000000000000022,
            "shown": "0.3%",
            "run": "opus-5-5-tips"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": false,
        "reason": "This run does not publish a cost per task."
      },
      "consistent": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 8,
            "repeated": 10,
            "share": 0.8,
            "shown": "8 of 10"
          }
        ],
        "close": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "identical": 7,
            "repeated": 10,
            "share": 0.7,
            "shown": "7 of 10"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "identical": 4,
            "repeated": 10,
            "share": 0.4,
            "shown": "4 of 10"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "identical": 7,
            "repeated": 10,
            "share": 0.7,
            "shown": "7 of 10"
          }
        ]
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "medianMs": 11372,
            "shown": "11.4 s"
          }
        ],
        "models": 6
      },
      "sentence": "In Claude Code, Claude Haiku 4.5 scores highest. Claude Opus 5, Claude Sonnet 5, Claude Fable 5.1 and Claude Opus 5.5 are too close to call."
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "codex-tips-run",
        "runLabel": "Codex run, tip by tip",
        "tier": "t1",
        "leaders": [
          {
            "model": "gpt-5.6-terra",
            "name": "GPT-5.6 Terra",
            "mean": 0.6235999999999999,
            "shown": "62.4%",
            "run": "codex-tips-run"
          }
        ],
        "close": [],
        "rangeAbsent": true
      },
      "cheapest": {
        "measured": false,
        "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In Codex, GPT-5.6 Terra scores highest."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t2"
      ],
      "entries": [
        {
          "kind": "absent",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.05,
          "shown": "5.0%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "best",
          "run": "tip-tier2-run",
          "repeat": "8 of 10",
          "repeatSetting": "at the vendor default",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.03666666666666667,
          "shown": "3.7%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "close",
          "run": "tip-tier2-run",
          "repeat": "10 of 10",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "in another run",
          "reason": null
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "tip-tier2-run",
        "label": "Harder tasks, tier 2",
        "date": "2026-09-19"
      },
      "joined": [
        {
          "id": "opus-5-5-tips",
          "label": "Claude Opus 5.5, tier 2",
          "date": "2026-09-25"
        }
      ],
      "tasks": 10,
      "tiers": [
        "t2"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.041666666666666664,
          "shown": "4.2%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "best",
          "run": "tip-tier2-run",
          "repeat": "7 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.0025000000000000022,
          "shown": "0.3%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "close",
          "run": "tip-tier2-run",
          "repeat": "4 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.0025000000000000022,
          "shown": "0.3%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "close",
          "run": "tip-tier2-run",
          "repeat": "8 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "mean": 0.0025000000000000022,
          "shown": "0.3%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "close",
          "run": "opus-5-5-tips",
          "repeat": "0 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.014166666666666671,
          "shown": "1.4%",
          "tier": "t2",
          "tierWord": "tier 2",
          "mark": "close",
          "run": "tip-tier2-run",
          "repeat": "7 of 10",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        }
      ]
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "tasks": 10,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "mean": 0.20600000000000002,
          "shown": "20.6%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-luna",
          "name": "GPT-5.6 Luna",
          "mean": 0.5768000000000001,
          "shown": "57.7%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-terra",
          "name": "GPT-5.6 Terra",
          "mean": 0.6235999999999999,
          "shown": "62.4%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-6-luna",
          "name": "GPT-6 Luna",
          "mean": 0.51,
          "shown": "51.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-6-sol",
          "name": "GPT-6 Sol",
          "word": "withheld",
          "reason": "Incomplete coverage at retry exhaustion; 1 registered units have no valid record. Resume only within registered retry policy."
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-opus-5",
      "name": "Claude Opus 5",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+93.1 percentage points (+90.2 to +95.9)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+93.3 percentage points (+88.2 to +98.5)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+77.4 percentage points (+67.9 to +86.9)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+86.5 percentage points (+76.7 to +96.3)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash Lite",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+95.4 percentage points (+88.7 to +102.1)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+99.8 percentage points (+99.3 to +100.2)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-5.6-terra",
      "name": "GPT-5.6 Terra",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+66.0 percentage points (+45.1 to +86.9)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-5.6-luna",
      "name": "GPT-5.6 Luna",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+62.6 percentage points (+38.3 to +86.8)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-6-sol",
      "name": "GPT-6 Sol",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+71.7 percentage points (+51.2 to +92.1)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-6-luna",
      "name": "GPT-6 Luna",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+79.3 percentage points (+68.2 to +90.3)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "tier 2",
      "tier": "t2"
    },
    {
      "kind": "tip",
      "instrument": "claude-code",
      "model": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "what": "Getting the length right",
      "href": "/tips/exact-length",
      "lift": "+99.8 percentage points (+99.3 to +100.2)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Claude Opus 5.5, tier 2",
      "tier": "t2"
    }
  ],
  "nothingLeftToMeasure": [],
  "honestStop": null,
  "tierNote": "The first tasks got too easy to tell the models apart, so we built a harder set, tier 2, and the scores here are from it. In Codex the harder set was not run, so those scores are from tier 1.",
  "versionPairs": [
    "On getting the length right, Claude Fable 5.1 scores 50% unaided against 76% for Claude Fable 5; scores lower."
  ]
}
