{
  "surface": "openaddict-job",
  "version": "job-v1",
  "generatedAt": "2026-10-03",
  "generatedAtMeans": "The newest run date this document reads. It is not the time the file was written.",
  "rule": "Every figure is read inside one job and one instrument and names the run it came from. Nothing is averaged across instruments, and a job measured on two instruments carries a separate entry for each. An absence is a reason, never a zero.",
  "job": "C16-cutoff-disclosure",
  "jobKind": "tip",
  "jobLabel": "Questions about recent events",
  "measure": "score with no tip applied",
  "scale": "unit",
  "url": "https://openaddict.com/jobs/recent-events",
  "sentence": "Answering about recent events means a question about something newer than the model knows, where the right answer says so. Each model was given the same tasks with no help, tested in the API, Claude Code and Codex.",
  "picks": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "launch",
        "label": "Main run, tip by tip",
        "date": "2026-08-17"
      },
      "joined": [
        {
          "id": "codex-api-twin",
          "label": "API twin for the Codex run",
          "date": "2026-09-07"
        }
      ],
      "best": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run, tip by tip",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.8555555555555556,
            "shown": "85.6%",
            "run": "launch"
          }
        ],
        "close": [],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": false,
        "reason": "The best model on this table publishes no cost per task, so a cheaper model cannot be priced against it."
      },
      "consistent": {
        "measured": false,
        "reason": "The models were asked under different settings, and a setting changes how often an answer repeats, so they are not ranked."
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "gemini-3.1-flash-lite",
            "name": "Gemini 3.1 Flash Lite",
            "medianMs": 563,
            "shown": "563 ms"
          }
        ],
        "models": 2
      },
      "sentence": "In the API, Claude Haiku 4.5 scores highest."
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "launch",
        "label": "Main run",
        "date": "2026-09-03"
      },
      "joined": [],
      "best": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run",
        "tier": "t1",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "mean": 0.9666666666666667,
            "shown": "96.7%",
            "run": "launch"
          }
        ],
        "close": [
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "mean": 0.9333333333333333,
            "shown": "93.3%",
            "run": "launch"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "mean": 0.9222222222222223,
            "shown": "92.2%",
            "run": "launch"
          },
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "mean": 0.8777777777777778,
            "shown": "87.8%",
            "run": "launch"
          },
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "mean": 0.8027777777777778,
            "shown": "80.3%",
            "run": "launch"
          }
        ],
        "rangeAbsent": false
      },
      "cheapest": {
        "measured": true,
        "run": "launch",
        "runLabel": "Main run",
        "pick": {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9666666666666667,
          "usd": 0.0021981,
          "shown": "$0.00220",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "launch",
          "runLabel": "Main run"
        },
        "best": {
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9666666666666667,
          "usd": 0.0021981,
          "shown": "$0.00220",
          "basis": "list-price",
          "label": "run on subscription; shown at API list price for comparison",
          "run": "launch",
          "runLabel": "Main run"
        },
        "pickIsBest": true,
        "basesDiffer": false
      },
      "consistent": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "condition": "no-sampling-control",
        "conditionLabel": "no sampling control here",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "identical": 28,
            "repeated": 30,
            "share": 0.9333333333333333,
            "shown": "28 of 30"
          }
        ],
        "close": [
          {
            "model": "claude-fable-5",
            "name": "Claude Fable 5",
            "identical": 20,
            "repeated": 30,
            "share": 0.6666666666666666,
            "shown": "20 of 30"
          },
          {
            "model": "claude-fable-5-1",
            "name": "Claude Fable 5.1",
            "identical": 20,
            "repeated": 30,
            "share": 0.6666666666666666,
            "shown": "20 of 30"
          },
          {
            "model": "claude-opus-5",
            "name": "Claude Opus 5",
            "identical": 23,
            "repeated": 30,
            "share": 0.7666666666666667,
            "shown": "23 of 30"
          },
          {
            "model": "claude-opus-5-5",
            "name": "Claude Opus 5.5",
            "identical": 24,
            "repeated": 30,
            "share": 0.8,
            "shown": "24 of 30"
          },
          {
            "model": "claude-sonnet-5",
            "name": "Claude Sonnet 5",
            "identical": 22,
            "repeated": 30,
            "share": 0.7333333333333333,
            "shown": "22 of 30"
          }
        ]
      },
      "fastest": {
        "measured": true,
        "run": "reliability-five-pass",
        "runLabel": "Each tip asked five times",
        "scope": "time per call, across all twelve tips asked five times",
        "leaders": [
          {
            "model": "claude-haiku-4-5",
            "name": "Claude Haiku 4.5",
            "medianMs": 11372,
            "shown": "11.4 s"
          }
        ],
        "models": 6
      },
      "sentence": "In Claude Code, Claude Haiku 4.5 scores highest. Claude Sonnet 5, Claude Fable 5.1, Claude Fable 5 and Claude Opus 5 are too close to call."
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "best": {
        "measured": false,
        "reason": "Only one model was measured on this run, so there is nothing to rank."
      },
      "cheapest": {
        "measured": false,
        "reason": "Only one model was measured on this run, so there is nothing to rank."
      },
      "consistent": {
        "measured": false,
        "reason": "No run asked these tasks more than once this way, so there is no repeat to compare."
      },
      "fastest": {
        "measured": false,
        "reason": "No run of this job timed its calls this way."
      },
      "sentence": "In Codex, only one model was measured on this run, so there is nothing to rank."
    }
  ],
  "scores": [
    {
      "instrument": "api",
      "instrumentLabel": "tested via API",
      "run": {
        "id": "launch",
        "label": "Main run, tip by tip",
        "date": "2026-08-17"
      },
      "joined": [
        {
          "id": "codex-api-twin",
          "label": "API twin for the Codex run",
          "date": "2026-09-07"
        }
      ],
      "tasks": 43,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.8555555555555556,
          "shown": "85.6%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "28 of 30",
          "repeatSetting": "temperature 0 asked for",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5-mini",
          "name": "GPT-5 mini",
          "mean": 0.4861111111111111,
          "shown": "48.6%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "launch",
          "repeat": "6 of 30",
          "repeatSetting": "at the vendor default",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gemini-3.1-flash-lite",
          "name": "Gemini 3.1 Flash Lite",
          "mean": 0.03333333333333333,
          "shown": "3.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "launch",
          "repeat": "30 of 30",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "mean": 0.275,
          "shown": "27.5%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-api-twin",
          "repeat": "25 of 30",
          "repeatSetting": "at temperature 0",
          "atCeiling": false
        }
      ]
    },
    {
      "instrument": "claude-code",
      "instrumentLabel": "tested in Claude Code",
      "run": {
        "id": "launch",
        "label": "Main run",
        "date": "2026-09-03"
      },
      "joined": [],
      "tasks": 30,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "figure",
          "model": "claude-haiku-4-5",
          "name": "Claude Haiku 4.5",
          "mean": 0.9666666666666667,
          "shown": "96.7%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "best",
          "run": "launch",
          "repeat": "28 of 30",
          "repeatSetting": "no sampling control here",
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-fable-5",
          "name": "Claude Fable 5",
          "mean": 0.8777777777777778,
          "shown": "87.8%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "20 of 30",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "figure",
          "model": "claude-fable-5-1",
          "name": "Claude Fable 5.1",
          "mean": 0.9222222222222223,
          "shown": "92.2%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "20 of 30",
          "repeatSetting": "no sampling control here",
          "atCeiling": true
        },
        {
          "kind": "figure",
          "model": "claude-opus-5",
          "name": "Claude Opus 5",
          "mean": 0.8027777777777778,
          "shown": "80.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "23 of 30",
          "repeatSetting": "no sampling control here",
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "claude-opus-5-5",
          "name": "Claude Opus 5.5",
          "word": "in another run",
          "reason": null
        },
        {
          "kind": "figure",
          "model": "claude-sonnet-5",
          "name": "Claude Sonnet 5",
          "mean": 0.9333333333333333,
          "shown": "93.3%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": "close",
          "run": "launch",
          "repeat": "22 of 30",
          "repeatSetting": "no sampling control here",
          "atCeiling": true
        }
      ]
    },
    {
      "instrument": "codex-cli",
      "instrumentLabel": "tested in Codex",
      "run": {
        "id": "codex-tips-run",
        "label": "Codex run, tip by tip",
        "date": "2026-10-03"
      },
      "joined": [],
      "tasks": 30,
      "tiers": [
        "t1"
      ],
      "entries": [
        {
          "kind": "absent",
          "model": "gpt-5.4-mini",
          "name": "GPT-5.4 mini",
          "word": "withheld",
          "reason": "model no longer served to this account after 38 of 60 units; two retries refused"
        },
        {
          "kind": "absent",
          "model": "gpt-5.6-luna",
          "name": "GPT-5.6 Luna",
          "word": "withheld",
          "reason": "6 units exhausted the single C8 retry after policy-blocked built-in tool attempts; no completed tool or MCP hit. Full one-pass coverage not reached."
        },
        {
          "kind": "figure",
          "model": "gpt-5.6-terra",
          "name": "GPT-5.6 Terra",
          "mean": 0.3,
          "shown": "30.0%",
          "tier": "t1",
          "tierWord": "tier 1",
          "mark": null,
          "run": "codex-tips-run",
          "repeat": "one pass, not measured",
          "repeatSetting": null,
          "atCeiling": false
        },
        {
          "kind": "absent",
          "model": "gpt-6-luna",
          "name": "GPT-6 Luna",
          "word": "withheld",
          "reason": "Incomplete coverage at retry exhaustion; 9 registered units have no valid record. Resume only within registered retry policy."
        },
        {
          "kind": "absent",
          "model": "gpt-6-sol",
          "name": "GPT-6 Sol",
          "word": "withheld",
          "reason": "Incomplete coverage at retry exhaustion; 15 registered units have no valid record. Resume only within registered retry policy."
        }
      ]
    }
  ],
  "helps": [
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gpt-5-mini",
      "name": "GPT-5 mini",
      "what": "Questions about recent events",
      "href": "/tips/cutoff-disclosure",
      "lift": "+41.1 percentage points (+24.6 to +57.6)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Main run",
      "tier": "t1"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gemini-3.1-flash-lite",
      "name": "Gemini 3.1 Flash Lite",
      "what": "Questions about recent events",
      "href": "/tips/cutoff-disclosure",
      "lift": "+30.0 percentage points (+13.3 to +46.7)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Main run",
      "tier": "t1"
    },
    {
      "kind": "tip",
      "instrument": "api",
      "model": "gpt-5.4-mini",
      "name": "GPT-5.4 mini",
      "what": "Questions about recent events",
      "href": "/tips/cutoff-disclosure",
      "lift": "+59.2 percentage points (+41.0 to +77.4)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "API twin for the Codex run",
      "tier": "t1"
    },
    {
      "kind": "tip",
      "instrument": "codex-cli",
      "model": "gpt-5.4-mini",
      "name": "GPT-5.4 mini",
      "what": "Questions about recent events",
      "href": "/tips/cutoff-disclosure",
      "lift": "+54.2 percentage points (+37.9 to +70.4)",
      "instruction": "no one-line instruction was tested on tips",
      "runLabel": "Codex run, replaying two earlier tests",
      "tier": "t1"
    }
  ],
  "nothingLeftToMeasure": [
    {
      "instrument": "claude-code",
      "model": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "unaided": "96.7%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "unaided": "92.2%",
      "tier": "t1"
    },
    {
      "instrument": "claude-code",
      "model": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "unaided": "93.3%",
      "tier": "t1"
    }
  ],
  "honestStop": null,
  "tierNote": "These scores are from the first tasks we built for this job, tier 1, the set the models are read on.",
  "versionPairs": [
    "On questions about recent events, Claude Fable 5.1 scores 92% unaided against 88% for Claude Fable 5; no measurable change."
  ]
}
