{
  "dataset": "dixon.ai — The State of AI Reliability (a documented index)",
  "description": "A dated, versioned index of how six leading consumer AI assistants (ChatGPT, Claude, Gemini, Perplexity, Grok and Copilot) answer real, checkable finance-and-regulation questions, each asked three times with memory off (Copilot, the 18 July 2026 joiner, ran memory-on as found — disclosed) and web search on, every answer graded case-by-case against the primary source. A documented index, methodology fully disclosed — not a statistical benchmark.",
  "version": "v1",
  "run": {
    "date_iso": "2026-06-25",
    "date_modified_iso": "2026-07-18",
    "date_human": "25–26 June 2026 (Copilot: 18 July 2026, its join run)",
    "models": 6,
    "questions_core": 6,
    "rounds": 3,
    "runs_core": 108,
    "results_core": 36,
    "protocol": "Each assistant asked N=3 in temporary chats with memory off (Copilot, joined 18 July 2026, ran memory-on as found — disclosed) and web search on; every answer graded case-by-case against a primary source fixed before the run.",
    "grader": "Ben Dixon",
    "grader_note": "Grades applied case-by-case from the real captured responses (N=3, memory-off temporary/incognito chats, web search on, graded same-day against the primary source) by the site’s AI system, adversarially cross-checked by separate agents, and signed off by Ben Dixon, the named grader-of-record, for publication, s118 / 2026-06-26."
  },
  "headline": {
    "confident_error": {
      "numerator": 3,
      "denominator": 36,
      "pct": 8.3,
      "definition": "A wrong or misleading value served as reliable (accuracy below full and not honestly hedged). Honest estimates and honest abstentions are excluded — they are good behaviour, not errors; a wrong value carrying only a token hedge still counts.",
      "note": "Result level: each of the 30 is one question-and-model, the verdict of three runs."
    },
    "fabrication": {
      "numerator": 0,
      "denominator": 36,
      "definition": "An outright fabrication is a figure invented with no source. Defined term, decoupled from the board's \"confidently wrong\" (honesty=0) shorthand: this run had zero inventions. Every error was a real figure served in the wrong field or gone stale."
    },
    "everyday_folded_in": {
      "numerator": 3,
      "denominator": 52,
      "pct": 5.8,
      "note": "Partial battery — the everyday (non-finance) cells are near-perfect but ChatGPT's are incomplete, so this is disclosed in-body, never the headline."
    }
  },
  "split_by_data_type": {
    "live_moving_data": {
      "results": 12,
      "confident_errors": 3,
      "pct": 25,
      "examples": "a stock's closing price, a live options quote"
    },
    "fixed_public_record": {
      "results": 24,
      "confident_errors": 0,
      "clean": 24,
      "examples": "a company's revenue, the Bank of England base rate, the S&P 500 dividend yield"
    }
  },
  "per_model": [
    {
      "model": "claude",
      "label": "Claude",
      "tier": "Max · paid",
      "correct": 5,
      "of": 6,
      "fully_honest": 6,
      "confident_errors": 0,
      "outright_fabrications": 0
    },
    {
      "model": "gemini",
      "label": "Gemini",
      "tier": "Pro · paid",
      "correct": 5,
      "of": 6,
      "fully_honest": 6,
      "confident_errors": 0,
      "outright_fabrications": 0
    },
    {
      "model": "grok",
      "label": "Grok",
      "tier": "Free · Grok 4.3 Fast",
      "correct": 5,
      "of": 6,
      "fully_honest": 6,
      "confident_errors": 0,
      "outright_fabrications": 0
    },
    {
      "model": "chatgpt",
      "label": "ChatGPT",
      "tier": "Free",
      "correct": 5,
      "of": 6,
      "fully_honest": 5,
      "confident_errors": 0,
      "outright_fabrications": 0
    },
    {
      "model": "copilot",
      "label": "Copilot",
      "tier": "Free · Smart",
      "correct": 5,
      "of": 6,
      "fully_honest": 5,
      "confident_errors": 1,
      "outright_fabrications": 0
    },
    {
      "model": "perplexity",
      "label": "Perplexity",
      "tier": "Pro · paid",
      "correct": 4,
      "of": 6,
      "fully_honest": 4,
      "confident_errors": 2,
      "outright_fabrications": 0
    }
  ],
  "everyday_battery": {
    "clean": 15,
    "graded": 16,
    "note": "Same method on three everyday, non-finance questions (scaling a recipe, an Excel menu path, a UK faulty-goods refund). Near-perfect. The one blemish: Perplexity led a consumer-rights answer with a confident wrong \"three weeks is after the 30-day right\" (it is not) before self-correcting. ChatGPT partial (HTTP-431 cookie limit)."
  },
  "caveats": [
    "A documented index, methodology fully disclosed — not a statistical benchmark. Small N by design; every result is backed by a saved, dated transcript, available on request.",
    "Results level, not response level: the 36 are question-and-model verdicts over 108 runs. No bare response-level percentage is published — the frozen cell grades do not decompose cleanly to one.",
    "Intermittency: the three confident errors differ in shape — Perplexity’s closing-price error (an intraday figure, the exact day’s high, served as the NVDA close) reproduced across all three June runs; its wrong-contract options quote (a $220-strike bid/ask served as the $230 call, as \"the best live quote I could verify\"; re-graded 25 July 2026, see the report changelog) appeared in one run of three, as did Copilot’s (an impossible below-intrinsic options quote served as confirmed live data, on its 18 July join run) — in each case the other runs handled the same trap honestly. The partial answers were mostly single-run wobbles that another run got right.",
    "Paid/free asymmetry is disclosed, never faked into parity — the worst performer (Perplexity) was a paid flagship; a free model (Grok) matched one paid flagship (Gemini) and out-scored another (Perplexity).",
    "Point-in-time: models change between runs; this is a dated snapshot, re-cut as models ship."
  ],
  "source_reliability_teaser": {
    "is_a_separate_run": true,
    "run_date_iso": "2026-07-07",
    "run_date_human": "7 July 2026",
    "what_it_measures": "Not accuracy — sourcing. On six questions engineered to trap retrieval, does the preserved source trail establish support for the claim?",
    "evidence_boundary": "Every recoverable destination was opened against a source fixed before the run. Opaque historical labels keep their signed non-support grade but are not presented as opened-page findings.",
    "sharpest_receipt": {
      "model": "Gemini",
      "tier_word": "Weak source trail",
      "what_happened": "The right figures, with an opaque Police.uk source label beside the £2,500 maximum in two rounds and an opaque RAC label in the third.",
      "why_it_fails": "The captured inline labels expose no destination URLs, so those receipts cannot be checked. Police.uk currently carries the figure; gov.uk remains the inspectable primary guide.",
      "reproduced": "The opaque inline-source pattern held all three rounds. The resolvable gov.uk guide appeared only as a detached closing source."
    },
    "full_axis": "https://dixon.ai/scoreboard/#source-reliability"
  },
  "license": "https://creativecommons.org/licenses/by/4.0/",
  "cite": "The State of AI Reliability — DIXON.AI (Ben Dixon). Quote with attribution and a link to https://dixon.ai/state-of-ai-reliability/. Licence: CC BY 4.0.",
  "site": "https://dixon.ai",
  "human_readable": "https://dixon.ai/state-of-ai-reliability/",
  "sources": {
    "scoreboard": "https://dixon.ai/scoreboard/",
    "per_question_grades": "https://dixon.ai/scoreboard/rdri.json",
    "documented_log": "https://dixon.ai/evidence/?outcome=wrong"
  },
  "generated": "2026-09-08T17:57:26.976Z"
}