{
  "observatory": "AI Error Observatory",
  "operator": "Axion Labs",
  "url": "https://errors.getaxionlabs.com",
  "dataset_version": "2026-09-01.v1",
  "updated": "2026-09-01",
  "provenance": "Every record carries claim, metric, named source, source_url, checked date and an evidence grade. Grades are defined in grade_key. Nothing is estimated or vendor-supplied.",
  "grade_key": {
    "MEASURED": "An independent party ran the test; we read the chart itself (including its own freshness stamp) and copied the numbers.",
    "REPORTED": "Published by a named party; we verified the publication exists and say who claims it — we did not re-run it.",
    "FIRST-PARTY": "Happened inside our own operation; logged in our own files on the day, quoted from those logs."
  },
  "classes": {
    "fabrication": {
      "name": "Fabrication",
      "line": "The model invents an answer and delivers it with full confidence."
    },
    "derailment": {
      "name": "Derailment",
      "line": "On long autonomous runs the agent loops, flails, and burns money without finishing."
    },
    "delivery-loss": {
      "name": "Delivery loss",
      "line": "The work succeeded — the answer was lost on the way back to you."
    },
    "misrouting": {
      "name": "Misrouting",
      "line": "The wrong model or effort level for the job — often costing more to do worse."
    },
    "stale-evidence": {
      "name": "Stale evidence",
      "line": "Decisions made on benchmark charts that quietly stopped updating."
    },
    "metric-misread": {
      "name": "Metric misread",
      "line": "The number was real — the conclusion drawn from it was not."
    }
  },
  "record_count": 14,
  "records": [
    {
      "id": "sonnet5-tb4-flail",
      "class": "derailment",
      "title": "The mid-tier model that scored worst AND billed most",
      "finding": "Left to work 330 autonomous terminal tasks, Claude Sonnet 5 (max effort) completed 12.42% — bottom of the board — while burning 21.6 billion tokens and $9,603.86, the highest bill of any entry. The top model spent $5,969 to score 51.8%. On long-horizon work a weaker model does not save money; it flails.",
      "metric": "12.42% completed · $9,603.86 · 21.6B tokens",
      "grade": "MEASURED",
      "source": "Terminal-Bench 4.0 leaderboard (Stanford/Harbor/Laude Institute), board updated 2026-08-29",
      "source_url": "https://www.tbench.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "best-agent-fails-half",
      "class": "derailment",
      "title": "The best agent on earth still fails ~half its first attempts",
      "finding": "The #1 agent-model pair on the freshest independent chart (Claude Code + Opus 5) completed 51.82% of 330 autonomous tasks at max effort. That is the CEILING of the field today: 48.2% of first attempts by the best available system fail.",
      "metric": "51.82% pass@1 (±3.39) — field maximum",
      "grade": "MEASURED",
      "source": "Terminal-Bench 4.0 leaderboard, board updated 2026-08-29",
      "source_url": "https://www.tbench.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "retry-beats-upgrade",
      "class": "derailment",
      "title": "Retrying beats upgrading — by 25 points",
      "finding": "Same model, same tasks, five attempts with a checker picking the winner: Opus 5 goes 51.8% → 69.7% (+17.9 points). Switching to the next-best model instead LOSES 7.3 points. Most agent failures are recoverable — if a machine-checkable verification gate exists to catch them.",
      "metric": "pass@1 51.8% → pass@5 69.7%",
      "grade": "MEASURED",
      "source": "Terminal-Bench 4.0 pass@k table, board updated 2026-08-29",
      "source_url": "https://www.tbench.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "speed-is-not-reliability",
      "class": "metric-misread",
      "title": "The fastest competent model is one of the worst agents",
      "finding": "Gemini 3.7 Flash streams at 300 tokens/second with a respectable intelligence index of 56 — and ranks 30th on LMArena’s agent board (score 0.0100 ±0.0090), scored on real sessions. Fast and smart does not equal a reliable agent.",
      "metric": "300 tok/s · index 56 · agent rank 30",
      "grade": "MEASURED",
      "source": "ArtificialAnalysis model index + LMArena agent leaderboard",
      "source_url": "https://lmarena.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "tool-hallucination-is-scored",
      "class": "fabrication",
      "title": "Agents inventing tools is now a measured, ranked failure",
      "finding": "LMArena’s agent board scores real sessions on five behavioural failure signals — including tool_hallucination, an agent claiming to have used tools or produced results that do not exist. Models the text arena ranks as near-equals separate widely on these signals.",
      "metric": "5 scored failure signals incl. tool_hallucination",
      "grade": "MEASURED",
      "source": "LMArena agent leaderboard (scored on real sessions, not votes)",
      "source_url": "https://lmarena.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "rank-one-fallacy",
      "class": "metric-misread",
      "title": "\"The #1 model\" is a statistical near-tie",
      "finding": "On LMArena’s overall text arena (7,922,078 human votes, 395 models, cutoff 2026-08-27) the top ELEVEN models span 16 Elo points. Picking a vendor because it holds rank 1 on this chart is reading noise — and the chart measures what humans prefer reading, not what completes a job.",
      "metric": "Top 11 within 16 Elo of each other",
      "grade": "MEASURED",
      "source": "LMArena overall text leaderboard, vote cutoff 2026-08-27",
      "source_url": "https://lmarena.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "more-reasoning-worse",
      "class": "metric-misread",
      "title": "Maximum reasoning effort measurably made the agent worse",
      "finding": "On LMArena’s agent board, Claude Opus 5 at HIGH effort outscores the same model at MAX effort (0.1388 vs 0.1200, 21,330 and 16,975 sessions). More thinking is not monotonically better — past a point it degrades real-session outcomes.",
      "metric": "High 0.1388 > Max 0.1200 (same model)",
      "grade": "MEASURED",
      "source": "LMArena agent leaderboard",
      "source_url": "https://lmarena.ai/leaderboard",
      "checked": "2026-08-31"
    },
    {
      "id": "wrong-axis-routing",
      "class": "misrouting",
      "title": "The \"cheaper tier\" was dumber AND 2.4x the price",
      "finding": "Measured cost-per-task: Claude Opus 5 at medium effort scores intelligence 59 for $0.72/task; Claude Sonnet 5 at max effort scores 55 for $1.72/task. Teams routing by model tier to save money are paying more for less — effort level, not model tier, is the real cost lever.",
      "metric": "Opus-medium 59 @ $0.72 vs Sonnet-max 55 @ $1.72",
      "grade": "MEASURED",
      "source": "ArtificialAnalysis Intelligence Index (cost-per-task methodology is AA’s own)",
      "source_url": "https://artificialanalysis.ai/leaderboards/models",
      "checked": "2026-08-31"
    },
    {
      "id": "stale-benchmark-citation",
      "class": "stale-evidence",
      "title": "Two famous benchmarks quietly stopped updating; citations kept flowing",
      "finding": "SWE-bench Verified’s newest submission is 2026-02-26 — six months stale. The aider polyglot leaderboard states on its own page \"last updated November 20, 2025\" — nine months stale. Both are still routinely cited as current evidence for model choices. The fix costs nothing: read the chart’s own freshness stamp before citing it.",
      "metric": "Newest entries: 2026-02-26 and 2025-11-20",
      "grade": "MEASURED",
      "source": "swebench.com embedded leaderboard data; aider.chat’s own page label",
      "source_url": "https://www.swebench.com/",
      "checked": "2026-08-31"
    },
    {
      "id": "popla-fabrication",
      "class": "fabrication",
      "title": "Asked about a parking adjudicator, the model invented immigration enforcement",
      "finding": "In our own field test, gpt-oss-120b (high reasoning effort, a widely used free-tier model) was asked what POPLA is. It returned a confident, detailed, wholly invented answer about UK immigration enforcement. POPLA is Parking on Private Land Appeals — the parking-ticket adjudicator. Since this test, every free-model output in our stack carries a verify-before-you-rely header.",
      "metric": "1 question · 1 confident answer · 0 true statements",
      "grade": "FIRST-PARTY",
      "source": "Axion Labs field test, logged same day in scripts/TOOLS.md (axion-readers, public repo)",
      "source_url": "https://github.com/endrezsoltdios-sketch/axion-readers",
      "checked": "2026-08-31"
    },
    {
      "id": "petition-inversion",
      "class": "metric-misread",
      "title": "A research agent ranked a market #1 — citing a petition to abolish it",
      "finding": "One of our own research sub-agents scored a market first for demand-pain. Its strongest evidence was a petition demanding the market be abolished. The signal was real, the reading was inverted — and it was delivered with full confidence. Since then: every load-bearing agent claim gets asked \"what would show this to be wrong?\" before it drives a decision.",
      "metric": "Top-ranked \"opportunity\" = abolition campaign",
      "grade": "FIRST-PARTY",
      "source": "Axion Labs session log, 31 Jul 2026, written into our operating rules the same day",
      "source_url": "https://github.com/endrezsoltdios-sketch/axion-readers",
      "checked": "2026-07-31"
    },
    {
      "id": "bulk-reprice-blindness",
      "class": "misrouting",
      "title": "An AI bulk operation cut a £14.99 product to £2.99",
      "finding": "An AI-driven bulk reprice applied a discount rule across a product catalogue without asking which items were actually in scope — cutting a £14.99 product to £2.99 live. It was reversed in five minutes for exactly one reason: the run logged every old value before applying. Dry-run by default and rollback logs turned an expensive mistake into a non-event.",
      "metric": "£14.99 → £2.99, recovered via pre-logged rollback values",
      "grade": "FIRST-PARTY",
      "source": "Axion Labs incident, 31 Jul 2026, codified same day as a standing dry-run law",
      "source_url": "https://github.com/endrezsoltdios-sketch/axion-readers",
      "checked": "2026-07-31"
    },
    {
      "id": "delivery-stage-loss",
      "class": "delivery-loss",
      "title": "77% of failing agent runs lose the answer at the delivery step",
      "finding": "APIFlow-Bench reports that in 77% of failing agent runs the underlying work succeeded — the answer was lost at the delivery stage, the last handover back to the caller. The costliest agent error is often not thinking; it is the envelope. Our own MCP gateway has asserted every tool result at the delivery boundary since 1 Sep 2026 because of this number.",
      "metric": "77% of failures = delivery stage",
      "grade": "REPORTED",
      "source": "APIFlow-Bench, arXiv 2608.29128 (paper claim; we did not re-run the benchmark)",
      "source_url": "https://arxiv.org/abs/2608.29128",
      "checked": "2026-09-01"
    },
    {
      "id": "agentic-cost-inversion",
      "class": "misrouting",
      "title": "On autonomous work, the cheap model cost 61% more than the frontier one",
      "finding": "Full-run totals on the same 330 tasks: Claude Sonnet 5 spent $9,603.86 to complete 12.4%; Claude Opus 5 spent $5,969.11 to complete 51.8%. Per completed task that is roughly $235 vs $35. \"Use the cheaper model\" is measured to be the expensive choice wherever the work is long-horizon and autonomous.",
      "metric": "~$235 vs ~$35 per completed task",
      "grade": "MEASURED",
      "source": "Terminal-Bench 4.0 leaderboard run costs, board updated 2026-08-29",
      "source_url": "https://www.tbench.ai/leaderboard",
      "checked": "2026-08-31"
    }
  ],
  "honesty": "v1 seed set (1 Sep 2026): our audit of live benchmark charts plus our own production logs. Not a real-time feed; grows only as records are verified.",
  "more_doors": "https://open.getaxionlabs.com/doors.json",
  "mcp_gateway": "https://mcp.getaxionlabs.com",
  "report_an_error": "mailto:hello@getaxionlabs.com (a named source and date are required to become a record)"
}