{
 "video": "https://modelfatigue.news/decision-models/",
 "numbers": [
  {
   "id": "n-test",
   "what": "Test messages every model answered (BANKING77 test split)",
   "value": 3080,
   "shown": "3,080",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-options",
   "what": "Options in every question: the 77 intents and none-of-these",
   "value": 78,
   "shown": "78",
   "whose": "Model Fatigue",
   "source": "Our protocol (FREEZE.md), fixed before any test message was sent",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-intents",
   "what": "BANKING77 intents in the test split",
   "value": 77,
   "shown": "77",
   "whose": "Model Fatigue",
   "source": "Our protocol (FREEZE.md), fixed before any test message was sent",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-dev",
   "what": "Training messages held out as the dev slice, where every threshold and recalibration was fitted",
   "value": 573,
   "shown": "573",
   "whose": "Model Fatigue",
   "source": "Our protocol (FREEZE.md), fixed before any test message was sent",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "budget-2",
   "what": "Error budget for the workload figures: share of wrong answers allowed among the accepted ones",
   "value": 0.02,
   "shown": "2%",
   "whose": "Model Fatigue",
   "source": "Our protocol (FREEZE.md), fixed before any test message was sent",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "retrieval-ms",
   "what": "Median time to find the five labelled examples for a message, locally (not in any model's latency)",
   "value": 0.00961,
   "shown": "10 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-served",
   "what": "Jev, model as served",
   "value": "jev-1.13.0",
   "shown": "jev-1.13.0",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc",
   "what": "Jev, share of the messages sorted correctly",
   "value": 0.9392857142857143,
   "shown": "93.9%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc-lo",
   "what": "Jev, accuracy, 95% interval, low",
   "value": 0.9302921633913691,
   "shown": "93.0",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc-hi",
   "what": "Jev, accuracy, 95% interval, high",
   "value": 0.9471848120797445,
   "shown": "94.7",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-none",
   "what": "Jev, times it answered none-of-these (always wrong here)",
   "value": 14,
   "shown": "14",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p50",
   "what": "Jev, median time per request from Berlin",
   "value": 0.235395,
   "shown": "235 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p95",
   "what": "Jev, 95th percentile time per request from Berlin",
   "value": 0.33507649999999994,
   "shown": "335 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p99",
   "what": "Jev, 99th percentile time per request from Berlin",
   "value": 0.44170570000000026,
   "shown": "442 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-per1k",
   "what": "Jev, dollars per 1,000 decisions (rate card × tokens counted)",
   "value": 0.07753194545454546,
   "shown": "$0.078",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-served",
   "what": "Jev (30 September pass), model as served",
   "value": "jev-1.13.0",
   "shown": "jev-1.13.0",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-acc",
   "what": "Jev (30 September pass), share of the messages sorted correctly",
   "value": 0.938961038961039,
   "shown": "93.9%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-acc-lo",
   "what": "Jev (30 September pass), accuracy, 95% interval, low",
   "value": 0.9299469172345147,
   "shown": "93.0",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-acc-hi",
   "what": "Jev (30 September pass), accuracy, 95% interval, high",
   "value": 0.9468815164956743,
   "shown": "94.7",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-none",
   "what": "Jev (30 September pass), times it answered none-of-these (always wrong here)",
   "value": 13,
   "shown": "13",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-p50",
   "what": "Jev (30 September pass), median time per request from Berlin",
   "value": 0.25251500000000004,
   "shown": "253 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-p95",
   "what": "Jev (30 September pass), 95th percentile time per request from Berlin",
   "value": 0.3062035,
   "shown": "306 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-p99",
   "what": "Jev (30 September pass), 99th percentile time per request from Berlin",
   "value": 0.4029340000000003,
   "shown": "403 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-per1k",
   "what": "Jev (30 September pass), dollars per 1,000 decisions (rate card × tokens counted)",
   "value": 0.07753194545454546,
   "shown": "$0.078",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-same",
   "what": "Jev (30 September pass), same answer on the repeat pass",
   "value": 199,
   "shown": "199",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-repeat-n",
   "what": "Jev (30 September pass), messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-served",
   "what": "GPT-6 Luna, model as served",
   "value": "gpt-6-luna",
   "shown": "gpt-6-luna",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc",
   "what": "GPT-6 Luna, share of the messages sorted correctly",
   "value": 0.9409090909090909,
   "shown": "94.1%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc-lo",
   "what": "GPT-6 Luna, accuracy, 95% interval, low",
   "value": 0.9320194212464704,
   "shown": "93.2",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc-hi",
   "what": "GPT-6 Luna, accuracy, 95% interval, high",
   "value": 0.9487002629292668,
   "shown": "94.9",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-none",
   "what": "GPT-6 Luna, times it answered none-of-these (always wrong here)",
   "value": 4,
   "shown": "4",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p50",
   "what": "GPT-6 Luna, median time per request from Berlin",
   "value": 1.0973549999999999,
   "shown": "1,097 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p95",
   "what": "GPT-6 Luna, 95th percentile time per request from Berlin",
   "value": 2.2347954999999993,
   "shown": "2,235 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p99",
   "what": "GPT-6 Luna, 99th percentile time per request from Berlin",
   "value": 3.348933400000001,
   "shown": "3,349 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-per1k",
   "what": "GPT-6 Luna, dollars per 1,000 decisions (rate card × tokens counted)",
   "value": 0.143331387987013,
   "shown": "$0.143",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-same",
   "what": "GPT-6 Luna, same answer on the repeat pass",
   "value": 197,
   "shown": "197",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-repeat-n",
   "what": "GPT-6 Luna, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-served",
   "what": "Kev-4B, model as served",
   "value": "jaredpalmer/kev-4b-20260924",
   "shown": "jaredpalmer/kev-4b-20260924",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-acc",
   "what": "Kev-4B, share of the messages sorted correctly",
   "value": 0.8772727272727273,
   "shown": "87.7%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-acc-lo",
   "what": "Kev-4B, accuracy, 95% interval, low",
   "value": 0.86521217033876,
   "shown": "86.5",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-acc-hi",
   "what": "Kev-4B, accuracy, 95% interval, high",
   "value": 0.8883933326157368,
   "shown": "88.8",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-none",
   "what": "Kev-4B, times it answered none-of-these (always wrong here)",
   "value": 252,
   "shown": "252",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-p50",
   "what": "Kev-4B, median time per request from Berlin",
   "value": 0.53989,
   "shown": "540 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-p95",
   "what": "Kev-4B, 95th percentile time per request from Berlin",
   "value": 0.8834519999999999,
   "shown": "883 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-p99",
   "what": "Kev-4B, 99th percentile time per request from Berlin",
   "value": 1.3720374000000004,
   "shown": "1,372 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-per1k",
   "what": "Kev-4B, dollars per 1,000 decisions (as billed)",
   "value": 0.04094848636363636,
   "shown": "$0.041",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-same",
   "what": "Kev-4B, same answer on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-repeat-n",
   "what": "Kev-4B, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-served",
   "what": "Clef, model as served",
   "value": "clef",
   "shown": "clef",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc",
   "what": "Clef, share of the messages sorted correctly",
   "value": 0.9506493506493506,
   "shown": "95.1%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc-lo",
   "what": "Clef, accuracy, 95% interval, low",
   "value": 0.9424225746200366,
   "shown": "94.2",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc-hi",
   "what": "Clef, accuracy, 95% interval, high",
   "value": 0.9577533617834414,
   "shown": "95.8",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-none",
   "what": "Clef, times it answered none-of-these (always wrong here)",
   "value": 0,
   "shown": "0",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p50",
   "what": "Clef, median time per request from Berlin",
   "value": 0.562495,
   "shown": "562 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p95",
   "what": "Clef, 95th percentile time per request from Berlin",
   "value": 1.1474874999999964,
   "shown": "1,147 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p99",
   "what": "Clef, 99th percentile time per request from Berlin",
   "value": 2.0007208000000003,
   "shown": "2,001 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-per1k",
   "what": "Clef, dollars per 1,000 decisions (rate card × tokens counted)",
   "value": 0.4797513506493506,
   "shown": "$0.480",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-same",
   "what": "Clef, same answer on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-repeat-n",
   "what": "Clef, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-served",
   "what": "Clef-flash, model as served",
   "value": "clef-flash",
   "shown": "clef-flash",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-acc",
   "what": "Clef-flash, share of the messages sorted correctly",
   "value": 0.9516233766233766,
   "shown": "95.2%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-acc-lo",
   "what": "Clef-flash, accuracy, 95% interval, low",
   "value": 0.9434670440163816,
   "shown": "94.3",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-acc-hi",
   "what": "Clef-flash, accuracy, 95% interval, high",
   "value": 0.9586545176098705,
   "shown": "95.9",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-none",
   "what": "Clef-flash, times it answered none-of-these (always wrong here)",
   "value": 0,
   "shown": "0",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-p50",
   "what": "Clef-flash, median time per request from Berlin",
   "value": 0.19639,
   "shown": "196 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-p95",
   "what": "Clef-flash, 95th percentile time per request from Berlin",
   "value": 1.013879999999998,
   "shown": "1,014 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-p99",
   "what": "Clef-flash, 99th percentile time per request from Berlin",
   "value": 2.1720025,
   "shown": "2,172 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-per1k",
   "what": "Clef-flash, dollars per 1,000 decisions (rate card × tokens counted)",
   "value": 0.17990675649350651,
   "shown": "$0.180",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-same",
   "what": "Clef-flash, same answer on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-repeat-n",
   "what": "Clef-flash, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-served",
   "what": "Kev-27B, model as served",
   "value": "jaredpalmer/kev-27b",
   "shown": "jaredpalmer/kev-27b",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-acc",
   "what": "Kev-27B, share of the messages sorted correctly",
   "value": 0.9383116883116883,
   "shown": "93.8%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-acc-lo",
   "what": "Kev-27B, accuracy, 95% interval, low",
   "value": 0.9292566262741926,
   "shown": "92.9",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-acc-hi",
   "what": "Kev-27B, accuracy, 95% interval, high",
   "value": 0.9462747239741469,
   "shown": "94.6",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-none",
   "what": "Kev-27B, times it answered none-of-these (always wrong here)",
   "value": 21,
   "shown": "21",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-p50",
   "what": "Kev-27B, median time per request from Berlin",
   "value": 0.528605,
   "shown": "529 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-p95",
   "what": "Kev-27B, 95th percentile time per request from Berlin",
   "value": 0.5508195,
   "shown": "551 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-p99",
   "what": "Kev-27B, 99th percentile time per request from Berlin",
   "value": 0.5976951000000001,
   "shown": "598 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-per1k",
   "what": "Kev-27B, dollars per 1,000 decisions (our GPU time, as Modal billed it)",
   "value": 0.7295102727351169,
   "shown": "$0.730",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-same",
   "what": "Kev-27B, same answer on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-repeat-n",
   "what": "Kev-27B, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-served",
   "what": "Kev-9B, model as served",
   "value": "jaredpalmer/kev-9b",
   "shown": "jaredpalmer/kev-9b",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-acc",
   "what": "Kev-9B, share of the messages sorted correctly",
   "value": 0.9133116883116883,
   "shown": "91.3%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-acc-lo",
   "what": "Kev-9B, accuracy, 95% interval, low",
   "value": 0.9028523242309728,
   "shown": "90.3",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-acc-hi",
   "what": "Kev-9B, accuracy, 95% interval, high",
   "value": 0.922741311966165,
   "shown": "92.3",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-none",
   "what": "Kev-9B, times it answered none-of-these (always wrong here)",
   "value": 110,
   "shown": "110",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-p50",
   "what": "Kev-9B, median time per request from Berlin",
   "value": 0.436515,
   "shown": "437 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-p95",
   "what": "Kev-9B, 95th percentile time per request from Berlin",
   "value": 0.5482069999999997,
   "shown": "548 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-p99",
   "what": "Kev-9B, 99th percentile time per request from Berlin",
   "value": 0.6647247000000001,
   "shown": "665 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-per1k",
   "what": "Kev-9B, dollars per 1,000 decisions (our GPU time, as Modal billed it)",
   "value": 0.6211176265005325,
   "shown": "$0.621",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-same",
   "what": "Kev-9B, same answer on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-repeat-n",
   "what": "Kev-9B, messages on the repeat pass",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-vj-jr",
   "what": "Messages Jev got right and GPT-6 Luna got wrong",
   "value": 27,
   "shown": "27",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-vj-ar",
   "what": "Messages GPT-6 Luna got right and Jev got wrong",
   "value": 33,
   "shown": "33",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-vj-jr",
   "what": "Messages Jev got right and Kev-4B got wrong",
   "value": 205,
   "shown": "205",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-vj-ar",
   "what": "Messages Kev-4B got right and Jev got wrong",
   "value": 15,
   "shown": "15",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-vj-jr",
   "what": "Messages Jev got right and Clef got wrong",
   "value": 19,
   "shown": "19",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-vj-ar",
   "what": "Messages Clef got right and Jev got wrong",
   "value": 54,
   "shown": "54",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-vj-jr",
   "what": "Messages Jev got right and Clef-flash got wrong",
   "value": 22,
   "shown": "22",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-vj-ar",
   "what": "Messages Clef-flash got right and Jev got wrong",
   "value": 60,
   "shown": "60",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-vj-jr",
   "what": "Messages Jev got right and Kev-27B got wrong",
   "value": 27,
   "shown": "27",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-vj-ar",
   "what": "Messages Kev-27B got right and Jev got wrong",
   "value": 24,
   "shown": "24",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-vj-jr",
   "what": "Messages Jev got right and Kev-9B got wrong",
   "value": 97,
   "shown": "97",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-vj-ar",
   "what": "Messages Kev-9B got right and Jev got wrong",
   "value": 17,
   "shown": "17",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-twice-same",
   "what": "Jev's two passes (30 September, 2 October), messages with the same answer",
   "value": 3073,
   "shown": "3,073",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-acc-nonone",
   "what": "Kev-27B, accuracy with none-of-these set aside (its top real intent)",
   "value": 0.9409090909090909,
   "shown": "94.1%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-acc-nonone",
   "what": "Kev-9B, accuracy with none-of-these set aside (its top real intent)",
   "value": 0.9363636363636364,
   "shown": "93.6%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-acc-nonone",
   "what": "Kev-4B, accuracy with none-of-these set aside (a reading chosen after seeing its dev behaviour)",
   "value": 0.9334415584415584,
   "shown": "93.3%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-then-acc",
   "what": "Jev, accuracy in jev-2's pass",
   "value": 0.9402597402597402,
   "shown": "94.0%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 pass of Jev on the same messages",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-then-p50",
   "what": "Jev, median time per request in jev-2's pass",
   "value": 0.31307,
   "shown": "313 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 pass of Jev on the same messages",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-drift-same",
   "what": "Jev, messages with the same answer as in jev-2's pass",
   "value": 3075,
   "shown": "3,075",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-then-acc",
   "what": "GPT-6 Luna, accuracy in jev-2's pass",
   "value": 0.939935064935065,
   "shown": "94.0%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 pass of GPT-6 Luna on the same messages",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-then-p50",
   "what": "GPT-6 Luna, median time per request in jev-2's pass",
   "value": 1.1963300000000001,
   "shown": "1,196 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 pass of GPT-6 Luna on the same messages",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-drift-same",
   "what": "GPT-6 Luna, messages with the same answer as in jev-2's pass",
   "value": 3047,
   "shown": "3,047",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-floor",
   "what": "Network floor to typesafe's host from the Mac mini, warm, median of 20 (start of Jev's run)",
   "value": 0.1708,
   "shown": "171 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-floor",
   "what": "Network floor to openai's host from the Mac mini, warm, median of 20 (start of GPT-6 Luna's run)",
   "value": 0.1696,
   "shown": "170 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-floor",
   "what": "Network floor to openrouter's host from the Mac mini, warm, median of 20 (start of Kev-4B's run)",
   "value": 0.0214,
   "shown": "21 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-floor",
   "what": "Network floor to cloudflare's host from the Mac mini, warm, median of 20 (start of Clef's run)",
   "value": 0.2037,
   "shown": "204 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-floor",
   "what": "Network floor to cloudflare's host from the Mac mini, warm, median of 20 (start of Clef-flash's run)",
   "value": 0.2037,
   "shown": "204 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-floor",
   "what": "Network floor to modal's host from the Mac mini, warm, median of 20 (start of Kev-27B's run)",
   "value": 0.1324,
   "shown": "132 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-floor",
   "what": "Network floor to modal's host from the Mac mini, warm, median of 20 (start of Kev-9B's run)",
   "value": 0.1324,
   "shown": "132 ms",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-server",
   "what": "Kev-27B, median model time reported by its own server (our arithmetic on the records)",
   "value": 0.1105,
   "shown": "110 ms",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-server",
   "what": "Kev-9B, median model time reported by its own server (our arithmetic on the records)",
   "value": 0.0354,
   "shown": "35 ms",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-cachew",
   "what": "GPT-6 Luna, calls billed for writing tokens to OpenAI's prompt cache (none read from it; our count on the records)",
   "value": 3032,
   "shown": "3,032",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-probe-n",
   "what": "Test messages sent to each Kev model as a latency probe before its settings were frozen (timing read, answers not scored)",
   "value": 40,
   "shown": "40",
   "whose": "Model Fatigue",
   "source": "Our protocol addendum for Kev-27B and Kev-9B (FREEZE-ADDENDUM-kev.md)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-per1k-c6",
   "what": "Kev-27B, dollars per 1,000 decisions at six requests at a time (the zero-shot pass)",
   "value": 0.1572897050265309,
   "shown": "$0.157",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-per1k-c6",
   "what": "Kev-9B, dollars per 1,000 decisions at six requests at a time (the zero-shot pass)",
   "value": 0.11191627653552004,
   "shown": "$0.112",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-ece",
   "what": "Jev, calibration error (ECE) of its confidence as returned",
   "value": 0.03540259740259737,
   "shown": "0.035",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-ece-re",
   "what": "Jev, calibration error after a fit on our dev slice",
   "value": 0.010092819653914644,
   "shown": "0.010",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-auroc",
   "what": "Jev, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8396526005053688,
   "shown": "0.840",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-thr",
   "what": "Jev, confidence threshold fixed on the dev slice",
   "value": 0.99,
   "shown": "0.990",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-cov",
   "what": "Jev, share of messages accepted automatically",
   "value": 0.8438311688311688,
   "shown": "84.4%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-risk",
   "what": "Jev, error among the accepted",
   "value": 0.020392458637937667,
   "shown": "2.04%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-risk-lo",
   "what": "Jev, error among the accepted, 95% interval, low",
   "value": 0.015624462633927156,
   "shown": "1.56",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-risk-hi",
   "what": "Jev, error among the accepted, 95% interval, high",
   "value": 0.02657618453568931,
   "shown": "2.66",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel2-n",
   "what": "Jev, answers stated 0.2 to 0.3: how many",
   "value": 1,
   "shown": "1",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel2-pred",
   "what": "Jev, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.28,
   "shown": "0.28",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel2-obs",
   "what": "Jev, answers stated 0.2 to 0.3: share right",
   "value": 0.0,
   "shown": "0.00",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel3-n",
   "what": "Jev, answers stated 0.3 to 0.4: how many",
   "value": 3,
   "shown": "3",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel3-pred",
   "what": "Jev, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.3466666666666667,
   "shown": "0.35",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel3-obs",
   "what": "Jev, answers stated 0.3 to 0.4: share right",
   "value": 0.6666666666666666,
   "shown": "0.67",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel4-n",
   "what": "Jev, answers stated 0.4 to 0.5: how many",
   "value": 17,
   "shown": "17",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel4-pred",
   "what": "Jev, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.44588235294117645,
   "shown": "0.45",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel4-obs",
   "what": "Jev, answers stated 0.4 to 0.5: share right",
   "value": 0.35294117647058826,
   "shown": "0.35",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel5-n",
   "what": "Jev, answers stated 0.5 to 0.6: how many",
   "value": 37,
   "shown": "37",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel5-pred",
   "what": "Jev, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5454054054054055,
   "shown": "0.55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel5-obs",
   "what": "Jev, answers stated 0.5 to 0.6: share right",
   "value": 0.4864864864864865,
   "shown": "0.49",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel6-n",
   "what": "Jev, answers stated 0.6 to 0.7: how many",
   "value": 39,
   "shown": "39",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel6-pred",
   "what": "Jev, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6448717948717948,
   "shown": "0.64",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel6-obs",
   "what": "Jev, answers stated 0.6 to 0.7: share right",
   "value": 0.4358974358974359,
   "shown": "0.44",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel7-n",
   "what": "Jev, answers stated 0.7 to 0.8: how many",
   "value": 46,
   "shown": "46",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel7-pred",
   "what": "Jev, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.751304347826087,
   "shown": "0.75",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel7-obs",
   "what": "Jev, answers stated 0.7 to 0.8: share right",
   "value": 0.6304347826086957,
   "shown": "0.63",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel8-n",
   "what": "Jev, answers stated 0.8 to 0.9: how many",
   "value": 81,
   "shown": "81",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel8-pred",
   "what": "Jev, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8523456790123456,
   "shown": "0.85",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel8-obs",
   "what": "Jev, answers stated 0.8 to 0.9: share right",
   "value": 0.7160493827160493,
   "shown": "0.72",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel9-n",
   "what": "Jev, answers stated 0.9 to 1.0: how many",
   "value": 2856,
   "shown": "2,856",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel9-pred",
   "what": "Jev, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9951995798319327,
   "shown": "1.00",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-rel9-obs",
   "what": "Jev, answers stated 0.9 to 1.0: share right",
   "value": 0.967436974789916,
   "shown": "0.97",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-ece",
   "what": "GPT-6 Luna, calibration error (ECE) of its confidence as returned",
   "value": 0.024792207792207907,
   "shown": "0.025",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-ece-re",
   "what": "GPT-6 Luna, calibration error after a fit on our dev slice",
   "value": 0.020785867881441718,
   "shown": "0.021",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-auroc",
   "what": "GPT-6 Luna, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8758209526843067,
   "shown": "0.876",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-thr",
   "what": "GPT-6 Luna, confidence threshold fixed on the dev slice",
   "value": 0.9,
   "shown": "0.900",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-cov",
   "what": "GPT-6 Luna, share of messages accepted automatically",
   "value": 0.9126623376623376,
   "shown": "91.3%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-risk",
   "what": "GPT-6 Luna, error among the accepted",
   "value": 0.029882604055496264,
   "shown": "2.99%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-risk-lo",
   "what": "GPT-6 Luna, error among the accepted, 95% interval, low",
   "value": 0.024201569019382828,
   "shown": "2.42",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-risk-hi",
   "what": "GPT-6 Luna, error among the accepted, 95% interval, high",
   "value": 0.03684683953049791,
   "shown": "3.68",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel3-n",
   "what": "GPT-6 Luna, answers stated 0.3 to 0.4: how many",
   "value": 2,
   "shown": "2",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel3-pred",
   "what": "GPT-6 Luna, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.375,
   "shown": "0.38",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel3-obs",
   "what": "GPT-6 Luna, answers stated 0.3 to 0.4: share right",
   "value": 0.5,
   "shown": "0.50",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel4-n",
   "what": "GPT-6 Luna, answers stated 0.4 to 0.5: how many",
   "value": 9,
   "shown": "9",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel4-pred",
   "what": "GPT-6 Luna, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.4633333333333333,
   "shown": "0.46",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel4-obs",
   "what": "GPT-6 Luna, answers stated 0.4 to 0.5: share right",
   "value": 0.3333333333333333,
   "shown": "0.33",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel5-n",
   "what": "GPT-6 Luna, answers stated 0.5 to 0.6: how many",
   "value": 22,
   "shown": "22",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel5-pred",
   "what": "GPT-6 Luna, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5627272727272727,
   "shown": "0.56",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel5-obs",
   "what": "GPT-6 Luna, answers stated 0.5 to 0.6: share right",
   "value": 0.3181818181818182,
   "shown": "0.32",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel6-n",
   "what": "GPT-6 Luna, answers stated 0.6 to 0.7: how many",
   "value": 29,
   "shown": "29",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel6-pred",
   "what": "GPT-6 Luna, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6493103448275863,
   "shown": "0.65",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel6-obs",
   "what": "GPT-6 Luna, answers stated 0.6 to 0.7: share right",
   "value": 0.5862068965517241,
   "shown": "0.59",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel7-n",
   "what": "GPT-6 Luna, answers stated 0.7 to 0.8: how many",
   "value": 69,
   "shown": "69",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel7-pred",
   "what": "GPT-6 Luna, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.752463768115942,
   "shown": "0.75",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel7-obs",
   "what": "GPT-6 Luna, answers stated 0.7 to 0.8: share right",
   "value": 0.5942028985507246,
   "shown": "0.59",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel8-n",
   "what": "GPT-6 Luna, answers stated 0.8 to 0.9: how many",
   "value": 138,
   "shown": "138",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel8-pred",
   "what": "GPT-6 Luna, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8593478260869566,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel8-obs",
   "what": "GPT-6 Luna, answers stated 0.8 to 0.9: share right",
   "value": 0.7391304347826086,
   "shown": "0.74",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel9-n",
   "what": "GPT-6 Luna, answers stated 0.9 to 1.0: how many",
   "value": 2811,
   "shown": "2,811",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel9-pred",
   "what": "GPT-6 Luna, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9844254713625046,
   "shown": "0.98",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-rel9-obs",
   "what": "GPT-6 Luna, answers stated 0.9 to 1.0: share right",
   "value": 0.9701173959445037,
   "shown": "0.97",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-ece",
   "what": "Kev-4B, calibration error (ECE) of its confidence as returned",
   "value": 0.11257100649350645,
   "shown": "0.113",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-ece-re",
   "what": "Kev-4B, calibration error after a fit on our dev slice",
   "value": 0.009428836399292085,
   "shown": "0.009",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-auroc",
   "what": "Kev-4B, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.9219713792252653,
   "shown": "0.922",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-thr",
   "what": "Kev-4B, confidence threshold fixed on the dev slice",
   "value": 0.6688,
   "shown": "0.669",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-cov",
   "what": "Kev-4B, share of messages accepted automatically",
   "value": 0.7428571428571429,
   "shown": "74.3%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-risk",
   "what": "Kev-4B, error among the accepted",
   "value": 0.017482517482517484,
   "shown": "1.75%",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-risk-lo",
   "what": "Kev-4B, error among the accepted, 95% interval, low",
   "value": 0.012864885436917476,
   "shown": "1.29",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-risk-hi",
   "what": "Kev-4B, error among the accepted, 95% interval, high",
   "value": 0.023717747498971292,
   "shown": "2.37",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel2-n",
   "what": "Kev-4B, answers stated 0.2 to 0.3: how many",
   "value": 16,
   "shown": "16",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel2-pred",
   "what": "Kev-4B, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.2686,
   "shown": "0.27",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel2-obs",
   "what": "Kev-4B, answers stated 0.2 to 0.3: share right",
   "value": 0.1875,
   "shown": "0.19",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel3-n",
   "what": "Kev-4B, answers stated 0.3 to 0.4: how many",
   "value": 147,
   "shown": "147",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel3-pred",
   "what": "Kev-4B, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.36238503401360544,
   "shown": "0.36",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel3-obs",
   "what": "Kev-4B, answers stated 0.3 to 0.4: share right",
   "value": 0.2925170068027211,
   "shown": "0.29",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel4-n",
   "what": "Kev-4B, answers stated 0.4 to 0.5: how many",
   "value": 264,
   "shown": "264",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel4-pred",
   "what": "Kev-4B, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.44819545454545456,
   "shown": "0.45",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel4-obs",
   "what": "Kev-4B, answers stated 0.4 to 0.5: share right",
   "value": 0.4810606060606061,
   "shown": "0.48",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel5-n",
   "what": "Kev-4B, answers stated 0.5 to 0.6: how many",
   "value": 201,
   "shown": "201",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel5-pred",
   "what": "Kev-4B, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5474368159203981,
   "shown": "0.55",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel5-obs",
   "what": "Kev-4B, answers stated 0.5 to 0.6: share right",
   "value": 0.7064676616915423,
   "shown": "0.71",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel6-n",
   "what": "Kev-4B, answers stated 0.6 to 0.7: how many",
   "value": 237,
   "shown": "237",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel6-pred",
   "what": "Kev-4B, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6500362869198313,
   "shown": "0.65",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel6-obs",
   "what": "Kev-4B, answers stated 0.6 to 0.7: share right",
   "value": 0.8776371308016878,
   "shown": "0.88",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel7-n",
   "what": "Kev-4B, answers stated 0.7 to 0.8: how many",
   "value": 369,
   "shown": "369",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel7-pred",
   "what": "Kev-4B, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.7540235772357723,
   "shown": "0.75",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel7-obs",
   "what": "Kev-4B, answers stated 0.7 to 0.8: share right",
   "value": 0.9539295392953929,
   "shown": "0.95",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel8-n",
   "what": "Kev-4B, answers stated 0.8 to 0.9: how many",
   "value": 797,
   "shown": "797",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel8-pred",
   "what": "Kev-4B, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8591007528230866,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel8-obs",
   "what": "Kev-4B, answers stated 0.8 to 0.9: share right",
   "value": 0.9824341279799247,
   "shown": "0.98",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel9-n",
   "what": "Kev-4B, answers stated 0.9 to 1.0: how many",
   "value": 1049,
   "shown": "1,049",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel9-pred",
   "what": "Kev-4B, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9299280266920877,
   "shown": "0.93",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-rel9-obs",
   "what": "Kev-4B, answers stated 0.9 to 1.0: share right",
   "value": 0.9952335557673975,
   "shown": "1.00",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-ece",
   "what": "Clef, calibration error (ECE) of its confidence as returned",
   "value": 0.029707012987013053,
   "shown": "0.030",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-ece-re",
   "what": "Clef, calibration error after a fit on our dev slice",
   "value": 0.014990557137148372,
   "shown": "0.015",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-auroc",
   "what": "Clef, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.9001181873741732,
   "shown": "0.900",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-thr",
   "what": "Clef, confidence threshold fixed on the dev slice",
   "value": 0.8034,
   "shown": "0.803",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-cov",
   "what": "Clef, share of messages accepted automatically",
   "value": 0.9321428571428572,
   "shown": "93.2%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-risk",
   "what": "Clef, error among the accepted",
   "value": 0.024033437826541274,
   "shown": "2.40%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-risk-lo",
   "what": "Clef, error among the accepted, 95% interval, low",
   "value": 0.019034914578768047,
   "shown": "1.90",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-risk-hi",
   "what": "Clef, error among the accepted, 95% interval, high",
   "value": 0.030304012477247837,
   "shown": "3.03",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel2-n",
   "what": "Clef, answers stated 0.2 to 0.3: how many",
   "value": 5,
   "shown": "5",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel2-pred",
   "what": "Clef, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.2811,
   "shown": "0.28",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel2-obs",
   "what": "Clef, answers stated 0.2 to 0.3: share right",
   "value": 0.4,
   "shown": "0.40",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel3-n",
   "what": "Clef, answers stated 0.3 to 0.4: how many",
   "value": 13,
   "shown": "13",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel3-pred",
   "what": "Clef, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.35826153846153846,
   "shown": "0.36",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel3-obs",
   "what": "Clef, answers stated 0.3 to 0.4: share right",
   "value": 0.6923076923076923,
   "shown": "0.69",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel4-n",
   "what": "Clef, answers stated 0.4 to 0.5: how many",
   "value": 23,
   "shown": "23",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel4-pred",
   "what": "Clef, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.4494304347826087,
   "shown": "0.45",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel4-obs",
   "what": "Clef, answers stated 0.4 to 0.5: share right",
   "value": 0.5652173913043478,
   "shown": "0.57",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel5-n",
   "what": "Clef, answers stated 0.5 to 0.6: how many",
   "value": 38,
   "shown": "38",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel5-pred",
   "what": "Clef, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.549507894736842,
   "shown": "0.55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel5-obs",
   "what": "Clef, answers stated 0.5 to 0.6: share right",
   "value": 0.5,
   "shown": "0.50",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel6-n",
   "what": "Clef, answers stated 0.6 to 0.7: how many",
   "value": 55,
   "shown": "55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel6-pred",
   "what": "Clef, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6482090909090911,
   "shown": "0.65",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel6-obs",
   "what": "Clef, answers stated 0.6 to 0.7: share right",
   "value": 0.5636363636363636,
   "shown": "0.56",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel7-n",
   "what": "Clef, answers stated 0.7 to 0.8: how many",
   "value": 69,
   "shown": "69",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel7-pred",
   "what": "Clef, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.7555376811594203,
   "shown": "0.76",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel7-obs",
   "what": "Clef, answers stated 0.7 to 0.8: share right",
   "value": 0.6956521739130435,
   "shown": "0.70",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel8-n",
   "what": "Clef, answers stated 0.8 to 0.9: how many",
   "value": 187,
   "shown": "187",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel8-pred",
   "what": "Clef, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8563213903743316,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel8-obs",
   "what": "Clef, answers stated 0.8 to 0.9: share right",
   "value": 0.8342245989304813,
   "shown": "0.83",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel9-n",
   "what": "Clef, answers stated 0.9 to 1.0: how many",
   "value": 2690,
   "shown": "2,690",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel9-pred",
   "what": "Clef, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9594422304832713,
   "shown": "0.96",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-rel9-obs",
   "what": "Clef, answers stated 0.9 to 1.0: share right",
   "value": 0.9851301115241635,
   "shown": "0.99",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-ece",
   "what": "Clef-flash, calibration error (ECE) of its confidence as returned",
   "value": 0.14252301948051946,
   "shown": "0.143",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-ece-re",
   "what": "Clef-flash, calibration error after a fit on our dev slice",
   "value": 0.005597865348406929,
   "shown": "0.006",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-auroc",
   "what": "Clef-flash, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8190472592216047,
   "shown": "0.819",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-thr",
   "what": "Clef-flash, confidence threshold fixed on the dev slice",
   "value": 0.7323,
   "shown": "0.732",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-cov",
   "what": "Clef-flash, share of messages accepted automatically",
   "value": 0.8168831168831169,
   "shown": "81.7%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-risk",
   "what": "Clef-flash, error among the accepted",
   "value": 0.021462639109697933,
   "shown": "2.15%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-risk-lo",
   "what": "Clef-flash, error among the accepted, 95% interval, low",
   "value": 0.01648687278154381,
   "shown": "1.65",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-risk-hi",
   "what": "Clef-flash, error among the accepted, 95% interval, high",
   "value": 0.027897504395180316,
   "shown": "2.79",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel2-n",
   "what": "Clef-flash, answers stated 0.2 to 0.3: how many",
   "value": 7,
   "shown": "7",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel2-pred",
   "what": "Clef-flash, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.2601,
   "shown": "0.26",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel2-obs",
   "what": "Clef-flash, answers stated 0.2 to 0.3: share right",
   "value": 0.14285714285714285,
   "shown": "0.14",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel3-n",
   "what": "Clef-flash, answers stated 0.3 to 0.4: how many",
   "value": 12,
   "shown": "12",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel3-pred",
   "what": "Clef-flash, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.34730833333333333,
   "shown": "0.35",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel3-obs",
   "what": "Clef-flash, answers stated 0.3 to 0.4: share right",
   "value": 0.3333333333333333,
   "shown": "0.33",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel4-n",
   "what": "Clef-flash, answers stated 0.4 to 0.5: how many",
   "value": 59,
   "shown": "59",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel4-pred",
   "what": "Clef-flash, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.4583101694915255,
   "shown": "0.46",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel4-obs",
   "what": "Clef-flash, answers stated 0.4 to 0.5: share right",
   "value": 0.6779661016949152,
   "shown": "0.68",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel5-n",
   "what": "Clef-flash, answers stated 0.5 to 0.6: how many",
   "value": 87,
   "shown": "87",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel5-pred",
   "what": "Clef-flash, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5560540229885057,
   "shown": "0.56",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel5-obs",
   "what": "Clef-flash, answers stated 0.5 to 0.6: share right",
   "value": 0.7816091954022989,
   "shown": "0.78",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel6-n",
   "what": "Clef-flash, answers stated 0.6 to 0.7: how many",
   "value": 241,
   "shown": "241",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel6-pred",
   "what": "Clef-flash, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6557560165975104,
   "shown": "0.66",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel6-obs",
   "what": "Clef-flash, answers stated 0.6 to 0.7: share right",
   "value": 0.8713692946058091,
   "shown": "0.87",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel7-n",
   "what": "Clef-flash, answers stated 0.7 to 0.8: how many",
   "value": 736,
   "shown": "736",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel7-pred",
   "what": "Clef-flash, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.75900625,
   "shown": "0.76",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel7-obs",
   "what": "Clef-flash, answers stated 0.7 to 0.8: share right",
   "value": 0.9510869565217391,
   "shown": "0.95",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel8-n",
   "what": "Clef-flash, answers stated 0.8 to 0.9: how many",
   "value": 1339,
   "shown": "1,339",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel8-pred",
   "what": "Clef-flash, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8537438386855862,
   "shown": "0.85",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel8-obs",
   "what": "Clef-flash, answers stated 0.8 to 0.9: share right",
   "value": 0.9783420463032113,
   "shown": "0.98",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel9-n",
   "what": "Clef-flash, answers stated 0.9 to 1.0: how many",
   "value": 599,
   "shown": "599",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel9-pred",
   "what": "Clef-flash, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9228242070116862,
   "shown": "0.92",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-rel9-obs",
   "what": "Clef-flash, answers stated 0.9 to 1.0: share right",
   "value": 0.998330550918197,
   "shown": "1.00",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-ece",
   "what": "Kev-27B, calibration error (ECE) of its confidence as returned",
   "value": 0.014207987012986964,
   "shown": "0.014",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-ece-re",
   "what": "Kev-27B, calibration error after a fit on our dev slice",
   "value": 0.008714457272506165,
   "shown": "0.009",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-auroc",
   "what": "Kev-27B, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.920481697322892,
   "shown": "0.920",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-thr",
   "what": "Kev-27B, confidence threshold fixed on the dev slice",
   "value": 0.7928,
   "shown": "0.793",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-cov",
   "what": "Kev-27B, share of messages accepted automatically",
   "value": 0.9103896103896104,
   "shown": "91.0%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-risk",
   "what": "Kev-27B, error among the accepted",
   "value": 0.02282453637660485,
   "shown": "2.28%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-risk-lo",
   "what": "Kev-27B, error among the accepted, 95% interval, low",
   "value": 0.017914905043793377,
   "shown": "1.79",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-risk-hi",
   "what": "Kev-27B, error among the accepted, 95% interval, high",
   "value": 0.0290398804398322,
   "shown": "2.90",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel1-n",
   "what": "Kev-27B, answers stated 0.1 to 0.2: how many",
   "value": 1,
   "shown": "1",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel1-pred",
   "what": "Kev-27B, answers stated 0.1 to 0.2: mean stated confidence",
   "value": 0.1888,
   "shown": "0.19",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel1-obs",
   "what": "Kev-27B, answers stated 0.1 to 0.2: share right",
   "value": 1.0,
   "shown": "1.00",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel2-n",
   "what": "Kev-27B, answers stated 0.2 to 0.3: how many",
   "value": 9,
   "shown": "9",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel2-pred",
   "what": "Kev-27B, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.2532888888888889,
   "shown": "0.25",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel2-obs",
   "what": "Kev-27B, answers stated 0.2 to 0.3: share right",
   "value": 0.2222222222222222,
   "shown": "0.22",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel3-n",
   "what": "Kev-27B, answers stated 0.3 to 0.4: how many",
   "value": 25,
   "shown": "25",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel3-pred",
   "what": "Kev-27B, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.3658760000000001,
   "shown": "0.37",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel3-obs",
   "what": "Kev-27B, answers stated 0.3 to 0.4: share right",
   "value": 0.36,
   "shown": "0.36",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel4-n",
   "what": "Kev-27B, answers stated 0.4 to 0.5: how many",
   "value": 51,
   "shown": "51",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel4-pred",
   "what": "Kev-27B, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.4537529411764705,
   "shown": "0.45",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel4-obs",
   "what": "Kev-27B, answers stated 0.4 to 0.5: share right",
   "value": 0.43137254901960786,
   "shown": "0.43",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel5-n",
   "what": "Kev-27B, answers stated 0.5 to 0.6: how many",
   "value": 55,
   "shown": "55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel5-pred",
   "what": "Kev-27B, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5475854545454546,
   "shown": "0.55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel5-obs",
   "what": "Kev-27B, answers stated 0.5 to 0.6: share right",
   "value": 0.45454545454545453,
   "shown": "0.45",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel6-n",
   "what": "Kev-27B, answers stated 0.6 to 0.7: how many",
   "value": 62,
   "shown": "62",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel6-pred",
   "what": "Kev-27B, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6482032258064516,
   "shown": "0.65",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel6-obs",
   "what": "Kev-27B, answers stated 0.6 to 0.7: share right",
   "value": 0.5645161290322581,
   "shown": "0.56",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel7-n",
   "what": "Kev-27B, answers stated 0.7 to 0.8: how many",
   "value": 75,
   "shown": "75",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel7-pred",
   "what": "Kev-27B, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.7496440000000001,
   "shown": "0.75",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel7-obs",
   "what": "Kev-27B, answers stated 0.7 to 0.8: share right",
   "value": 0.7733333333333333,
   "shown": "0.77",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel8-n",
   "what": "Kev-27B, answers stated 0.8 to 0.9: how many",
   "value": 204,
   "shown": "204",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel8-pred",
   "what": "Kev-27B, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8589098039215687,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel8-obs",
   "what": "Kev-27B, answers stated 0.8 to 0.9: share right",
   "value": 0.8578431372549019,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel9-n",
   "what": "Kev-27B, answers stated 0.9 to 1.0: how many",
   "value": 2598,
   "shown": "2,598",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel9-pred",
   "what": "Kev-27B, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9753343341031563,
   "shown": "0.98",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-rel9-obs",
   "what": "Kev-27B, answers stated 0.9 to 1.0: share right",
   "value": 0.9865280985373364,
   "shown": "0.99",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-ece",
   "what": "Kev-9B, calibration error (ECE) of its confidence as returned",
   "value": 0.03842435064935068,
   "shown": "0.038",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-ece-re",
   "what": "Kev-9B, calibration error after a fit on our dev slice",
   "value": 0.012262876471652633,
   "shown": "0.012",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-auroc",
   "what": "Kev-9B, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.9057099794826321,
   "shown": "0.906",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-thr",
   "what": "Kev-9B, confidence threshold fixed on the dev slice",
   "value": 0.8127,
   "shown": "0.813",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-cov",
   "what": "Kev-9B, share of messages accepted automatically",
   "value": 0.826948051948052,
   "shown": "82.7%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-risk",
   "what": "Kev-9B, error among the accepted",
   "value": 0.024734982332155476,
   "shown": "2.47%",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-risk-lo",
   "what": "Kev-9B, error among the accepted, 95% interval, low",
   "value": 0.019380968427526332,
   "shown": "1.94",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-risk-hi",
   "what": "Kev-9B, error among the accepted, 95% interval, high",
   "value": 0.03152050659938243,
   "shown": "3.15",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel2-n",
   "what": "Kev-9B, answers stated 0.2 to 0.3: how many",
   "value": 2,
   "shown": "2",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel2-pred",
   "what": "Kev-9B, answers stated 0.2 to 0.3: mean stated confidence",
   "value": 0.2871,
   "shown": "0.29",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel2-obs",
   "what": "Kev-9B, answers stated 0.2 to 0.3: share right",
   "value": 0.5,
   "shown": "0.50",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel3-n",
   "what": "Kev-9B, answers stated 0.3 to 0.4: how many",
   "value": 21,
   "shown": "21",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel3-pred",
   "what": "Kev-9B, answers stated 0.3 to 0.4: mean stated confidence",
   "value": 0.36151428571428573,
   "shown": "0.36",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel3-obs",
   "what": "Kev-9B, answers stated 0.3 to 0.4: share right",
   "value": 0.23809523809523808,
   "shown": "0.24",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel4-n",
   "what": "Kev-9B, answers stated 0.4 to 0.5: how many",
   "value": 88,
   "shown": "88",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel4-pred",
   "what": "Kev-9B, answers stated 0.4 to 0.5: mean stated confidence",
   "value": 0.4609920454545455,
   "shown": "0.46",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel4-obs",
   "what": "Kev-9B, answers stated 0.4 to 0.5: share right",
   "value": 0.375,
   "shown": "0.38",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel5-n",
   "what": "Kev-9B, answers stated 0.5 to 0.6: how many",
   "value": 102,
   "shown": "102",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel5-pred",
   "what": "Kev-9B, answers stated 0.5 to 0.6: mean stated confidence",
   "value": 0.5521333333333334,
   "shown": "0.55",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel5-obs",
   "what": "Kev-9B, answers stated 0.5 to 0.6: share right",
   "value": 0.47058823529411764,
   "shown": "0.47",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel6-n",
   "what": "Kev-9B, answers stated 0.6 to 0.7: how many",
   "value": 115,
   "shown": "115",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel6-pred",
   "what": "Kev-9B, answers stated 0.6 to 0.7: mean stated confidence",
   "value": 0.6540373913043479,
   "shown": "0.65",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel6-obs",
   "what": "Kev-9B, answers stated 0.6 to 0.7: share right",
   "value": 0.6434782608695652,
   "shown": "0.64",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel7-n",
   "what": "Kev-9B, answers stated 0.7 to 0.8: how many",
   "value": 180,
   "shown": "180",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel7-pred",
   "what": "Kev-9B, answers stated 0.7 to 0.8: mean stated confidence",
   "value": 0.7548805555555556,
   "shown": "0.75",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel7-obs",
   "what": "Kev-9B, answers stated 0.7 to 0.8: share right",
   "value": 0.8055555555555556,
   "shown": "0.81",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel8-n",
   "what": "Kev-9B, answers stated 0.8 to 0.9: how many",
   "value": 440,
   "shown": "440",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel8-pred",
   "what": "Kev-9B, answers stated 0.8 to 0.9: mean stated confidence",
   "value": 0.8604661363636363,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel8-obs",
   "what": "Kev-9B, answers stated 0.8 to 0.9: share right",
   "value": 0.9204545454545454,
   "shown": "0.92",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel9-n",
   "what": "Kev-9B, answers stated 0.9 to 1.0: how many",
   "value": 2132,
   "shown": "2,132",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel9-pred",
   "what": "Kev-9B, answers stated 0.9 to 1.0: mean stated confidence",
   "value": 0.9565132270168856,
   "shown": "0.96",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-rel9-obs",
   "what": "Kev-9B, answers stated 0.9 to 1.0: share right",
   "value": 0.9859287054409006,
   "shown": "0.99",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di-f1",
   "what": "Jev, macro-F1 on the Decision Index's zero-shot question",
   "value": 0.8000046642892913,
   "shown": "80.0",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di-f1-lo",
   "what": "Jev, zero-shot macro-F1, 95% interval, low",
   "value": 0.7860837415265793,
   "shown": "78.6",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di-f1-hi",
   "what": "Jev, zero-shot macro-F1, 95% interval, high",
   "value": 0.8108486777677364,
   "shown": "81.1",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di-acc",
   "what": "Jev, accuracy on the zero-shot question",
   "value": 0.8071428571428572,
   "shown": "80.7%",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di-f1",
   "what": "Clef, macro-F1 on the Decision Index's zero-shot question",
   "value": 0.9420316998167463,
   "shown": "94.2",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di-f1-lo",
   "what": "Clef, zero-shot macro-F1, 95% interval, low",
   "value": 0.9329378000053707,
   "shown": "93.3",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di-f1-hi",
   "what": "Clef, zero-shot macro-F1, 95% interval, high",
   "value": 0.9494195074154366,
   "shown": "94.9",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di-acc",
   "what": "Clef, accuracy on the zero-shot question",
   "value": 0.9422077922077922,
   "shown": "94.2%",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-di-f1",
   "what": "Clef-flash, macro-F1 on the Decision Index's zero-shot question",
   "value": 0.9084736076902837,
   "shown": "90.8",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-di-f1-lo",
   "what": "Clef-flash, zero-shot macro-F1, 95% interval, low",
   "value": 0.8977984284255756,
   "shown": "89.8",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-di-f1-hi",
   "what": "Clef-flash, zero-shot macro-F1, 95% interval, high",
   "value": 0.9173864837464847,
   "shown": "91.7",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-di-acc",
   "what": "Clef-flash, accuracy on the zero-shot question",
   "value": 0.9087662337662338,
   "shown": "90.9%",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-di-f1",
   "what": "Kev-27B, macro-F1 on the Decision Index's zero-shot question",
   "value": 0.8640871097889219,
   "shown": "86.4",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-di-f1-lo",
   "what": "Kev-27B, zero-shot macro-F1, 95% interval, low",
   "value": 0.8514143596341035,
   "shown": "85.1",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-di-f1-hi",
   "what": "Kev-27B, zero-shot macro-F1, 95% interval, high",
   "value": 0.8740878848855513,
   "shown": "87.4",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-di-acc",
   "what": "Kev-27B, accuracy on the zero-shot question",
   "value": 0.8681818181818182,
   "shown": "86.8%",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-di-f1",
   "what": "Kev-9B, macro-F1 on the Decision Index's zero-shot question",
   "value": 0.8302597361284337,
   "shown": "83.0",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-di-f1-lo",
   "what": "Kev-9B, zero-shot macro-F1, 95% interval, low",
   "value": 0.8157288132697301,
   "shown": "81.6",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-di-f1-hi",
   "what": "Kev-9B, zero-shot macro-F1, 95% interval, high",
   "value": 0.8409507773641868,
   "shown": "84.1",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-di-acc",
   "what": "Kev-9B, accuracy on the zero-shot question",
   "value": 0.8331168831168831,
   "shown": "83.3%",
   "whose": "Model Fatigue",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev2-jev-zs",
   "what": "Jev without examples on our 78-option question, jev-2's pass",
   "value": 0.7834415584415585,
   "shown": "78.3%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-claim",
   "what": "Clef, BANKING77 macro-F1 in Cloudflare's launch table",
   "value": 94.2,
   "shown": "94.20",
   "whose": "Cloudflare",
   "source": "Cloudflare, Clef and Clef-flash launch post",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-claim",
   "what": "Clef-flash, BANKING77 macro-F1 in Cloudflare's launch table",
   "value": 90.93,
   "shown": "90.93",
   "whose": "Cloudflare",
   "source": "Cloudflare, Clef and Clef-flash launch post",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-claim",
   "what": "Jev, BANKING77 macro-F1 in Cloudflare's launch table",
   "value": 79.74,
   "shown": "79.74",
   "whose": "Cloudflare",
   "source": "Cloudflare, Clef and Clef-flash launch post",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-claim",
   "what": "Kev 9B (the v1 checkpoint), BANKING77 macro-F1 on the Decision Index, the board's own run",
   "value": 0.8483,
   "shown": "84.83",
   "whose": "The Decision Index (multimodalart)",
   "source": "The community Decision Index, edition 0.2.1 (its data file)",
   "url": "https://huggingface.co/spaces/multimodalart/jev-decision-index",
   "read": "2026-10-02T15:32",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-mem-orig",
   "what": "Kev-4B, zero-shot accuracy on training messages",
   "value": 0.635,
   "shown": "63.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-mem-para",
   "what": "Kev-4B, zero-shot accuracy on paraphrases of them",
   "value": 0.555,
   "shown": "55.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-mem-orig",
   "what": "Clef, zero-shot accuracy on training messages",
   "value": 0.945,
   "shown": "94.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-mem-para",
   "what": "Clef, zero-shot accuracy on paraphrases of them",
   "value": 0.87,
   "shown": "87.0%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-mem-orig",
   "what": "Clef-flash, zero-shot accuracy on training messages",
   "value": 0.955,
   "shown": "95.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-mem-para",
   "what": "Clef-flash, zero-shot accuracy on paraphrases of them",
   "value": 0.835,
   "shown": "83.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-mem-orig",
   "what": "Kev-27B, zero-shot accuracy on training messages",
   "value": 0.795,
   "shown": "79.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-mem-para",
   "what": "Kev-27B, zero-shot accuracy on paraphrases of them",
   "value": 0.655,
   "shown": "65.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-mem-orig",
   "what": "Kev-9B, zero-shot accuracy on training messages",
   "value": 0.74,
   "shown": "74.0%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-mem-para",
   "what": "Kev-9B, zero-shot accuracy on paraphrases of them",
   "value": 0.625,
   "shown": "62.5%",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-mem-orig",
   "what": "Jev, zero-shot accuracy on training messages (jev-2's probe)",
   "value": 0.725,
   "shown": "72.5%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of Jev",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-mem-para",
   "what": "Jev, zero-shot accuracy on paraphrases (jev-2's probe)",
   "value": 0.68,
   "shown": "68.0%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of Jev",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-mem-gap",
   "what": "Jev, originals minus paraphrases, points (jev-2's probe)",
   "value": 4.5,
   "shown": "+4.5 points",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of Jev",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-mem-lo",
   "what": "Jev, gap, 95% interval, low (jev-2's probe)",
   "value": 0.5,
   "shown": "+0.5",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of Jev",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-mem-hi",
   "what": "Jev, gap, 95% interval, high (jev-2's probe)",
   "value": 8.5,
   "shown": "+8.5",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of Jev",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-mem-orig",
   "what": "GPT-6 Luna, zero-shot accuracy on training messages (jev-2's probe)",
   "value": 0.765,
   "shown": "76.5%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of GPT-6 Luna",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-mem-para",
   "what": "GPT-6 Luna, zero-shot accuracy on paraphrases (jev-2's probe)",
   "value": 0.715,
   "shown": "71.5%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of GPT-6 Luna",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-mem-gap",
   "what": "GPT-6 Luna, originals minus paraphrases, points (jev-2's probe)",
   "value": 5.0,
   "shown": "+5.0 points",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of GPT-6 Luna",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-mem-lo",
   "what": "GPT-6 Luna, gap, 95% interval, low (jev-2's probe)",
   "value": 1.0,
   "shown": "+1.0",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of GPT-6 Luna",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-mem-hi",
   "what": "GPT-6 Luna, gap, 95% interval, high (jev-2's probe)",
   "value": 9.5,
   "shown": "+9.5",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), jev-2's memorisation probe of GPT-6 Luna",
   "url": null,
   "read": "2026-09-24",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-n",
   "what": "Training messages in the probe (and as many paraphrases)",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "knn-orig",
   "what": "Five-nearest-neighbour vote (no memory possible), originals",
   "value": 0.945,
   "shown": "94.5%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), for jev-2's memorisation yardstick",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "knn-para",
   "what": "Five-nearest-neighbour vote, paraphrases",
   "value": 0.795,
   "shown": "79.5%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), for jev-2's memorisation yardstick",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "knn-gap",
   "what": "Five-nearest-neighbour vote, gap, points",
   "value": 15.0,
   "shown": "+15.0 points",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), for jev-2's memorisation yardstick",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "knn-lo",
   "what": "Five-nearest-neighbour vote, gap, 95% interval, low",
   "value": 10.0,
   "shown": "+10.0",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), for jev-2's memorisation yardstick",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "knn-hi",
   "what": "Five-nearest-neighbour vote, gap, 95% interval, high",
   "value": 20.5,
   "shown": "+20.5",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "Our results write-up (RESULTS.md), for jev-2's memorisation yardstick",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-trained-acc",
   "what": "A trained classifier (bge-small + logistic regression), accuracy",
   "value": 0.9314935064935065,
   "shown": "93.1%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-trained-acc-lo",
   "what": "A trained classifier (bge-small + logistic regression), accuracy, 95% interval, low",
   "value": 0.9220238855693331,
   "shown": "92.2",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-trained-acc-hi",
   "what": "A trained classifier (bge-small + logistic regression), accuracy, 95% interval, high",
   "value": 0.9398880881195878,
   "shown": "94.0",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-trained-p50",
   "what": "A trained classifier (bge-small + logistic regression), median time per request (on the Mac, no network)",
   "value": 0.008969999999999999,
   "shown": "9 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-1nn-acc",
   "what": "The single nearest labelled example, accuracy",
   "value": 0.925974025974026,
   "shown": "92.6%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-1nn-acc-lo",
   "what": "The single nearest labelled example, accuracy, 95% interval, low",
   "value": 0.9161875300142004,
   "shown": "91.6",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-1nn-acc-hi",
   "what": "The single nearest labelled example, accuracy, 95% interval, high",
   "value": 0.9346992340790008,
   "shown": "93.5",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-1nn-p50",
   "what": "The single nearest labelled example, median time per request (on the Mac, no network)",
   "value": 0.00961,
   "shown": "10 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-ling-acc",
   "what": "Ling 3.0 Flash, accuracy",
   "value": 0.937987012987013,
   "shown": "93.8%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-ling-acc-lo",
   "what": "Ling 3.0 Flash, accuracy, 95% interval, low",
   "value": 0.9289115804852008,
   "shown": "92.9",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-ling-acc-hi",
   "what": "Ling 3.0 Flash, accuracy, 95% interval, high",
   "value": 0.9459712280222141,
   "shown": "94.6",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-ling-p50",
   "what": "Ling 3.0 Flash, median time per request from Berlin",
   "value": 0.723005,
   "shown": "723 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-ling-per1k",
   "what": "Ling 3.0 Flash, dollars per 1,000 decisions, as billed",
   "value": 0.019790154545454543,
   "shown": "$0.020",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-dsv41-acc",
   "what": "DeepSeek V4.1 Flash, accuracy",
   "value": 0.936038961038961,
   "shown": "93.6%",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-dsv41-acc-lo",
   "what": "DeepSeek V4.1 Flash, accuracy, 95% interval, low",
   "value": 0.9268426713666059,
   "shown": "92.7",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-dsv41-acc-hi",
   "what": "DeepSeek V4.1 Flash, accuracy, 95% interval, high",
   "value": 0.9441488866952606,
   "shown": "94.4",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-dsv41-p50",
   "what": "DeepSeek V4.1 Flash, median time per request from Berlin",
   "value": 0.95337,
   "shown": "953 ms",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-dsv41-per1k",
   "what": "DeepSeek V4.1 Flash, dollars per 1,000 decisions, as billed",
   "value": 0.03787527272727272,
   "shown": "$0.038",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "M37's jev-2 run on the same protocol and the same 3,080 messages",
   "url": "https://www.youtube.com/watch?v=MwGATggBw-c",
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ref-trained-pool",
   "what": "Labelled training messages the trained classifier learned from (and the pool the five examples come from)",
   "value": 9405,
   "shown": "9,405",
   "whose": "Model Fatigue (M37's jev-2 run)",
   "source": "jev-2's protocol (its FREEZE.md), a copy kept with this page's sources in our repository",
   "url": null,
   "read": "2026-09-21",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-tok",
   "what": "Jev, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 1845.9987012987012,
   "shown": "1,846",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-0930-tok",
   "what": "Jev (30 September pass), input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 1845.9987012987012,
   "shown": "1,846",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-tok",
   "what": "GPT-6 Luna, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 1066.0217532467532,
   "shown": "1,066",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-tok",
   "what": "Kev-4B, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 974.9639610389611,
   "shown": "975",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-tok",
   "what": "Clef, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 1998.963961038961,
   "shown": "1,999",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-tok",
   "what": "Clef-flash, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 1998.963961038961,
   "shown": "1,999",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-tok",
   "what": "Kev-27B, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 974.9639610389611,
   "shown": "975",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-tok",
   "what": "Kev-9B, input tokens counted per request (our arithmetic: the pass's total over the messages)",
   "value": 974.9639610389611,
   "shown": "975",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-vj",
   "what": "GPT-6 Luna minus Jev, points (Jev pass of 30 September)",
   "value": 0.19480519480519481,
   "shown": "+0.2 points",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-vj-lo",
   "what": "GPT-6 Luna minus Jev, 95% interval, low",
   "value": -0.3246753246753247,
   "shown": "−0.3",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-vj-hi",
   "what": "GPT-6 Luna minus Jev, 95% interval, high",
   "value": 0.6818181818181818,
   "shown": "+0.7",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-vj",
   "what": "Kev-4B minus Jev, points (Jev pass of 30 September)",
   "value": -6.1688311688311686,
   "shown": "−6.2 points",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-vj-lo",
   "what": "Kev-4B minus Jev, 95% interval, low",
   "value": -7.11038961038961,
   "shown": "−7.1",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-vj-hi",
   "what": "Kev-4B minus Jev, 95% interval, high",
   "value": -5.292207792207792,
   "shown": "−5.3",
   "whose": "Model Fatigue",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-vj",
   "what": "Clef minus Jev, points (same-day Jev pass)",
   "value": 1.1363636363636365,
   "shown": "+1.1 points",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-vj-lo",
   "what": "Clef minus Jev, 95% interval, low",
   "value": 0.6168831168831169,
   "shown": "+0.6",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-vj-hi",
   "what": "Clef minus Jev, 95% interval, high",
   "value": 1.6883116883116882,
   "shown": "+1.7",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-vj",
   "what": "Clef-flash minus Jev, points (same-day Jev pass)",
   "value": 1.2337662337662338,
   "shown": "+1.2 points",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-vj-lo",
   "what": "Clef-flash minus Jev, 95% interval, low",
   "value": 0.6818181818181818,
   "shown": "+0.7",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-vj-hi",
   "what": "Clef-flash minus Jev, 95% interval, high",
   "value": 1.8181818181818181,
   "shown": "+1.8",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-vj",
   "what": "Kev-27B minus Jev, points (same-day Jev pass)",
   "value": -0.09740259740259741,
   "shown": "−0.1 points",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-vj-lo",
   "what": "Kev-27B minus Jev, 95% interval, low",
   "value": -0.551948051948052,
   "shown": "−0.6",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-vj-hi",
   "what": "Kev-27B minus Jev, 95% interval, high",
   "value": 0.35714285714285715,
   "shown": "+0.4",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-vj",
   "what": "Kev-9B minus Jev, points (same-day Jev pass)",
   "value": -2.5974025974025974,
   "shown": "−2.6 points",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-vj-lo",
   "what": "Kev-9B minus Jev, 95% interval, low",
   "value": -3.279220779220779,
   "shown": "−3.3",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-vj-hi",
   "what": "Kev-9B minus Jev, 95% interval, high",
   "value": -1.9155844155844155,
   "shown": "−1.9",
   "whose": "Model Fatigue",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-w-ref1k",
   "what": "Jev, messages per 1,000 sent to a person (our arithmetic)",
   "value": 156.16883116883116,
   "shown": "156",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-w-ref1k",
   "what": "GPT-6 Luna, messages per 1,000 sent to a person (our arithmetic)",
   "value": 87.33766233766234,
   "shown": "87",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-w-ref1k",
   "what": "Kev-4B, messages per 1,000 sent to a person (our arithmetic)",
   "value": 257.14285714285717,
   "shown": "257",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 30 September: Jev, GPT-6 Luna, Kev-4B",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-w-ref1k",
   "what": "Clef, messages per 1,000 sent to a person (our arithmetic)",
   "value": 67.85714285714286,
   "shown": "68",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-w-ref1k",
   "what": "Clef-flash, messages per 1,000 sent to a person (our arithmetic)",
   "value": 183.11688311688312,
   "shown": "183",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Clef, Clef-flash and a same-day Jev pass",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-w-ref1k",
   "what": "Kev-27B, messages per 1,000 sent to a person (our arithmetic)",
   "value": 89.6103896103896,
   "shown": "90",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-w-ref1k",
   "what": "Kev-9B, messages per 1,000 sent to a person (our arithmetic)",
   "value": 173.05194805194805,
   "shown": "173",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our run of 2 October: Kev-27B and Kev-9B on our own H100s (Modal, eu-north)",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-gain",
   "what": "Jev, accuracy on our question minus accuracy on the Decision Index's question, points (our arithmetic)",
   "value": 13.214285714285712,
   "shown": "+13.2 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-gain",
   "what": "Clef, accuracy on our question minus accuracy on the Decision Index's question, points (our arithmetic)",
   "value": 0.8441558441558472,
   "shown": "+0.8 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-gain",
   "what": "Clef-flash, accuracy on our question minus accuracy on the Decision Index's question, points (our arithmetic)",
   "value": 4.285714285714281,
   "shown": "+4.3 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-gain",
   "what": "Kev-27B, accuracy on our question minus accuracy on the Decision Index's question, points (our arithmetic)",
   "value": 7.012987012987015,
   "shown": "+7.0 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-gain",
   "what": "Kev-9B, accuracy on our question minus accuracy on the Decision Index's question, points (our arithmetic)",
   "value": 8.019480519480515,
   "shown": "+8.0 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our replication of the Decision Index's zero-shot BANKING77 question, 2 October",
   "url": null,
   "read": "2026-10-02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-mem-gap",
   "what": "Kev-4B, originals minus paraphrases, points",
   "value": 8.0,
   "shown": "+8.0 points",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-mem-lo",
   "what": "Kev-4B, gap, 95% interval, low",
   "value": 2.5,
   "shown": "+2.5",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-4b-mem-hi",
   "what": "Kev-4B, gap, 95% interval, high",
   "value": 14.000000000000002,
   "shown": "+14.0",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-mem-gap",
   "what": "Clef, originals minus paraphrases, points",
   "value": 7.5,
   "shown": "+7.5 points",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-mem-lo",
   "what": "Clef, gap, 95% interval, low",
   "value": 4.0,
   "shown": "+4.0",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-mem-hi",
   "what": "Clef, gap, 95% interval, high",
   "value": 11.5,
   "shown": "+11.5",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-mem-gap",
   "what": "Clef-flash, originals minus paraphrases, points",
   "value": 12.0,
   "shown": "+12.0 points",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-mem-lo",
   "what": "Clef-flash, gap, 95% interval, low",
   "value": 7.5,
   "shown": "+7.5",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-flash-mem-hi",
   "what": "Clef-flash, gap, 95% interval, high",
   "value": 16.5,
   "shown": "+16.5",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-mem-gap",
   "what": "Kev-27B, originals minus paraphrases, points",
   "value": 14.000000000000002,
   "shown": "+14.0 points",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-mem-lo",
   "what": "Kev-27B, gap, 95% interval, low",
   "value": 9.0,
   "shown": "+9.0",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-27b-mem-hi",
   "what": "Kev-27B, gap, 95% interval, high",
   "value": 19.0,
   "shown": "+19.0",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-mem-gap",
   "what": "Kev-9B, originals minus paraphrases, points",
   "value": 11.5,
   "shown": "+11.5 points",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-mem-lo",
   "what": "Kev-9B, gap, 95% interval, low",
   "value": 6.5,
   "shown": "+6.5",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "kev-9b-mem-hi",
   "what": "Kev-9B, gap, 95% interval, high",
   "value": 17.0,
   "shown": "+17.0",
   "whose": "Model Fatigue",
   "source": "Our memorisation probe: 200 training messages and paraphrases of them",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  }
 ]
}