{
 "video": "https://modelfatigue.news/v/decisions-api/",
 "numbers": [
  {
   "id": "n-test",
   "what": "Banking77 test messages every model answered",
   "value": 3080,
   "shown": "3,080",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-reasons",
   "what": "Reasons a message can be sorted into",
   "value": 77,
   "shown": "77",
   "whose": "Model Fatigue",
   "source": "Our frozen protocol for the bank-message test (FREEZE.md)",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-options",
   "what": "Options in each question: the 77 reasons plus none of these",
   "value": 78,
   "shown": "78",
   "whose": "Model Fatigue",
   "source": "Our results write-up (RESULTS.md, \"OpenAI's Decisions API (added 2026-10-07)\")",
   "url": null,
   "read": "2026-10-07",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-shots",
   "what": "Already-sorted example messages shown with each message",
   "value": 5,
   "shown": "5",
   "whose": "Model Fatigue",
   "source": "Our frozen protocol for the bank-message test (FREEZE.md)",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dev-n",
   "what": "Messages in the separate set every model's cut-off was set on (not test messages)",
   "value": 573,
   "shown": "573",
   "whose": "Model Fatigue",
   "source": "Our pass of the Decisions API over the separate set of 573 messages its cut-off was set on (not test messages)",
   "url": null,
   "read": "2026-10-07T01:09",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-acc",
   "what": "the Decisions API, share of the test messages sorted correctly",
   "value": 0.935064935064935,
   "shown": "93.5%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-acc-lo",
   "what": "the Decisions API, share right, 95% interval, low",
   "value": 0.9258090740348918,
   "shown": "92.6",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-acc-hi",
   "what": "the Decisions API, share right, 95% interval, high",
   "value": 0.9432368588042007,
   "shown": "94.3",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-none",
   "what": "the Decisions API, times it answered none of these",
   "value": 4,
   "shown": "4",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-p50",
   "what": "the Decisions API, median time per request from Berlin",
   "value": 0.23868,
   "shown": "239 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-p95",
   "what": "the Decisions API, 95th percentile time per request from Berlin",
   "value": 0.37005249999999973,
   "shown": "370 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-p99",
   "what": "the Decisions API, 99th percentile time per request from Berlin",
   "value": 0.5943527000000001,
   "shown": "594 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-per1k",
   "what": "the Decisions API, dollars per 1,000 decisions (price list × the tokens each response reports)",
   "value": 0.11360217532467533,
   "shown": "$0.114",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-thr",
   "what": "the Decisions API, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.99,
   "shown": "0.99",
   "whose": "Model Fatigue",
   "source": "Our pass of the Decisions API over the separate set of 573 messages its cut-off was set on (not test messages)",
   "url": null,
   "read": "2026-10-07T01:09",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-cov",
   "what": "the Decisions API, share of the test messages it handled without a person",
   "value": 0.8396103896103896,
   "shown": "84.0%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-risk",
   "what": "the Decisions API, wrong among the messages it handled without a person",
   "value": 0.021268368136117557,
   "shown": "2.13%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-ref",
   "what": "the Decisions API, messages per 1,000 sent to a person",
   "value": 160,
   "shown": "160",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-ece",
   "what": "the Decisions API, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.04079220779220774,
   "shown": "0.041",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-auroc",
   "what": "the Decisions API, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8459574652777778,
   "shown": "0.85",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc",
   "what": "Jev, share of the test messages sorted correctly",
   "value": 0.938961038961039,
   "shown": "93.9%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc-lo",
   "what": "Jev, share right, 95% interval, low",
   "value": 0.9299469172345147,
   "shown": "93.0",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-acc-hi",
   "what": "Jev, share right, 95% interval, high",
   "value": 0.9468815164956743,
   "shown": "94.7",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-none",
   "what": "Jev, times it answered none of these",
   "value": 13,
   "shown": "13",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p50",
   "what": "Jev, median time per request from Berlin",
   "value": 0.24234,
   "shown": "242 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p95",
   "what": "Jev, 95th percentile time per request from Berlin",
   "value": 0.30670649999999994,
   "shown": "307 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-p99",
   "what": "Jev, 99th percentile time per request from Berlin",
   "value": 0.38633740000000005,
   "shown": "386 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-per1k",
   "what": "Jev, dollars per 1,000 decisions (price list × the tokens each response reports)",
   "value": 0.07753194545454546,
   "shown": "$0.078",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-thr",
   "what": "Jev, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.99,
   "shown": "0.99",
   "whose": "Model Fatigue",
   "source": "Our passes of Jev and GPT-6 Luna over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-09-30T21:28",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-cov",
   "what": "Jev, share of the test messages it handled without a person",
   "value": 0.8409090909090909,
   "shown": "84.1%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-risk",
   "what": "Jev, wrong among the messages it handled without a person",
   "value": 0.020463320463320462,
   "shown": "2.05%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-ref",
   "what": "Jev, messages per 1,000 sent to a person",
   "value": 159,
   "shown": "159",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-ece",
   "what": "Jev, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.03520129870129873,
   "shown": "0.035",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-auroc",
   "what": "Jev, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8457354845354758,
   "shown": "0.85",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc",
   "what": "GPT-6 Luna through the Responses API, share of the test messages sorted correctly",
   "value": 0.9415584415584416,
   "shown": "94.2%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc-lo",
   "what": "GPT-6 Luna through the Responses API, share right, 95% interval, low",
   "value": 0.9327108134839683,
   "shown": "93.3",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-acc-hi",
   "what": "GPT-6 Luna through the Responses API, share right, 95% interval, high",
   "value": 0.9493059541736182,
   "shown": "94.9",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-none",
   "what": "GPT-6 Luna through the Responses API, times it answered none of these",
   "value": 5,
   "shown": "5",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p50",
   "what": "GPT-6 Luna through the Responses API, median time per request from Berlin",
   "value": 1.04669,
   "shown": "1,047 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p95",
   "what": "GPT-6 Luna through the Responses API, 95th percentile time per request from Berlin",
   "value": 1.6039169999999994,
   "shown": "1,604 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p99",
   "what": "GPT-6 Luna through the Responses API, 99th percentile time per request from Berlin",
   "value": 2.372687800000001,
   "shown": "2,373 ms",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p50s",
   "what": "GPT-6 Luna through the Responses API, median time per request from Berlin, in seconds",
   "value": 1.04669,
   "shown": "1.05 s",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p95s",
   "what": "GPT-6 Luna through the Responses API, 95th percentile time per request from Berlin, in seconds",
   "value": 1.6039169999999994,
   "shown": "1.60 s",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-p99s",
   "what": "GPT-6 Luna through the Responses API, 99th percentile time per request from Berlin, in seconds",
   "value": 2.372687800000001,
   "shown": "2.37 s",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-per1k",
   "what": "GPT-6 Luna through the Responses API, dollars per 1,000 decisions (price list × the tokens each response reports)",
   "value": 0.14332879058441558,
   "shown": "$0.143",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-thr",
   "what": "GPT-6 Luna through the Responses API, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.9,
   "shown": "0.90",
   "whose": "Model Fatigue",
   "source": "Our passes of Jev and GPT-6 Luna over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-09-30T21:28",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-cov",
   "what": "GPT-6 Luna through the Responses API, share of the test messages it handled without a person",
   "value": 0.9087662337662338,
   "shown": "90.9%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-risk",
   "what": "GPT-6 Luna through the Responses API, wrong among the messages it handled without a person",
   "value": 0.02750982493747767,
   "shown": "2.75%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-ref",
   "what": "GPT-6 Luna through the Responses API, messages per 1,000 sent to a person",
   "value": 91,
   "shown": "91",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-ece",
   "what": "GPT-6 Luna through the Responses API, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.02408441558441566,
   "shown": "0.024",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-auroc",
   "what": "GPT-6 Luna through the Responses API, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8631024904214559,
   "shown": "0.86",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc",
   "what": "Clef, share of the test messages sorted correctly",
   "value": 0.9506493506493506,
   "shown": "95.1%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc-lo",
   "what": "Clef, share right, 95% interval, low",
   "value": 0.9424225746200366,
   "shown": "94.2",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-acc-hi",
   "what": "Clef, share right, 95% interval, high",
   "value": 0.9577533617834414,
   "shown": "95.8",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-none",
   "what": "Clef, times it answered none of these",
   "value": 0,
   "shown": "0",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p50",
   "what": "Clef, median time per request from Berlin",
   "value": 0.562495,
   "shown": "562 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p95",
   "what": "Clef, 95th percentile time per request from Berlin",
   "value": 1.1474874999999964,
   "shown": "1,147 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-p99",
   "what": "Clef, 99th percentile time per request from Berlin",
   "value": 2.0007208000000003,
   "shown": "2,001 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-per1k",
   "what": "Clef, dollars per 1,000 decisions (price list × the tokens each response reports)",
   "value": 0.4797513506493506,
   "shown": "$0.480",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-thr",
   "what": "Clef, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.8034,
   "shown": "0.80",
   "whose": "Model Fatigue",
   "source": "Our passes of Clef, Clef-flash, Kev-27B and Kev-9B over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-10-02T15:38",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-cov",
   "what": "Clef, share of the test messages it handled without a person",
   "value": 0.9321428571428572,
   "shown": "93.2%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-risk",
   "what": "Clef, wrong among the messages it handled without a person",
   "value": 0.024033437826541274,
   "shown": "2.40%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-ref",
   "what": "Clef, messages per 1,000 sent to a person",
   "value": 68,
   "shown": "68",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-ece",
   "what": "Clef, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.029707012987013053,
   "shown": "0.030",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-auroc",
   "what": "Clef, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.9001181873741732,
   "shown": "0.90",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-acc",
   "what": "Clef-flash, share of the test messages sorted correctly",
   "value": 0.9516233766233766,
   "shown": "95.2%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-acc-lo",
   "what": "Clef-flash, share right, 95% interval, low",
   "value": 0.9434670440163816,
   "shown": "94.3",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-acc-hi",
   "what": "Clef-flash, share right, 95% interval, high",
   "value": 0.9586545176098705,
   "shown": "95.9",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-none",
   "what": "Clef-flash, times it answered none of these",
   "value": 0,
   "shown": "0",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-p50",
   "what": "Clef-flash, median time per request from Berlin",
   "value": 0.19639,
   "shown": "196 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-p95",
   "what": "Clef-flash, 95th percentile time per request from Berlin",
   "value": 1.013879999999998,
   "shown": "1,014 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-p99",
   "what": "Clef-flash, 99th percentile time per request from Berlin",
   "value": 2.1720025,
   "shown": "2,172 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-per1k",
   "what": "Clef-flash, dollars per 1,000 decisions (price list × the tokens each response reports)",
   "value": 0.17990675649350651,
   "shown": "$0.180",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-thr",
   "what": "Clef-flash, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.7323,
   "shown": "0.73",
   "whose": "Model Fatigue",
   "source": "Our passes of Clef, Clef-flash, Kev-27B and Kev-9B over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-10-02T15:38",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-cov",
   "what": "Clef-flash, share of the test messages it handled without a person",
   "value": 0.8168831168831169,
   "shown": "81.7%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-risk",
   "what": "Clef-flash, wrong among the messages it handled without a person",
   "value": 0.021462639109697933,
   "shown": "2.15%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-ref",
   "what": "Clef-flash, messages per 1,000 sent to a person",
   "value": 183,
   "shown": "183",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-ece",
   "what": "Clef-flash, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.14252301948051946,
   "shown": "0.143",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-auroc",
   "what": "Clef-flash, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.8190472592216047,
   "shown": "0.82",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-acc",
   "what": "Kev-27B, share of the test messages sorted correctly",
   "value": 0.9383116883116883,
   "shown": "93.8%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-acc-lo",
   "what": "Kev-27B, share right, 95% interval, low",
   "value": 0.9292566262741926,
   "shown": "92.9",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-acc-hi",
   "what": "Kev-27B, share right, 95% interval, high",
   "value": 0.9462747239741469,
   "shown": "94.6",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-none",
   "what": "Kev-27B, times it answered none of these",
   "value": 21,
   "shown": "21",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-p50",
   "what": "Kev-27B, median time per request from Berlin",
   "value": 0.528605,
   "shown": "529 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-p95",
   "what": "Kev-27B, 95th percentile time per request from Berlin",
   "value": 0.5508195,
   "shown": "551 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-p99",
   "what": "Kev-27B, 99th percentile time per request from Berlin",
   "value": 0.5976951000000001,
   "shown": "598 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-thr",
   "what": "Kev-27B, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.7928,
   "shown": "0.79",
   "whose": "Model Fatigue",
   "source": "Our passes of Clef, Clef-flash, Kev-27B and Kev-9B over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-10-02T15:38",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-cov",
   "what": "Kev-27B, share of the test messages it handled without a person",
   "value": 0.9103896103896104,
   "shown": "91.0%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-risk",
   "what": "Kev-27B, wrong among the messages it handled without a person",
   "value": 0.02282453637660485,
   "shown": "2.28%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-ref",
   "what": "Kev-27B, messages per 1,000 sent to a person",
   "value": 90,
   "shown": "90",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-ece",
   "what": "Kev-27B, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.014207987012986964,
   "shown": "0.014",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-auroc",
   "what": "Kev-27B, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.920481697322892,
   "shown": "0.92",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-acc",
   "what": "Kev-9B, share of the test messages sorted correctly",
   "value": 0.9133116883116883,
   "shown": "91.3%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-acc-lo",
   "what": "Kev-9B, share right, 95% interval, low",
   "value": 0.9028523242309728,
   "shown": "90.3",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-acc-hi",
   "what": "Kev-9B, share right, 95% interval, high",
   "value": 0.922741311966165,
   "shown": "92.3",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-none",
   "what": "Kev-9B, times it answered none of these",
   "value": 110,
   "shown": "110",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-p50",
   "what": "Kev-9B, median time per request from Berlin",
   "value": 0.436515,
   "shown": "437 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-p95",
   "what": "Kev-9B, 95th percentile time per request from Berlin",
   "value": 0.5482069999999997,
   "shown": "548 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-p99",
   "what": "Kev-9B, 99th percentile time per request from Berlin",
   "value": 0.6647247000000001,
   "shown": "665 ms",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-thr",
   "what": "Kev-9B, cut-off on its stated confidence, set on the separate set for a 2% error target",
   "value": 0.8127,
   "shown": "0.81",
   "whose": "Model Fatigue",
   "source": "Our passes of Clef, Clef-flash, Kev-27B and Kev-9B over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-10-02T15:38",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-cov",
   "what": "Kev-9B, share of the test messages it handled without a person",
   "value": 0.826948051948052,
   "shown": "82.7%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-risk",
   "what": "Kev-9B, wrong among the messages it handled without a person",
   "value": 0.024734982332155476,
   "shown": "2.47%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-ref",
   "what": "Kev-9B, messages per 1,000 sent to a person",
   "value": 173,
   "shown": "173",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-ece",
   "what": "Kev-9B, calibration error of its stated confidence (ECE; lower is closer)",
   "value": 0.03842435064935068,
   "shown": "0.038",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-auroc",
   "what": "Kev-9B, how well its confidence separates its right answers from its wrong ones (AUROC)",
   "value": 0.9057099794826321,
   "shown": "0.91",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-risk-lo",
   "what": "the Decisions API, wrong among handled, 95% interval, low",
   "value": 0.016376577961557774,
   "shown": "1.6%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-risk-hi",
   "what": "the Decisions API, wrong among handled, 95% interval, high",
   "value": 0.027580396086584788,
   "shown": "2.8%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-risk-lo",
   "what": "Jev, wrong among handled, 95% interval, low",
   "value": 0.015678870856910845,
   "shown": "1.6%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-risk-hi",
   "what": "Jev, wrong among handled, 95% interval, high",
   "value": 0.026668202302837993,
   "shown": "2.7%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-wrong-let-through",
   "what": "The Decisions API's wrong answers at 0.99 or above, which the 2% cut-off would let through without a person",
   "value": 55,
   "shown": "55",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-repeat-same",
   "what": "Messages sent to the Decisions API a second time that got the same answer",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-repeat-n",
   "what": "Messages sent to the Decisions API a second time",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dvj-jr",
   "what": "Messages Jev got right and the Decisions API got wrong",
   "value": 38,
   "shown": "38",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dvj-dr",
   "what": "Messages the Decisions API got right and Jev got wrong",
   "value": 26,
   "shown": "26",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "lvj-jr",
   "what": "Messages Jev got right and Luna got wrong",
   "value": 23,
   "shown": "23",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "lvj-lr",
   "what": "Messages Luna got right and Jev got wrong",
   "value": 31,
   "shown": "31",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph-lr",
   "what": "Messages Luna got right and the Decisions API got wrong (post hoc)",
   "value": 41,
   "shown": "41",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph-dr",
   "what": "Messages the Decisions API got right and Luna got wrong (post hoc)",
   "value": 21,
   "shown": "21",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "speedup",
   "what": "Luna's median time over the Decisions API's, same messages, same night (our arithmetic)",
   "value": 4.385327635327635,
   "shown": "4.4×",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "openai-150ms",
   "what": "OpenAI's time per request for the Decisions API (launch clip)",
   "value": 0.15,
   "shown": "150 ms",
   "whose": "OpenAI",
   "source": "OpenAI Developers' Decisions API launch clip, its on-screen text",
   "url": "https://x.com/OpenAIDevs/status/2105003318917697873",
   "read": "2026-09-29",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "openai-1-6s",
   "what": "OpenAI's time per request for the Responses API (launch clip)",
   "value": 1.6,
   "shown": "1.6 s",
   "whose": "OpenAI",
   "source": "OpenAI Developers' Decisions API launch clip, its on-screen text",
   "url": "https://x.com/OpenAIDevs/status/2105003318917697873",
   "read": "2026-09-29",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "openai-10x",
   "what": "OpenAI's speed-up, Decisions API over the Responses API (launch clip)",
   "value": 10,
   "shown": "10×",
   "whose": "OpenAI",
   "source": "OpenAI Developers' Decisions API launch clip, its on-screen text",
   "url": "https://x.com/OpenAIDevs/status/2105003318917697873",
   "read": "2026-09-29",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "openai-10k",
   "what": "Customer requests in OpenAI's launch clip",
   "value": 10000,
   "shown": "10,000",
   "whose": "OpenAI",
   "source": "OpenAI Developers' Decisions API launch clip, its on-screen text",
   "url": "https://x.com/OpenAIDevs/status/2105003318917697873",
   "read": "2026-09-29",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "docs-10x",
   "what": "OpenAI's speed-up in the Decisions guide",
   "value": 10,
   "shown": "10×",
   "whose": "OpenAI",
   "source": "OpenAI, the Decisions API guide",
   "url": "https://developers.openai.com/api/docs/guides/decisions",
   "read": "2026-10-07T01:05",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "under-150",
   "what": "Decisions calls that came back to Berlin in under 150 ms",
   "value": 0,
   "shown": "0",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "fastest",
   "what": "The fastest Decisions call from Berlin",
   "value": 0.16462,
   "shown": "165 ms",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "floor-start",
   "what": "The trip from Berlin to OpenAI's API and back with no model work, median, before the run",
   "value": 0.1661,
   "shown": "166 ms",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "floor-end",
   "what": "The same trip, median, after the run",
   "value": 0.1666,
   "shown": "167 ms",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-minus-floor",
   "what": "The Decisions API's median time minus the trip, before the run (our arithmetic)",
   "value": 0.07308,
   "shown": "73 ms",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "run-window",
   "what": "When the timed run went out (first network-floor check to the last)",
   "value": "23:13 to 00:39 UTC",
   "shown": "23:13 to 00:39 UTC",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "win-jev",
   "what": "When Jev's test pass ran",
   "value": "23:14 to 23:27 UTC",
   "shown": "23:14 to 23:27 UTC",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "win-decisions",
   "what": "When the Decisions API's test pass ran",
   "value": "23:27 to 23:40 UTC",
   "shown": "23:27 to 23:40 UTC",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "win-luna",
   "what": "When Luna's test pass ran",
   "value": "23:41 to 00:38 UTC",
   "shown": "23:41 to 00:38 UTC",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "rate-dec",
   "what": "The Decisions API's price per million input tokens (nothing else billed)",
   "value": 0.1,
   "shown": "$0.10",
   "whose": "OpenAI",
   "source": "OpenAI, the Decisions API guide",
   "url": "https://developers.openai.com/api/docs/guides/decisions",
   "read": "2026-10-07T01:05",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "rate-jev",
   "what": "Jev's price per million input tokens (TypeSafe's rate card)",
   "value": 0.042,
   "shown": "$0.042",
   "whose": "Model Fatigue",
   "source": "Our frozen protocol for the bank-message test (FREEZE.md)",
   "url": null,
   "read": "2026-09-30",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-tok",
   "what": "the Decisions API, input tokens per request as its service counted them, average",
   "value": 1136.0217532467532,
   "shown": "1,136",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-tok",
   "what": "Jev, input tokens per request as its service counted them, average",
   "value": 1845.9987012987012,
   "shown": "1,846",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "luna-tok",
   "what": "GPT-6 Luna through the Responses API, input tokens per request as its service counted them, average",
   "value": 1066.0217532467532,
   "shown": "1,066",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-tok",
   "what": "Clef, input tokens per request as its service counted them, average",
   "value": 1998.963961038961,
   "shown": "1,999",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-tok",
   "what": "Clef-flash, input tokens per request as its service counted them, average",
   "value": 1998.963961038961,
   "shown": "1,999",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "cost-ratio",
   "what": "The Decisions API's cost per decision over Jev's (our arithmetic)",
   "value": 1.4652305531437582,
   "shown": "1.5×",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-run-cost",
   "what": "The Decisions API test pass, all 3,080 messages, at the price list",
   "value": 0.3498947,
   "shown": "$0.35",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-99",
   "what": "Decisions answers stated at 0.99 or 1.00",
   "value": 2586,
   "shown": "2,586",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "n-100",
   "what": "Decisions answers stated at 1.00",
   "value": 2146,
   "shown": "2,146",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "bin90-n",
   "what": "Decisions answers stated at 0.90 or more",
   "value": 2902,
   "shown": "2,902",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "bin90-mean",
   "what": "The Decisions API's average stated confidence on those answers",
   "value": 0.9940902825637491,
   "shown": "0.994",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "bin90-right",
   "what": "Share of those answers that were right",
   "value": 0.9607167470709855,
   "shown": "96.1%",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "top-dec-dev",
   "what": "The Decisions API, separate set: wrong among its answers stated at 1.00",
   "value": 0.010582010582010581,
   "shown": "1.06%",
   "whose": "Model Fatigue",
   "source": "Our pass of the Decisions API over the separate set of 573 messages its cut-off was set on (not test messages)",
   "url": null,
   "read": "2026-10-07T01:09",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "top-jev-dev",
   "what": "Jev, separate set: wrong among its answers stated at 1.00",
   "value": 0.018561484918793503,
   "shown": "1.86%",
   "whose": "Model Fatigue",
   "source": "Our passes of Jev and GPT-6 Luna over the separate set of 573 messages their cut-offs were set on (not test messages)",
   "url": null,
   "read": "2026-09-30T21:28",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "top-dec-test",
   "what": "The Decisions API, test: wrong among its answers stated at 1.00",
   "value": 0.015843429636533086,
   "shown": "1.58%",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "top-jev-test",
   "what": "Jev, test: wrong among its answers stated at 1.00",
   "value": 0.01589242053789731,
   "shown": "1.59%",
   "whose": "Model Fatigue",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "target",
   "what": "The error target every cut-off was set for: wrong answers among those a model handles alone",
   "value": 0.02,
   "shown": "2%",
   "whose": "Model Fatigue",
   "source": "Our protocol addendum for the Decisions API, committed before the first test call (commit 23057db1)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "target-strict",
   "what": "The stricter target no setting could meet for the Decisions API or Jev",
   "value": 0.01,
   "shown": "1%",
   "whose": "Model Fatigue",
   "source": "Our results write-up (RESULTS.md, \"OpenAI's Decisions API (added 2026-10-07)\")",
   "url": null,
   "read": "2026-10-07",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-decisions",
   "what": "Message A: the Decisions API's probability for the answer it picked",
   "value": 0.36,
   "shown": "36%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-decisions-2",
   "what": "Message A: the Decisions API's probability for its second answer",
   "value": 0.19,
   "shown": "19%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-jev",
   "what": "Message A: Jev's probability for the answer it picked",
   "value": 0.6,
   "shown": "60%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-jev-2",
   "what": "Message A: Jev's probability for its second answer",
   "value": 0.22,
   "shown": "22%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-clef",
   "what": "Message A: Clef's probability for the answer it picked",
   "value": 0.5761,
   "shown": "58%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-a-clef-2",
   "what": "Message A: Clef's probability for its second answer",
   "value": 0.1003,
   "shown": "10%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-decisions",
   "what": "Message B: the Decisions API's probability for the answer it picked",
   "value": 0.84,
   "shown": "84%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-decisions-2",
   "what": "Message B: the Decisions API's probability for its second answer",
   "value": 0.06,
   "shown": "6%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-jev",
   "what": "Message B: Jev's probability for the answer it picked",
   "value": 0.48,
   "shown": "48%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-jev-2",
   "what": "Message B: Jev's probability for its second answer",
   "value": 0.36,
   "shown": "36%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-clef",
   "what": "Message B: Clef's probability for the answer it picked",
   "value": 0.34,
   "shown": "34%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-b-clef-2",
   "what": "Message B: Clef's probability for its second answer",
   "value": 0.2386,
   "shown": "24%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-decisions",
   "what": "Message C: the Decisions API's probability for the answer it picked",
   "value": 0.99,
   "shown": "99%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-decisions-2",
   "what": "Message C: the Decisions API's probability for its second answer",
   "value": 0.01,
   "shown": "1%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-jev",
   "what": "Message C: Jev's probability for the answer it picked",
   "value": 0.94,
   "shown": "94%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-jev-2",
   "what": "Message C: Jev's probability for its second answer",
   "value": 0.06,
   "shown": "6%",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-clef",
   "what": "Message C: Clef's probability for the answer it picked",
   "value": 0.7409,
   "shown": "74%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "sbs-c-clef-2",
   "what": "Message C: Clef's probability for its second answer",
   "value": 0.166,
   "shown": "17%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "all-right",
   "what": "Test messages the Decisions API, Jev and Clef all got right",
   "value": 2841,
   "shown": "2,841",
   "whose": "Model Fatigue, arithmetic on our runs",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "all-wrong",
   "what": "Test messages all three got wrong",
   "value": 125,
   "shown": "125",
   "whose": "Model Fatigue, arithmetic on our runs",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "opt-max",
   "what": "The most options the Decisions API accepted in one question",
   "value": 255,
   "shown": "255",
   "whose": "Model Fatigue",
   "source": "Our calls to the Decisions API with 1 to 1,000 options, to find its limit",
   "url": null,
   "read": "2026-10-07T01:08",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "opt-refused",
   "what": "Options in the question the API refused as too many",
   "value": 256,
   "shown": "256",
   "whose": "Model Fatigue",
   "source": "Our calls to the Decisions API with 1 to 1,000 options, to find its limit",
   "url": null,
   "read": "2026-10-07T01:08",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "opt-min",
   "what": "The fewest options the API accepted in one question",
   "value": 2,
   "shown": "2",
   "whose": "Model Fatigue",
   "source": "Our calls to the Decisions API with 1 to 1,000 options, to find its limit",
   "url": null,
   "read": "2026-10-07T01:08",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-di",
   "what": "the Decisions API, share right with the message alone (the Decision Index's question)",
   "value": 0.7769480519480519,
   "shown": "77.7%",
   "whose": "Model Fatigue",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-di-f1",
   "what": "the Decisions API, macro-F1 with the message alone",
   "value": 0.7694673649101087,
   "shown": "76.95",
   "whose": "Model Fatigue",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dec-gain",
   "what": "the Decisions API, share right on our question (five examples) minus on the message alone, points",
   "value": 15.81,
   "shown": "+15.8 points",
   "whose": "Model Fatigue, arithmetic on our runs",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di",
   "what": "Jev, share right with the message alone (the Decision Index's question)",
   "value": 0.8071428571428572,
   "shown": "80.7%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-di-f1",
   "what": "Jev, macro-F1 with the message alone",
   "value": 0.8000046642892913,
   "shown": "80.00",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "jev-gain",
   "what": "Jev, share right on our question (five examples) minus on the message alone, points",
   "value": 13.18,
   "shown": "+13.2 points",
   "whose": "Model Fatigue, arithmetic on our runs",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di",
   "what": "Clef, share right with the message alone (the Decision Index's question)",
   "value": 0.9422077922077922,
   "shown": "94.2%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-di-f1",
   "what": "Clef, macro-F1 with the message alone",
   "value": 0.9420316998167463,
   "shown": "94.20",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "clef-gain",
   "what": "Clef, share right on our question (five examples) minus on the message alone, points",
   "value": 0.84,
   "shown": "+0.8 points",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-di",
   "what": "Clef-flash, share right with the message alone (the Decision Index's question)",
   "value": 0.9087662337662338,
   "shown": "90.9%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-di-f1",
   "what": "Clef-flash, macro-F1 with the message alone",
   "value": 0.9084736076902837,
   "shown": "90.85",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-gain",
   "what": "Clef-flash, share right on our question (five examples) minus on the message alone, points",
   "value": 4.29,
   "shown": "+4.3 points",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-di",
   "what": "Kev-27B, share right with the message alone (the Decision Index's question)",
   "value": 0.8681818181818182,
   "shown": "86.8%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-di-f1",
   "what": "Kev-27B, macro-F1 with the message alone",
   "value": 0.8640871097889219,
   "shown": "86.41",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k27-gain",
   "what": "Kev-27B, share right on our question (five examples) minus on the message alone, points",
   "value": 7.01,
   "shown": "+7.0 points",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-di",
   "what": "Kev-9B, share right with the message alone (the Decision Index's question)",
   "value": 0.8331168831168831,
   "shown": "83.3%",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-di-f1",
   "what": "Kev-9B, macro-F1 with the message alone",
   "value": 0.8302597361284337,
   "shown": "83.03",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "k9-gain",
   "what": "Kev-9B, share right on our question (five examples) minus on the message alone, points",
   "value": 8.02,
   "shown": "+8.0 points",
   "whose": "Model Fatigue",
   "source": "Our test of Clef, Clef-flash, Kev-27B and Kev-9B on the same messages (the Clef vs Jev video's run)",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "di-options",
   "what": "Options in the Decision Index's question (the reasons, without none of these)",
   "value": 77,
   "shown": "77",
   "whose": "Model Fatigue",
   "source": "Our results write-up (RESULTS.md, \"OpenAI's Decisions API (added 2026-10-07)\")",
   "url": null,
   "read": "2026-10-07",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "cf-clef",
   "what": "Clef on BANKING77 in Cloudflare's table (macro-F1)",
   "value": 94.2,
   "shown": "94.20",
   "whose": "Cloudflare",
   "source": "Cloudflare, launch post for Clef and Clef-flash",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "cf-jev",
   "what": "Jev on BANKING77 in Cloudflare's table (macro-F1)",
   "value": 79.74,
   "shown": "79.74",
   "whose": "Cloudflare",
   "source": "Cloudflare, launch post for Clef and Clef-flash",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-n",
   "what": "Messages from the public collection's training part, asked as written and reworded",
   "value": 200,
   "shown": "200",
   "whose": "Model Fatigue",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-orig",
   "what": "The Decisions API, share right on the original messages (no examples)",
   "value": 0.695,
   "shown": "69.5%",
   "whose": "Model Fatigue",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-para",
   "what": "The Decisions API, share right on the same messages reworded",
   "value": 0.69,
   "shown": "69.0%",
   "whose": "Model Fatigue",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dvj",
   "what": "The Decisions API minus Jev, share right, same messages, same night",
   "value": -0.38961038961038963,
   "shown": "−0.39 points",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dvj-lo",
   "what": "The Decisions API minus Jev, 95% interval, low",
   "value": -0.9090909090909091,
   "shown": "−0.91",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "dvj-hi",
   "what": "The Decisions API minus Jev, 95% interval, high",
   "value": 0.12987012987012986,
   "shown": "+0.13",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "lvj",
   "what": "GPT-6 Luna through the Responses API minus Jev, share right, same messages, same night",
   "value": 0.2597402597402597,
   "shown": "+0.26 points",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "lvj-lo",
   "what": "Luna minus Jev, 95% interval, low",
   "value": -0.22727272727272727,
   "shown": "−0.23",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "lvj-hi",
   "what": "Luna minus Jev, 95% interval, high",
   "value": 0.7467532467532467,
   "shown": "+0.75",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph",
   "what": "Luna through the Responses API minus the Decisions API, same messages (post hoc: compared after the run, not in the frozen plan)",
   "value": 0.6493506493506493,
   "shown": "+0.65 points",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph-lo",
   "what": "Luna minus the Decisions API, 95% interval, low (post hoc)",
   "value": 0.16233766233766234,
   "shown": "+0.16",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph-hi",
   "what": "Luna minus the Decisions API, 95% interval, high (post hoc)",
   "value": 1.1688311688311688,
   "shown": "+1.17",
   "whose": "Model Fatigue",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "ph-ratio",
   "what": "Luna-only right answers per Decisions-only right answer (post hoc, our arithmetic)",
   "value": 1.9523809523809523,
   "shown": "2.0",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our test, the night the Decisions API opened: Jev, the Decisions API and GPT-6 Luna through the Responses API, 3,080 Banking77 test messages, five sorted examples each, 78 options, one request at a time from a Mac mini in Berlin",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "server-p50",
   "what": "Time OpenAI's servers report spending on each Decisions call (openai-processing-ms), median",
   "value": 0.065,
   "shown": "65 ms",
   "whose": "OpenAI, read from each response",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "server-p95",
   "what": "OpenAI's reported server time per Decisions call, 95th percentile",
   "value": 0.144,
   "shown": "144 ms",
   "whose": "OpenAI, read from each response",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "server-p99",
   "what": "OpenAI's reported server time per Decisions call, 99th percentile",
   "value": 0.3670500000000002,
   "shown": "367 ms",
   "whose": "OpenAI, read from each response",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "server-rest",
   "what": "Our time per Decisions call minus OpenAI's reported server time, median (our arithmetic)",
   "value": 0.17044,
   "shown": "170 ms",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "rate-ratio",
   "what": "The Decisions API's price per token over Jev's (our arithmetic)",
   "value": 2.380952380952381,
   "shown": "2.4×",
   "whose": "Model Fatigue, arithmetic on the two price lists",
   "source": "Our arithmetic on two price lists: OpenAI's Decisions guide (read 7 Oct 2026, 01:05 CEST) and Jev's rate card in our frozen protocol (30 Sep 2026)",
   "url": null,
   "read": "2026-10-07T01:05",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "tok-ratio",
   "what": "Jev's token count per request over the Decisions API's (our arithmetic)",
   "value": 1.6249677402944371,
   "shown": "1.6×",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "flash-cost-ratio",
   "what": "Clef-flash's cost per decision over Jev's (our arithmetic)",
   "value": 2.320421026955659,
   "shown": "2.3×",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "share-99",
   "what": "Share of Decisions answers stated at 0.99 or 1.00 (our arithmetic)",
   "value": 0.8396103896103896,
   "shown": "84%",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "bin90-share",
   "what": "Share of Decisions answers stated at 0.90 or more (our arithmetic)",
   "value": 0.9422077922077922,
   "shown": "94%",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic on the raw call records (derived.json, this page's own file)",
   "url": null,
   "read": "2026-10-07T01:13",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "di-gap",
   "what": "Jev minus the Decisions API with the message alone (our arithmetic)",
   "value": 3.019480519480522,
   "shown": "3.0 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our arithmetic across the two nights' passes: the Decisions API, Jev and Luna on 7 October, Clef, Clef-flash, the Kev models and every model's message-alone pass but the Decisions API's on 2 October",
   "url": null,
   "read": "2026-10-02T16:02",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "cf-gap",
   "what": "Clef minus Jev on BANKING77 in Cloudflare's table (our arithmetic)",
   "value": 14.460000000000008,
   "shown": "14.5 points",
   "whose": "Model Fatigue, arithmetic on Cloudflare's numbers",
   "source": "Cloudflare, launch post for Clef and Clef-flash",
   "url": "https://blog.cloudflare.com/clef-decision-models/",
   "read": "2026-10-02T15:31",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-gap",
   "what": "Originals minus rewordings (our arithmetic)",
   "value": 0.5,
   "shown": "+0.5 points",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-lo",
   "what": "Originals minus rewordings, 95% interval, low",
   "value": -3.5000000000000004,
   "shown": "−3.5",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  },
  {
   "id": "mem-hi",
   "what": "Originals minus rewordings, 95% interval, high",
   "value": 5.0,
   "shown": "+5.0",
   "whose": "Model Fatigue, arithmetic on our run",
   "source": "Our passes of the Decisions API without examples, and the memorisation check (about six requests at a time, from another machine; no times quoted from them)",
   "url": null,
   "read": "2026-10-07T02:43",
   "moves": false,
   "caveat": null,
   "now": null,
   "now_shown": null,
   "now_read": null
  }
 ]
}