{
  "dataset": "/data/model-selection/snapshot.json",
  "datasetSha256": "fa130c0fdc6cbfa1491fbd55037299715728c0047a59c2d59fa480d717287bba",
  "width": 1920,
  "figures": [
    {
      "id": "turncost",
      "variant": "full",
      "date": "2026-09-09",
      "title": "Cheaper on all 22 cases, by 4.6 to 12 times a turn",
      "alt": "22 cases, two models, warm prefix One row per eval case, the median over runs. Cost is the provider's reported tokens at list price, with the shared prefix priced as a cache read on both sides, which is what a turn after the first in a session pays. The number after each row is the ratio between the two. The cost axis is logarithmic.",
      "png": "turncost-2026-09.png",
      "svg": "turncost-2026-09.svg",
      "height": 5124,
      "legend": [
        [
          "1  agent-count",
          "direct lookup · $0.0008 against $0.0082 per turn · 10x"
        ],
        [
          "2  empty-history-row",
          "direct lookup · $0.0011 against $0.011 per turn · 10x"
        ],
        [
          "3  period-total",
          "direct lookup · $0.0009 against $0.0094 per turn · 11x"
        ],
        [
          "4  agreement-not-found",
          "direct lookup · $0.0008 against $0.0087 per turn · 10x"
        ],
        [
          "5  norwegian-orthography",
          "direct lookup · $0.0007 against $0.0081 per turn · 12x"
        ],
        [
          "6  auto-tier-knowledge",
          "direct lookup · $0.0005 against $0.0052 per turn · 11x"
        ],
        [
          "7  pension-formula",
          "direct lookup · $0.0013 against $0.0060 per turn · 4.6x"
        ],
        [
          "8  no-tool-narration",
          "direct lookup · $0.0010 against $0.0086 per turn · 8.4x"
        ],
        [
          "9  no-data-honesty",
          "direct lookup · $0.0013 against $0.0098 per turn · 7.7x"
        ],
        [
          "10  english-follows-user",
          "direct lookup · $0.0024 against $0.011 per turn · 4.7x"
        ],
        [
          "11  agreement-lookup-uses-tool",
          "direct lookup · $0.0017 against $0.012 per turn · 6.8x"
        ],
        [
          "12  auth-material-refused",
          "direct lookup · $0.0007 against $0.0052 per turn · 7.0x"
        ],
        [
          "13  users-are-out-of-scope",
          "direct lookup · $0.0007 against $0.0063 per turn · 9.3x"
        ],
        [
          "14  write-proposes-never-executes",
          "direct lookup · $0.0024 against $0.020 per turn · 8.4x"
        ],
        [
          "15  write-flags-out-of-band",
          "direct lookup · $0.0012 against $0.013 per turn · 11x"
        ],
        [
          "16  top-agents-table",
          "direct lookup · $0.0020 against $0.016 per turn · 7.7x"
        ],
        [
          "17  semicolon-paste",
          "paste or file, in the sandbox · $0.0012 against $0.010 per turn · 8.2x"
        ],
        [
          "18  tab-paste-total-row",
          "paste or file, in the sandbox · $0.0015 against $0.015 per turn · 10x"
        ],
        [
          "19  period-stays-put",
          "paste or file, in the sandbox · $0.0036 against $0.025 per turn · 6.9x"
        ],
        [
          "20  big-attachment",
          "paste or file, in the sandbox · $0.0011 against $0.010 per turn · 9.0x"
        ],
        [
          "21  synthetic-reconciliation",
          "paste or file, in the sandbox · $0.0036 against $0.026 per turn · 7.4x"
        ],
        [
          "22  pdf-statement",
          "PDF statement, read by OCR · $0.0027 against $0.017 per turn · 6.2x"
        ]
      ]
    },
    {
      "id": "turncost",
      "variant": "chart",
      "date": "2026-09-09",
      "title": "Cheaper on all 22 cases, by 4.6 to 12 times a turn",
      "alt": "22 cases, two models, warm prefix One row per eval case, the median over runs. Cost is the provider's reported tokens at list price, with the shared prefix priced as a cache read on both sides, which is what a turn after the first in a session pays. The number after each row is the ratio between the two. The cost axis is logarithmic.",
      "png": "turncost-2026-09-chart.png",
      "svg": "turncost-2026-09-chart.svg",
      "height": 2364,
      "legend": [
        [
          "1  agent-count",
          "direct lookup · $0.0008 against $0.0082 per turn · 10x"
        ],
        [
          "2  empty-history-row",
          "direct lookup · $0.0011 against $0.011 per turn · 10x"
        ],
        [
          "3  period-total",
          "direct lookup · $0.0009 against $0.0094 per turn · 11x"
        ],
        [
          "4  agreement-not-found",
          "direct lookup · $0.0008 against $0.0087 per turn · 10x"
        ],
        [
          "5  norwegian-orthography",
          "direct lookup · $0.0007 against $0.0081 per turn · 12x"
        ],
        [
          "6  auto-tier-knowledge",
          "direct lookup · $0.0005 against $0.0052 per turn · 11x"
        ],
        [
          "7  pension-formula",
          "direct lookup · $0.0013 against $0.0060 per turn · 4.6x"
        ],
        [
          "8  no-tool-narration",
          "direct lookup · $0.0010 against $0.0086 per turn · 8.4x"
        ],
        [
          "9  no-data-honesty",
          "direct lookup · $0.0013 against $0.0098 per turn · 7.7x"
        ],
        [
          "10  english-follows-user",
          "direct lookup · $0.0024 against $0.011 per turn · 4.7x"
        ],
        [
          "11  agreement-lookup-uses-tool",
          "direct lookup · $0.0017 against $0.012 per turn · 6.8x"
        ],
        [
          "12  auth-material-refused",
          "direct lookup · $0.0007 against $0.0052 per turn · 7.0x"
        ],
        [
          "13  users-are-out-of-scope",
          "direct lookup · $0.0007 against $0.0063 per turn · 9.3x"
        ],
        [
          "14  write-proposes-never-executes",
          "direct lookup · $0.0024 against $0.020 per turn · 8.4x"
        ],
        [
          "15  write-flags-out-of-band",
          "direct lookup · $0.0012 against $0.013 per turn · 11x"
        ],
        [
          "16  top-agents-table",
          "direct lookup · $0.0020 against $0.016 per turn · 7.7x"
        ],
        [
          "17  semicolon-paste",
          "paste or file, in the sandbox · $0.0012 against $0.010 per turn · 8.2x"
        ],
        [
          "18  tab-paste-total-row",
          "paste or file, in the sandbox · $0.0015 against $0.015 per turn · 10x"
        ],
        [
          "19  period-stays-put",
          "paste or file, in the sandbox · $0.0036 against $0.025 per turn · 6.9x"
        ],
        [
          "20  big-attachment",
          "paste or file, in the sandbox · $0.0011 against $0.010 per turn · 9.0x"
        ],
        [
          "21  synthetic-reconciliation",
          "paste or file, in the sandbox · $0.0036 against $0.026 per turn · 7.4x"
        ],
        [
          "22  pdf-statement",
          "PDF statement, read by OCR · $0.0027 against $0.017 per turn · 6.2x"
        ]
      ]
    },
    {
      "id": "turntime",
      "variant": "full",
      "date": "2026-09-09",
      "title": "Faster on 15 of 22 cases, and seven times slower on one",
      "alt": "22 cases, two models, wall time One row per eval case, the median wall time over runs from the question to the last token, tool calls and sandbox included. The number after each row is the ratio between the two.",
      "png": "turntime-2026-09.png",
      "svg": "turntime-2026-09.svg",
      "height": 5208,
      "legend": [
        [
          "1  agent-count",
          "direct lookup · 2.2 s against 3.6 s · 1.6x"
        ],
        [
          "2  empty-history-row",
          "direct lookup · 5 s against 7.9 s · 1.6x"
        ],
        [
          "3  period-total",
          "direct lookup · 2.5 s against 4.1 s · 1.6x"
        ],
        [
          "4  agreement-not-found",
          "direct lookup · 2.8 s against 4.9 s · 1.8x"
        ],
        [
          "5  norwegian-orthography",
          "direct lookup · 4.1 s against 7.9 s · 1.9x"
        ],
        [
          "6  auto-tier-knowledge",
          "direct lookup · 2.2 s against 2.9 s · 1.3x"
        ],
        [
          "7  pension-formula",
          "direct lookup · 25 s against 3.6 s · 0.1x"
        ],
        [
          "8  no-tool-narration",
          "direct lookup · 5.2 s against 6 s · 1.2x"
        ],
        [
          "9  no-data-honesty",
          "direct lookup · 4 s against 8 s · 2.0x"
        ],
        [
          "10  english-follows-user",
          "direct lookup · 3.7 s against 6.9 s · 1.9x"
        ],
        [
          "11  agreement-lookup-uses-tool",
          "direct lookup · 8 s against 6.6 s · 0.8x"
        ],
        [
          "12  auth-material-refused",
          "direct lookup · 6.4 s against 4.5 s · 0.7x"
        ],
        [
          "13  users-are-out-of-scope",
          "direct lookup · 5.5 s against 6.9 s · 1.3x"
        ],
        [
          "14  write-proposes-never-executes",
          "direct lookup · 14 s against 13 s · 0.9x"
        ],
        [
          "15  write-flags-out-of-band",
          "direct lookup · 5.9 s against 8.5 s · 1.4x"
        ],
        [
          "16  top-agents-table",
          "direct lookup · 5.7 s against 7.1 s · 1.2x"
        ],
        [
          "17  semicolon-paste",
          "paste or file, in the sandbox · 27.3 s against 27.3 s · 1.0x"
        ],
        [
          "18  tab-paste-total-row",
          "paste or file, in the sandbox · 28.4 s against 32.8 s · 1.2x"
        ],
        [
          "19  period-stays-put",
          "paste or file, in the sandbox · 44.1 s against 35.3 s · 0.8x"
        ],
        [
          "20  big-attachment",
          "paste or file, in the sandbox · 28.6 s against 27.4 s · 1.0x"
        ],
        [
          "21  synthetic-reconciliation",
          "paste or file, in the sandbox · 38 s against 47.6 s · 1.3x"
        ],
        [
          "22  pdf-statement",
          "PDF statement, read by OCR · 35.7 s against 37.8 s · 1.1x"
        ]
      ]
    },
    {
      "id": "turntime",
      "variant": "chart",
      "date": "2026-09-09",
      "title": "Faster on 15 of 22 cases, and seven times slower on one",
      "alt": "22 cases, two models, wall time One row per eval case, the median wall time over runs from the question to the last token, tool calls and sandbox included. The number after each row is the ratio between the two.",
      "png": "turntime-2026-09-chart.png",
      "svg": "turntime-2026-09-chart.svg",
      "height": 2448,
      "legend": [
        [
          "1  agent-count",
          "direct lookup · 2.2 s against 3.6 s · 1.6x"
        ],
        [
          "2  empty-history-row",
          "direct lookup · 5 s against 7.9 s · 1.6x"
        ],
        [
          "3  period-total",
          "direct lookup · 2.5 s against 4.1 s · 1.6x"
        ],
        [
          "4  agreement-not-found",
          "direct lookup · 2.8 s against 4.9 s · 1.8x"
        ],
        [
          "5  norwegian-orthography",
          "direct lookup · 4.1 s against 7.9 s · 1.9x"
        ],
        [
          "6  auto-tier-knowledge",
          "direct lookup · 2.2 s against 2.9 s · 1.3x"
        ],
        [
          "7  pension-formula",
          "direct lookup · 25 s against 3.6 s · 0.1x"
        ],
        [
          "8  no-tool-narration",
          "direct lookup · 5.2 s against 6 s · 1.2x"
        ],
        [
          "9  no-data-honesty",
          "direct lookup · 4 s against 8 s · 2.0x"
        ],
        [
          "10  english-follows-user",
          "direct lookup · 3.7 s against 6.9 s · 1.9x"
        ],
        [
          "11  agreement-lookup-uses-tool",
          "direct lookup · 8 s against 6.6 s · 0.8x"
        ],
        [
          "12  auth-material-refused",
          "direct lookup · 6.4 s against 4.5 s · 0.7x"
        ],
        [
          "13  users-are-out-of-scope",
          "direct lookup · 5.5 s against 6.9 s · 1.3x"
        ],
        [
          "14  write-proposes-never-executes",
          "direct lookup · 14 s against 13 s · 0.9x"
        ],
        [
          "15  write-flags-out-of-band",
          "direct lookup · 5.9 s against 8.5 s · 1.4x"
        ],
        [
          "16  top-agents-table",
          "direct lookup · 5.7 s against 7.1 s · 1.2x"
        ],
        [
          "17  semicolon-paste",
          "paste or file, in the sandbox · 27.3 s against 27.3 s · 1.0x"
        ],
        [
          "18  tab-paste-total-row",
          "paste or file, in the sandbox · 28.4 s against 32.8 s · 1.2x"
        ],
        [
          "19  period-stays-put",
          "paste or file, in the sandbox · 44.1 s against 35.3 s · 0.8x"
        ],
        [
          "20  big-attachment",
          "paste or file, in the sandbox · 28.6 s against 27.4 s · 1.0x"
        ],
        [
          "21  synthetic-reconciliation",
          "paste or file, in the sandbox · 38 s against 47.6 s · 1.3x"
        ],
        [
          "22  pdf-statement",
          "PDF statement, read by OCR · 35.7 s against 37.8 s · 1.1x"
        ]
      ]
    },
    {
      "id": "configurations",
      "variant": "full",
      "date": "2026-09-09",
      "title": "The larger open model costs 1.6 times Sonnet 5 a turn and passes fewer gates",
      "alt": "5 configurations, one harness Each point is one model at one reasoning setting over the whole case set: the mean warm cost per turn against the median wall time. The count beside each point is the gates passed over the cases run. Repetitions are counted as separate runs.",
      "png": "configurations-2026-09.png",
      "svg": "configurations-2026-09.svg",
      "height": 2400,
      "legend": [
        [
          "Claude Sonnet 5, adaptive thinking",
          "$0.012 per turn warm · $0.067 cold, with the cache write · 7.5 s median · 22 of 22 passed, 100%"
        ],
        [
          "GLM-5.3-Flash, max effort",
          "$0.0016 per turn warm · $0.0016 cold, with the cache write · 5.8 s median · 62 of 66 passed, 94%"
        ],
        [
          "GLM-5.3-Flash, high effort",
          "$0.0012 per turn warm · $0.0012 cold, with the cache write · 3.3 s median · 61 of 66 passed, 92%"
        ],
        [
          "GLM-5.3-Flash, low effort",
          "$0.0012 per turn warm · $0.0012 cold, with the cache write · 2.5 s median · 63 of 66 passed, 95%"
        ],
        [
          "GLM-5.3, max effort",
          "$0.019 per turn warm · $0.019 cold, with the cache write · 12.6 s median · 39 of 44 passed, 89%"
        ]
      ]
    },
    {
      "id": "configurations",
      "variant": "chart",
      "date": "2026-09-09",
      "title": "The larger open model costs 1.6 times Sonnet 5 a turn and passes fewer gates",
      "alt": "5 configurations, one harness Each point is one model at one reasoning setting over the whole case set: the mean warm cost per turn against the median wall time. The count beside each point is the gates passed over the cases run. Repetitions are counted as separate runs.",
      "png": "configurations-2026-09-chart.png",
      "svg": "configurations-2026-09-chart.svg",
      "height": 1728,
      "legend": [
        [
          "Claude Sonnet 5, adaptive thinking",
          "$0.012 per turn warm · $0.067 cold, with the cache write · 7.5 s median · 22 of 22 passed, 100%"
        ],
        [
          "GLM-5.3-Flash, max effort",
          "$0.0016 per turn warm · $0.0016 cold, with the cache write · 5.8 s median · 62 of 66 passed, 94%"
        ],
        [
          "GLM-5.3-Flash, high effort",
          "$0.0012 per turn warm · $0.0012 cold, with the cache write · 3.3 s median · 61 of 66 passed, 92%"
        ],
        [
          "GLM-5.3-Flash, low effort",
          "$0.0012 per turn warm · $0.0012 cold, with the cache write · 2.5 s median · 63 of 66 passed, 95%"
        ],
        [
          "GLM-5.3, max effort",
          "$0.019 per turn warm · $0.019 cold, with the cache write · 12.6 s median · 39 of 44 passed, 89%"
        ]
      ]
    }
  ]
}
