{
  "dataset": "/data/on-prem/snapshot.json",
  "datasetSha256": "6e17b9bece295a57a7c327c1ae73b5a53b6f33f9be44be06d1b31e8de78d18b0",
  "width": 1920,
  "figures": [
    {
      "id": "batching",
      "variant": "full",
      "date": "2026-08-07",
      "title": "Same GPU, 9.8 times the tokens at a fifth of the speed per user",
      "alt": "5 per-user speeds, one B200 One NVIDIA B200 serving Kimi K2.6. Each point is the best configuration InferenceX measured at that per-user speed, on a single-turn request of 8k input and 1k output tokens. The price is the cost per million tokens at SemiAnalysis’s estimate of what a large cloud pays to own the GPU, $1.73 an hour.",
      "png": "batching-2026-09.png",
      "svg": "batching-2026-09.svg",
      "height": 2400,
      "legend": [
        [
          "30 tokens a second per user",
          "4,732 tokens a second per GPU · $0.10 per million tokens"
        ],
        [
          "50 tokens a second per user",
          "3,449 tokens a second per GPU · $0.14 per million tokens"
        ],
        [
          "75 tokens a second per user",
          "2,265 tokens a second per GPU · $0.21 per million tokens"
        ],
        [
          "100 tokens a second per user",
          "1,510 tokens a second per GPU · $0.32 per million tokens"
        ],
        [
          "150 tokens a second per user",
          "485 tokens a second per GPU · $0.99 per million tokens"
        ]
      ]
    },
    {
      "id": "batching",
      "variant": "chart",
      "date": "2026-08-07",
      "title": "Same GPU, 9.8 times the tokens at a fifth of the speed per user",
      "alt": "5 per-user speeds, one B200 One NVIDIA B200 serving Kimi K2.6. Each point is the best configuration InferenceX measured at that per-user speed, on a single-turn request of 8k input and 1k output tokens. The price is the cost per million tokens at SemiAnalysis’s estimate of what a large cloud pays to own the GPU, $1.73 an hour.",
      "png": "batching-2026-09-chart.png",
      "svg": "batching-2026-09-chart.svg",
      "height": 1728,
      "legend": [
        [
          "30 tokens a second per user",
          "4,732 tokens a second per GPU · $0.10 per million tokens"
        ],
        [
          "50 tokens a second per user",
          "3,449 tokens a second per GPU · $0.14 per million tokens"
        ],
        [
          "75 tokens a second per user",
          "2,265 tokens a second per GPU · $0.21 per million tokens"
        ],
        [
          "100 tokens a second per user",
          "1,510 tokens a second per GPU · $0.32 per million tokens"
        ],
        [
          "150 tokens a second per user",
          "485 tokens a second per GPU · $0.99 per million tokens"
        ]
      ]
    },
    {
      "id": "prices",
      "variant": "full",
      "date": "2026-09-08",
      "title": "Same B200, five prices from $1.73 to $12.36 an hour",
      "alt": "5 prices for one B200 hour The same NVIDIA B200, priced by who owns it. Rental rows are on-demand or capacity-block list prices, retrieved 8 September 2026. Owned rows are costs: a listed eight-GPU server written off over three years plus power at the US industrial rate, and SemiAnalysis’s estimate at hyperscaler scale. AWS prints $12.355.",
      "png": "prices-2026-09.png",
      "svg": "prices-2026-09.svg",
      "height": 2520,
      "legend": [
        [
          "AWS capacity block",
          "$12.36 per GPU hour · list price to rent · p6-b200.48xlarge, US East, $98.84 an hour for eight"
        ],
        [
          "RunPod on-demand, secure cloud",
          "$6.79 per GPU hour · list price to rent · Community cloud is $5.98"
        ],
        [
          "Lambda on-demand",
          "$6.69 per GPU hour · list price to rent · Eight-GPU instance"
        ],
        [
          "Owned, small buyer, server and power only",
          "$2.07 per GPU hour · cost of owning · Derived below: the listed server over 36 months plus power at the US industrial rate. No colocation, staff, spares or networking."
        ],
        [
          "Owned at hyperscaler volume, SemiAnalysis estimate",
          "$1.73 per GPU hour · cost of owning · SemiAnalysis AI Cloud TCO model, 'Owning at Large Hyperscaler Volume'"
        ]
      ]
    },
    {
      "id": "prices",
      "variant": "chart",
      "date": "2026-09-08",
      "title": "Same B200, five prices from $1.73 to $12.36 an hour",
      "alt": "5 prices for one B200 hour The same NVIDIA B200, priced by who owns it. Rental rows are on-demand or capacity-block list prices, retrieved 8 September 2026. Owned rows are costs: a listed eight-GPU server written off over three years plus power at the US industrial rate, and SemiAnalysis’s estimate at hyperscaler scale. AWS prints $12.355.",
      "png": "prices-2026-09-chart.png",
      "svg": "prices-2026-09-chart.svg",
      "height": 1800,
      "legend": [
        [
          "AWS capacity block",
          "$12.36 per GPU hour · list price to rent · p6-b200.48xlarge, US East, $98.84 an hour for eight"
        ],
        [
          "RunPod on-demand, secure cloud",
          "$6.79 per GPU hour · list price to rent · Community cloud is $5.98"
        ],
        [
          "Lambda on-demand",
          "$6.69 per GPU hour · list price to rent · Eight-GPU instance"
        ],
        [
          "Owned, small buyer, server and power only",
          "$2.07 per GPU hour · cost of owning · Derived below: the listed server over 36 months plus power at the US industrial rate. No colocation, staff, spares or networking."
        ],
        [
          "Owned at hyperscaler volume, SemiAnalysis estimate",
          "$1.73 per GPU hour · cost of owning · SemiAnalysis AI Cloud TCO model, 'Owning at Large Hyperscaler Volume'"
        ]
      ]
    },
    {
      "id": "latency",
      "variant": "full",
      "date": "2026-04-07",
      "title": "The slowest 1% of API calls wait 1.7 to 9.9 times the median",
      "alt": "10 models, 24 hours of pings Time to first token, median to 99th percentile. ModelStats’ 24 hours of pings every ten minutes from one US datacenter, April 2026, on the models it tracked then. The ratio is the 99th percentile over the median. The tick is the single slowest request in the window. This is the shape of the tail, not a current ranking.",
      "png": "latency-2026-09.png",
      "svg": "latency-2026-09.svg",
      "height": 3276,
      "legend": [
        [
          "Gemini 2.5 Flash Lite",
          "median 311 ms · 99th percentile 3.1 s · slowest 3.3 s · 9.9x the median at the 99th percentile"
        ],
        [
          "Gemini 2.5 Flash",
          "median 477 ms · 99th percentile 1.3 s · slowest 1.9 s · 2.7x the median at the 99th percentile"
        ],
        [
          "GPT-4o Mini",
          "median 536 ms · 99th percentile 4.5 s · slowest 5.9 s · 8.3x the median at the 99th percentile"
        ],
        [
          "Claude Haiku 4.5",
          "median 541 ms · 99th percentile 4.1 s · slowest 11.6 s · 7.6x the median at the 99th percentile"
        ],
        [
          "GPT-4o",
          "median 639 ms · 99th percentile 4.5 s · slowest 7.0 s · 7.1x the median at the 99th percentile"
        ],
        [
          "GPT-4.1 Mini",
          "median 661 ms · 99th percentile 2.2 s · slowest 4.3 s · 3.3x the median at the 99th percentile"
        ],
        [
          "GPT-4.1",
          "median 726 ms · 99th percentile 1.9 s · slowest 4.8 s · 2.7x the median at the 99th percentile"
        ],
        [
          "Claude Sonnet 4.6",
          "median 941 ms · 99th percentile 4.8 s · slowest 5.5 s · 5.1x the median at the 99th percentile"
        ],
        [
          "DeepSeek V3",
          "median 1.5 s · 99th percentile 2.5 s · slowest 3.1 s · 1.7x the median at the 99th percentile"
        ],
        [
          "Claude Opus 4.6",
          "median 1.6 s · 99th percentile 3.8 s · slowest 5.3 s · 2.3x the median at the 99th percentile"
        ]
      ]
    },
    {
      "id": "latency",
      "variant": "chart",
      "date": "2026-04-07",
      "title": "The slowest 1% of API calls wait 1.7 to 9.9 times the median",
      "alt": "10 models, 24 hours of pings Time to first token, median to 99th percentile. ModelStats’ 24 hours of pings every ten minutes from one US datacenter, April 2026, on the models it tracked then. The ratio is the 99th percentile over the median. The tick is the single slowest request in the window. This is the shape of the tail, not a current ranking.",
      "png": "latency-2026-09-chart.png",
      "svg": "latency-2026-09-chart.svg",
      "height": 1956,
      "legend": [
        [
          "Gemini 2.5 Flash Lite",
          "median 311 ms · 99th percentile 3.1 s · slowest 3.3 s · 9.9x the median at the 99th percentile"
        ],
        [
          "Gemini 2.5 Flash",
          "median 477 ms · 99th percentile 1.3 s · slowest 1.9 s · 2.7x the median at the 99th percentile"
        ],
        [
          "GPT-4o Mini",
          "median 536 ms · 99th percentile 4.5 s · slowest 5.9 s · 8.3x the median at the 99th percentile"
        ],
        [
          "Claude Haiku 4.5",
          "median 541 ms · 99th percentile 4.1 s · slowest 11.6 s · 7.6x the median at the 99th percentile"
        ],
        [
          "GPT-4o",
          "median 639 ms · 99th percentile 4.5 s · slowest 7.0 s · 7.1x the median at the 99th percentile"
        ],
        [
          "GPT-4.1 Mini",
          "median 661 ms · 99th percentile 2.2 s · slowest 4.3 s · 3.3x the median at the 99th percentile"
        ],
        [
          "GPT-4.1",
          "median 726 ms · 99th percentile 1.9 s · slowest 4.8 s · 2.7x the median at the 99th percentile"
        ],
        [
          "Claude Sonnet 4.6",
          "median 941 ms · 99th percentile 4.8 s · slowest 5.5 s · 5.1x the median at the 99th percentile"
        ],
        [
          "DeepSeek V3",
          "median 1.5 s · 99th percentile 2.5 s · slowest 3.1 s · 1.7x the median at the 99th percentile"
        ],
        [
          "Claude Opus 4.6",
          "median 1.6 s · 99th percentile 3.8 s · slowest 5.3 s · 2.3x the median at the 99th percentile"
        ]
      ]
    },
    {
      "id": "crossover",
      "variant": "full",
      "date": "2026-09-08",
      "title": "Owning costs the same as the API at 308 million tokens a day",
      "alt": "3 crossings, one owned node One eight-B200 node against three ways of paying per token. Owning is the listed server over three years plus power, nothing else, so its line is flat. The three rising lines are Kimi K2.6’s list price, the same price with every input token a cache hit, and renting the same GPUs at Lambda by the hour. The node’s capacity is 2.4 billion tokens a day, past the right edge.",
      "png": "crossover-2026-09.png",
      "svg": "crossover-2026-09.svg",
      "height": 2472,
      "legend": [
        [
          "1  API list price",
          "308M tokens a day · $12,077 a month · 13% of the node"
        ],
        [
          "2  API, every input token cached",
          "677M tokens a day · $12,077 a month · 28% of the node"
        ],
        [
          "3  Renting at Lambda, hours as used",
          "737M tokens a day · $12,077 a month · 31% of the node"
        ],
        [
          "Owning, flat",
          "$12,077 a month at any volume up to 2.38B tokens a day"
        ]
      ]
    },
    {
      "id": "crossover",
      "variant": "chart",
      "date": "2026-09-08",
      "title": "Owning costs the same as the API at 308 million tokens a day",
      "alt": "3 crossings, one owned node One eight-B200 node against three ways of paying per token. Owning is the listed server over three years plus power, nothing else, so its line is flat. The three rising lines are Kimi K2.6’s list price, the same price with every input token a cache hit, and renting the same GPUs at Lambda by the hour. The node’s capacity is 2.4 billion tokens a day, past the right edge.",
      "png": "crossover-2026-09-chart.png",
      "svg": "crossover-2026-09-chart.svg",
      "height": 1872,
      "legend": [
        [
          "1  API list price",
          "308M tokens a day · $12,077 a month · 13% of the node"
        ],
        [
          "2  API, every input token cached",
          "677M tokens a day · $12,077 a month · 28% of the node"
        ],
        [
          "3  Renting at Lambda, hours as used",
          "737M tokens a day · $12,077 a month · 31% of the node"
        ],
        [
          "Owning, flat",
          "$12,077 a month at any volume up to 2.38B tokens a day"
        ]
      ]
    },
    {
      "id": "utilization",
      "variant": "full",
      "date": "2026-09-08",
      "title": "The owned token costs 2.6 times the API at 5% utilization and breaks even at 13%",
      "alt": "5 utilizations, one owned node What the owned token costs at the utilizations that actually occur: the node’s $12,077 a month spread over the tokens it serves. Each bar is anchored: Cast AI’s measured average across tens of thousands of clusters, the two break-even points, the half that 83% of enterprises in VentureBeat’s survey do not reach, and full use.",
      "png": "utilization-2026-09.png",
      "svg": "utilization-2026-09.svg",
      "height": 2400,
      "legend": [
        [
          "5%, Cast AI’s measured average",
          "$3.33 per million tokens · 2.6x the API price"
        ],
        [
          "13%, breaks even with the API",
          "$1.29 per million tokens · 1.0x the API price"
        ],
        [
          "31%, breaks even with renting at Lambda",
          "$0.54 per million tokens · 0.4x the API price"
        ],
        [
          "50%, the surveys’ ceiling",
          "$0.33 per million tokens · 0.3x the API price"
        ],
        [
          "100%, never sleeps",
          "$0.17 per million tokens · 0.1x the API price"
        ]
      ]
    },
    {
      "id": "utilization",
      "variant": "chart",
      "date": "2026-09-08",
      "title": "The owned token costs 2.6 times the API at 5% utilization and breaks even at 13%",
      "alt": "5 utilizations, one owned node What the owned token costs at the utilizations that actually occur: the node’s $12,077 a month spread over the tokens it serves. Each bar is anchored: Cast AI’s measured average across tens of thousands of clusters, the two break-even points, the half that 83% of enterprises in VentureBeat’s survey do not reach, and full use.",
      "png": "utilization-2026-09-chart.png",
      "svg": "utilization-2026-09-chart.svg",
      "height": 1728,
      "legend": [
        [
          "5%, Cast AI’s measured average",
          "$3.33 per million tokens · 2.6x the API price"
        ],
        [
          "13%, breaks even with the API",
          "$1.29 per million tokens · 1.0x the API price"
        ],
        [
          "31%, breaks even with renting at Lambda",
          "$0.54 per million tokens · 0.4x the API price"
        ],
        [
          "50%, the surveys’ ceiling",
          "$0.33 per million tokens · 0.3x the API price"
        ],
        [
          "100%, never sleeps",
          "$0.17 per million tokens · 0.1x the API price"
        ]
      ]
    }
  ]
}
