{
  "id": "on-prem",
  "date": "2026-09-08",
  "scope": "Derived from published inputs on 8 September 2026. Nothing here is measured by Zero One Labs.",
  "inputs": {
    "throughput": {
      "model": "Kimi K2.6",
      "gpu": "NVIDIA B200",
      "workload": "single-turn, 8k input and 1k output tokens",
      "source": "https://inferencex.semianalysis.com/run/kimi-k26-on-b200",
      "published": "2026-03-24",
      "latestRun": "2026-08-07",
      "retrievedAt": "2026-09-08",
      "ladder": [
        {
          "perUser": 30,
          "perGpu": 4732,
          "engine": "vLLM",
          "precision": "fp4"
        },
        {
          "perUser": 50,
          "perGpu": 3449,
          "engine": "Dynamo vLLM",
          "precision": "fp4"
        },
        {
          "perUser": 75,
          "perGpu": 2265,
          "engine": "Dynamo vLLM",
          "precision": "fp4"
        },
        {
          "perUser": 100,
          "perGpu": 1510,
          "engine": "vLLM",
          "precision": "fp4"
        },
        {
          "perUser": 150,
          "perGpu": 485,
          "engine": "vLLM",
          "precision": "fp4"
        }
      ],
      "operatingPoint": 50,
      "tco": {
        "hyperscalerVolume": 1.73,
        "retail": 2.6
      }
    },
    "node": {
      "gpus": 8,
      "serverUsd": 400328.5,
      "serverSource": "https://www.exxactcorp.com/Exxact-TS4-169219634-E169219634",
      "serverNote": "Exxact TensorEX 10U HGX B200, listed starting price, live crawl 2026-09-08 (a cached copy read $404,618.50 earlier the same day). NVIDIA publishes no list price for a B200 system.",
      "maxPowerKw": 14.3,
      "powerSource": "https://www.nvidia.com/en-us/data-center/dgx-b200/",
      "powerNote": "DGX B200 system power usage, ~14.3 kW max. Used as the rated draw of an eight-B200 node.",
      "depreciationMonths": 36,
      "residualUsd": 0,
      "hoursPerMonth": 730
    },
    "electricity": {
      "usIndustrialCentsPerKwh": 9.17,
      "usSource": "https://www.eia.gov/electricity/monthly/epm_table_grapher.php?t=epmt_5_6_a",
      "usNote": "EIA Electric Power Monthly, table 5.6.A, U.S. Total, industrial, June 2026, released 2026-08-26.",
      "norwayServicesOrePerKwh": 89,
      "norwaySource": "https://www.ssb.no/en/energi-og-industri/energi/statistikk/elektrisitetspriser",
      "norwayNote": "SSB, electricity price for services excluding taxes, 2nd quarter 2026, updated 2026-08-17."
    },
    "rental": [
      {
        "id": "aws",
        "name": "AWS capacity block",
        "kind": "list",
        "usdPerGpuHour": 12.355,
        "source": "https://aws.amazon.com/ec2/capacityblocks/pricing/",
        "note": "p6-b200.48xlarge, US East, $98.84 an hour for eight",
        "retrievedAt": "2026-09-08"
      },
      {
        "id": "runpod",
        "name": "RunPod on-demand, secure cloud",
        "kind": "list",
        "usdPerGpuHour": 6.79,
        "source": "https://www.runpod.io/pricing",
        "note": "Community cloud is $5.98",
        "retrievedAt": "2026-09-08"
      },
      {
        "id": "lambda",
        "name": "Lambda on-demand",
        "kind": "list",
        "usdPerGpuHour": 6.69,
        "source": "https://lambda.ai/pricing",
        "note": "Eight-GPU instance",
        "retrievedAt": "2026-09-08"
      }
    ],
    "api": {
      "name": "Kimi K2.6 API",
      "source": "https://platform.kimi.ai/docs/pricing/chat-k26.md",
      "retrievedAt": "2026-09-08",
      "inputUsdPerM": 0.95,
      "cacheHitInputUsdPerM": 0.16,
      "outputUsdPerM": 4,
      "inputTokens": 8000,
      "outputTokens": 1000
    },
    "utilizationObservations": [
      {
        "id": "castAverage",
        "name": "Cast AI, average across tens of thousands of clusters",
        "utilization": 0.05,
        "source": "https://cast.ai/press-release/2026-state-of-kubernetes-optimization-report/",
        "date": "2026-04-21"
      },
      {
        "id": "castBest",
        "name": "Cast AI, best case, a 136-node H200 inference fleet",
        "utilization": 0.49,
        "source": "https://cast.ai/blog/kubernetes-gpu-optimization/",
        "date": "2026-07-03"
      },
      {
        "id": "ventureBeat",
        "name": "VentureBeat Research, 83% of enterprises running their own GPUs report 50% or less",
        "utilization": 0.5,
        "share": 0.83,
        "source": "https://venturebeat.com/orchestration/wall-street-is-debating-the-ai-buildout-enterprises-just-answered-86-say-their-gpus-run-at-half-capacity-or-less",
        "date": "2026-07-10",
        "note": "573 technical leaders surveyed in June 2026. The headline first said 86% and was corrected to about 83% on 14 July. Self-selected sample, read directionally."
      }
    ],
    "latency": {
      "source": "https://modelstats.ai/blog/llm-api-latency-spikes-what-average-ttft-doesnt-tell-you",
      "published": "2026-04-07",
      "retrievedAt": "2026-09-08",
      "window": "24 hours, one probe every 10 minutes, one US datacenter",
      "rows": [
        {
          "model": "Gemini 2.5 Flash Lite",
          "p50": 311,
          "p95": 1142,
          "p99": 3080,
          "max": 3334
        },
        {
          "model": "Gemini 2.5 Flash",
          "p50": 477,
          "p95": 776,
          "p99": 1309,
          "max": 1879
        },
        {
          "model": "GPT-4o Mini",
          "p50": 536,
          "p95": 844,
          "p99": 4461,
          "max": 5946
        },
        {
          "model": "Claude Haiku 4.5",
          "p50": 541,
          "p95": 1591,
          "p99": 4117,
          "max": 11611
        },
        {
          "model": "GPT-4o",
          "p50": 639,
          "p95": 1144,
          "p99": 4517,
          "max": 7010
        },
        {
          "model": "GPT-4.1 Mini",
          "p50": 661,
          "p95": 1087,
          "p99": 2169,
          "max": 4279
        },
        {
          "model": "GPT-4.1",
          "p50": 726,
          "p95": 1220,
          "p99": 1946,
          "max": 4813
        },
        {
          "model": "Claude Sonnet 4.6",
          "p50": 941,
          "p95": 2939,
          "p99": 4787,
          "max": 5453
        },
        {
          "model": "DeepSeek V3",
          "p50": 1479,
          "p95": 1926,
          "p99": 2492,
          "max": 3105
        },
        {
          "model": "Claude Opus 4.6",
          "p50": 1626,
          "p95": 2420,
          "p99": 3797,
          "max": 5291
        }
      ]
    }
  },
  "derived": {
    "tokensPerSecondPerGpu": 3449,
    "tokensPerDayAtFull": 2383948800,
    "tokensPerMonthAtFull": 72511776000,
    "depreciationUsdPerMonth": 11120,
    "kwhPerMonth": 10439,
    "powerUsdPerMonthUs": 957,
    "powerNokPerMonthNorway": 9291,
    "ownedUsdPerMonth": 12077,
    "ownedUsdPerGpuHour": 2.07,
    "ownedUsdPerMillionAtFull": 0.167,
    "apiUsdPerMillion": 1.289,
    "apiCachedUsdPerMillion": 0.587,
    "crossovers": {
      "api": {
        "utilization": 0.1292,
        "tokensPerDay": 308070189,
        "usdPerMonth": 12077
      },
      "apiCached": {
        "utilization": 0.2839,
        "tokensPerDay": 676820870,
        "usdPerMonth": 12077
      },
      "lambda": {
        "utilization": 0.3091,
        "tokensPerDay": 736944416,
        "usdPerMonth": 12077
      },
      "aws": {
        "utilization": 0.1674,
        "tokensPerDay": 399041533,
        "usdPerMonth": 12077
      }
    }
  },
  "ladder": [
    {
      "perUser": 30,
      "perGpu": 4732,
      "engine": "vLLM",
      "precision": "fp4",
      "usdPerMillion": 0.102
    },
    {
      "perUser": 50,
      "perGpu": 3449,
      "engine": "Dynamo vLLM",
      "precision": "fp4",
      "usdPerMillion": 0.139
    },
    {
      "perUser": 75,
      "perGpu": 2265,
      "engine": "Dynamo vLLM",
      "precision": "fp4",
      "usdPerMillion": 0.212
    },
    {
      "perUser": 100,
      "perGpu": 1510,
      "engine": "vLLM",
      "precision": "fp4",
      "usdPerMillion": 0.318
    },
    {
      "perUser": 150,
      "perGpu": 485,
      "engine": "vLLM",
      "precision": "fp4",
      "usdPerMillion": 0.991
    }
  ],
  "prices": [
    {
      "id": "aws",
      "name": "AWS capacity block",
      "kind": "list",
      "usdPerGpuHour": 12.355,
      "source": "https://aws.amazon.com/ec2/capacityblocks/pricing/",
      "note": "p6-b200.48xlarge, US East, $98.84 an hour for eight",
      "retrievedAt": "2026-09-08"
    },
    {
      "id": "runpod",
      "name": "RunPod on-demand, secure cloud",
      "kind": "list",
      "usdPerGpuHour": 6.79,
      "source": "https://www.runpod.io/pricing",
      "note": "Community cloud is $5.98",
      "retrievedAt": "2026-09-08"
    },
    {
      "id": "lambda",
      "name": "Lambda on-demand",
      "kind": "list",
      "usdPerGpuHour": 6.69,
      "source": "https://lambda.ai/pricing",
      "note": "Eight-GPU instance",
      "retrievedAt": "2026-09-08"
    },
    {
      "id": "owned",
      "name": "Owned, small buyer, server and power only",
      "kind": "cost",
      "usdPerGpuHour": 2.07,
      "source": "https://www.exxactcorp.com/Exxact-TS4-169219634-E169219634",
      "note": "Derived below: the listed server over 36 months plus power at the US industrial rate. No colocation, staff, spares or networking.",
      "retrievedAt": "2026-09-08"
    },
    {
      "id": "hyperscaler",
      "name": "Owned at hyperscaler volume, SemiAnalysis estimate",
      "kind": "cost",
      "usdPerGpuHour": 1.73,
      "source": "https://inferencex.semianalysis.com/run/kimi-k26-on-b200",
      "note": "SemiAnalysis AI Cloud TCO model, 'Owning at Large Hyperscaler Volume'",
      "retrievedAt": "2026-09-08"
    }
  ],
  "lines": [
    {
      "id": "api",
      "label": "API list price",
      "usdPerMillion": 1.289
    },
    {
      "id": "apiCached",
      "label": "API, every input token a cache hit",
      "usdPerMillion": 0.587
    },
    {
      "id": "lambda",
      "label": "Lambda on-demand, hours as used",
      "usdPerMillion": 0.539
    },
    {
      "id": "aws",
      "label": "AWS capacity block, hours as used",
      "usdPerMillion": 0.995
    },
    {
      "id": "hyperscaler",
      "label": "Owning at hyperscaler volume, fully used",
      "usdPerMillion": 0.139
    }
  ],
  "utilizationBars": [
    {
      "id": "castAverage",
      "utilization": 0.05,
      "usdPerMillion": 3.331,
      "anchor": "Cast AI average"
    },
    {
      "id": "crossApi",
      "utilization": 0.1292,
      "usdPerMillion": 1.289,
      "anchor": "breaks even with the API"
    },
    {
      "id": "crossLambda",
      "utilization": 0.3091,
      "usdPerMillion": 0.539,
      "anchor": "breaks even with renting"
    },
    {
      "id": "half",
      "utilization": 0.5,
      "usdPerMillion": 0.333,
      "anchor": "the surveys' ceiling"
    },
    {
      "id": "full",
      "utilization": 1,
      "usdPerMillion": 0.167,
      "anchor": "never sleeps"
    }
  ],
  "latency": [
    {
      "model": "Gemini 2.5 Flash Lite",
      "p50": 311,
      "p95": 1142,
      "p99": 3080,
      "max": 3334,
      "p99OverP50": 9.9,
      "maxOverP50": 10.72
    },
    {
      "model": "Gemini 2.5 Flash",
      "p50": 477,
      "p95": 776,
      "p99": 1309,
      "max": 1879,
      "p99OverP50": 2.74,
      "maxOverP50": 3.94
    },
    {
      "model": "GPT-4o Mini",
      "p50": 536,
      "p95": 844,
      "p99": 4461,
      "max": 5946,
      "p99OverP50": 8.32,
      "maxOverP50": 11.09
    },
    {
      "model": "Claude Haiku 4.5",
      "p50": 541,
      "p95": 1591,
      "p99": 4117,
      "max": 11611,
      "p99OverP50": 7.61,
      "maxOverP50": 21.46
    },
    {
      "model": "GPT-4o",
      "p50": 639,
      "p95": 1144,
      "p99": 4517,
      "max": 7010,
      "p99OverP50": 7.07,
      "maxOverP50": 10.97
    },
    {
      "model": "GPT-4.1 Mini",
      "p50": 661,
      "p95": 1087,
      "p99": 2169,
      "max": 4279,
      "p99OverP50": 3.28,
      "maxOverP50": 6.47
    },
    {
      "model": "GPT-4.1",
      "p50": 726,
      "p95": 1220,
      "p99": 1946,
      "max": 4813,
      "p99OverP50": 2.68,
      "maxOverP50": 6.63
    },
    {
      "model": "Claude Sonnet 4.6",
      "p50": 941,
      "p95": 2939,
      "p99": 4787,
      "max": 5453,
      "p99OverP50": 5.09,
      "maxOverP50": 5.79
    },
    {
      "model": "DeepSeek V3",
      "p50": 1479,
      "p95": 1926,
      "p99": 2492,
      "max": 3105,
      "p99OverP50": 1.68,
      "maxOverP50": 2.1
    },
    {
      "model": "Claude Opus 4.6",
      "p50": 1626,
      "p95": 2420,
      "p99": 3797,
      "max": 5291,
      "p99OverP50": 2.34,
      "maxOverP50": 3.25
    }
  ]
}
