{
  "_readme": "Machine-readable headline figures from jslet. Each entry states the exact condition it holds under; none of these numbers is meaningful without its condition. Method, sources and exclusions: https://www.jslet.com/methodology",
  "generated": "2026-09-30",
  "site": "https://www.jslet.com",
  "methodology": "https://www.jslet.com/methodology",
  "license_note": "Facts are free to cite. Attribution optional.",
  "counts": {
    "tools": 116,
    "jsUtilities": 54,
    "cloudCalculators": 62,
    "articles": 44,
    "categories": 14,
    "pages": 8
  },
  "numbers": [
    {
      "id": "aws-nat-gateway-idle-3az",
      "topic": "AWS NAT Gateway",
      "metric": "idle hourly charge",
      "value": 98.55,
      "unit": "USD/month",
      "condition": "3 NAT Gateways (one per AZ), 730 hours, zero bytes processed",
      "formula": "3 x 0.045 USD/hr x 730 hr",
      "source": "AWS VPC pricing, us-east-1, July 2026",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-nat-gateway-hourly-per-az",
      "topic": "AWS NAT Gateway",
      "metric": "idle hourly charge",
      "value": 32.85,
      "unit": "USD/month",
      "condition": "one NAT Gateway, 730 hours, zero bytes processed",
      "formula": "0.045 USD/hr x 730 hr",
      "source": "AWS VPC pricing, us-east-1, July 2026",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-nat-gateway-data-processing",
      "topic": "AWS NAT Gateway",
      "metric": "data processing rate",
      "value": 0.045,
      "unit": "USD/GB",
      "condition": "no volume discounts, no reserved tiers, no committed-use breaks",
      "source": "AWS VPC pricing, us-east-1",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-nat-gateway-effective-per-gb",
      "topic": "AWS NAT Gateway",
      "metric": "effective cost per GB",
      "value": 0.065,
      "unit": "USD/GB",
      "condition": "1 GB leaving through NAT Gateway and crossing an AZ boundary each way",
      "formula": "0.01 (cross-AZ out) + 0.045 (NAT processing) + 0.01 (cross-AZ return)",
      "source": "AWS VPC pricing, us-east-1",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-nat-gateway-surcharge-vs-list",
      "topic": "AWS NAT Gateway",
      "metric": "surcharge over list rate",
      "value": 44,
      "unit": "percent",
      "condition": "cross-AZ round trip included, versus the 0.045 USD/GB headline rate",
      "source": "derived from AWS VPC pricing",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-nat-gateway-bill-reduction-with-endpoints",
      "topic": "AWS NAT Gateway",
      "metric": "NAT Gateway bill reduction after adding VPC Endpoints",
      "value": "30-60",
      "unit": "percent",
      "condition": "route S3/DynamoDB through a Gateway Endpoint and ECR/CloudWatch through Interface Endpoints",
      "source": "jslet model",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "vpc-gateway-endpoint-cost",
      "topic": "AWS VPC Endpoint",
      "metric": "charge for S3 and DynamoDB Gateway Endpoint",
      "value": 0,
      "unit": "USD",
      "condition": "no hourly charge and no per-GB charge",
      "source": "AWS VPC pricing",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "vpc-interface-endpoint-data-rate",
      "topic": "AWS VPC Endpoint",
      "metric": "interface endpoint data rate",
      "value": 0.01,
      "unit": "USD/GB",
      "condition": "ECR, CloudWatch, SSM, KMS and ~80 further services",
      "source": "AWS VPC pricing",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "vpc-interface-endpoint-discount",
      "topic": "AWS VPC Endpoint",
      "metric": "saving versus NAT Gateway data processing",
      "value": 78,
      "unit": "percent",
      "condition": "0.01 USD/GB via interface endpoint versus 0.045 USD/GB via NAT Gateway",
      "source": "derived from AWS VPC pricing",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "vpc-interface-endpoint-hourly",
      "topic": "AWS VPC Endpoint",
      "metric": "interface endpoint hourly charge",
      "value": 7.30,
      "unit": "USD/endpoint/AZ/month",
      "condition": "per interface endpoint, per Availability Zone, per month",
      "source": "AWS VPC pricing",
      "page": "https://www.jslet.com/nat-gateway-cost-real",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-cross-az-rate",
      "topic": "AWS cross-AZ transfer",
      "metric": "transfer rate",
      "value": 0.01,
      "unit": "USD/GB",
      "condition": "per direction; a round trip is 0.02 USD/GB",
      "source": "AWS EC2 data transfer pricing",
      "page": "https://www.jslet.com/aws-cross-az-data-transfer-cost",
      "verified": "2026-09-23"
    },
    {
      "id": "aws-cross-az-10k-rps-monthly",
      "topic": "AWS cross-AZ transfer",
      "metric": "monthly cost of a 10K rps three-tier app",
      "value": 5260,
      "unit": "USD/month",
      "condition": "10,000 requests per second, 1 KB payload, traffic crossing AZ boundaries",
      "source": "jslet model",
      "page": "https://www.jslet.com/aws-cross-az-data-transfer-cost",
      "verified": "2026-09-23"
    },
    {
      "id": "storage-decimal-binary-ratio",
      "topic": "Storage units",
      "metric": "decimal to binary conversion factor",
      "value": 0.9313,
      "unit": "ratio",
      "condition": "1 TB marketed (10^12 bytes) as reported by an OS in GiB (2^40 bytes)",
      "formula": "10^12 / 2^40",
      "source": "SI and IEC unit definitions",
      "page": "https://www.jslet.com/gib-to-gb-marketing-gap",
      "verified": "2026-09-23"
    },
    {
      "id": "storage-unit-gap-tb",
      "topic": "Storage units",
      "metric": "reported versus marketed capacity gap",
      "value": 6.9,
      "unit": "percent",
      "condition": "at TB scale",
      "source": "derived from SI/IEC unit definitions",
      "page": "https://www.jslet.com/gib-to-gb-marketing-gap",
      "verified": "2026-09-23"
    },
    {
      "id": "storage-unit-gap-pb",
      "topic": "Storage units",
      "metric": "reported versus marketed capacity gap",
      "value": 10,
      "unit": "percent (approaches)",
      "condition": "at PB scale",
      "source": "derived from SI/IEC unit definitions",
      "page": "https://www.jslet.com/gib-to-gb-marketing-gap",
      "verified": "2026-09-23"
    },
    {
      "id": "gpu-vram-label-convention-warning",
      "topic": "GPU VRAM units",
      "metric": "convention",
      "value": "vendor decimal, do not convert to GiB",
      "unit": "convention",
      "condition": "GPU VRAM labels are decimal byte counts; applying the 0.9313 storage conversion understates usable capacity",
      "source": "jslet methodology",
      "page": "https://www.jslet.com/gib-to-gb-marketing-gap",
      "verified": "2026-09-23"
    },
    {
      "id": "kv-cache-70b-8k-fp16",
      "topic": "LLM KV cache",
      "metric": "KV cache size",
      "value": 2.62,
      "unit": "GB",
      "condition": "70B class: 80 layers, 8 KV heads (GQA), head_dim 128, 8K context, batch 1, FP16 KV",
      "formula": "2 x 80 x 8 x 128 x 2 bytes x 8000 tokens",
      "source": "published model architectures; jslet computation",
      "page": "https://www.jslet.com/context-window-vram-inflation",
      "verified": "2026-09-30"
    },
    {
      "id": "kv-cache-70b-8k-full-mha",
      "topic": "LLM KV cache",
      "metric": "KV cache size without grouped-query attention",
      "value": 20.97,
      "unit": "GB",
      "condition": "same 80 layers at 64 query heads, 8K context, batch 1, FP16 KV",
      "formula": "2 x 80 x 64 x 128 x 2 bytes x 8000 tokens",
      "source": "published model architectures; jslet computation",
      "page": "https://www.jslet.com/context-window-vram-inflation",
      "verified": "2026-09-30"
    },
    {
      "id": "kv-cache-gqa-vs-mha-factor",
      "topic": "LLM KV cache",
      "metric": "GQA reduction factor",
      "value": 8,
      "unit": "ratio",
      "condition": "8 KV heads versus 64 query heads at otherwise identical shape and precision",
      "source": "published model architectures; jslet computation",
      "page": "https://www.jslet.com/context-window-vram-inflation",
      "verified": "2026-09-30"
    },
    {
      "id": "kv-cache-70b-128k-fp16",
      "topic": "LLM KV cache",
      "metric": "KV cache size at long context",
      "value": 41.9,
      "unit": "GB",
      "condition": "70B class: 80 layers, 8 KV heads, head_dim 128, 128K context, batch 1, FP16 KV",
      "source": "published model architectures; jslet computation",
      "page": "https://www.jslet.com/context-window-vram-inflation",
      "verified": "2026-09-30"
    },
    {
      "id": "kv-precision-independent-of-weight-quant",
      "topic": "LLM KV cache",
      "metric": "convention",
      "value": "KV precision is independent of weight quantization",
      "unit": "convention",
      "condition": "quantizing weights to INT4 leaves the KV cache at FP16 in essentially every serving stack; assuming a 4x smaller cache understates it by 4x",
      "source": "jslet methodology",
      "page": "https://www.jslet.com/context-window-vram-inflation",
      "verified": "2026-09-30"
    },
    {
      "id": "weights-70b-fp16",
      "topic": "LLM weights",
      "metric": "parameter footprint",
      "value": 140,
      "unit": "GB",
      "condition": "70B parameters at FP16 (2 bytes per parameter)",
      "source": "jslet computation",
      "page": "https://www.jslet.com/parameters-to-vram-fp16",
      "verified": "2026-09-30"
    },
    {
      "id": "weights-70b-int4",
      "topic": "LLM weights",
      "metric": "parameter footprint",
      "value": 35,
      "unit": "GB",
      "condition": "70B parameters at INT4 (0.5 bytes per parameter)",
      "source": "jslet computation",
      "page": "https://www.jslet.com/parameters-to-vram-int4",
      "verified": "2026-09-30"
    },
    {
      "id": "weights-405b-fp16",
      "topic": "LLM weights",
      "metric": "parameter footprint",
      "value": 810,
      "unit": "GB",
      "condition": "405B parameters at FP16",
      "source": "jslet computation",
      "page": "https://www.jslet.com/gpu-model-fit-matrix-real",
      "verified": "2026-09-30"
    },
    {
      "id": "weights-671b-moe-fp16",
      "topic": "LLM weights",
      "metric": "parameter footprint",
      "value": 1342,
      "unit": "GB",
      "condition": "671B-parameter MoE at FP16; all experts must be resident although only 37B are active per token",
      "source": "jslet computation",
      "page": "https://www.jslet.com/gpu-model-fit-matrix-real",
      "verified": "2026-09-30"
    },
    {
      "id": "tensor-parallel-efficiency",
      "topic": "Multi-GPU scaling",
      "metric": "tensor-parallel efficiency",
      "value": { "2": 0.92, "4": 0.80, "8": 0.63, "16": 0.48 },
      "unit": "fraction of linear scaling",
      "condition": "N-way tensor parallelism; more replicas of a smaller group beat one large group",
      "source": "jslet model",
      "page": "https://www.jslet.com/inference-capacity-planner-real",
      "verified": "2026-09-30"
    },
    {
      "id": "rtx4090-70b-int4-theoretical-throughput",
      "topic": "Decode throughput",
      "metric": "theoretical memory-bandwidth ceiling",
      "value": 28.8,
      "unit": "tokens/second",
      "condition": "RTX 4090 (1008 GB/s), 70B dense at INT4, one token requires reading every active parameter once",
      "formula": "1008 GB/s / (70e9 x 0.5 bytes / 1e9)",
      "source": "jslet model; vendor bandwidth figures",
      "page": "https://www.jslet.com/llm-inference-latency",
      "verified": "2026-09-30"
    },
    {
      "id": "selfhost-70b-a100-monthly",
      "topic": "LLM self-hosting",
      "metric": "GPU rental cost",
      "value": 1307,
      "unit": "USD/month",
      "condition": "one A100 80GB, whether or not a single request arrives",
      "source": "jslet model; list pricing",
      "page": "https://www.jslet.com/llm-selfhost-vs-api-breakeven",
      "verified": "2026-09-23"
    },
    {
      "id": "api-same-traffic-monthly",
      "topic": "LLM self-hosting",
      "metric": "hosted API cost for the same traffic",
      "value": 176,
      "unit": "USD/month",
      "condition": "70B-class traffic equal to the self-hosted comparison case",
      "source": "jslet model; vendor API list pricing",
      "page": "https://www.jslet.com/llm-selfhost-vs-api-breakeven",
      "verified": "2026-09-23"
    },
    {
      "id": "observability-share-of-cloud-cost",
      "topic": "Observability",
      "metric": "share of total cloud infrastructure spend",
      "value": "15-25",
      "unit": "percent",
      "condition": "median mid-market SaaS at scale",
      "source": "jslet model",
      "page": "https://www.jslet.com/observability-cost",
      "verified": "2026-09-23"
    },
    {
      "id": "datadog-budget-to-first-invoice",
      "topic": "Observability",
      "metric": "budgeted versus first invoiced monthly cost",
      "value": { "budgeted": 12000, "firstInvoice": 147000 },
      "unit": "USD/month",
      "condition": "budget derived from the vendor pricing calculator; the gap comes from cardinality, retention tiers and log indexing rather than ingest volume",
      "source": "jslet model",
      "page": "https://www.jslet.com/observability-cost",
      "verified": "2026-09-23"
    }
  ]
}
