{
  "slug": "llm-api-token-cost",
  "title": "LLM API Token Cost Calculator",
  "heading": "LLM API Token Cost Calculator",
  "category": "financial",
  "url": "https://www.revenuelab.fyi/toolbox/llm-api-token-cost",
  "summary": "Estimate monthly spend on GPT/Claude/Gemini-style token-based API pricing.",
  "description": "LLM APIs bill separately for input tokens (your prompt plus any retrieved context) and output tokens (the model's response), usually at different rates since output generation is more compute-intensive per token than input processing. This calculator takes your average input and output token counts per request, separate per-million-token rates for each, and monthly request volume to compute total spend, plus a per-request cost that's useful for pricing your own product on top of the API. As a practical estimation note, English text averages roughly 4 characters or about 0.75 words per token, so a 1,000-word prompt is roughly 1,300-1,400 tokens — use your provider's actual tokenizer for precise counts since this varies by model family, but the 4-characters-per-token rule gets you within about 10-15% for rough budgeting.",
  "formula": "Cost = (input tokens ÷ 1,000,000 × input rate + output tokens ÷ 1,000,000 × output rate) × requests per month.",
  "dateModified": "2026-09-30",
  "run_url": "https://www.revenuelab.fyi/api/public/calc?tool=llm-api-token-cost",
  "inputs": [
    {
      "id": "inputTokens",
      "label": "Avg input tokens per request",
      "kind": "number",
      "hint": null,
      "default": 1500,
      "unit": null,
      "min": 0,
      "max": null
    },
    {
      "id": "outputTokens",
      "label": "Avg output tokens per request",
      "kind": "number",
      "hint": null,
      "default": 500,
      "unit": null,
      "min": 0,
      "max": null
    },
    {
      "id": "inputRate",
      "label": "Input rate",
      "kind": "number",
      "hint": null,
      "default": 3,
      "unit": "$/1M tokens",
      "min": 0,
      "max": null
    },
    {
      "id": "outputRate",
      "label": "Output rate",
      "kind": "number",
      "hint": null,
      "default": 15,
      "unit": "$/1M tokens",
      "min": 0,
      "max": null
    },
    {
      "id": "requestsPerMonth",
      "label": "Requests per month",
      "kind": "number",
      "hint": null,
      "default": 200000,
      "unit": null,
      "min": 0,
      "max": null
    },
    {
      "id": "cacheDiscountPct",
      "label": "Cached input discount",
      "kind": "number",
      "hint": null,
      "default": 0,
      "unit": "%",
      "min": 0,
      "max": 95
    }
  ],
  "outputs": [
    {
      "id": "monthlyCost",
      "label": "Total monthly API cost",
      "format": "currency",
      "hint": null,
      "primary": true
    },
    {
      "id": "costPerRequest",
      "label": "Cost per request",
      "format": "currency",
      "hint": null,
      "primary": false
    },
    {
      "id": "inputCostPerReq",
      "label": "Total input token cost",
      "format": "currency",
      "hint": null,
      "primary": false
    },
    {
      "id": "outputCostPerReq",
      "label": "Total output token cost",
      "format": "currency",
      "hint": null,
      "primary": false
    }
  ],
  "worked_example": {
    "inputs": [
      "Avg input tokens per request: 1500",
      "Avg output tokens per request: 500",
      "Input rate: 3 $/1M tokens",
      "Output rate: 15 $/1M tokens",
      "Requests per month: 200000",
      "Cached input discount: 0 %"
    ],
    "outputs": [
      "Total monthly API cost: $2,400.00",
      "Cost per request: $0.012",
      "Total input token cost: $900.00",
      "Total output token cost: $1,500.00"
    ]
  },
  "how_to": {
    "title": "How to use this",
    "steps": [
      "Enter avg input tokens per request.",
      "Enter avg output tokens per request.",
      "Enter input rate ($/1M tokens).",
      "Enter output rate ($/1M tokens).",
      "Enter requests per month.",
      "Enter cached input discount (%).",
      "Read your total monthly api cost on the right — it updates as you type.",
      "Hit Share to keep the scenario or send it to someone."
    ]
  },
  "scenarios": [
    {
      "name": "Conservative",
      "description": "Lower-end numbers — what if things land soft?",
      "values": {
        "inputTokens": 900,
        "outputTokens": 300,
        "inputRate": 1.7999999999999998,
        "outputRate": 9,
        "requestsPerMonth": 120000,
        "cacheDiscountPct": 0
      }
    },
    {
      "name": "Typical",
      "description": "Defaults — the most common real-world setup.",
      "values": {
        "inputTokens": 1500,
        "outputTokens": 500,
        "inputRate": 3,
        "outputRate": 15,
        "requestsPerMonth": 200000,
        "cacheDiscountPct": 0
      }
    },
    {
      "name": "Ambitious",
      "description": "Higher-end numbers — what if things really pop?",
      "values": {
        "inputTokens": 2400,
        "outputTokens": 800,
        "inputRate": 4.800000000000001,
        "outputRate": 24,
        "requestsPerMonth": 320000,
        "cacheDiscountPct": 0
      }
    }
  ],
  "limitations": [
    "Results are estimates before tax, fees, and inflation unless an input explicitly covers them.",
    "Rates are treated as fixed for the whole period — variable-rate products will drift from this projection.",
    "This is educational maths, not financial advice. Check anything contractual with the lender or your accountant."
  ],
  "faq": [
    {
      "q": "Why is output token pricing so much higher than input?",
      "a": "Generating output tokens requires a sequential forward pass per token (autoregressive decoding), while input tokens can be processed in parallel during the prefill phase, making output generation more compute-intensive per token. Most providers price output at 3-5x the input rate as a result, which is why long, verbose responses cost disproportionately more than long, information-dense prompts."
    },
    {
      "q": "How much does prompt caching actually save?",
      "a": "Providers offering prompt/context caching typically discount repeated input tokens (a long system prompt or shared document context reused across requests) by 50-90%, since the model doesn't need to reprocess the cached prefix. For applications with a large shared system prompt or knowledge base, caching can cut total input cost dramatically even though it does nothing for the output side."
    },
    {
      "q": "How do I estimate token count before I have real usage data?",
      "a": "Use the roughly 4-characters-per-token (English) or 0.75-tokens-per-word heuristic on sample prompts and expected responses, then validate against your provider's actual tokenizer (OpenAI's tiktoken, Anthropic's token counting endpoint) once you have real traffic. Non-English languages and code often tokenize less efficiently — expect 20-40% more tokens per character than the English heuristic suggests."
    }
  ],
  "related": [
    "https://www.revenuelab.fyi/toolbox/gpu-training-cost",
    "https://www.revenuelab.fyi/toolbox/vector-db-storage-cost",
    "https://www.revenuelab.fyi/toolbox/serverless-invocation-cost"
  ],
  "license": "CC-BY-4.0",
  "citation": "RevenueLab — LLM API Token Cost Calculator (https://www.revenuelab.fyi/toolbox/llm-api-token-cost)"
}