{
  "slug": "gpu-training-cost",
  "title": "GPU Training Cost Calculator",
  "heading": "GPU Model Training Cost Calculator",
  "category": "financial",
  "url": "https://www.revenuelab.fyi/toolbox/gpu-training-cost",
  "summary": "Estimate the cloud GPU bill for a training run before you start it.",
  "description": "GPU training cost is driven by three multiplied factors: how many GPU-hours the job needs (GPU count × wall-clock hours), the hourly rate for that GPU type, and whether you're paying on-demand or spot/preemptible pricing. This calculator computes total training cost from those inputs and also reports cost per training epoch if you provide epoch count, which is the number teams actually use to decide whether to keep iterating on hyperparameters or cut losses on a run that isn't converging. Spot/preemptible GPU instances typically cost 50-90% less than on-demand but can be reclaimed by the cloud provider mid-job, so they only make sense with checkpointing that lets a job resume rather than restart — factor that operational requirement in before assuming the spot discount is free money, since a job that gets preempted repeatedly without good checkpointing can end up costing more in wasted compute than it saved on rate.",
  "formula": "Total cost = GPU count × hourly rate per GPU × training hours × (1 − spot discount%).",
  "dateModified": "2026-09-30",
  "run_url": "https://www.revenuelab.fyi/api/public/calc?tool=gpu-training-cost",
  "inputs": [
    {
      "id": "gpuCount",
      "label": "Number of GPUs",
      "kind": "number",
      "hint": null,
      "default": 8,
      "unit": null,
      "min": 1,
      "max": null
    },
    {
      "id": "hourlyRate",
      "label": "Hourly rate per GPU",
      "kind": "number",
      "hint": null,
      "default": 2.5,
      "unit": "$",
      "min": 0,
      "max": null
    },
    {
      "id": "trainingHours",
      "label": "Total training wall-clock hours",
      "kind": "number",
      "hint": null,
      "default": 72,
      "unit": null,
      "min": 0.1,
      "max": null
    },
    {
      "id": "spotDiscountPct",
      "label": "Spot/preemptible discount",
      "kind": "number",
      "hint": null,
      "default": 0,
      "unit": "%",
      "min": 0,
      "max": 90
    },
    {
      "id": "epochs",
      "label": "Training epochs (optional)",
      "kind": "number",
      "hint": null,
      "default": 10,
      "unit": null,
      "min": 0,
      "max": null
    }
  ],
  "outputs": [
    {
      "id": "totalCost",
      "label": "Total training cost",
      "format": "currency",
      "hint": null,
      "primary": true
    },
    {
      "id": "gpuHours",
      "label": "Total GPU-hours consumed",
      "format": "number",
      "hint": null,
      "primary": false
    },
    {
      "id": "costPerEpoch",
      "label": "Cost per epoch",
      "format": "currency",
      "hint": null,
      "primary": false
    },
    {
      "id": "effectiveRate",
      "label": "Effective hourly rate per GPU",
      "format": "currency",
      "hint": null,
      "primary": false
    }
  ],
  "worked_example": {
    "inputs": [
      "Number of GPUs: 8",
      "Hourly rate per GPU: 2.5 $",
      "Total training wall-clock hours: 72",
      "Spot/preemptible discount: 0 %",
      "Training epochs (optional): 10"
    ],
    "outputs": [
      "Total training cost: $1,440.00",
      "Total GPU-hours consumed: 576",
      "Cost per epoch: $144.00",
      "Effective hourly rate per GPU: $2.50"
    ]
  },
  "how_to": {
    "title": "How to use this",
    "steps": [
      "Enter number of gpus.",
      "Enter hourly rate per gpu ($).",
      "Enter total training wall-clock hours.",
      "Enter spot/preemptible discount (%).",
      "Enter training epochs (optional).",
      "Read your total training cost on the right — it updates as you type.",
      "Hit Share to keep the scenario or send it to someone."
    ]
  },
  "scenarios": [
    {
      "name": "Conservative",
      "description": "Lower-end numbers — what if things land soft?",
      "values": {
        "gpuCount": 5,
        "hourlyRate": 1.5,
        "trainingHours": 43,
        "spotDiscountPct": 0,
        "epochs": 6
      }
    },
    {
      "name": "Typical",
      "description": "Defaults — the most common real-world setup.",
      "values": {
        "gpuCount": 8,
        "hourlyRate": 2.5,
        "trainingHours": 72,
        "spotDiscountPct": 0,
        "epochs": 10
      }
    },
    {
      "name": "Ambitious",
      "description": "Higher-end numbers — what if things really pop?",
      "values": {
        "gpuCount": 13,
        "hourlyRate": 4,
        "trainingHours": 115,
        "spotDiscountPct": 0,
        "epochs": 16
      }
    }
  ],
  "limitations": [
    "Results are estimates before tax, fees, and inflation unless an input explicitly covers them.",
    "Rates are treated as fixed for the whole period — variable-rate products will drift from this projection.",
    "This is educational maths, not financial advice. Check anything contractual with the lender or your accountant."
  ],
  "faq": [
    {
      "q": "Is renting GPUs from a specialty cloud cheaper than AWS/GCP/Azure?",
      "a": "Often yes, sometimes substantially — specialty GPU cloud providers frequently price high-end accelerators 30-60% below the big three hyperscalers because they don't bundle the same breadth of managed services and enterprise support. For a pure training job with no other cloud service dependencies, it's worth comparing rates directly since the discount can be the difference between a run being affordable or not."
    },
    {
      "q": "How risky is using spot instances for a multi-day training run?",
      "a": "Reclamation rates vary by GPU type, region, and demand, but multi-day jobs on popular GPU types can see preemption multiple times over their run. This is only safe with frequent checkpointing (every 15-30 minutes or every N steps) so a preempted job resumes near where it left off rather than restarting from scratch — without that, spot savings can be wiped out by wasted repeated computation."
    },
    {
      "q": "Why does GPU count matter separately from GPU-hours?",
      "a": "Multi-GPU training introduces communication overhead (gradient synchronization across GPUs/nodes) that means going from 1 to 8 GPUs rarely gives a full 8x speedup — realistic scaling efficiency is often 70-90% depending on interconnect (NVLink vs. standard networking) and model architecture. Model your actual measured scaling efficiency, not a naive linear assumption, or you'll underestimate total training hours needed."
    }
  ],
  "related": [
    "https://www.revenuelab.fyi/toolbox/llm-api-token-cost",
    "https://www.revenuelab.fyi/toolbox/vector-db-storage-cost",
    "https://www.revenuelab.fyi/toolbox/cloud-vm-monthly-cost"
  ],
  "license": "CC-BY-4.0",
  "citation": "RevenueLab — GPU Training Cost Calculator (https://www.revenuelab.fyi/toolbox/gpu-training-cost)"
}