VRAM Calculator — C++ source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// vram-calculator — C++ port: estimate the VRAM an LLM needs (weights + KV cache).
#include <cmath>
#include <cstdio>
#include <stdexcept>
#include <string>
#include <vector>
// Supported weight quantizations. bytesPerParam = bytes stored per weight
// (Q4_K_M = 4.85 bits/weight, the llama.cpp mix).
enum class Quant { Fp32, Fp16, Bf16, Int8, Int4, Q4KM };
constexpr double bytesPerParam(Quant q) {
switch (q) {
case Quant::Fp32: return 4;
case Quant::Fp16:
case Quant::Bf16: return 2;
case Quant::Int8: return 1;
case Quant::Int4: return 0.5;
case Quant::Q4KM: return 4.85 / 8;
}
return 0; // unreachable; silences -Wreturn-type on some toolchains
}
// Attention architecture and serving knobs; every field must be >= 1. The
// defaults are a modern GQA-style layout — Llama-2-70B overrides layers to
// 80 with the same GQA shape.
struct VramOptions {
int layers = 32; // transformer layers (blocks)
int kvHeads = 8; // key/value heads after GQA
int headDim = 128; // dimension of one attention head
int batch = 1; // sequences served concurrently; multiplies the KV cache
int kvBytes = 2; // bytes per KV element (fp16 K and V tensors)
};
// One VRAM estimate, in decimal gigabytes (GB = 10^9, matching how "70B"
// and GPU sizes are quoted).
struct VramBreakdown {
Quant quant;
double bytesPerParam;
double weightsGB;
double kvCacheGB;
double totalGB() const { return weightsGB + kvCacheGB; }
};
// A GPU memory tier, and one card scored against a total.
struct GpuCard { std::string name; double sizeGB; };
struct GpuFit { std::string name; double sizeGB; bool fits; double headroomGB; };
constexpr double GB = 1e9;
// Common GPU memory tiers, from consumer boards to datacenter cards.
const std::vector<GpuCard>& gpuCards() {
static const std::vector<GpuCard> cards = {
{"RTX 3060 Ti / RTX 4060 / RX 7600", 8},
{"RTX 3060 12 GB / RTX 4070", 12},
{"RTX 4060 Ti 16 GB / RTX 5080", 16},
{"RTX 3090 / RTX 4090", 24},
{"RTX A6000 / L40S", 48},
{"A100 80 GB / H100 / H200", 80},
};
return cards;
}
// Estimate the VRAM footprint of a model: weights plus KV cache.
// weightsGB = paramsB × bytesPerParam
// kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9
// Throws std::invalid_argument on paramsB ≤ 0, negative context, or any
// option below 1 (context 0 is allowed — no context, no cache).
VramBreakdown vram(double paramsB, Quant quant, long long context, const VramOptions& opts = {}) {
if (!std::isfinite(paramsB) || paramsB <= 0)
throw std::invalid_argument("paramsB must be finite > 0");
if (context < 0) throw std::invalid_argument("context must be >= 0");
if (opts.layers < 1 || opts.kvHeads < 1 || opts.headDim < 1 || opts.batch < 1 || opts.kvBytes < 1)
throw std::invalid_argument("every option must be >= 1");
double weightsGB = paramsB * 1e9 * bytesPerParam(quant) / GB;
double kvCacheGB = 2.0 * opts.layers * context * opts.kvHeads * opts.headDim * opts.kvBytes * opts.batch / GB;
return {quant, bytesPerParam(quant), weightsGB, kvCacheGB};
}
// Score every card against a total footprint. `fits` is inclusive: a total
// exactly equal to the card size fits (headroom 0).
std::vector<GpuFit> gpuFits(double totalGB, const std::vector<GpuCard>& cards = gpuCards()) {
if (!std::isfinite(totalGB) || totalGB < 0)
throw std::invalid_argument("totalGB must be finite >= 0");
std::vector<GpuFit> out;
out.reserve(cards.size());
for (const auto& c : cards)
out.push_back({c.name, c.sizeGB, c.sizeGB >= totalGB, c.sizeGB - totalGB});
return out;
}
int main() {
// The lib's canonical vectors: a 70B Llama-2-shape build (80 layers) and
// an 8B default-arch build, then the GPU fit table for the 8B total.
VramOptions llama70;
llama70.layers = 80;
auto r70 = vram(70, Quant::Q4KM, 4096, llama70);
auto r8 = vram(8, Quant::Fp16, 8192);
std::printf("70B q4_K_M @4096 (80 layers): %.2f GB weights + %.2f GB KV = %.2f GB total\n",
r70.weightsGB, r70.kvCacheGB, r70.totalGB());
std::printf("8B fp16 @8192: %.2f GB weights + %.2f GB KV = %.2f GB total\n",
r8.weightsGB, r8.kvCacheGB, r8.totalGB());
std::printf("GPU fit for %.2f GB:\n", r8.totalGB());
for (const auto& f : gpuFits(r8.totalGB()))
std::printf(" %-36s %s (%.2f GB %s)\n", f.name.c_str(), f.fits ? "fits" : "too small",
std::fabs(f.headroomGB), f.fits ? "headroom" : "short");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →