Skip to content

VRAM Calculator — C++ source

Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// vram-calculator — C++ port: estimate the VRAM an LLM needs (weights + KV cache).
#include <cmath>
#include <cstdio>
#include <stdexcept>
#include <string>
#include <vector>

// Supported weight quantizations. bytesPerParam = bytes stored per weight
// (Q4_K_M = 4.85 bits/weight, the llama.cpp mix).
enum class Quant { Fp32, Fp16, Bf16, Int8, Int4, Q4KM };

constexpr double bytesPerParam(Quant q) {
    switch (q) {
        case Quant::Fp32: return 4;
        case Quant::Fp16:
        case Quant::Bf16: return 2;
        case Quant::Int8: return 1;
        case Quant::Int4: return 0.5;
        case Quant::Q4KM: return 4.85 / 8;
    }
    return 0; // unreachable; silences -Wreturn-type on some toolchains
}

// Attention architecture and serving knobs; every field must be >= 1. The
// defaults are a modern GQA-style layout — Llama-2-70B overrides layers to
// 80 with the same GQA shape.
struct VramOptions {
    int layers = 32;   // transformer layers (blocks)
    int kvHeads = 8;   // key/value heads after GQA
    int headDim = 128; // dimension of one attention head
    int batch = 1;     // sequences served concurrently; multiplies the KV cache
    int kvBytes = 2;   // bytes per KV element (fp16 K and V tensors)
};

// One VRAM estimate, in decimal gigabytes (GB = 10^9, matching how "70B"
// and GPU sizes are quoted).
struct VramBreakdown {
    Quant quant;
    double bytesPerParam;
    double weightsGB;
    double kvCacheGB;
    double totalGB() const { return weightsGB + kvCacheGB; }
};

// A GPU memory tier, and one card scored against a total.
struct GpuCard { std::string name; double sizeGB; };
struct GpuFit { std::string name; double sizeGB; bool fits; double headroomGB; };

constexpr double GB = 1e9;

// Common GPU memory tiers, from consumer boards to datacenter cards.
const std::vector<GpuCard>& gpuCards() {
    static const std::vector<GpuCard> cards = {
        {"RTX 3060 Ti / RTX 4060 / RX 7600", 8},
        {"RTX 3060 12 GB / RTX 4070", 12},
        {"RTX 4060 Ti 16 GB / RTX 5080", 16},
        {"RTX 3090 / RTX 4090", 24},
        {"RTX A6000 / L40S", 48},
        {"A100 80 GB / H100 / H200", 80},
    };
    return cards;
}

// Estimate the VRAM footprint of a model: weights plus KV cache.
//   weightsGB = paramsB × bytesPerParam
//   kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9
// Throws std::invalid_argument on paramsB ≤ 0, negative context, or any
// option below 1 (context 0 is allowed — no context, no cache).
VramBreakdown vram(double paramsB, Quant quant, long long context, const VramOptions& opts = {}) {
    if (!std::isfinite(paramsB) || paramsB <= 0)
        throw std::invalid_argument("paramsB must be finite > 0");
    if (context < 0) throw std::invalid_argument("context must be >= 0");
    if (opts.layers < 1 || opts.kvHeads < 1 || opts.headDim < 1 || opts.batch < 1 || opts.kvBytes < 1)
        throw std::invalid_argument("every option must be >= 1");

    double weightsGB = paramsB * 1e9 * bytesPerParam(quant) / GB;
    double kvCacheGB = 2.0 * opts.layers * context * opts.kvHeads * opts.headDim * opts.kvBytes * opts.batch / GB;
    return {quant, bytesPerParam(quant), weightsGB, kvCacheGB};
}

// Score every card against a total footprint. `fits` is inclusive: a total
// exactly equal to the card size fits (headroom 0).
std::vector<GpuFit> gpuFits(double totalGB, const std::vector<GpuCard>& cards = gpuCards()) {
    if (!std::isfinite(totalGB) || totalGB < 0)
        throw std::invalid_argument("totalGB must be finite >= 0");
    std::vector<GpuFit> out;
    out.reserve(cards.size());
    for (const auto& c : cards)
        out.push_back({c.name, c.sizeGB, c.sizeGB >= totalGB, c.sizeGB - totalGB});
    return out;
}

int main() {
    // The lib's canonical vectors: a 70B Llama-2-shape build (80 layers) and
    // an 8B default-arch build, then the GPU fit table for the 8B total.
    VramOptions llama70;
    llama70.layers = 80;
    auto r70 = vram(70, Quant::Q4KM, 4096, llama70);
    auto r8 = vram(8, Quant::Fp16, 8192);
    std::printf("70B q4_K_M @4096 (80 layers): %.2f GB weights + %.2f GB KV = %.2f GB total\n",
                r70.weightsGB, r70.kvCacheGB, r70.totalGB());
    std::printf("8B  fp16    @8192:            %.2f GB weights + %.2f GB KV = %.2f GB total\n",
                r8.weightsGB, r8.kvCacheGB, r8.totalGB());
    std::printf("GPU fit for %.2f GB:\n", r8.totalGB());
    for (const auto& f : gpuFits(r8.totalGB()))
        std::printf("  %-36s %s (%.2f GB %s)\n", f.name.c_str(), f.fits ? "fits" : "too small",
                    std::fabs(f.headroomGB), f.fits ? "headroom" : "short");
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →