Skip to content

VRAM Calculator — C# source

Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// vram-calculator — C# port: estimate the VRAM an LLM needs (weights + KV cache).

using System;
using System.Collections.Generic;
using System.Linq;

namespace VramCalculatorDemo;

/// <summary>Supported weight quantizations.</summary>
public enum Quant { Fp32, Fp16, Bf16, Int8, Int4, Q4KM }

public static class QuantInfo
{
    /// <summary>Bytes stored per weight for each format
    /// (Q4_K_M = 4.85 bits/weight, the llama.cpp mix).</summary>
    public static double BytesPerParam(Quant q) => q switch
    {
        Quant.Fp32 => 4,
        Quant.Fp16 or Quant.Bf16 => 2,
        Quant.Int8 => 1,
        Quant.Int4 => 0.5,
        Quant.Q4KM => 4.85 / 8,
        _ => throw new ArgumentException($"unknown quantization {q}"),
    };
}

/// <summary>Architecture and serving knobs; every field must be >= 1.
/// Defaults form the modern GQA-style layout (Llama-2-70B overrides
/// Layers to 80).</summary>
public record VramOptions(
    int Layers = 32, int KvHeads = 8, int HeadDim = 128, int Batch = 1, int KvBytes = 2);

/// <summary>One VRAM estimate, in decimal gigabytes (GB = 10^9, matching
/// how "70B" and GPU sizes are quoted).</summary>
public record VramBreakdown(Quant Quant, double BytesPerParam, double WeightsGB, double KvCacheGB)
{
    public double TotalGB => WeightsGB + KvCacheGB;
}

/// <summary>A GPU memory tier, and one card scored against a total.</summary>
public record GpuCard(string Name, double SizeGB);
public record GpuFit(string Name, double SizeGB, bool Fits, double HeadroomGB);

public static class Vram
{
    private const double GB = 1e9;

    /// <summary>Common GPU memory tiers, from consumer boards to datacenter cards.</summary>
    public static readonly IReadOnlyList<GpuCard> GpuCards = new GpuCard[]
    {
        new("RTX 3060 Ti / RTX 4060 / RX 7600", 8),
        new("RTX 3060 12 GB / RTX 4070", 12),
        new("RTX 4060 Ti 16 GB / RTX 5080", 16),
        new("RTX 3090 / RTX 4090", 24),
        new("RTX A6000 / L40S", 48),
        new("A100 80 GB / H100 / H200", 80),
    };

    /// <summary>
    /// Estimate the VRAM footprint of a model: weights plus KV cache.
    /// weightsGB = paramsB × bytesPerParam;
    /// kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9.
    /// Throws ArgumentException on paramsB ≤ 0, negative context, or any
    /// option below 1 (context 0 is allowed — no context, no cache).
    /// </summary>
    public static VramBreakdown Estimate(double paramsB, Quant quant, int context, VramOptions? opts = null)
    {
        if (!double.IsFinite(paramsB) || paramsB <= 0)
            throw new ArgumentException($"paramsB must be finite > 0 (got {paramsB})");
        if (context < 0) throw new ArgumentException($"context must be >= 0 (got {context})");
        VramOptions o = opts ?? new();
        if (o.Layers < 1 || o.KvHeads < 1 || o.HeadDim < 1 || o.Batch < 1 || o.KvBytes < 1)
            throw new ArgumentException($"every option must be >= 1 (got {o})");

        double bpp = QuantInfo.BytesPerParam(quant);
        double weightsGB = paramsB * 1e9 * bpp / GB;
        double kvCacheGB = 2.0 * o.Layers * context * o.KvHeads * o.HeadDim * o.KvBytes * o.Batch / GB;
        return new VramBreakdown(quant, bpp, weightsGB, kvCacheGB);
    }

    /// <summary>Score every card against a total footprint. Fits is
    /// inclusive: a total exactly equal to the card size fits (headroom 0).</summary>
    public static IEnumerable<GpuFit> GpuFits(double totalGB, IReadOnlyList<GpuCard>? cards = null)
    {
        if (!double.IsFinite(totalGB) || totalGB < 0)
            throw new ArgumentException($"totalGB must be finite >= 0 (got {totalGB})");
        return (cards ?? GpuCards).Select(c => new GpuFit(c.Name, c.SizeGB, c.SizeGB >= totalGB, c.SizeGB - totalGB));
    }
}

public static class Program
{
    public static void Main()
    {
        // The lib's canonical vectors: a 70B Llama-2-shape build (80 layers)
        // and an 8B default-arch build, then the GPU fit table for the 8B total.
        var r70 = Vram.Estimate(70, Quant.Q4KM, 4096, new VramOptions(Layers: 80));
        var r8 = Vram.Estimate(8, Quant.Fp16, 8192);
        Console.WriteLine($"70B q4_K_M @4096 (80 layers): {r70.WeightsGB:F2} GB weights + {r70.KvCacheGB:F2} GB KV = {r70.TotalGB:F2} GB total");
        Console.WriteLine($"8B  fp16    @8192:            {r8.WeightsGB:F2} GB weights + {r8.KvCacheGB:F2} GB KV = {r8.TotalGB:F2} GB total");
        Console.WriteLine($"GPU fit for {r8.TotalGB:F2} GB:");
        foreach (var f in Vram.GpuFits(r8.TotalGB))
            Console.WriteLine($"  {f.Name,-36} {(f.Fits ? "fits" : "too small")} ({Math.Abs(f.HeadroomGB):F2} GB {(f.Fits ? "headroom" : "short")})");
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →