VRAM Calculator — C# source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// vram-calculator — C# port: estimate the VRAM an LLM needs (weights + KV cache).
using System;
using System.Collections.Generic;
using System.Linq;
namespace VramCalculatorDemo;
/// <summary>Supported weight quantizations.</summary>
public enum Quant { Fp32, Fp16, Bf16, Int8, Int4, Q4KM }
public static class QuantInfo
{
/// <summary>Bytes stored per weight for each format
/// (Q4_K_M = 4.85 bits/weight, the llama.cpp mix).</summary>
public static double BytesPerParam(Quant q) => q switch
{
Quant.Fp32 => 4,
Quant.Fp16 or Quant.Bf16 => 2,
Quant.Int8 => 1,
Quant.Int4 => 0.5,
Quant.Q4KM => 4.85 / 8,
_ => throw new ArgumentException($"unknown quantization {q}"),
};
}
/// <summary>Architecture and serving knobs; every field must be >= 1.
/// Defaults form the modern GQA-style layout (Llama-2-70B overrides
/// Layers to 80).</summary>
public record VramOptions(
int Layers = 32, int KvHeads = 8, int HeadDim = 128, int Batch = 1, int KvBytes = 2);
/// <summary>One VRAM estimate, in decimal gigabytes (GB = 10^9, matching
/// how "70B" and GPU sizes are quoted).</summary>
public record VramBreakdown(Quant Quant, double BytesPerParam, double WeightsGB, double KvCacheGB)
{
public double TotalGB => WeightsGB + KvCacheGB;
}
/// <summary>A GPU memory tier, and one card scored against a total.</summary>
public record GpuCard(string Name, double SizeGB);
public record GpuFit(string Name, double SizeGB, bool Fits, double HeadroomGB);
public static class Vram
{
private const double GB = 1e9;
/// <summary>Common GPU memory tiers, from consumer boards to datacenter cards.</summary>
public static readonly IReadOnlyList<GpuCard> GpuCards = new GpuCard[]
{
new("RTX 3060 Ti / RTX 4060 / RX 7600", 8),
new("RTX 3060 12 GB / RTX 4070", 12),
new("RTX 4060 Ti 16 GB / RTX 5080", 16),
new("RTX 3090 / RTX 4090", 24),
new("RTX A6000 / L40S", 48),
new("A100 80 GB / H100 / H200", 80),
};
/// <summary>
/// Estimate the VRAM footprint of a model: weights plus KV cache.
/// weightsGB = paramsB × bytesPerParam;
/// kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9.
/// Throws ArgumentException on paramsB ≤ 0, negative context, or any
/// option below 1 (context 0 is allowed — no context, no cache).
/// </summary>
public static VramBreakdown Estimate(double paramsB, Quant quant, int context, VramOptions? opts = null)
{
if (!double.IsFinite(paramsB) || paramsB <= 0)
throw new ArgumentException($"paramsB must be finite > 0 (got {paramsB})");
if (context < 0) throw new ArgumentException($"context must be >= 0 (got {context})");
VramOptions o = opts ?? new();
if (o.Layers < 1 || o.KvHeads < 1 || o.HeadDim < 1 || o.Batch < 1 || o.KvBytes < 1)
throw new ArgumentException($"every option must be >= 1 (got {o})");
double bpp = QuantInfo.BytesPerParam(quant);
double weightsGB = paramsB * 1e9 * bpp / GB;
double kvCacheGB = 2.0 * o.Layers * context * o.KvHeads * o.HeadDim * o.KvBytes * o.Batch / GB;
return new VramBreakdown(quant, bpp, weightsGB, kvCacheGB);
}
/// <summary>Score every card against a total footprint. Fits is
/// inclusive: a total exactly equal to the card size fits (headroom 0).</summary>
public static IEnumerable<GpuFit> GpuFits(double totalGB, IReadOnlyList<GpuCard>? cards = null)
{
if (!double.IsFinite(totalGB) || totalGB < 0)
throw new ArgumentException($"totalGB must be finite >= 0 (got {totalGB})");
return (cards ?? GpuCards).Select(c => new GpuFit(c.Name, c.SizeGB, c.SizeGB >= totalGB, c.SizeGB - totalGB));
}
}
public static class Program
{
public static void Main()
{
// The lib's canonical vectors: a 70B Llama-2-shape build (80 layers)
// and an 8B default-arch build, then the GPU fit table for the 8B total.
var r70 = Vram.Estimate(70, Quant.Q4KM, 4096, new VramOptions(Layers: 80));
var r8 = Vram.Estimate(8, Quant.Fp16, 8192);
Console.WriteLine($"70B q4_K_M @4096 (80 layers): {r70.WeightsGB:F2} GB weights + {r70.KvCacheGB:F2} GB KV = {r70.TotalGB:F2} GB total");
Console.WriteLine($"8B fp16 @8192: {r8.WeightsGB:F2} GB weights + {r8.KvCacheGB:F2} GB KV = {r8.TotalGB:F2} GB total");
Console.WriteLine($"GPU fit for {r8.TotalGB:F2} GB:");
foreach (var f in Vram.GpuFits(r8.TotalGB))
Console.WriteLine($" {f.Name,-36} {(f.Fits ? "fits" : "too small")} ({Math.Abs(f.HeadroomGB):F2} GB {(f.Fits ? "headroom" : "short")})");
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →