VRAM Calculator — C source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* vram-calculator — C port: estimate the VRAM an LLM needs (weights + KV
cache). C11, stdlib only. Port of src/lib/vramCalculator.ts — same math
as this dir's typescript.ts; validation collapses to a 0/-1 status since
C has no exceptions. */
#include <math.h>
#include <stdio.h>
/* Quantizations, in the order the UI lists them. kQuantBytes[i] = bytes
stored per weight (Q4_K_M = 4.85 bits/weight, the llama.cpp mix). */
typedef enum { VRAM_FP32, VRAM_FP16, VRAM_BF16, VRAM_INT8, VRAM_INT4, VRAM_Q4_KM } vram_quant;
static const double kQuantBytes[] = {4, 2, 2, 1, 0.5, 4.85 / 8};
/* Attention architecture and serving knobs; every field must be >= 1. The
defaults are a modern GQA-style layout — e.g. Llama-2-70B overrides
layers to 80 with the same GQA shape. */
typedef struct {
int layers; /* transformer layers (blocks) — default 32 */
int kv_heads; /* key/value heads after GQA — default 8 */
int head_dim; /* dimension of one attention head — default 128 */
int batch; /* sequences served concurrently; multiplies the KV cache */
int kv_bytes; /* bytes per KV element (fp16 K and V tensors) */
} vram_opts;
static vram_opts vram_opts_default(void) {
vram_opts o = {32, 8, 128, 1, 2};
return o;
}
/* One VRAM estimate, in decimal gigabytes (GB = 10^9, matching how "70B"
and GPU sizes are quoted). */
typedef struct {
double bytes_per_param;
double weights_gb;
double kv_cache_gb;
double total_gb;
} vram_breakdown;
/* Estimate the VRAM footprint:
weights = params_b * 1e9 * bytes_per_param / 1e9
kv = 2 * layers * context * kv_heads * head_dim * kv_bytes * batch / 1e9
Returns 0, or -1 when params_b <= 0, context < 0, or an option < 1
(context 0 is allowed — no context, no cache). NULL opts = defaults. */
static int vram_calc(double params_b, vram_quant q, long context,
const vram_opts *opts_in, vram_breakdown *out) {
if (!isfinite(params_b) || params_b <= 0) return -1;
if (context < 0) return -1;
vram_opts o = opts_in ? *opts_in : vram_opts_default();
if (o.layers < 1 || o.kv_heads < 1 || o.head_dim < 1 || o.batch < 1 || o.kv_bytes < 1) return -1;
out->bytes_per_param = kQuantBytes[q];
out->weights_gb = params_b * kQuantBytes[q]; /* the 1e9s cancel */
out->kv_cache_gb = 2.0 * o.layers * context * o.kv_heads * o.head_dim * o.kv_bytes * o.batch / 1e9;
out->total_gb = out->weights_gb + out->kv_cache_gb;
return 0;
}
/* Common GPU memory tiers, from consumer boards to datacenter cards. */
static const char *kGpuNames[] = {
"RTX 3060 Ti / RTX 4060 / RX 7600", "RTX 3060 12 GB / RTX 4070",
"RTX 4060 Ti 16 GB / RTX 5080", "RTX 3090 / RTX 4090",
"RTX A6000 / L40S", "A100 80 GB / H100 / H200",
};
static const double kGpuSizes[] = {8, 12, 16, 24, 48, 80};
enum { GPU_COUNT = 6 };
int main(void) {
/* The lib's canonical vectors: a 70B Llama-2-shape build (80 layers) and
an 8B default-arch build, then the GPU fit table for the 8B total. */
vram_breakdown r70, r8;
vram_opts llama70 = {80, 8, 128, 1, 2};
if (vram_calc(70, VRAM_Q4_KM, 4096, &llama70, &r70) != 0 ||
vram_calc(8, VRAM_FP16, 8192, NULL, &r8) != 0)
return 1;
printf("70B q4_K_M @4096 (80 layers): %.2f GB weights + %.2f GB KV = %.2f GB total\n",
r70.weights_gb, r70.kv_cache_gb, r70.total_gb);
printf("8B fp16 @8192: %.2f GB weights + %.2f GB KV = %.2f GB total\n",
r8.weights_gb, r8.kv_cache_gb, r8.total_gb);
printf("GPU fit for %.2f GB:\n", r8.total_gb);
for (int i = 0; i < GPU_COUNT; i++) {
double headroom = kGpuSizes[i] - r8.total_gb; /* inclusive: 0 fits */
printf(" %-36s %s (%.2f GB %s)\n", kGpuNames[i], headroom >= 0 ? "fits" : "too small",
fabs(headroom), headroom >= 0 ? "headroom" : "short");
}
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →