VRAM Calculator — TypeScript source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* Pure VRAM math for the VRAM Calculator. Sizes use decimal gigabytes
* (GB = 10^9 bytes) throughout, matching how parameter counts ("70B") and
* GPU marketing sizes are quoted.
*
* Two components are estimated:
* - Weights: params × bytes-per-param (the quantization's footprint).
* - KV cache: 2 (K and V) × layers × context × kvHeads × headDim × batch
* × bytes per KV element.
* Activations and CUDA-context overhead are not modeled — treat the total
* as a floor and leave headroom on the card.
*/
/** Supported weight quantizations. */
export type Quant = 'fp32' | 'fp16' | 'bf16' | 'int8' | 'int4' | 'q4_K_M';
/** Every valid Quant, in the order the UI lists them. */
export const QUANTS: readonly Quant[] = ['fp32', 'fp16', 'bf16', 'int8', 'int4', 'q4_K_M'];
export interface QuantInfo {
/** Select label. */
label: string;
/** Bytes stored per weight (Q4_K_M = 4.85 bits/weight, the llama.cpp mix). */
bytesPerParam: number;
}
/** Quantization table: bytes per weight for each format. */
export const QUANT_INFO: Record<Quant, QuantInfo> = {
fp32: { label: 'FP32 · 32-bit float', bytesPerParam: 4 },
fp16: { label: 'FP16 · 16-bit float', bytesPerParam: 2 },
bf16: { label: 'BF16 · bfloat16', bytesPerParam: 2 },
int8: { label: 'INT8 · 8-bit', bytesPerParam: 1 },
int4: { label: 'INT4 · 4-bit', bytesPerParam: 0.5 },
q4_K_M: { label: 'Q4_K_M · GGUF ~4.85 bpw', bytesPerParam: 4.85 / 8 },
};
/** Quick-pick parameter counts (billions) shared by the UI. */
export const PARAM_PRESETS: readonly number[] = [0.5, 1.5, 7, 8, 13, 70];
/**
* Default attention architecture: a modern GQA-style layout (32 layers,
* 8 KV heads, 128-dim heads). Override per model — e.g. Llama-2-70B uses
* 80 layers with the same GQA shape.
*/
export const DEFAULT_ARCH = { layers: 32, kvHeads: 8, headDim: 128 } as const;
/** Bytes per KV-cache element (fp16 K and V tensors) unless overridden. */
export const DEFAULT_KV_BYTES = 2;
export interface VramOptions {
/** Transformer layers (blocks). Default 32. */
layers?: number;
/** Key/value heads after GQA. Default 8. */
kvHeads?: number;
/** Dimension of one attention head. Default 128. */
headDim?: number;
/** Sequences served concurrently; multiplies the KV cache. Default 1. */
batch?: number;
/** Bytes per KV-cache element. Default 2 (fp16). */
kvBytes?: number;
}
export interface VramBreakdown {
quant: Quant;
/** Bytes per weight for the chosen quantization. */
bytesPerParam: number;
/** Weights footprint in GB. */
weightsGB: number;
/** KV-cache footprint in GB. */
kvCacheGB: number;
/** weightsGB + kvCacheGB. */
totalGB: number;
}
/** Decimal gigabytes. */
const GB = 1e9;
/** Throws unless `value` is a finite number ≥ min (min itself allowed). */
function requireFiniteMin(name: string, value: number, min: number): number {
if (!Number.isFinite(value) || value < min) {
throw new Error(`${name} must be a finite number ≥ ${min} (got ${value})`);
}
return value;
}
/**
* Estimate the VRAM footprint of a model: weights plus KV cache.
*
* weightsGB = paramsB × bytesPerParam
* kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9
*
* Throws on paramsB ≤ 0, an unknown quantization, negative context, or any
* option below 1 (context 0 is allowed — no context, no cache).
*/
export function vram(
paramsB: number,
quant: Quant,
context: number,
opts: VramOptions = {},
): VramBreakdown {
if (!Number.isFinite(paramsB) || paramsB <= 0) {
throw new Error(`paramsB must be a finite number > 0 (got ${paramsB})`);
}
const info: QuantInfo | undefined = (QUANT_INFO as Record<string, QuantInfo | undefined>)[quant];
if (info === undefined) {
throw new Error(`unknown quantization "${String(quant)}" — expected one of ${QUANTS.join(', ')}`);
}
requireFiniteMin('context', context, 0);
const layers = requireFiniteMin('layers', opts.layers ?? DEFAULT_ARCH.layers, 1);
const kvHeads = requireFiniteMin('kvHeads', opts.kvHeads ?? DEFAULT_ARCH.kvHeads, 1);
const headDim = requireFiniteMin('headDim', opts.headDim ?? DEFAULT_ARCH.headDim, 1);
const batch = requireFiniteMin('batch', opts.batch ?? 1, 1);
const kvBytes = requireFiniteMin('kvBytes', opts.kvBytes ?? DEFAULT_KV_BYTES, 1);
const weightsGB = (paramsB * 1e9 * info.bytesPerParam) / GB;
const kvCacheGB =
(2 * layers * context * kvHeads * headDim * kvBytes * batch) / GB;
return {
quant,
bytesPerParam: info.bytesPerParam,
weightsGB,
kvCacheGB,
totalGB: weightsGB + kvCacheGB,
};
}
export interface GpuCard {
/** Card names consumers recognize, grouped per memory tier. */
name: string;
sizeGB: number;
}
/** Common GPU memory tiers, from consumer boards to datacenter cards. */
export const GPU_CARDS: readonly GpuCard[] = [
{ name: 'RTX 3060 Ti / RTX 4060 / RX 7600', sizeGB: 8 },
{ name: 'RTX 3060 12 GB / RTX 4070', sizeGB: 12 },
{ name: 'RTX 4060 Ti 16 GB / RTX 5080', sizeGB: 16 },
{ name: 'RTX 3090 / RTX 4090', sizeGB: 24 },
{ name: 'RTX A6000 / L40S', sizeGB: 48 },
{ name: 'A100 80 GB / H100 / H200', sizeGB: 80 },
];
export interface GpuFit {
name: string;
sizeGB: number;
/** True when the total fits with zero or more GB to spare. */
fits: boolean;
/** sizeGB − totalGB; negative when the card is too small. */
headroomGB: number;
}
/**
* Score every card against a total footprint. `fits` is inclusive: a total
* exactly equal to the card size fits (headroom 0). Pass a custom `cards`
* list to score other tiers.
*/
export function gpuFits(totalGB: number, cards: readonly GpuCard[] = GPU_CARDS): GpuFit[] {
requireFiniteMin('totalGB', totalGB, 0);
return cards.map((c) => {
const headroomGB = c.sizeGB - totalGB;
return { name: c.name, sizeGB: c.sizeGB, fits: headroomGB >= 0, headroomGB };
});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →