VRAM Calculator — JavaScript source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* VRAM Calculator — estimate the VRAM an LLM needs (weights + KV cache).
*
* Language: JavaScript (ES2022+, ES module; runs unmodified in Node 18+
* and modern browsers)
* Source: CosmoDev polyglot showcase port of the VRAM Calculator tool,
* ported from src/lib/vramCalculator.ts (the canonical TypeScript
* implementation).
* Tool page: https://dev.cosmolabs.org/tools/vram-calculator
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Sizes use decimal gigabytes (GB = 10^9 bytes). Two components are
* estimated: weights (params × bytes-per-param) and the KV cache
* (2 × layers × context × kvHeads × headDim × batch × bytes per KV
* element). Activations and CUDA-context overhead are not modeled — treat
* the total as a floor and leave headroom on the card.
*/
/** Every supported quantization id, in the order the UI lists them. */
export const QUANTS = ['fp32', 'fp16', 'bf16', 'int8', 'int4', 'q4_K_M'];
/** Quantization table: bytes stored per weight for each format
* (Q4_K_M = 4.85 bits/weight, the llama.cpp mix). */
export const QUANT_BYTES = {
fp32: 4,
fp16: 2,
bf16: 2,
int8: 1,
int4: 0.5,
q4_K_M: 4.85 / 8,
};
/** Default attention architecture: a modern GQA-style layout (32 layers,
* 8 KV heads, 128-dim heads). Override per model. */
export const DEFAULT_ARCH = { layers: 32, kvHeads: 8, headDim: 128 };
/** Bytes per KV-cache element (fp16 K and V tensors) unless overridden. */
export const DEFAULT_KV_BYTES = 2;
/** Decimal gigabytes. */
const GB = 1e9;
/** Throws unless `value` is a finite number ≥ min (min itself allowed). */
function requireFiniteMin(name, value, min) {
if (!Number.isFinite(value) || value < min) {
throw new Error(`${name} must be a finite number ≥ ${min} (got ${value})`);
}
return value;
}
/**
* Estimate the VRAM footprint of a model: weights plus KV cache.
*
* weightsGB = paramsB × bytesPerParam
* kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9
*
* `opts` keys (all optional): layers (default 32), kvHeads (8), headDim
* (128), batch (1), kvBytes (2). Throws on paramsB ≤ 0, an unknown
* quantization, negative context, or any option below 1 (context 0 is
* allowed — no context, no cache).
*/
export function vram(paramsB, quant, context, opts = {}) {
if (!Number.isFinite(paramsB) || paramsB <= 0) {
throw new Error(`paramsB must be a finite number > 0 (got ${paramsB})`);
}
const bytesPerParam = QUANT_BYTES[quant];
if (bytesPerParam === undefined) {
throw new Error(`unknown quantization "${String(quant)}" — expected one of ${QUANTS.join(', ')}`);
}
requireFiniteMin('context', context, 0);
const layers = requireFiniteMin('layers', opts.layers ?? DEFAULT_ARCH.layers, 1);
const kvHeads = requireFiniteMin('kvHeads', opts.kvHeads ?? DEFAULT_ARCH.kvHeads, 1);
const headDim = requireFiniteMin('headDim', opts.headDim ?? DEFAULT_ARCH.headDim, 1);
const batch = requireFiniteMin('batch', opts.batch ?? 1, 1);
const kvBytes = requireFiniteMin('kvBytes', opts.kvBytes ?? DEFAULT_KV_BYTES, 1);
const weightsGB = (paramsB * 1e9 * bytesPerParam) / GB;
const kvCacheGB = (2 * layers * context * kvHeads * headDim * kvBytes * batch) / GB;
return { quant, bytesPerParam, weightsGB, kvCacheGB, totalGB: weightsGB + kvCacheGB };
}
/** Common GPU memory tiers, from consumer boards to datacenter cards.
* Each entry: `{ name, sizeGB }`. */
export const GPU_CARDS = [
{ name: 'RTX 3060 Ti / RTX 4060 / RX 7600', sizeGB: 8 },
{ name: 'RTX 3060 12 GB / RTX 4070', sizeGB: 12 },
{ name: 'RTX 4060 Ti 16 GB / RTX 5080', sizeGB: 16 },
{ name: 'RTX 3090 / RTX 4090', sizeGB: 24 },
{ name: 'RTX A6000 / L40S', sizeGB: 48 },
{ name: 'A100 80 GB / H100 / H200', sizeGB: 80 },
];
/**
* Score every card against a total footprint. `fits` is inclusive: a total
* exactly equal to the card size fits (headroom 0). Each row is
* `{ name, sizeGB, fits, headroomGB }` with headroomGB = sizeGB − totalGB
* (negative when the card is too small). Pass a custom `cards` list to
* score other tiers.
*/
export function gpuFits(totalGB, cards = GPU_CARDS) {
requireFiniteMin('totalGB', totalGB, 0);
return cards.map((c) => {
const headroomGB = c.sizeGB - totalGB;
return { name: c.name, sizeGB: c.sizeGB, fits: headroomGB >= 0, headroomGB };
});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →