Skip to content

VRAM Calculator — TypeScript source

Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * Pure VRAM math for the VRAM Calculator. Sizes use decimal gigabytes
 * (GB = 10^9 bytes) throughout, matching how parameter counts ("70B") and
 * GPU marketing sizes are quoted.
 *
 * Two components are estimated:
 * - Weights: params × bytes-per-param (the quantization's footprint).
 * - KV cache: 2 (K and V) × layers × context × kvHeads × headDim × batch
 *   × bytes per KV element.
 * Activations and CUDA-context overhead are not modeled — treat the total
 * as a floor and leave headroom on the card.
 */

/** Supported weight quantizations. */
export type Quant = 'fp32' | 'fp16' | 'bf16' | 'int8' | 'int4' | 'q4_K_M';

/** Every valid Quant, in the order the UI lists them. */
export const QUANTS: readonly Quant[] = ['fp32', 'fp16', 'bf16', 'int8', 'int4', 'q4_K_M'];

export interface QuantInfo {
  /** Select label. */
  label: string;
  /** Bytes stored per weight (Q4_K_M = 4.85 bits/weight, the llama.cpp mix). */
  bytesPerParam: number;
}

/** Quantization table: bytes per weight for each format. */
export const QUANT_INFO: Record<Quant, QuantInfo> = {
  fp32: { label: 'FP32 · 32-bit float', bytesPerParam: 4 },
  fp16: { label: 'FP16 · 16-bit float', bytesPerParam: 2 },
  bf16: { label: 'BF16 · bfloat16', bytesPerParam: 2 },
  int8: { label: 'INT8 · 8-bit', bytesPerParam: 1 },
  int4: { label: 'INT4 · 4-bit', bytesPerParam: 0.5 },
  q4_K_M: { label: 'Q4_K_M · GGUF ~4.85 bpw', bytesPerParam: 4.85 / 8 },
};

/** Quick-pick parameter counts (billions) shared by the UI. */
export const PARAM_PRESETS: readonly number[] = [0.5, 1.5, 7, 8, 13, 70];

/**
 * Default attention architecture: a modern GQA-style layout (32 layers,
 * 8 KV heads, 128-dim heads). Override per model — e.g. Llama-2-70B uses
 * 80 layers with the same GQA shape.
 */
export const DEFAULT_ARCH = { layers: 32, kvHeads: 8, headDim: 128 } as const;

/** Bytes per KV-cache element (fp16 K and V tensors) unless overridden. */
export const DEFAULT_KV_BYTES = 2;

export interface VramOptions {
  /** Transformer layers (blocks). Default 32. */
  layers?: number;
  /** Key/value heads after GQA. Default 8. */
  kvHeads?: number;
  /** Dimension of one attention head. Default 128. */
  headDim?: number;
  /** Sequences served concurrently; multiplies the KV cache. Default 1. */
  batch?: number;
  /** Bytes per KV-cache element. Default 2 (fp16). */
  kvBytes?: number;
}

export interface VramBreakdown {
  quant: Quant;
  /** Bytes per weight for the chosen quantization. */
  bytesPerParam: number;
  /** Weights footprint in GB. */
  weightsGB: number;
  /** KV-cache footprint in GB. */
  kvCacheGB: number;
  /** weightsGB + kvCacheGB. */
  totalGB: number;
}

/** Decimal gigabytes. */
const GB = 1e9;

/** Throws unless `value` is a finite number ≥ min (min itself allowed). */
function requireFiniteMin(name: string, value: number, min: number): number {
  if (!Number.isFinite(value) || value < min) {
    throw new Error(`${name} must be a finite number ≥ ${min} (got ${value})`);
  }
  return value;
}

/**
 * Estimate the VRAM footprint of a model: weights plus KV cache.
 *
 * weightsGB = paramsB × bytesPerParam
 * kvCacheGB = 2 × layers × context × kvHeads × headDim × kvBytes × batch / 1e9
 *
 * Throws on paramsB ≤ 0, an unknown quantization, negative context, or any
 * option below 1 (context 0 is allowed — no context, no cache).
 */
export function vram(
  paramsB: number,
  quant: Quant,
  context: number,
  opts: VramOptions = {},
): VramBreakdown {
  if (!Number.isFinite(paramsB) || paramsB <= 0) {
    throw new Error(`paramsB must be a finite number > 0 (got ${paramsB})`);
  }
  const info: QuantInfo | undefined = (QUANT_INFO as Record<string, QuantInfo | undefined>)[quant];
  if (info === undefined) {
    throw new Error(`unknown quantization "${String(quant)}" — expected one of ${QUANTS.join(', ')}`);
  }
  requireFiniteMin('context', context, 0);
  const layers = requireFiniteMin('layers', opts.layers ?? DEFAULT_ARCH.layers, 1);
  const kvHeads = requireFiniteMin('kvHeads', opts.kvHeads ?? DEFAULT_ARCH.kvHeads, 1);
  const headDim = requireFiniteMin('headDim', opts.headDim ?? DEFAULT_ARCH.headDim, 1);
  const batch = requireFiniteMin('batch', opts.batch ?? 1, 1);
  const kvBytes = requireFiniteMin('kvBytes', opts.kvBytes ?? DEFAULT_KV_BYTES, 1);

  const weightsGB = (paramsB * 1e9 * info.bytesPerParam) / GB;
  const kvCacheGB =
    (2 * layers * context * kvHeads * headDim * kvBytes * batch) / GB;
  return {
    quant,
    bytesPerParam: info.bytesPerParam,
    weightsGB,
    kvCacheGB,
    totalGB: weightsGB + kvCacheGB,
  };
}

export interface GpuCard {
  /** Card names consumers recognize, grouped per memory tier. */
  name: string;
  sizeGB: number;
}

/** Common GPU memory tiers, from consumer boards to datacenter cards. */
export const GPU_CARDS: readonly GpuCard[] = [
  { name: 'RTX 3060 Ti / RTX 4060 / RX 7600', sizeGB: 8 },
  { name: 'RTX 3060 12 GB / RTX 4070', sizeGB: 12 },
  { name: 'RTX 4060 Ti 16 GB / RTX 5080', sizeGB: 16 },
  { name: 'RTX 3090 / RTX 4090', sizeGB: 24 },
  { name: 'RTX A6000 / L40S', sizeGB: 48 },
  { name: 'A100 80 GB / H100 / H200', sizeGB: 80 },
];

export interface GpuFit {
  name: string;
  sizeGB: number;
  /** True when the total fits with zero or more GB to spare. */
  fits: boolean;
  /** sizeGB − totalGB; negative when the card is too small. */
  headroomGB: number;
}

/**
 * Score every card against a total footprint. `fits` is inclusive: a total
 * exactly equal to the card size fits (headroom 0). Pass a custom `cards`
 * list to score other tiers.
 */
export function gpuFits(totalGB: number, cards: readonly GpuCard[] = GPU_CARDS): GpuFit[] {
  requireFiniteMin('totalGB', totalGB, 0);
  return cards.map((c) => {
    const headroomGB = c.sizeGB - totalGB;
    return { name: c.name, sizeGB: c.sizeGB, fits: headroomGB >= 0, headroomGB };
  });
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →