VRAM Calculator — PHP source
Estimate the VRAM an LLM needs — weights by quantization plus the KV cache for your context and batch — and see which consumer and datacenter GPUs hold it.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* vram_calculator — estimate the VRAM an LLM needs (weights + KV cache).
*
* Language: PHP (8.1+, standard library only)
* Source: CosmoDev polyglot showcase port of the VRAM Calculator tool,
* ported from src/lib/vramCalculator.ts (the canonical TypeScript
* implementation).
* Tool page: https://dev.cosmolabs.org/tools/vram-calculator
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Sizes use decimal gigabytes (GB = 10^9 bytes). Two components are
* estimated: weights (params × bytes-per-param) and the KV cache
* (2 × layers × context × kvHeads × headDim × batch × bytes per KV
* element). Activations and CUDA-context overhead are not modeled — treat
* the total as a floor and leave headroom on the card.
*/
declare(strict_types=1);
/**
* Every supported quantization id, in the order the UI lists them.
*/
const VRAM_QUANTS = ['fp32', 'fp16', 'bf16', 'int8', 'int4', 'q4_K_M'];
/**
* Quantization table: bytes stored per weight for each format
* (q4_K_M = 4.85 bits/weight, the llama.cpp mix).
*/
const VRAM_BYTES_PER_PARAM = [
'fp32' => 4.0,
'fp16' => 2.0,
'bf16' => 2.0,
'int8' => 1.0,
'int4' => 0.5,
'q4_K_M' => 4.85 / 8.0,
];
/**
* Default attention architecture: a modern GQA-style layout (32 layers,
* 8 KV heads, 128-dim heads). Override per model.
*/
const VRAM_DEFAULT_LAYERS = 32;
const VRAM_DEFAULT_KV_HEADS = 8;
const VRAM_DEFAULT_HEAD_DIM = 128;
/**
* Bytes per KV-cache element (fp16 K and V tensors) unless overridden.
*/
const VRAM_DEFAULT_KV_BYTES = 2.0;
/**
* A decimal gigabyte.
*/
const VRAM_GB = 1e9;
/**
* Common GPU memory tiers, from consumer boards to datacenter cards.
* Each row: ['name' => string, 'sizeGB' => float].
*/
const VRAM_GPU_CARDS = [
['name' => 'RTX 3060 Ti / RTX 4060 / RX 7600', 'sizeGB' => 8.0],
['name' => 'RTX 3060 12 GB / RTX 4070', 'sizeGB' => 12.0],
['name' => 'RTX 4060 Ti 16 GB / RTX 5080', 'sizeGB' => 16.0],
['name' => 'RTX 3090 / RTX 4090', 'sizeGB' => 24.0],
['name' => 'RTX A6000 / L40S', 'sizeGB' => 48.0],
['name' => 'A100 80 GB / H100 / H200', 'sizeGB' => 80.0],
];
/**
* Throw unless the value is finite and >= $min ($min itself allowed).
*/
function vram_require_finite_min(string $name, float $value, float $min): float
{
if (!is_finite($value) || $value < $min) {
throw new InvalidArgumentException("{$name} must be a finite number >= {$min} (got {$value})");
}
return $value;
}
/**
* Estimate the VRAM footprint of a model: weights plus KV cache.
*
* weightsGB = paramsB * bytesPerParam
* kvCacheGB = 2 * layers * context * kvHeads * headDim * kvBytes * batch / 1e9
*
* $opts keys (all optional): layers (default 32), kvHeads (8), headDim
* (128), batch (1), kvBytes (2). Returns an array with keys quant,
* bytesPerParam, weightsGB, kvCacheGB, totalGB. Throws InvalidArgumentException
* on paramsB <= 0, an unknown quantization, negative context, or any option
* below 1 (context 0 is allowed — no context, no cache).
*/
function vram(float $paramsB, string $quant, float $context, ?array $opts = null): array
{
if (!is_finite($paramsB) || $paramsB <= 0.0) {
throw new InvalidArgumentException("paramsB must be a finite number > 0 (got {$paramsB})");
}
if (!isset(VRAM_BYTES_PER_PARAM[$quant])) {
$expected = implode(', ', VRAM_QUANTS);
throw new InvalidArgumentException("unknown quantization \"{$quant}\" — expected one of {$expected}");
}
$bytesPerParam = VRAM_BYTES_PER_PARAM[$quant];
$opts ??= [];
$layers = vram_require_finite_min('layers', (float)($opts['layers'] ?? VRAM_DEFAULT_LAYERS), 1.0);
$kvHeads = vram_require_finite_min('kvHeads', (float)($opts['kvHeads'] ?? VRAM_DEFAULT_KV_HEADS), 1.0);
$headDim = vram_require_finite_min('headDim', (float)($opts['headDim'] ?? VRAM_DEFAULT_HEAD_DIM), 1.0);
$batch = vram_require_finite_min('batch', (float)($opts['batch'] ?? 1), 1.0);
$kvBytes = vram_require_finite_min('kvBytes', (float)($opts['kvBytes'] ?? VRAM_DEFAULT_KV_BYTES), 1.0);
vram_require_finite_min('context', $context, 0.0);
$weightsGB = $paramsB * 1e9 * $bytesPerParam / VRAM_GB;
$kvCacheGB = 2.0 * $layers * $context * $kvHeads * $headDim * $kvBytes * $batch / VRAM_GB;
return [
'quant' => $quant,
'bytesPerParam' => $bytesPerParam,
'weightsGB' => $weightsGB,
'kvCacheGB' => $kvCacheGB,
'totalGB' => $weightsGB + $kvCacheGB,
];
}
/**
* Score every card against a total footprint. `fits` is inclusive: a total
* exactly equal to the card size fits (headroom 0). Returns a list of rows
* with keys name, sizeGB, fits, headroomGB; headroomGB = sizeGB − totalGB
* (negative when the card is too small). Pass a custom $cards list to score
* other tiers.
*/
function gpu_fits(float $totalGB, ?array $cards = null): array
{
vram_require_finite_min('totalGB', $totalGB, 0.0);
$cards ??= VRAM_GPU_CARDS;
$rows = [];
foreach ($cards as $card) {
$headroomGB = $card['sizeGB'] - $totalGB;
$rows[] = [
'name' => $card['name'],
'sizeGB' => $card['sizeGB'],
'fits' => $headroomGB >= 0.0,
'headroomGB' => $headroomGB,
];
}
return $rows;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →