Image Token Calculator — Python source
Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""Image Token Calculator — estimate the vision token cost of an image using
OpenAI-style tile math.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the Image Token Calculator
tool, ported from src/lib/imageTokenCalculator.ts (the canonical
TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/image-token-calculator
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; raises ValueError on invalid input (as the TS
reference throws).
- Functionally equivalent to the TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no pip packages).
Rounding note: the TS reference uses Math.round (half up). Python's built-in
round() is banker's rounding, so _round_half_up reproduces the TS behavior.
"""
from __future__ import annotations
import math
from dataclasses import dataclass
from typing import Literal, Tuple, Union
DetailLevel = Literal["low", "high", "auto"]
#: Fixed token cost of the low-resolution image view.
LOW_DETAIL_TOKENS = 85
#: Token cost of one high-resolution 512 px tile.
TILE_TOKENS = 170
#: Images are first scaled to fit inside this square.
MAX_SIDE = 2048
#: Then the shortest side is capped at this length.
MAX_SHORT_SIDE = 768
#: Tile edge length in pixels.
TILE_SIZE = 512
#: Both dimensions at or under this -> 'auto' stays low detail.
AUTO_LOW_MAX = 512
@dataclass(frozen=True)
class TokenBreakdown:
"""Mirrors the TokenBreakdown interface in the TS lib."""
detail: Literal["low", "high"]
#: Dimensions after the high-detail downscaling pipeline (identity for low).
scaled_width: int
scaled_height: int
#: 512 px tiles along each axis (both 1 in low detail).
tiles_x: int
tiles_y: int
#: Total 512 px tiles used (tiles_x * tiles_y).
tiles: int
#: Fixed base cost of the low-resolution view, in tokens.
base: int
#: Extra tokens for the high-resolution tile views (0 in low detail).
detail_tokens: int
#: Total estimated tokens: base + detail_tokens.
total: int
def _round_half_up(x: float) -> int:
"""JS Math.round parity: round half up (values here are always positive)."""
return math.floor(x + 0.5)
def _reject_non_dimension(name: str, value: object) -> None:
"""Raise ValueError unless value is an int (not bool) greater than zero."""
if isinstance(value, bool) or not isinstance(value, int):
raise ValueError("Width and height must be whole pixels")
if value <= 0:
raise ValueError("Width and height must be greater than zero")
def preprocess_image(width: int, height: int) -> Tuple[int, int]:
"""Scale (width, height) per the vision preprocessing pipeline:
1. fit inside a MAX_SIDE x MAX_SIDE square (longest side capped), then
2. cap the shortest side at MAX_SHORT_SIDE.
Aspect ratio is preserved; each step is skipped when already satisfied.
"""
longest = max(width, height)
if longest > MAX_SIDE:
scale = MAX_SIDE / longest
width = max(1, _round_half_up(width * scale))
height = max(1, _round_half_up(height * scale))
shortest = min(width, height)
if shortest > MAX_SHORT_SIDE:
scale = MAX_SHORT_SIDE / shortest
width = max(1, _round_half_up(width * scale))
height = max(1, _round_half_up(height * scale))
return width, height
def image_tokens(
width: int, height: int, detail: DetailLevel = "auto"
) -> TokenBreakdown:
"""Estimate the token cost of a width x height image at a detail level.
- 'low': fixed LOW_DETAIL_TOKENS, whatever the size.
- 'high': the image is downscaled by preprocess_image, tiled into
TILE_SIZE squares, and each tile costs TILE_TOKENS on top of the base.
- 'auto': low when both dimensions are <= AUTO_LOW_MAX, else high.
Raises ValueError for non-positive/non-integer dimensions or an unknown
detail level.
"""
_reject_non_dimension("width", width)
_reject_non_dimension("height", height)
if detail in ("low", "high"):
resolved: Literal["low", "high"] = detail
elif detail == "auto":
resolved = "low" if width <= AUTO_LOW_MAX and height <= AUTO_LOW_MAX else "high"
else:
raise ValueError(f"Unknown detail level: {detail!r}")
if resolved == "low":
return TokenBreakdown(
detail="low",
scaled_width=width,
scaled_height=height,
tiles_x=1,
tiles_y=1,
tiles=1,
base=LOW_DETAIL_TOKENS,
detail_tokens=0,
total=LOW_DETAIL_TOKENS,
)
scaled_width, scaled_height = preprocess_image(width, height)
tiles_x = math.ceil(scaled_width / TILE_SIZE)
tiles_y = math.ceil(scaled_height / TILE_SIZE)
tiles = tiles_x * tiles_y
detail_tokens = tiles * TILE_TOKENS
return TokenBreakdown(
detail="high",
scaled_width=scaled_width,
scaled_height=scaled_height,
tiles_x=tiles_x,
tiles_y=tiles_y,
tiles=tiles,
base=LOW_DETAIL_TOKENS,
detail_tokens=detail_tokens,
total=LOW_DETAIL_TOKENS + detail_tokens,
)
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →