Skip to content

Image Token Calculator — Python source

Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Image Token Calculator — estimate the vision token cost of an image using
OpenAI-style tile math.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Image Token Calculator
          tool, ported from src/lib/imageTokenCalculator.ts (the canonical
          TypeScript implementation).
Tool page: https://dev.cosmolabs.org/tools/image-token-calculator
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; raises ValueError on invalid input (as the TS
    reference throws).
  - Functionally equivalent to the TS reference: same inputs -> same outputs.
  - Self-contained: stdlib only (no pip packages).

Rounding note: the TS reference uses Math.round (half up). Python's built-in
round() is banker's rounding, so _round_half_up reproduces the TS behavior.
"""

from __future__ import annotations

import math
from dataclasses import dataclass
from typing import Literal, Tuple, Union

DetailLevel = Literal["low", "high", "auto"]

#: Fixed token cost of the low-resolution image view.
LOW_DETAIL_TOKENS = 85
#: Token cost of one high-resolution 512 px tile.
TILE_TOKENS = 170
#: Images are first scaled to fit inside this square.
MAX_SIDE = 2048
#: Then the shortest side is capped at this length.
MAX_SHORT_SIDE = 768
#: Tile edge length in pixels.
TILE_SIZE = 512
#: Both dimensions at or under this -> 'auto' stays low detail.
AUTO_LOW_MAX = 512


@dataclass(frozen=True)
class TokenBreakdown:
    """Mirrors the TokenBreakdown interface in the TS lib."""

    detail: Literal["low", "high"]
    #: Dimensions after the high-detail downscaling pipeline (identity for low).
    scaled_width: int
    scaled_height: int
    #: 512 px tiles along each axis (both 1 in low detail).
    tiles_x: int
    tiles_y: int
    #: Total 512 px tiles used (tiles_x * tiles_y).
    tiles: int
    #: Fixed base cost of the low-resolution view, in tokens.
    base: int
    #: Extra tokens for the high-resolution tile views (0 in low detail).
    detail_tokens: int
    #: Total estimated tokens: base + detail_tokens.
    total: int


def _round_half_up(x: float) -> int:
    """JS Math.round parity: round half up (values here are always positive)."""
    return math.floor(x + 0.5)


def _reject_non_dimension(name: str, value: object) -> None:
    """Raise ValueError unless value is an int (not bool) greater than zero."""
    if isinstance(value, bool) or not isinstance(value, int):
        raise ValueError("Width and height must be whole pixels")
    if value <= 0:
        raise ValueError("Width and height must be greater than zero")


def preprocess_image(width: int, height: int) -> Tuple[int, int]:
    """Scale (width, height) per the vision preprocessing pipeline:

    1. fit inside a MAX_SIDE x MAX_SIDE square (longest side capped), then
    2. cap the shortest side at MAX_SHORT_SIDE.

    Aspect ratio is preserved; each step is skipped when already satisfied.
    """
    longest = max(width, height)
    if longest > MAX_SIDE:
        scale = MAX_SIDE / longest
        width = max(1, _round_half_up(width * scale))
        height = max(1, _round_half_up(height * scale))
    shortest = min(width, height)
    if shortest > MAX_SHORT_SIDE:
        scale = MAX_SHORT_SIDE / shortest
        width = max(1, _round_half_up(width * scale))
        height = max(1, _round_half_up(height * scale))
    return width, height


def image_tokens(
    width: int, height: int, detail: DetailLevel = "auto"
) -> TokenBreakdown:
    """Estimate the token cost of a width x height image at a detail level.

    - 'low': fixed LOW_DETAIL_TOKENS, whatever the size.
    - 'high': the image is downscaled by preprocess_image, tiled into
      TILE_SIZE squares, and each tile costs TILE_TOKENS on top of the base.
    - 'auto': low when both dimensions are <= AUTO_LOW_MAX, else high.

    Raises ValueError for non-positive/non-integer dimensions or an unknown
    detail level.
    """
    _reject_non_dimension("width", width)
    _reject_non_dimension("height", height)

    if detail in ("low", "high"):
        resolved: Literal["low", "high"] = detail
    elif detail == "auto":
        resolved = "low" if width <= AUTO_LOW_MAX and height <= AUTO_LOW_MAX else "high"
    else:
        raise ValueError(f"Unknown detail level: {detail!r}")

    if resolved == "low":
        return TokenBreakdown(
            detail="low",
            scaled_width=width,
            scaled_height=height,
            tiles_x=1,
            tiles_y=1,
            tiles=1,
            base=LOW_DETAIL_TOKENS,
            detail_tokens=0,
            total=LOW_DETAIL_TOKENS,
        )

    scaled_width, scaled_height = preprocess_image(width, height)
    tiles_x = math.ceil(scaled_width / TILE_SIZE)
    tiles_y = math.ceil(scaled_height / TILE_SIZE)
    tiles = tiles_x * tiles_y
    detail_tokens = tiles * TILE_TOKENS
    return TokenBreakdown(
        detail="high",
        scaled_width=scaled_width,
        scaled_height=scaled_height,
        tiles_x=tiles_x,
        tiles_y=tiles_y,
        tiles=tiles,
        base=LOW_DETAIL_TOKENS,
        detail_tokens=detail_tokens,
        total=LOW_DETAIL_TOKENS + detail_tokens,
    )

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →