Image Token Calculator — Zig source
Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! Image Token Calculator — estimate the vision token cost of an image using
//! OpenAI-style tile math.
//!
//! Language: Zig (0.13, standard library only)
//! Source: CosmoDev polyglot showcase port of the Image Token Calculator
//! tool, ported from src/lib/imageTokenCalculator.ts (the canonical
//! TypeScript implementation).
//! Live at: https://dev.cosmolabs.org/tools/image-token-calculator
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; invalid input is an error union value, never a
//! panic.
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no third-party packages).
//!
//! Rounding note: the TS reference uses Math.round (half up); shrink spells it
//! as @floor(x + 0.5) for exact parity.
const std = @import("std");
/// Fixed token cost of the low-resolution image view.
pub const low_detail_tokens: u32 = 85;
/// Token cost of one high-resolution 512 px tile.
pub const tile_tokens: u32 = 170;
/// Images are first scaled to fit inside this square.
pub const max_side: u32 = 2048;
/// Then the shortest side is capped at this length.
pub const max_short_side: u32 = 768;
/// Tile edge length in pixels.
pub const tile_size: u32 = 512;
/// Both dimensions at or under this -> `auto` stays low detail.
pub const auto_low_max: u32 = 512;
/// Requested detail mode of an image (`auto` mirrors the TS default; it is a
/// Zig keyword, hence the `@"auto"` spelling).
pub const DetailLevel = enum {
low,
high,
@"auto",
};
/// Why an estimate was rejected (the TS reference throws for both).
pub const ImageTokenError = error{
/// Width/height must be integers greater than zero.
InvalidDimensions,
/// The detail level string was not "low", "high", or "auto".
UnknownDetail,
};
/// A width x height pair returned by `preprocessImage`.
pub const Dimensions = struct {
width: u32,
height: u32,
};
/// Mirrors the TokenBreakdown interface in the TS lib.
pub const TokenBreakdown = struct {
/// Detail level actually applied ("low" or "high").
detail: []const u8,
/// Dimensions after the high-detail downscaling pipeline (identity for low).
scaled_width: u32,
scaled_height: u32,
/// 512 px tiles along each axis (both 1 in low detail).
tiles_x: u32,
tiles_y: u32,
/// Total 512 px tiles used (tiles_x * tiles_y).
tiles: u32,
/// Fixed base cost of the low-resolution view, in tokens.
base: u32,
/// Extra tokens for the high-resolution tile views (0 in low detail).
detail_tokens: u32,
/// Total estimated tokens: base + detail_tokens.
total: u32,
};
/// JS Math.round parity, floored at 1 px: half up, never zero.
fn shrink(side: u32, scale: f64) u32 {
const v = @floor(@as(f64, @floatFromInt(side)) * scale + 0.5);
if (v < 1.0) return 1;
return @intFromFloat(v);
}
/// ceil(n / d) for positive integers, without floating point.
fn ceilDiv(n: u32, d: u32) u32 {
return (n + d - 1) / d;
}
/// Scale (width, height) per the vision preprocessing pipeline:
/// 1. fit inside a `max_side` x `max_side` square (longest side capped), then
/// 2. cap the shortest side at `max_short_side`.
///
/// Aspect ratio is preserved; each step is skipped when already satisfied.
pub fn preprocessImage(width: u32, height: u32) Dimensions {
var w = width;
var h = height;
const longest = @max(w, h);
if (longest > max_side) {
const scale = @as(f64, @floatFromInt(max_side)) / @as(f64, @floatFromInt(longest));
w = shrink(w, scale);
h = shrink(h, scale);
}
const shortest = @min(w, h);
if (shortest > max_short_side) {
const scale = @as(f64, @floatFromInt(max_short_side)) / @as(f64, @floatFromInt(shortest));
w = shrink(w, scale);
h = shrink(h, scale);
}
return .{ .width = w, .height = h };
}
/// Estimate the token cost of a `width` x `height` image at the given detail
/// level.
///
/// - `.low`: fixed `low_detail_tokens`, whatever the size.
/// - `.high`: the image is downscaled by `preprocessImage`, tiled into
/// `tile_size` squares, and each tile costs `tile_tokens` on top of the
/// base.
/// - `.@"auto"`: low when both dimensions are <= `auto_low_max`, otherwise
/// high.
pub fn imageTokens(width: u32, height: u32, detail: DetailLevel) ImageTokenError!TokenBreakdown {
if (width == 0 or height == 0) return error.InvalidDimensions;
const resolved_high = switch (detail) {
.low => false,
.high => true,
.@"auto" => width > auto_low_max or height > auto_low_max,
};
if (!resolved_high) {
return .{
.detail = "low",
.scaled_width = width,
.scaled_height = height,
.tiles_x = 1,
.tiles_y = 1,
.tiles = 1,
.base = low_detail_tokens,
.detail_tokens = 0,
.total = low_detail_tokens,
};
}
const scaled = preprocessImage(width, height);
const tiles_x = ceilDiv(scaled.width, tile_size);
const tiles_y = ceilDiv(scaled.height, tile_size);
const tiles = tiles_x * tiles_y;
const detail_tokens = tiles * tile_tokens;
return .{
.detail = "high",
.scaled_width = scaled.width,
.scaled_height = scaled.height,
.tiles_x = tiles_x,
.tiles_y = tiles_y,
.tiles = tiles,
.base = low_detail_tokens,
.detail_tokens = detail_tokens,
.total = low_detail_tokens + detail_tokens,
};
}
/// Parse a detail-level string ("low" | "high" | "auto") — the bridge from
/// the TS string union to `DetailLevel`.
pub fn parseDetail(detail: []const u8) ImageTokenError!DetailLevel {
if (std.mem.eql(u8, detail, "low")) return .low;
if (std.mem.eql(u8, detail, "high")) return .high;
if (std.mem.eql(u8, detail, "auto")) return .@"auto";
return error.UnknownDetail;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →