Image Token Calculator — Swift source
Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.
This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Image Token Calculator — estimate the vision token cost of an image using
// OpenAI-style tile math.
//
// Language: Swift (5.9, standard library only)
// Source: CosmoDev polyglot showcase port of the Image Token Calculator
// tool, ported from src/lib/imageTokenCalculator.ts (the canonical
// TypeScript implementation).
// Live at: https://dev.cosmolabs.org/tools/image-token-calculator
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; invalid input is a thrown error, never a trap.
// - Functionally equivalent to the TS reference: same inputs -> same outputs.
// - Self-contained: the Swift standard library only (no packages).
//
// Rounding note: the TS reference uses Math.round (half up); shrink spells it
// as floor(x + 0.5) for exact parity.
/// Requested detail mode of an image (`auto` mirrors the TS default).
public enum DetailLevel: String {
case low
case high
case auto
}
/// A width x height pair returned by `ImageTokenCalculator.preprocessImage`.
public struct Dimensions: Equatable, Sendable {
public let width: Int
public let height: Int
public init(width: Int, height: Int) {
self.width = width
self.height = height
}
}
/// Why an estimate was rejected: non-positive dimensions or an unknown detail
/// level (the TS reference throws for both).
public enum ImageTokenError: Error, Equatable, Sendable {
/// Width/height must be integers greater than zero.
case invalidDimensions
/// The detail level string was not "low", "high", or "auto".
case unknownDetail(String)
}
/// Mirrors the `TokenBreakdown` interface in the TS lib.
public struct TokenBreakdown: Equatable, Sendable {
/// Detail level actually applied ("low" or "high").
public let detail: String
/// Dimensions after the high-detail downscaling pipeline (identity for low).
public let scaledWidth: Int
public let scaledHeight: Int
/// 512 px tiles along each axis (both 1 in low detail).
public let tilesX: Int
public let tilesY: Int
/// Total 512 px tiles used (`tilesX * tilesY`).
public let tiles: Int
/// Fixed base cost of the low-resolution view, in tokens.
public let base: Int
/// Extra tokens for the high-resolution tile views (0 in low detail).
public let detailTokens: Int
/// Total estimated tokens: `base + detailTokens`.
public let total: Int
public init(
detail: String,
scaledWidth: Int,
scaledHeight: Int,
tilesX: Int,
tilesY: Int,
tiles: Int,
base: Int,
detailTokens: Int,
total: Int
) {
self.detail = detail
self.scaledWidth = scaledWidth
self.scaledHeight = scaledHeight
self.tilesX = tilesX
self.tilesY = tilesY
self.tiles = tiles
self.base = base
self.detailTokens = detailTokens
self.total = total
}
}
/// Image Token Calculator — estimate the vision token cost of an image using
/// OpenAI-style tile math.
public enum ImageTokenCalculator {
/// Fixed token cost of the low-resolution image view.
public static let lowDetailTokens = 85
/// Token cost of one high-resolution 512 px tile.
public static let tileTokens = 170
/// Images are first scaled to fit inside this square.
public static let maxSide = 2048
/// Then the shortest side is capped at this length.
public static let maxShortSide = 768
/// Tile edge length in pixels.
public static let tileSize = 512
/// Both dimensions at or under this -> `.auto` stays low detail.
public static let autoLowMax = 512
/// JS `Math.round` parity, floored at 1 px: half up, never zero.
private static func shrink(_ side: Int, scale: Double) -> Int {
let v = (Double(side) * scale + 0.5).rounded(.down)
return v < 1 ? 1 : Int(v)
}
/// ceil(n / d) for positive integers, without floating point.
private static func ceilDiv(_ n: Int, _ d: Int) -> Int {
(n + d - 1) / d
}
/// Scale `(width, height)` per the vision preprocessing pipeline:
/// 1. fit inside a `maxSide` x `maxSide` square (longest side capped), then
/// 2. cap the shortest side at `maxShortSide`.
///
/// Aspect ratio is preserved; each step is skipped when already satisfied.
public static func preprocessImage(width: Int, height: Int) -> Dimensions {
var w = width
var h = height
let longest = max(w, h)
if longest > maxSide {
let scale = Double(maxSide) / Double(longest)
w = shrink(w, scale: scale)
h = shrink(h, scale: scale)
}
let shortest = min(w, h)
if shortest > maxShortSide {
let scale = Double(maxShortSide) / Double(shortest)
w = shrink(w, scale: scale)
h = shrink(h, scale: scale)
}
return Dimensions(width: w, height: h)
}
/// Estimate the token cost of a `width` x `height` image at the given
/// detail level.
///
/// - `.low`: fixed `lowDetailTokens`, whatever the size.
/// - `.high`: the image is downscaled by `preprocessImage(width:height:)`,
/// tiled into `tileSize` squares, and each tile costs `tileTokens` on
/// top of the base.
/// - `.auto`: low when both dimensions are <= `autoLowMax`, otherwise high.
public static func imageTokens(
width: Int,
height: Int,
detail: DetailLevel
) throws -> TokenBreakdown {
guard width > 0, height > 0 else {
throw ImageTokenError.invalidDimensions
}
let resolvedHigh: Bool
switch detail {
case .low:
resolvedHigh = false
case .high:
resolvedHigh = true
case .auto:
resolvedHigh = width > autoLowMax || height > autoLowMax
}
if !resolvedHigh {
return TokenBreakdown(
detail: "low",
scaledWidth: width,
scaledHeight: height,
tilesX: 1,
tilesY: 1,
tiles: 1,
base: lowDetailTokens,
detailTokens: 0,
total: lowDetailTokens
)
}
let scaled = preprocessImage(width: width, height: height)
let tilesX = ceilDiv(scaled.width, tileSize)
let tilesY = ceilDiv(scaled.height, tileSize)
let tiles = tilesX * tilesY
let detailTokens = tiles * tileTokens
return TokenBreakdown(
detail: "high",
scaledWidth: scaled.width,
scaledHeight: scaled.height,
tilesX: tilesX,
tilesY: tilesY,
tiles: tiles,
base: lowDetailTokens,
detailTokens: detailTokens,
total: lowDetailTokens + detailTokens
)
}
/// Parse a detail-level string ("low" | "high" | "auto") — the bridge from
/// the TS string union to `DetailLevel`.
public static func parseDetail(_ detail: String) throws -> DetailLevel {
guard let level = DetailLevel(rawValue: detail) else {
throw ImageTokenError.unknownDetail(detail)
}
return level
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →