Skip to content

Image Token Calculator — Swift source

Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.

This is the Swift implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Image Token Calculator — estimate the vision token cost of an image using
// OpenAI-style tile math.
//
// Language: Swift (5.9, standard library only)
// Source:   CosmoDev polyglot showcase port of the Image Token Calculator
//           tool, ported from src/lib/imageTokenCalculator.ts (the canonical
//           TypeScript implementation).
// Live at:  https://dev.cosmolabs.org/tools/image-token-calculator
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; invalid input is a thrown error, never a trap.
//   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//   - Self-contained: the Swift standard library only (no packages).
//
// Rounding note: the TS reference uses Math.round (half up); shrink spells it
// as floor(x + 0.5) for exact parity.

/// Requested detail mode of an image (`auto` mirrors the TS default).
public enum DetailLevel: String {
    case low
    case high
    case auto
}

/// A width x height pair returned by `ImageTokenCalculator.preprocessImage`.
public struct Dimensions: Equatable, Sendable {
    public let width: Int
    public let height: Int

    public init(width: Int, height: Int) {
        self.width = width
        self.height = height
    }
}

/// Why an estimate was rejected: non-positive dimensions or an unknown detail
/// level (the TS reference throws for both).
public enum ImageTokenError: Error, Equatable, Sendable {
    /// Width/height must be integers greater than zero.
    case invalidDimensions
    /// The detail level string was not "low", "high", or "auto".
    case unknownDetail(String)
}

/// Mirrors the `TokenBreakdown` interface in the TS lib.
public struct TokenBreakdown: Equatable, Sendable {
    /// Detail level actually applied ("low" or "high").
    public let detail: String
    /// Dimensions after the high-detail downscaling pipeline (identity for low).
    public let scaledWidth: Int
    public let scaledHeight: Int
    /// 512 px tiles along each axis (both 1 in low detail).
    public let tilesX: Int
    public let tilesY: Int
    /// Total 512 px tiles used (`tilesX * tilesY`).
    public let tiles: Int
    /// Fixed base cost of the low-resolution view, in tokens.
    public let base: Int
    /// Extra tokens for the high-resolution tile views (0 in low detail).
    public let detailTokens: Int
    /// Total estimated tokens: `base + detailTokens`.
    public let total: Int

    public init(
        detail: String,
        scaledWidth: Int,
        scaledHeight: Int,
        tilesX: Int,
        tilesY: Int,
        tiles: Int,
        base: Int,
        detailTokens: Int,
        total: Int
    ) {
        self.detail = detail
        self.scaledWidth = scaledWidth
        self.scaledHeight = scaledHeight
        self.tilesX = tilesX
        self.tilesY = tilesY
        self.tiles = tiles
        self.base = base
        self.detailTokens = detailTokens
        self.total = total
    }
}

/// Image Token Calculator — estimate the vision token cost of an image using
/// OpenAI-style tile math.
public enum ImageTokenCalculator {
    /// Fixed token cost of the low-resolution image view.
    public static let lowDetailTokens = 85
    /// Token cost of one high-resolution 512 px tile.
    public static let tileTokens = 170
    /// Images are first scaled to fit inside this square.
    public static let maxSide = 2048
    /// Then the shortest side is capped at this length.
    public static let maxShortSide = 768
    /// Tile edge length in pixels.
    public static let tileSize = 512
    /// Both dimensions at or under this -> `.auto` stays low detail.
    public static let autoLowMax = 512

    /// JS `Math.round` parity, floored at 1 px: half up, never zero.
    private static func shrink(_ side: Int, scale: Double) -> Int {
        let v = (Double(side) * scale + 0.5).rounded(.down)
        return v < 1 ? 1 : Int(v)
    }

    /// ceil(n / d) for positive integers, without floating point.
    private static func ceilDiv(_ n: Int, _ d: Int) -> Int {
        (n + d - 1) / d
    }

    /// Scale `(width, height)` per the vision preprocessing pipeline:
    /// 1. fit inside a `maxSide` x `maxSide` square (longest side capped), then
    /// 2. cap the shortest side at `maxShortSide`.
    ///
    /// Aspect ratio is preserved; each step is skipped when already satisfied.
    public static func preprocessImage(width: Int, height: Int) -> Dimensions {
        var w = width
        var h = height
        let longest = max(w, h)
        if longest > maxSide {
            let scale = Double(maxSide) / Double(longest)
            w = shrink(w, scale: scale)
            h = shrink(h, scale: scale)
        }
        let shortest = min(w, h)
        if shortest > maxShortSide {
            let scale = Double(maxShortSide) / Double(shortest)
            w = shrink(w, scale: scale)
            h = shrink(h, scale: scale)
        }
        return Dimensions(width: w, height: h)
    }

    /// Estimate the token cost of a `width` x `height` image at the given
    /// detail level.
    ///
    /// - `.low`: fixed `lowDetailTokens`, whatever the size.
    /// - `.high`: the image is downscaled by `preprocessImage(width:height:)`,
    ///   tiled into `tileSize` squares, and each tile costs `tileTokens` on
    ///   top of the base.
    /// - `.auto`: low when both dimensions are <= `autoLowMax`, otherwise high.
    public static func imageTokens(
        width: Int,
        height: Int,
        detail: DetailLevel
    ) throws -> TokenBreakdown {
        guard width > 0, height > 0 else {
            throw ImageTokenError.invalidDimensions
        }

        let resolvedHigh: Bool
        switch detail {
        case .low:
            resolvedHigh = false
        case .high:
            resolvedHigh = true
        case .auto:
            resolvedHigh = width > autoLowMax || height > autoLowMax
        }

        if !resolvedHigh {
            return TokenBreakdown(
                detail: "low",
                scaledWidth: width,
                scaledHeight: height,
                tilesX: 1,
                tilesY: 1,
                tiles: 1,
                base: lowDetailTokens,
                detailTokens: 0,
                total: lowDetailTokens
            )
        }

        let scaled = preprocessImage(width: width, height: height)
        let tilesX = ceilDiv(scaled.width, tileSize)
        let tilesY = ceilDiv(scaled.height, tileSize)
        let tiles = tilesX * tilesY
        let detailTokens = tiles * tileTokens
        return TokenBreakdown(
            detail: "high",
            scaledWidth: scaled.width,
            scaledHeight: scaled.height,
            tilesX: tilesX,
            tilesY: tilesY,
            tiles: tiles,
            base: lowDetailTokens,
            detailTokens: detailTokens,
            total: lowDetailTokens + detailTokens
        )
    }

    /// Parse a detail-level string ("low" | "high" | "auto") — the bridge from
    /// the TS string union to `DetailLevel`.
    public static func parseDetail(_ detail: String) throws -> DetailLevel {
        guard let level = DetailLevel(rawValue: detail) else {
            throw ImageTokenError.unknownDetail(detail)
        }
        return level
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →