Image Token Calculator — Kotlin source
Estimate the vision token cost of an image before sending it to an LLM - low/high/auto detail modes, the 512px tile math, the 2048/768 downscaling steps, and a full base + tiles + detail breakdown. Runs entirely in your browser.
This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Image Token Calculator — estimate the vision token cost of an image using
// OpenAI-style tile math.
//
// Language: Kotlin (1.9, standard library only)
// Source: CosmoDev polyglot showcase port of the Image Token Calculator
// tool, ported from src/lib/imageTokenCalculator.ts (the canonical
// TypeScript implementation).
// Live at: https://dev.cosmolabs.org/tools/image-token-calculator
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; invalid input throws IllegalArgumentException
// via require() (as the TS reference throws).
// - Functionally equivalent to the TS reference: same inputs -> same outputs.
// - Self-contained: the Kotlin stdlib only (no external dependencies).
//
// Rounding note: the TS reference uses Math.round (half up); shrink spells it
// as floor(x + 0.5) for exact parity.
package org.cosmolabs.cosmodev.polyglot
import kotlin.math.floor
import kotlin.math.max
/** Requested detail mode of an image ([DetailLevel.AUTO] mirrors the TS default). */
enum class DetailLevel { LOW, HIGH, AUTO }
/** A width x height pair returned by [ImageTokenCalculator.preprocessImage]. */
data class Dimensions(val width: Int, val height: Int)
/** Mirrors the TokenBreakdown interface in the TS lib. */
data class TokenBreakdown(
/** Detail level actually applied ("low" or "high"; AUTO resolves to one). */
val detail: String,
/** Dimensions after the high-detail downscaling pipeline (identity for low). */
val scaledWidth: Int,
val scaledHeight: Int,
/** 512 px tiles along each axis (both 1 in low detail). */
val tilesX: Int,
val tilesY: Int,
/** Total 512 px tiles used (tilesX * tilesY). */
val tiles: Int,
/** Fixed base cost of the low-resolution view, in tokens. */
val base: Int,
/** Extra tokens for the high-resolution tile views (0 in low detail). */
val detailTokens: Int,
/** Total estimated tokens: base + detailTokens. */
val total: Int,
)
/** Image Token Calculator — estimate the vision token cost of an image using
* OpenAI-style tile math. */
object ImageTokenCalculator {
/** Fixed token cost of the low-resolution image view. */
const val LOW_DETAIL_TOKENS: Int = 85
/** Token cost of one high-resolution 512 px tile. */
const val TILE_TOKENS: Int = 170
/** Images are first scaled to fit inside this square. */
const val MAX_SIDE: Int = 2048
/** Then the shortest side is capped at this length. */
const val MAX_SHORT_SIDE: Int = 768
/** Tile edge length in pixels. */
const val TILE_SIZE: Int = 512
/** Both dimensions at or under this -> [DetailLevel.AUTO] stays low detail. */
const val AUTO_LOW_MAX: Int = 512
/** JS Math.round parity, floored at 1 px: half up, never zero. */
private fun shrink(side: Int, scale: Double): Int =
max(1, floor(side * scale + 0.5).toInt())
/** ceil(n / d) for positive integers, without floating point. */
private fun ceilDiv(n: Int, d: Int): Int = (n + d - 1) / d
/**
* Scale (width, height) per the vision preprocessing pipeline:
* 1. fit inside a [MAX_SIDE] x [MAX_SIDE] square (longest side capped), then
* 2. cap the shortest side at [MAX_SHORT_SIDE].
*
* Aspect ratio is preserved; each step is skipped when already satisfied.
*/
fun preprocessImage(width: Int, height: Int): Dimensions {
var w = width
var h = height
val longest = maxOf(w, h)
if (longest > MAX_SIDE) {
val scale = MAX_SIDE.toDouble() / longest
w = shrink(w, scale)
h = shrink(h, scale)
}
val shortest = minOf(w, h)
if (shortest > MAX_SHORT_SIDE) {
val scale = MAX_SHORT_SIDE.toDouble() / shortest
w = shrink(w, scale)
h = shrink(h, scale)
}
return Dimensions(w, h)
}
/**
* Estimate the token cost of a width x height image at the given detail
* level.
*
* - [DetailLevel.LOW]: fixed [LOW_DETAIL_TOKENS], whatever the size.
* - [DetailLevel.HIGH]: the image is downscaled by [preprocessImage],
* tiled into [TILE_SIZE] squares, and each tile costs [TILE_TOKENS] on
* top of the base.
* - [DetailLevel.AUTO]: low when both dimensions are <= [AUTO_LOW_MAX],
* otherwise high.
*
* @throws IllegalArgumentException for non-positive dimensions or an
* unknown detail level
*/
fun imageTokens(width: Int, height: Int, detail: DetailLevel): TokenBreakdown {
require(width > 0 && height > 0) {
"Width and height must be greater than zero"
}
val resolvedHigh = when (detail) {
DetailLevel.LOW -> false
DetailLevel.HIGH -> true
DetailLevel.AUTO -> width > AUTO_LOW_MAX || height > AUTO_LOW_MAX
}
if (!resolvedHigh) {
return TokenBreakdown(
detail = "low",
scaledWidth = width,
scaledHeight = height,
tilesX = 1,
tilesY = 1,
tiles = 1,
base = LOW_DETAIL_TOKENS,
detailTokens = 0,
total = LOW_DETAIL_TOKENS,
)
}
val scaled = preprocessImage(width, height)
val tilesX = ceilDiv(scaled.width, TILE_SIZE)
val tilesY = ceilDiv(scaled.height, TILE_SIZE)
val tiles = tilesX * tilesY
val detailTokens = tiles * TILE_TOKENS
return TokenBreakdown(
detail = "high",
scaledWidth = scaled.width,
scaledHeight = scaled.height,
tilesX = tilesX,
tilesY = tilesY,
tiles = tiles,
base = LOW_DETAIL_TOKENS,
detailTokens = detailTokens,
total = LOW_DETAIL_TOKENS + detailTokens,
)
}
/**
* Parse a detail-level string ("low" | "high" | "auto") — the bridge from
* the TS string union to [DetailLevel].
*
* @throws IllegalArgumentException if the string is not a known level
*/
fun parseDetail(detail: String): DetailLevel = when (detail) {
"low" -> DetailLevel.LOW
"high" -> DetailLevel.HIGH
"auto" -> DetailLevel.AUTO
else -> throw IllegalArgumentException("Unknown detail level: $detail")
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →