Media Metadata Inspector — Ruby source
Inspect audio, video and image files at the byte level — MP4 boxes, ID3 tags, WAV chunks, GIF headers — parsed by our own readers, entirely in your browser.
This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.
# Media Metadata Inspector — read container metadata straight from the bytes.
#
# Language: Ruby (3.1+, standard library only)
# Source: CosmoDev polyglot showcase port of the media-metadata-inspector
# tool, ported from src/lib/media-meta.ts (the canonical TypeScript
# implementation).
# License: display source — part of CosmoDev's polyglot tool pages.
#
# Pure byte-slicing: no gems, no ffmpeg shelling out. Sniffs the container
# signature (MP4 "ftyp" box, "ID3", RIFF/WAVE, GIF8, WebM EBML magic), then
# walks that container for duration, dimensions, and tags. Deterministic;
# nil only for inputs too short for a signature.
#
# Endianness trap: MP4 and ID3 fields are big-endian, while RIFF/WAVE fields
# and GIF dimensions are little-endian — the readers below are named for the
# byte order they produce, because mixing them up is the classic bug here.
module MediaMeta
# kind: 'mp4' | 'mp3' | 'wav' | 'gif' | 'webm' | 'unknown'.
# Nulls mean "container parsed, field not present".
Meta = Struct.new(:kind, :duration_sec, :width, :height, :title, :artist, :album,
keyword_init: true)
module_function
# ---------------------------------------------------------------- readers --
def ascii(b, start, len)
(start...start + len).map { |i| b.getbyte(i).chr }.join
end
# unpack templates: N = u32 big-endian, V = u32 little-endian, v = u16
# little-endian. Using getbyte/unpack (not String#[]) keeps every read
# encoding-agnostic over the binary input.
def u32be(b, o) = b.byteslice(o, 4).unpack1('N')
def u32le(b, o) = b.byteslice(o, 4).unpack1('V')
def u16le(b, o) = b.byteslice(o, 2).unpack1('v')
# ID3 syncsafe integer: four 7-bit groups, so the high bit of every byte is
# free to keep MPEG audio frame syncs unambiguous.
def syncsafe(b, o)
(b.getbyte(o) << 21) | (b.getbyte(o + 1) << 14) |
(b.getbyte(o + 2) << 7) | b.getbyte(o + 3)
end
# ------------------------------------------------------------- MP4 boxes --
# MP4 box walk: payload ranges [start, end) for every box of a wanted type.
def find_boxes(b, start, end_, wanted)
found = []
pos = start
while pos + 8 <= end_
size = u32be(b, pos)
return found if size == 1 # 64-bit sizes unsupported — stop safely.
size = end_ - pos if size.zero? # size 0: box extends to the end
return found if size < 8 || pos + size > end_
type = ascii(b, pos + 4, 4)
found << [pos + 8, pos + size] if wanted.include?(type)
pos += size
end
found
end
# MP4: moov -> mvhd gives duration; the first trak's tkhd gives dimensions.
def probe_mp4(b)
meta = Meta.new(kind: 'mp4', duration_sec: nil, width: nil, height: nil)
find_boxes(b, 0, b.bytesize, ['moov']).each do |m_start, m_end|
find_boxes(b, m_start, m_end, ['mvhd']).each do |s, e|
version = b.getbyte(s)
if version.zero? && e - s >= 20
timescale = u32be(b, s + 12)
meta.duration_sec = u32be(b, s + 16) / timescale.to_f if timescale.positive?
elsif version == 1 && e - s >= 32
timescale = u32be(b, s + 20)
# 64-bit duration: high and low halves of the u64, each read as u32.
duration = u32be(b, s + 24) * 4_294_967_296 + u32be(b, s + 28)
meta.duration_sec = duration / timescale.to_f if timescale.positive?
end
end
find_boxes(b, m_start, m_end, ['trak']).each do |t_start, t_end|
tkhd = find_boxes(b, t_start, t_end, ['tkhd']).first
next unless tkhd && b.getbyte(tkhd[0]).zero? && tkhd[1] - tkhd[0] >= 84
# tkhd stores width/height as 16.16 fixed point — divide by 65536.
meta.width = u32be(b, tkhd[0] + 76) / 65_536.0
meta.height = u32be(b, tkhd[0] + 80) / 65_536.0
break # first track is enough
end
end
meta
end
# ------------------------------------------------------------ ID3v2 text --
# Decode one ID3 text frame (first byte selects the encoding).
def decode_id3_text(b, start, len)
return +'' if len <= 0
encoding = b.getbyte(start)
text = b.byteslice(start + 1, len - 1)
case encoding
when 3 # UTF-8 (ID3v2.4); stop at the null terminator.
z = text.index("\x00")
(z ? text.byteslice(0, z) : text).force_encoding(Encoding::UTF_8)
when 1, 2
# UTF-16: encoding 1 carries a BOM (sniff endianness); 2 is BOM-less BE.
bom = encoding == 1 && [0xff, 0xfe].include?(text.getbyte(0))
little = bom ? text.getbyte(0) == 0xff : false
codes = []
i = bom ? 2 : 0
while i + 2 <= text.bytesize
code = if little
text.getbyte(i) | (text.getbyte(i + 1) << 8)
else
(text.getbyte(i) << 8) | text.getbyte(i + 1)
end
break if code.zero?
codes << code
i += 2
end
codes.pack('U*')
else
# ISO-8859-1 / ASCII range: appending the byte as a codepoint to a
# UTF-8 String performs the latin1 -> UTF-8 widening for free.
out = +''
text.each_byte do |byte|
break if byte.zero?
out << byte
end
out
end
end
# ID3v2: walk frames, keep TIT2 (title), TPE1 (artist), TALB (album).
def probe_id3(b)
meta = Meta.new(kind: 'mp3', duration_sec: nil, width: nil, height: nil)
version = b.getbyte(3)
tag_size = syncsafe(b, 6)
pos = 10
end_ = [10 + tag_size, b.bytesize].min
while pos + 10 <= end_
id = ascii(b, pos, 4)
break unless id.match?(/\A[A-Z0-9]{4}\z/)
# ID3v2.4 frame sizes are syncsafe; v2.2/v2.3 are plain big-endian.
size = version == 4 ? syncsafe(b, pos + 4) : u32be(b, pos + 4)
break if size.zero?
# Clamp declared size to the tag end: a corrupt frame stays inside the tag.
avail = [end_ - (pos + 10), 0].max
text = decode_id3_text(b, pos + 10, [size, avail].min)
meta.title = text if id == 'TIT2'
meta.artist = text if id == 'TPE1'
meta.album = text if id == 'TALB'
pos += 10 + size
end
meta
end
# ------------------------------------------------------------------- WAV --
# WAV: fmt chunk gives byteRate; data chunk size / byteRate = duration.
def probe_wav(b)
meta = Meta.new(kind: 'wav', duration_sec: nil, width: nil, height: nil)
pos = 12 # past RIFF size + WAVE
byte_rate = 0
while pos + 8 <= b.bytesize
id = ascii(b, pos, 4)
size = u32le(b, pos + 4)
if id == 'fmt ' && pos + 8 + 16 <= b.bytesize
byte_rate = u32le(b, pos + 16) # byteRate sits at fmt payload offset 8
elsif id == 'data' && byte_rate.positive?
meta.duration_sec = size / byte_rate.to_f
break
end
pos += 8 + size + (size % 2) # chunks are word-aligned
end
meta
end
# ----------------------------------------------------------------- probe --
# Sniff the container and parse its metadata. nil when input is under 12
# bytes — too short for any signature.
def probe(bytes)
return nil if bytes.bytesize < 12
b = bytes.b # 8-bit copy so slicing never trips over the encoding
return probe_mp4(b) if ascii(b, 4, 4) == 'ftyp' # MP4 / MOV
return probe_id3(b) if ascii(b, 0, 3) == 'ID3'
if ascii(b, 0, 4) == 'RIFF' && ascii(b, 8, 4) == 'WAVE'
return probe_wav(b)
end
if ascii(b, 0, 4) == 'GIF8'
return Meta.new(kind: 'gif', duration_sec: nil,
width: u16le(b, 6), height: u16le(b, 8))
end
if b.getbyte(0) == 0x1a && b.getbyte(1) == 0x45 && b.getbyte(2) == 0xdf && b.getbyte(3) == 0xa3
return Meta.new(kind: 'webm', duration_sec: nil, width: nil, height: nil) # EBML magic
end
Meta.new(kind: 'unknown', duration_sec: nil, width: nil, height: nil)
end
end
# Usage:
# meta = MediaMeta.probe(File.binread('clip.mp4'))
# puts meta.kind, meta.duration_sec, meta.width, meta.height, meta.title
Also available in 9 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →