EXIF & Metadata Stripper — Ruby source
View and strip GPS, camera, date, and software metadata from photos entirely in your browser. Download a clean copy.
This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.
# EXIF Stripper — EXIF / XMP / IPTC metadata reader and remover for JPEG and
# PNG images.
#
# Language: Ruby (3.x, standard library only — no image gem, no ImageMagick)
# Source: CosmoDev polyglot showcase port of the EXIF Stripper tool, ported
# from src/lib/exif-stripper.ts (the canonical TypeScript
# implementation).
# License: display source — part of CosmoDev's polyglot tool pages.
#
# The binary containers are parsed by hand, exactly as the TS reference does:
# - JPEG: length-prefixed markers walked from the SOI. APP1 (0xFFE1) carries
# either "Exif\0\0" + a TIFF IFD tree (byte order taken from the TIFF BOM)
# or an XMP packet; APP13 (0xFFED) carries IPTC ("Photoshop 3.0\0").
# - PNG: chunk walk. tEXt / iTXt text chunks and the eXIf chunk (same TIFF
# structure) are the metadata carriers.
#
# Stripping rebuilds the file *without* those segments and copies every other
# byte verbatim, so JPEG stripping is lossless — the entropy-coded scan data is
# never re-encoded.
#
# Input and output are binary Ruby Strings (the equivalent of the TS
# ArrayBuffer). Every read is bounds-checked: a truncated or corrupted file
# yields whatever was parsed so far instead of raising.
module ExifStripper
# Everything the parser can surface, plus the rebuilt metadata-free image.
Metadata = Struct.new(
:camera_make,
:camera_model,
:software,
:date_time,
:gps_latitude, # decimal degrees, negative for S / W
:gps_longitude,
:image_width,
:image_height,
:orientation, # EXIF orientation 1-8
:exposure_time, # seconds (0.004 = 1/250 s)
:f_number,
:iso,
:focal_length, # millimetres
:stripped, # binary String: input with all metadata removed
keyword_init: true
)
# --- format signatures ------------------------------------------------------
PNG_SIGNATURE = "\x89PNG\r\n\x1a\n".b.freeze
EXIF_SIGNATURE = "Exif\0\0".b.freeze
IPTC_SIGNATURE = "Photoshop 3.0\0".b.freeze
XMP_SIGNATURES = [
'http://ns.adobe.com/xap/1.0/'.b.freeze,
'http://ns.adobe.com/xmp/extension/'.b.freeze
].freeze
# --- TIFF / EXIF tag numbers ------------------------------------------------
# EXIF field type -> byte size of one component.
TYPE_SIZE = { 1 => 1, 2 => 1, 3 => 2, 4 => 4, 5 => 8, 7 => 1, 9 => 4, 10 => 8 }.freeze
TYPE_ASCII = 2
TYPE_SHORT = 3
TYPE_LONG = 4
TYPE_RATIONAL = 5
# IFD0
TAG_IMAGE_WIDTH = 0x0100
TAG_IMAGE_HEIGHT = 0x0101
TAG_MAKE = 0x010f
TAG_MODEL = 0x0110
TAG_ORIENTATION = 0x0112
TAG_SOFTWARE = 0x0131
TAG_DATE_TIME = 0x0132
TAG_EXIF_IFD_POINTER = 0x8769
TAG_GPS_IFD_POINTER = 0x8825
# Exif SubIFD
TAG_EXPOSURE_TIME = 0x829a
TAG_F_NUMBER = 0x829d
TAG_ISO = 0x8827
TAG_DATE_TIME_ORIGINAL = 0x9003
TAG_FOCAL_LENGTH = 0x920a
# GPS IFD
TAG_GPS_LAT_REF = 0x0001
TAG_GPS_LAT = 0x0002
TAG_GPS_LON_REF = 0x0003
TAG_GPS_LON = 0x0004
# One JPEG marker segment. +finish+ is exclusive; +payload_start+ is the first
# byte after the 2-byte length field.
Segment = Struct.new(:marker, :start, :finish, :payload_start, keyword_init: true)
# A TIFF block: the raw bytes plus the byte order taken from its BOM.
Tiff = Struct.new(:bytes, :little, keyword_init: true)
class << self
# --- public API -----------------------------------------------------------
# Rebuild +data+ without any metadata segment. Image data is copied
# byte-for-byte, so JPEG stripping is lossless.
def strip_metadata(data)
bytes = binary(data)
raise ArgumentError, 'File is too small to be a valid image.' if bytes.bytesize < 8
return strip_jpeg(bytes) if jpeg?(bytes)
return strip_png(bytes) if png?(bytes)
raise ArgumentError, 'Unsupported format: only JPEG and PNG images are supported.'
end
# Read every supported metadata field out of +data+ and attach a stripped
# copy of the image. Unknown or malformed fields are simply left nil.
def parse_metadata(data)
bytes = binary(data)
raise ArgumentError, 'File is too small to be a valid image.' if bytes.bytesize < 8
unless jpeg?(bytes) || png?(bytes)
raise ArgumentError, 'Unsupported format: only JPEG and PNG images are supported.'
end
meta = Metadata.new(stripped: strip_metadata(bytes))
jpeg?(bytes) ? parse_jpeg(bytes, meta) : parse_png(bytes, meta)
meta
end
# Degrees / minutes / seconds plus a hemisphere reference to signed decimal
# degrees. Missing components count as zero.
def format_gps_coordinate(degrees, ref)
deg, min, sec = Array(degrees).values_at(0, 1, 2).map { |v| v || 0 }
decimal = deg + (min / 60.0) + (sec / 3600.0)
%w[S W].include?(ref) ? -decimal : decimal
end
private
# --- byte helpers ---------------------------------------------------------
def binary(data)
data.to_s.dup.force_encoding(Encoding::BINARY)
end
def u16(bytes, off, little = false)
chunk = bytes.byteslice(off, 2)
return nil if chunk.nil? || chunk.bytesize < 2
chunk.unpack1(little ? 'v' : 'n')
end
def u32(bytes, off, little = false)
chunk = bytes.byteslice(off, 4)
return nil if chunk.nil? || chunk.bytesize < 4
chunk.unpack1(little ? 'V' : 'N')
end
def signature_at?(bytes, off, signature)
return false if off.negative? || off + signature.bytesize > bytes.bytesize
bytes.byteslice(off, signature.bytesize) == signature
end
def jpeg?(bytes)
bytes.bytesize >= 2 && bytes.getbyte(0) == 0xff && bytes.getbyte(1) == 0xd8
end
def png?(bytes)
bytes.bytesize >= 8 && bytes.byteslice(0, 8) == PNG_SIGNATURE
end
# --- JPEG segment walk ----------------------------------------------------
# Walk length-prefixed markers from just after the SOI up to and including
# the SOS marker, whose entropy-coded data runs to the EOI. Returns nil when
# the structure is corrupted mid-walk.
def collect_jpeg_segments(bytes)
segments = []
size = bytes.bytesize
off = 2
while off + 2 <= size
return nil unless bytes.getbyte(off) == 0xff
if bytes.getbyte(off + 1) == 0xff
off += 1 # padding fill byte before the real marker
next
end
marker = bytes.getbyte(off + 1)
# Standalone markers carry no length field (TEM, RSTn, SOI, EOI).
if marker == 0x01 || (0xd0..0xd9).cover?(marker)
segments << Segment.new(marker: marker, start: off, finish: off + 2, payload_start: off + 2)
break if marker == 0xd9
off += 2
next
end
return nil if off + 4 > size
seg_size = u16(bytes, off + 2) # includes the two length bytes themselves
return nil if seg_size.nil? || seg_size < 2
finish = off + 2 + seg_size
return nil if finish > size
segments << Segment.new(marker: marker, start: off, finish: finish, payload_start: off + 4)
break if marker == 0xda # SOS — entropy data follows, no more segments
off = finish
end
segments
end
def jpeg_metadata_segment?(bytes, seg)
if seg.marker == 0xe1
return signature_at?(bytes, seg.payload_start, EXIF_SIGNATURE) ||
XMP_SIGNATURES.any? { |sig| signature_at?(bytes, seg.payload_start, sig) }
end
# APP13 is IPTC ("Photoshop 3.0\0" + 8BIM records) — always metadata.
seg.marker == 0xed && signature_at?(bytes, seg.payload_start, IPTC_SIGNATURE)
end
# --- TIFF / IFD parsing ---------------------------------------------------
# Where an entry's value lives: inline in the 4-byte value field when it
# fits, else at the recorded offset. Returns nil when out of bounds.
def value_offset(tiff, entry, type, count)
size = (TYPE_SIZE[type] || 1) * count
off = size <= 4 ? entry + 8 : u32(tiff.bytes, entry + 8, tiff.little)
return nil if off.nil? || off.negative? || off + size > tiff.bytes.bytesize
off
end
def read_ascii(tiff, entry, type, count)
return nil if type != TYPE_ASCII || count < 1
off = value_offset(tiff, entry, type, count)
return nil if off.nil?
text = tiff.bytes.byteslice(off, count).to_s
.force_encoding(Encoding::UTF_8)
.scrub('')
.sub(/\0+\z/, '')
.strip
text.empty? ? nil : text
end
def read_number(tiff, entry, type, count)
return nil if count < 1
off = value_offset(tiff, entry, type, count)
return nil if off.nil?
case type
when TYPE_SHORT then u16(tiff.bytes, off, tiff.little)
when TYPE_LONG then u32(tiff.bytes, off, tiff.little)
end
end
def read_rationals(tiff, entry, type, count)
return nil if type != TYPE_RATIONAL || count < 1
off = value_offset(tiff, entry, type, count)
return nil if off.nil?
Array.new(count) do |i|
num = u32(tiff.bytes, off + (i * 8), tiff.little)
den = u32(tiff.bytes, off + (i * 8) + 4, tiff.little)
den.nil? || den.zero? ? 0.0 : num / den.to_f
end
end
def parse_tiff(tiff_bytes, meta)
return if tiff_bytes.nil? || tiff_bytes.bytesize < 8
case tiff_bytes.byteslice(0, 2)
when 'II'.b then little = true
when 'MM'.b then little = false
else return
end
tiff = Tiff.new(bytes: tiff_bytes, little: little)
return unless u16(tiff_bytes, 2, little) == 42 # TIFF magic
read_ifd(tiff, u32(tiff_bytes, 4, little), meta)
end
# Iterate the 12-byte entries of one IFD, yielding tag / type / count / entry
# offset. Returns without yielding when the directory is out of bounds.
def each_ifd_entry(tiff, offset)
return if offset.nil? || offset < 8 || offset + 2 > tiff.bytes.bytesize
count = u16(tiff.bytes, offset, tiff.little)
return if count.nil?
return if offset + 2 + (count * 12) > tiff.bytes.bytesize
count.times do |i|
entry = offset + 2 + (i * 12)
yield(u16(tiff.bytes, entry, tiff.little),
u16(tiff.bytes, entry + 2, tiff.little),
u32(tiff.bytes, entry + 4, tiff.little),
entry)
end
end
def read_ifd(tiff, offset, meta)
each_ifd_entry(tiff, offset) do |tag, type, count, entry|
case tag
when TAG_IMAGE_WIDTH then meta.image_width ||= read_number(tiff, entry, type, count)
when TAG_IMAGE_HEIGHT then meta.image_height ||= read_number(tiff, entry, type, count)
when TAG_MAKE then meta.camera_make ||= read_ascii(tiff, entry, type, count)
when TAG_MODEL then meta.camera_model ||= read_ascii(tiff, entry, type, count)
when TAG_ORIENTATION then meta.orientation ||= read_number(tiff, entry, type, count)
when TAG_SOFTWARE then meta.software ||= read_ascii(tiff, entry, type, count)
when TAG_DATE_TIME then meta.date_time ||= read_ascii(tiff, entry, type, count)
when TAG_EXIF_IFD_POINTER
sub = read_number(tiff, entry, TYPE_LONG, 1)
read_exif_ifd(tiff, sub, meta) unless sub.nil?
when TAG_GPS_IFD_POINTER
gps = read_number(tiff, entry, TYPE_LONG, 1)
read_gps_ifd(tiff, gps, meta) unless gps.nil?
end
end
end
def read_exif_ifd(tiff, offset, meta)
each_ifd_entry(tiff, offset) do |tag, type, count, entry|
case tag
when TAG_EXPOSURE_TIME
meta.exposure_time ||= read_rationals(tiff, entry, type, count)&.first
when TAG_F_NUMBER
meta.f_number ||= read_rationals(tiff, entry, type, count)&.first
when TAG_ISO
meta.iso ||= read_number(tiff, entry, type, count)
when TAG_DATE_TIME_ORIGINAL
# Phone photos usually carry the real capture time here only.
meta.date_time ||= read_ascii(tiff, entry, type, count)
when TAG_FOCAL_LENGTH
meta.focal_length ||= read_rationals(tiff, entry, type, count)&.first
end
end
end
def read_gps_ifd(tiff, offset, meta)
lat_ref = 'N'
lon_ref = 'E'
lat = nil
lon = nil
each_ifd_entry(tiff, offset) do |tag, type, count, entry|
case tag
when TAG_GPS_LAT_REF
ref = read_ascii(tiff, entry, type, count)
lat_ref = ref[0].upcase if ref
when TAG_GPS_LAT
lat = read_rationals(tiff, entry, type, count)
when TAG_GPS_LON_REF
ref = read_ascii(tiff, entry, type, count)
lon_ref = ref[0].upcase if ref
when TAG_GPS_LON
lon = read_rationals(tiff, entry, type, count)
end
end
meta.gps_latitude ||= format_gps_coordinate(lat, lat_ref) if lat
meta.gps_longitude ||= format_gps_coordinate(lon, lon_ref) if lon
end
# XMP is XML; the only field worth surfacing is the editing software, which
# appears as e.g. <xmp:CreatorTool>Pixelmator Pro</xmp:CreatorTool>.
def parse_xmp(payload, meta)
text = payload.dup.force_encoding(Encoding::UTF_8).scrub('')
match = text.match(%r{:(?:CreatorTool|Software)>([^<]+)<})
meta.software ||= match[1].strip if match
end
# --- JPEG parse -----------------------------------------------------------
def parse_jpeg(bytes, meta)
(collect_jpeg_segments(bytes) || []).each do |seg|
if seg.marker == 0xe1
payload = bytes.byteslice(seg.payload_start, seg.finish - seg.payload_start).to_s
if signature_at?(bytes, seg.payload_start, EXIF_SIGNATURE)
parse_tiff(payload.byteslice(EXIF_SIGNATURE.bytesize..), meta)
elsif XMP_SIGNATURES.any? { |sig| signature_at?(bytes, seg.payload_start, sig) }
parse_xmp(payload, meta)
end
elsif sof?(seg)
# Start-of-frame header: precision(1), height(2), width(2), big-endian.
meta.image_height ||= u16(bytes, seg.payload_start + 1)
meta.image_width ||= u16(bytes, seg.payload_start + 3)
end
end
end
# SOF0-SOF15 minus the huffman/arithmetic/DNL markers that share the range.
def sof?(seg)
(0xc0..0xcf).cover?(seg.marker) &&
![0xc4, 0xc8, 0xcc].include?(seg.marker) &&
seg.finish - seg.payload_start >= 5
end
# --- PNG parse ------------------------------------------------------------
def parse_png(bytes, meta)
each_png_chunk(bytes, stop_at_iend: true) do |type, data, off, _length|
case type
when 'IHDR'
meta.image_width ||= u32(bytes, off + 8)
meta.image_height ||= u32(bytes, off + 12)
when 'tEXt', 'iTXt'
parse_png_text_chunk(type, data, meta)
when 'eXIf'
parse_tiff(data, meta)
end
end
end
# Walk PNG chunks (length, type, data, CRC) until the first corrupted length
# — or until IEND when +stop_at_iend+ is set, which is what the reader wants
# and the stripper does not (trailing bytes after IEND are preserved).
# Yields type, data, chunk offset and data length; returns the offset the
# walk stopped at.
def each_png_chunk(bytes, stop_at_iend: false)
off = 8
size = bytes.bytesize
while off + 8 <= size
length = u32(bytes, off)
break if length.nil? || length > size - off - 12
type = bytes.byteslice(off + 4, 4).to_s
yield(type, bytes.byteslice(off + 8, length).to_s, off, length)
off += 12 + length
break if stop_at_iend && type == 'IEND'
end
off
end
def parse_png_text_chunk(kind, data, meta)
nul = data.index("\0")
return if nul.nil? || nul < 1
return unless data.byteslice(0, nul) == 'Software'
if kind == 'tEXt'
text = data.byteslice(nul + 1..).to_s
else
# iTXt: keyword\0 compressionFlag(1) compressionMethod(1) languageTag\0
# translatedKeyword\0 text(utf-8). Only uncompressed text is read.
return unless data.getbyte(nul + 1) == 0
pos = nul + 3
2.times do # skip the language tag and the translated keyword
nxt = data.index("\0", pos)
return if nxt.nil?
pos = nxt + 1
end
text = data.byteslice(pos..).to_s
end
value = text.force_encoding(Encoding::UTF_8).scrub('').strip
meta.software ||= value unless value.empty?
end
# --- strip ----------------------------------------------------------------
def strip_jpeg(bytes)
segments = collect_jpeg_segments(bytes)
return bytes.dup if segments.nil? # corrupted walk — copy verbatim
parts = [bytes.byteslice(0, 2)]
segments.each do |seg|
if seg.marker == 0xda
# SOS header + entropy data + EOI are copied verbatim to the end.
parts << bytes.byteslice(seg.start..)
break
end
next if jpeg_metadata_segment?(bytes, seg)
parts << bytes.byteslice(seg.start, seg.finish - seg.start)
end
parts.join.b
end
def strip_png(bytes)
parts = [bytes.byteslice(0, 8)]
# tEXt / iTXt / eXIf are the metadata carriers; every other chunk (IHDR,
# PLTE, IDAT, ...) is copied byte-for-byte, CRC included.
tail = each_png_chunk(bytes) do |type, _data, off, length|
parts << bytes.byteslice(off, 12 + length) unless %w[tEXt iTXt eXIf].include?(type)
end
parts << bytes.byteslice(tail..) if tail < bytes.bytesize
parts.join.b
end
end
end
Also available in 8 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →