Every language
15 implementations, copy-ready. One at a time with syntax highlighting, or all inline.
SQLSQLrunnable
-- PostgreSQL: split per character, remember positions, re-aggregate DESC.
WITH graphemes AS (
SELECT c.ch, c.ord
FROM regexp_split_to_table(E'cafe\u0301 Foo', '') WITH ORDINALITY AS c(ch, ord)
)
SELECT string_agg(ch, '' ORDER BY ord DESC) AS reversed;
-- ooF ´efac — chars here are code points, not clusters: see notesPostgreSQL characters are CODE POINTS, so the combining accent still strands — no mainstream SQL dialect ships grapheme-aware reversal. MySQL REVERSE() is likewise code-point-wise; SQL Server's REVERSE() is UTF-16 code-unit-wise and can split surrogate pairs. WITH ORDINALITY (PG 9.4+) supplies the position column.
JSJavaScript
// Intl.Segmenter (Node 16+, all modern browsers) segments into grapheme
// clusters; [...s] iterates code points and strands combining marks.
const reverse = (s) =>
[...new Intl.Segmenter('en', { granularity: 'grapheme' }).segment(s)]
.map(({ segment }) => segment)
.reverse()
.join('');
console.log(reverse('cafe\u0301 Foo')); // ooF éfac — the é stayed gluedThe [...s].reverse().join('') idiom you see everywhere reverses CODE POINTS: valid UTF-8, but U+0301 detaches from its base and ZWJ emoji rip apart. On Node <16 or old browsers, the grapheme-splitter npm package is the drop-in.
TSTypeScript
function reverseGraphemes(s: string): string {
const segments = new Intl.Segmenter('en', { granularity: 'grapheme' }).segment(s);
return [...segments].map(({ segment }) => segment).reverse().join('');
}
// the decomposed é (e + U+0301) must survive as ONE cluster
console.log(reverseGraphemes('cafe\u0301 Foo') === 'ooF e\u0301fac'); // trueNeeds a lib with Intl.Segmenter typings (ES2022+ in tsconfig lib, TS 4.7+). The assertion is the spec: reversed output keeps the cluster e+U+0301 intact — s.split('').reverse().join('') fails it.
GoGo
package main
import (
"fmt"
"strings"
"github.com/rivo/uniseg"
)
func reverseGraphemes(s string) string {
var clusters []string
g := uniseg.NewGraphemes(s)
for g.Next() {
clusters = append(clusters, g.Str())
}
for i, j := 0, len(clusters)-1; i < j; i, j = i+1, j-1 {
clusters[i], clusters[j] = clusters[j], clusters[i]
}
return strings.Join(clusters, "")
}
func main() {
fmt.Println(reverseGraphemes("cafe\u0301 Foo")) // ooF éfac
}go get github.com/rivo/uniseg — the UAX #29 clusterer the terminal-UI ecosystem standardizes on. Stdlib has no segmentation: []rune reversal is code-point-wise and strands U+0301 from its 'e'. uniseg v0.4+ also offers the Step() iterator when allocation matters.
RsRust
use unicode_segmentation::UnicodeSegmentation;
fn main() {
let s = "cafe\u{0301} Foo";
// .chars().rev().collect() is code-point-wise: valid UTF-8, but
// U+0301 detaches from its base. graphemes(true) = extended clusters.
let reversed: String = s.graphemes(true).rev().collect();
assert_eq!(reversed, "ooF e\u{0301}fac");
println!("{reversed}");
}unicode-segmentation = "1" in Cargo.toml — the de-facto UAX #29 crate (std deliberately ships no segmentation). The true flag selects extended grapheme clusters, which is what keeps prefixing combining marks with their base.
PHPPHP
<?php
function reverseGraphemes(string $s): string {
$clusters = [];
$offset = 0; // byte position, advanced by reference each call
while ($offset < strlen($s)) {
$clusters[] = grapheme_extract($s, 1, GRAPHEME_EXTR_COUNT, $offset, $offset);
}
return implode('', array_reverse($clusters));
}
echo reverseGraphemes("cafe\u{0301} Foo"), PHP_EOL; // ooF éfacRequires the intl extension — grapheme_* is not ext/standard. The constant is GRAPHEME_EXTR_COUNT; the longer GRAPHEME_EXTRACT_* spelling is removed in PHP 8.5. strrev() is byte-wise: on UTF-8 it can emit invalid sequences, not merely strand an accent.
PyPython
import grapheme
def reverse_graphemes(s: str) -> str:
# graphemes() yields clusters; reversed() needs a sequence,
# so materialize the generator before reversing it.
return "".join(reversed(list(grapheme.graphemes(s))))
assert reverse_graphemes("cafe\u0301 Foo") == "ooF e\u0301fac"
print(reverse_graphemes("cafe\u0301 Foo")) # ooF éfacpip install grapheme — the stdlib offers nothing: s[::-1] reverses code points. If you already depend on the regex package, regex.findall(r'\X', s)[::-1] is the same cluster split.
CC
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* Return a malloc'd copy of s with its CODE POINTS (not bytes)
reversed — caller frees. A UTF-8 code point begins at any byte
that is not a 0b10xxxxxx continuation byte; walk backwards. */
char *reverse_utf8(const char *s) {
size_t len = strlen(s);
char *out = malloc(len + 1);
if (out == NULL) return NULL;
size_t o = 0;
size_t i = len;
while (i > 0) {
size_t start = i - 1;
while (start > 0 && (((unsigned char)s[start] & 0xC0) == 0x80))
start--; /* skip continuation bytes */
memcpy(out + o, s + start, i - start);
o += i - start;
i = start;
}
out[o] = '\0';
return out;
}
int main(void) {
char *r = reverse_utf8("cafe\u0301 Foo");
if (r != NULL) {
printf("%s\n", r); /* ooF ´efac — the accent strands: see notes */
free(r);
}
return 0;
}C has no stdlib grapheme segmentation — this is code-point-safe only: multi-byte sequences stay valid, but U+0301 detaches from its base (that stranded ´ in the output). Full cluster rules are UAX #29: use ICU's ubrk_open(UBRK_CHARACTER, ...) and reverse between break offsets.
C++C++
#include <iostream>
#include <string>
#include <string_view>
#include <vector>
// Reverse by code point: a code point begins at any byte that is
// not a 0b10xxxxxx continuation byte. std::string iterators are
// BYTE-wise — std::reverse would emit invalid UTF-8.
std::string reverse_utf8(std::string_view s) {
std::vector<std::string_view> cps;
for (std::size_t i = 0; i < s.size();) {
unsigned char c = static_cast<unsigned char>(s[i]);
std::size_t n = c >= 0xF0 ? 4 : c >= 0xE0 ? 3 : c >= 0xC0 ? 2 : 1;
cps.push_back(s.substr(i, n));
i += n;
}
std::string out;
for (auto it = cps.rbegin(); it != cps.rend(); ++it) out += *it;
return out;
}
int main() {
std::cout << reverse_utf8("cafe\u0301 Foo") << '\n'; // ooF ´efac
}Code-point safe, not grapheme safe — the combining accent strands exactly like the C impl. The C++ answer for real clusters is ICU (unicode/ubrk.h, UBRK_CHARACTER BreakIterator) reversing between boundaries; u32string round-trips are the same idea with a deprecated codecvt.
C#C#
using System.Globalization;
static string Reverse(string s)
{
var clusters = new List<string>();
var it = StringInfo.GetTextElementEnumerator(s);
while (it.MoveNext())
{
clusters.Add((string)it.Current!); // text element = grapheme cluster
}
clusters.Reverse();
return string.Concat(clusters);
}
Console.WriteLine(Reverse("cafe\u0301 Foo")); // ooF éfacStringInfo text elements are grapheme clusters — base + combining marks + surrogate pairs stay whole. The viral new string(s.Reverse().ToArray()) snippet reverses UTF-16 CODE UNITS and can split a surrogate pair (any astral emoji). .NET 5+ alternative: StringInfo.ParseCombiningCharacters for the offsets.
JvJava
import java.text.BreakIterator;
import java.util.ArrayList;
import java.util.Collections;
import java.util.List;
public class ReverseUnicode {
// StringBuilder.reverse() keeps surrogate PAIRS intact but still
// detaches combining marks — the character BreakIterator segments
// user-perceived characters (grapheme clusters).
static String reverse(String s) {
BreakIterator it = BreakIterator.getCharacterInstance();
it.setText(s);
List<String> clusters = new ArrayList<>();
int start = it.first();
for (int end = it.next(); end != BreakIterator.DONE; start = end, end = it.next()) {
clusters.add(s.substring(start, end));
}
Collections.reverse(clusters);
return String.join("", clusters);
}
public static void main(String[] args) {
System.out.println(reverse("cafe\u0301 Foo")); // ooF éfac
}
}The JDK character BreakIterator handles marks and surrogate pairs (verified: e+U+0301 survives) but predates emoji sequences — ZWJ families can still split; ICU4J's RuleBasedBreakIterator is the strict UAX #29 upgrade on the JVM.
SwSwift
let s = "cafe\u{301} Foo"
// Swift's Character IS an extended grapheme cluster, so the obvious
// one-liner is already Unicode-correct — the one language where
// naive wins.
print(String(s.reversed())) // ooF éfacCorrect by construction: String is a collection of Characters (clusters), so reversed() never splits e + U+0301 or a ZWJ family emoji. The trap is dropping to raw views — s.utf8.reversed() is byte-wise, s.unicodeScalars.reversed() is code-point-wise.
KtKotlin
import java.text.BreakIterator
fun reverseGraphemes(s: String): String {
val it = BreakIterator.getCharacterInstance().apply { setText(s) }
val clusters = buildList {
var start = it.first()
var end = it.next()
while (end != BreakIterator.DONE) {
add(s.substring(start, end))
start = end
end = it.next()
}
}
return clusters.asReversed().joinToString("")
}
fun main() {
println(reverseGraphemes("cafe\u0301 Foo")) // ooF éfac
}s.reversed() delegates to StringBuilder.reverse — surrogate pairs survive, combining marks do not. The JDK character BreakIterator is grapheme-ish: marks and pairs are safe, ZWJ emoji sequences are not — ICU4J on the JVM for strict UAX #29.
RbRuby
s = "cafe\u0301 Foo"
# .reverse is code-point-wise: U+0301 lands BEFORE its 'e'.
# grapheme_clusters (Ruby 2.5+) splits on extended clusters.
puts s.grapheme_clusters.reverse.join # ooF éfacgrapheme_clusters implements extended clusters, so ZWJ emoji sequences stay whole too. On a UTF-8 string .reverse emits valid but accent-stranded output; on an ASCII-8BIT (binary) string it flips raw bytes and corrupts multi-byte sequences.
ZigZig
const std = @import("std");
pub fn main() void {
const s = "cafe\u{301} Foo";
// Walk the UTF-8 bytes backwards: a code point begins at any
// byte that is NOT a 0b10xxxxxx continuation byte.
var out: [128]u8 = undefined;
var len: usize = 0;
var i: usize = s.len;
while (i > 0) {
var start = i - 1;
while (start > 0 and (s[start] & 0b1100_0000) == 0b1000_0000) start -= 1;
const cp = s[start..i];
@memcpy(out[len .. len + cp.len], cp);
len += cp.len;
i = start;
}
std.debug.print("{s}\n", .{out[0..len]}); // ooF ´efac
}Code-point safe, not grapheme safe — the combining accent strands, same caveat as the C/C++ impls; cluster rules (UAX #29) are far beyond the stdlib's scope. The fixed 128-byte buffer suits a demo; std.ArrayList for unbounded input.