Every language
12 implementations, copy-ready. One at a time with syntax highlighting, or all inline.
SQLSQLrunnable
WITH s(text) AS (VALUES ('Crème Brûlée 🇺🇸 flag'))
SELECT CASE WHEN char_length(text) > 10
THEN left(text, 9) || '…' -- 9 chars + ellipsis = 10 budget
ELSE text
END AS truncated
FROM s;left(s, n) and char_length(s) count characters (code points in Postgres) — correct for ordinary accented text, wrong for emoji families: '🇺🇸' counts as 2 and can still be cut in half for astral pairs. Postgres has no in-core grapheme segmenter; run this as-is in the playground and keep emoji-safe truncation in the app layer.
JSJavaScript
function truncateGrapheme(s, max) {
if (max <= 0) return '';
const seg = new Intl.Segmenter('en', { granularity: 'grapheme' });
const clusters = Array.from(seg.segment(s), (g) => g.segment);
if (clusters.length <= max) return s;
return clusters.slice(0, max - 1).join('') + '…';
}
truncateGrapheme('👨👩👧👦 family 🇺🇸', 8); // '👨👩👧👦 famil…'s.slice(0, 10) cuts by UTF-16 code units — it splits an emoji's surrogate pair in half and prints the lone-surrogate � you see in the wild. Intl.Segmenter walks real grapheme boundaries (Node 16+, all modern browsers); the ellipsis takes the max-th slot so the total stays exactly max.
TSTypeScript
export function truncateGrapheme(s: string, max: number): string {
if (max <= 0) return '';
const seg = new Intl.Segmenter('en', { granularity: 'grapheme' });
const clusters = Array.from(seg.segment(s), (g) => g.segment);
if (clusters.length <= max) return s;
return clusters.slice(0, max - 1).join('') + '…';
}Same engine call with a typed (s: string, max: number) => string contract — pure string in, string out, trivially testable. Needs lib es2022.intl (or newer) in tsconfig for the Intl.Segmenter typings.
GoGo
import (
"strings"
"github.com/rivo/uniseg"
)
// Truncate cuts s to at most max grapheme clusters, ellipsis inside the budget.
func Truncate(s string, max int) string {
if max <= 0 {
return ""
}
var b strings.Builder
n := 0
state := -1
for len(s) > 0 {
var cluster string
cluster, s, _, state = uniseg.FirstGraphemeClusterInString(s, state)
if n == max-1 && len(s) > 0 {
b.WriteString("…") // ellipsis instead of the max-th cluster
break
}
b.WriteString(cluster)
n++
}
return b.String()
}[]rune gets code points, NOT graphemes — a flag 🇺🇸 is two runes and a family 👨👩👧👦 is seven. The stdlib has no grapheme iterator; rivo/uniseg is the canonical package. FirstGraphemeClusterInString (string twin of FirstGraphemeClusterIn) steps clusters with zero allocations; state carries across calls.
RsRust
use unicode_segmentation::UnicodeSegmentation;
fn truncate_grapheme(s: &str, max: usize) -> String {
if max == 0 {
return String::new();
}
if s.graphemes(true).count() <= max {
return s.to_string();
}
s.graphemes(true).take(max - 1).collect::<String>() + "…"
}.chars() yields scalar values, not graphemes — flags arrive as two chars and families as several. unicode-segmentation is the de-facto standard crate; graphemes(true) selects EXTENDED clusters, the human-perceived kind (the flag or family counts as one).
PHPPHP
function truncateGrapheme(string $s, int $max): string
{
if ($max <= 0) return '';
preg_match_all('/\X/u', $s, $m); // \X = one grapheme cluster (PCRE2)
$clusters = $m[0];
if (count($clusters) <= $max) return $s;
return implode('', array_slice($clusters, 0, $max - 1)) . '…';
}mb_substr counts code points — it cuts a flag in half; mb_strcut cuts by bytes but at character boundaries — safer than substr() yet still not grapheme-aware. The /\X regex (PCRE2) is the pure-stdlib grapheme matcher; ext-intl also ships grapheme_extract() if you already load ICU.
PyPython
# stdlib core — counts CODE POINTS, fine for text, wrong for emoji:
def truncate_codepoints(s: str, max: int) -> str:
if len(s) <= max:
return s
return s[:max - 1] + '…'
# grapheme-correct (third-party: pip install grapheme):
import grapheme
def truncate_grapheme(s: str, max: int) -> str:
if grapheme.length(s) <= max:
return s
return grapheme.truncate(s, max - 1) + '…'len(s) and slicing count code points — 'ñ' is 1 but '👨👩👧👦' is 7 and '🇺🇸' is 2. The stdlib has no grapheme API; the grapheme package (grapheme.length / grapheme.truncate) is the usual answer, shown second with the honest stdlib version above it.
C#C#
using System.Collections.Generic;
using System.Globalization;
static string TruncateGrapheme(string s, int max)
{
if (max <= 0) return "";
var starts = new List<int>();
var el = StringInfo.GetTextElementEnumerator(s);
while (el.MoveNext())
starts.Add(el.ElementIndex); // start of each text element
if (starts.Count <= max) return s;
return s.Substring(0, starts[max - 1]) + "…";
}StringInfo is .NET's grapheme layer: new StringInfo(s).LengthInTextElements counts them; GetTextElementEnumerator().ElementIndex gives each element's start, so slicing at starts[max - 1] keeps max-1 elements. Surrogate pairs are handled; ZWJ families split — same imperfect-by-UWC-generation limit as Java.
JvJava
import java.text.BreakIterator;
static String truncateGrapheme(String s, int max) {
if (max <= 0) return "";
BreakIterator it = BreakIterator.getCharacterInstance();
it.setText(s);
int boundary = it.first();
int count = 0;
while (count < max - 1) { // walk max-1 clusters
int next = it.next();
if (next == BreakIterator.DONE) return s;
boundary = next;
count++;
}
int after = it.next(); // is there a max-th cluster?
if (after == BreakIterator.DONE || after == s.length()) return s;
return s.substring(0, boundary) + "…";
}BreakIterator.getCharacterInstance() is the old stdlib segmenter — it predates emoji ZWJ sequences, so families like 👨👩👧👦 split into person + child pieces. Correct for ordinary text and combining marks; grapheme-exact emoji work needs ICU4J's RuleBasedBreakIterator.
SwSwift
func truncateGrapheme(_ s: String, max: Int) -> String {
if max <= 0 { return "" }
if s.count <= max { return s } // .count IS grapheme count
return s.prefix(max - 1) + "…"
}The inversion: Swift is the language where the naive code is right. String is a Collection of Character (extended grapheme clusters), so count, prefix(_:) and iteration are all human-visible-character based — '👨👩👧👦'.count == 1 with no import and no library.
KtKotlin
import java.text.BreakIterator
fun truncateGrapheme(s: String, max: Int): String {
if (max <= 0) return ""
val it = BreakIterator.getCharacterInstance().apply { setText(s) }
var boundary = it.first()
var count = 0
while (count < max - 1) {
val next = it.next()
if (next == BreakIterator.DONE) return s
boundary = next
count++
}
val after = it.next()
return if (after == BreakIterator.DONE || after == s.length) s
else s.substring(0, boundary) + "…"
}The same JVM BreakIterator under a cleaner wrapper — apply { setText(s) } scopes setup, and the boundary walk is a plain loop. Inherits Java's caveat: surrogate pairs handled, ZWJ emoji families imperfect.
RbRuby
def truncate_grapheme(s, max)
return '' if max <= 0
clusters = s.scan(/\X/) # \X = one extended grapheme cluster
return s if clusters.size <= max
clusters.take(max - 1).join + '…'
endString#each_char yields code points, so s[0, 10] splits emoji mid-surrogate. /\X is Ruby's one-regex stdlib answer — scan(/\X/) decomposes the string exactly the way users count characters, no gem required.