Sort Lines & Remove Duplicates — Go source
Alphabetize, reverse, shuffle, dedupe, or length-sort lines of text. Supports case-insensitive and natural sorting (file2 before file10).
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package sortlines is the Go twin of CosmoDev's src/lib/sortLines.ts (dual
// source: the web lib is TypeScript, the CLI lib is Go — kept in lock-step).
// Pure + deterministic, never panics. The table-driven tests in
// sort-lines_test.go share vectors with src/lib/sortLines.test.ts so the two
// implementations are held to the same contract.
//
// Behavior mirrors the TS lib exactly: split the input on newlines, apply the
// requested sort mode (with optional trim / case sensitivity / natural ordering
// / shuffle seed), and join the result back to text. The Options struct uses
// zero-value defaults matching the TS defaults; CaseSensitive and Seed are
// pointers so their non-zero defaults (true and 1) stay distinguishable from an
// explicit false / 0 — see Options.
package sortlines
import (
"math"
"sort"
"strconv"
"strings"
)
// Mode selects the line operation. Values mirror the SortMode string union in
// src/lib/sortLines.ts so the cosmodev CLI can map a flag straight to a Mode.
type Mode string
const (
ModeAsc Mode = "asc"
ModeDesc Mode = "desc"
ModeLengthAsc Mode = "length-asc"
ModeLengthDesc Mode = "length-desc"
ModeReverse Mode = "reverse"
ModeShuffle Mode = "shuffle"
ModeUnique Mode = "unique"
)
// Options configures SortLines. The zero value (Options{}) matches the TS
// default (sortLines(input, mode) with no options): case sensitive, no trim,
// no natural ordering, seed 1.
//
// CaseSensitive is *bool because the TS default is true (not the bool zero
// value): nil → case sensitive; a non-nil value (including false) is used
// verbatim. Seed is *int for the same reason (TS default 1): nil → 1. Trim and
// Natural default to false, which equals the bool zero value, so they need no
// pointer — the same pattern slugify's Options uses for Separator vs the rest.
type Options struct {
CaseSensitive *bool // nil → true (default); non-nil used verbatim
Trim bool // default false
Natural bool // default false
Seed *int // nil → 1 (default); non-nil used verbatim
}
// Result is the output of SortLines, mirroring the SortResult interface in the
// TS lib.
type Result struct {
Lines []string
Text string
RemovedDuplicates int
}
// Mulberry32 is a small deterministic PRNG returning floats in [0, 1). It is
// the Go twin of mulberry32() in src/lib/sortLines.ts and produces the same
// sequence for a given seed: every step is 32-bit arithmetic, matching the JS
// bit operations exactly (>>> becomes uint32 shift, Math.imul becomes the low
// 32 bits of the product, |0 truncation is uint32 wraparound).
func Mulberry32(seed uint32) func() float64 {
a := seed
return func() float64 {
a += 0x6d2b79f5 // wraps mod 2^32, matching (a + 0x6d2b79f5) | 0
t := imul(a^(a>>15), 1|a)
t = (t + imul(t^(t>>7), 61|t)) ^ t
return float64(t^(t>>14)) / 4294967296.0
}
}
// imul is the low-32-bit product of two uint32 values, mirroring JS Math.imul
// (the truncated bits are identical whether the operands are read as signed or
// unsigned, so uint32 math reproduces it exactly).
func imul(a, b uint32) uint32 {
return uint32(uint64(a) * uint64(b))
}
// naturalChunks splits s into alternating digit / non-digit runs, mirroring the
// /(\d+|\D+)/g tokenizer in the TS naturalCompare. Only ASCII [0-9] counts as a
// digit, matching JS \d (without the unicode flag). The empty string yields
// [""] because TS match() returns null and the ?? fallback wraps [ax].
func naturalChunks(s string) []string {
if s == "" {
return []string{""}
}
runes := []rune(s)
chunks := make([]string, 0, len(runes))
for i := 0; i < len(runes); {
digit := runes[i] >= '0' && runes[i] <= '9'
j := i + 1
for j < len(runes) && (runes[j] >= '0' && runes[j] <= '9') == digit {
j++
}
chunks = append(chunks, string(runes[i:j]))
i = j
}
return chunks
}
// isDigitChunk reports whether chunk begins with an ASCII digit (mirrors
// /^\d/.test(chunk)). It is safe on the empty chunk.
func isDigitChunk(chunk string) bool {
return len(chunk) > 0 && chunk[0] >= '0' && chunk[0] <= '9'
}
// compareDigitChunks compares two all-digit runs by numeric value, mirroring
// parseInt(a) - parseInt(b). On int overflow it falls back to length-then-
// lexical order (still numeric for equal-length runs).
func compareDigitChunks(a, b string) int {
na, errA := strconv.Atoi(a)
nb, errB := strconv.Atoi(b)
if errA == nil && errB == nil {
switch {
case na < nb:
return -1
case na > nb:
return 1
default:
return 0
}
}
if len(a) != len(b) {
return len(a) - len(b)
}
switch {
case a < b:
return -1
case a > b:
return 1
default:
return 0
}
}
// naturalCompare is the Go twin of naturalCompare() in the TS lib: it splits
// both strings into text/number chunks and compares chunk-by-chunk.
func naturalCompare(a, b string, caseSensitive bool) int {
ax, bx := a, b
if !caseSensitive {
ax = strings.ToLower(a)
bx = strings.ToLower(b)
}
aa := naturalChunks(ax)
bb := naturalChunks(bx)
n := min(len(aa), len(bb))
for i := 0; i < n; i++ {
ca, cb := aa[i], bb[i]
an := isDigitChunk(ca)
bn := isDigitChunk(cb)
if an != bn {
// One side is numeric at this position: compare the raw chunks.
if ca < cb {
return -1
}
return 1
}
if an {
if c := compareDigitChunks(ca, cb); c != 0 {
return c
}
} else if ca != cb {
if ca < cb {
return -1
}
return 1
}
}
return len(aa) - len(bb)
}
// reverseStrings returns a reversed copy of s, mirroring [...lines].reverse().
func reverseStrings(s []string) []string {
out := make([]string, len(s))
for i, v := range s {
out[len(s)-1-i] = v
}
return out
}
// SortLines is the Go twin of sortLines() in src/lib/sortLines.ts. It splits
// input on newlines, applies the requested mode with the given options, and
// returns the lines, the joined text, and a duplicate count. It must agree with
// the TS lib on every shared vector and never panics.
func SortLines(input string, mode Mode, opts Options) Result {
caseSensitive := true
if opts.CaseSensitive != nil {
caseSensitive = *opts.CaseSensitive
}
seed := 1
if opts.Seed != nil {
seed = *opts.Seed
}
lines := strings.Split(input, "\n")
if opts.Trim {
for i := range lines {
lines[i] = strings.TrimSpace(lines[i])
}
}
norm := func(s string) string {
if caseSensitive {
return s
}
return strings.ToLower(s)
}
removedDuplicates := 0
switch mode {
case ModeUnique:
seen := make(map[string]struct{}, len(lines))
out := make([]string, 0, len(lines))
for _, l := range lines {
key := norm(l)
if _, ok := seen[key]; ok {
removedDuplicates++
} else {
seen[key] = struct{}{}
out = append(out, l)
}
}
lines = out
case ModeShuffle:
rng := Mulberry32(uint32(seed))
arr := make([]string, len(lines))
copy(arr, lines)
for i := len(arr) - 1; i > 0; i-- {
j := int(math.Floor(rng() * float64(i+1)))
arr[i], arr[j] = arr[j], arr[i]
}
lines = arr
case ModeReverse:
lines = reverseStrings(lines)
case ModeLengthAsc, ModeLengthDesc:
sorted := make([]string, len(lines))
copy(sorted, lines)
// Stable ascending by length (ties keep original order), then reverse
// for desc — exactly the TS map→sort→map + conditional reverse.
sort.SliceStable(sorted, func(a, b int) bool {
return len(sorted[a]) < len(sorted[b])
})
if mode == ModeLengthDesc {
sorted = reverseStrings(sorted)
}
lines = sorted
default: // ModeAsc, ModeDesc (any unknown mode → desc, matching TS)
dir := 1
if mode != ModeAsc {
dir = -1
}
sorted := make([]string, len(lines))
copy(sorted, lines)
sort.SliceStable(sorted, func(a, b int) bool {
x, y := sorted[a], sorted[b]
var c int
if opts.Natural {
c = naturalCompare(x, y, caseSensitive)
} else {
nx, ny := norm(x), norm(y)
switch {
case nx < ny:
c = -1
case nx > ny:
c = 1
}
}
return c*dir < 0
})
lines = sorted
}
return Result{
Lines: lines,
Text: strings.Join(lines, "\n"),
RemovedDuplicates: removedDuplicates,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →