Skip to content

Find & Replace — C source

Find and replace text with literal or regular-expression matching, global replace, case sensitivity, whole-word, and capture-group substitution. Live match counter.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * Find & replace with literal or regex matching, $-substitution
 * ($1 backrefs, $&, $$), case sensitivity, whole-word, and global modes.
 *
 * Language: C (C11, POSIX <regex.h> — ISO C has no regex engine; POSIX
 *           regcomp/regexec with ERE is the native choice)
 * Source:   CosmoDev polyglot showcase port of the find-replace tool,
 *           ported from src/lib/findReplace.ts (the canonical TypeScript
 *           implementation).
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Mirrors the live lib: a literal find string is ERE-escaped and matched
 * verbatim; an isRegex find is compiled as-is. POSIX ERE has no \b, so
 * wholeWord is enforced at match sites instead — a match is accepted only
 * when both sides sit at word boundaries (or string bounds), which is what
 * the lib's \b wrap computes. !caseSensitive maps to REG_ICASE; multiline
 * (the JS m flag) maps to REG_NEWLINE, which anchors ^ and $ at newlines
 * like JS m does (and, matching JS's default, stops `.` from crossing a
 * newline). Invalid patterns are reported as the regcomp error string —
 * the port never exits on them — and an empty find is a no-op.
 *
 * Replacement $-substitution is applied by expand_replacement (not ERE's
 * native \1 syntax) so it matches JavaScript's String.replace exactly for
 * the realistic cases: $$ -> $, $& -> whole match, $1..$99 -> capture group
 * (literal "$<digits>" when out of range). JS's $` and $' are unsupported.
 */

#include <ctype.h>
#include <regex.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

typedef struct {
    int is_regex;
    int case_sensitive;
    int whole_word;
    int global_;
    int multiline;
} fr_options;

/* Mirror of the TypeScript lib's FindReplaceOptions defaults. */
fr_options fr_options_default(void) {
    fr_options o;
    o.is_regex = 0;
    o.case_sensitive = 1;
    o.whole_word = 0;
    o.global_ = 1;
    o.multiline = 0;
    return o;
}

typedef struct {
    char *result;  /* malloc'd */
    size_t matches;
    char *error;   /* malloc'd, or NULL when none */
} fr_result;

void fr_result_free(fr_result *r) {
    if (r == NULL) {
        return;
    }
    free(r->result);
    free(r->error);
    r->result = NULL;
    r->error = NULL;
}

/* ---- growable output buffer ---- */

typedef struct {
    char *data;
    size_t len;
    size_t cap;
} fr_buf;

static int fr_buf_grow(fr_buf *b, size_t extra) {
    if (b->len + extra + 1 <= b->cap) {
        return 0;
    }
    size_t cap = b->cap ? b->cap : 64;
    while (cap < b->len + extra + 1) {
        cap *= 2;
    }
    char *p = realloc(b->data, cap);
    if (p == NULL) {
        return -1;
    }
    b->data = p;
    b->cap = cap;
    return 0;
}

static int fr_buf_push(fr_buf *b, const char *s, size_t n) {
    if (fr_buf_grow(b, n) != 0) {
        return -1;
    }
    memcpy(b->data + b->len, s, n);
    b->len += n;
    b->data[b->len] = '\0';
    return 0;
}

/* ---- pattern helpers ---- */

static int is_ere_meta(unsigned char c) {
    return c != '\0' && strchr(".^$*+?()[]{}|\\", c) != NULL;
}

/* Escape ERE metacharacters so a literal find string matches verbatim. */
static char *escape_ere(const char *s) {
    size_t n = strlen(s);
    char *out = malloc(2 * n + 1);
    if (out == NULL) {
        return NULL;
    }
    char *p = out;
    for (size_t i = 0; i < n; i++) {
        unsigned char c = (unsigned char)s[i];
        if (is_ere_meta(c)) {
            *p++ = '\\';
        }
        *p++ = (char)c;
    }
    *p = '\0';
    return out;
}

/* Assemble the pattern: escape literals, wrap \b..\b for whole-word.
 * POSIX ERE has no \b, so the wrap is skipped here and boundaries are
 * enforced at match sites instead (see at_word_boundary). */
static char *build_pattern(const char *find, const fr_options *o) {
    char *body = o->is_regex ? strdup(find) : escape_ere(find);
    if (body == NULL) {
        return NULL;
    }
    if (!o->whole_word) {
        return body;
    }
    size_t n = strlen(body);
    char *wrapped = malloc(n + 1);
    if (wrapped == NULL) {
        free(body);
        return NULL;
    }
    memcpy(wrapped, body, n + 1); /* \b checks are at match sites, not in-pattern */
    free(body);
    return wrapped;
}

/* Count unescaped '(' — ERE has no non-capturing groups, so every open
 * paren in a valid pattern opens a capture group. */
static size_t count_ere_groups(const char *pattern) {
    size_t n = 0;
    for (const char *p = pattern; *p != '\0'; p++) {
        if (*p == '\\' && p[1] != '\0') {
            p++;
        } else if (*p == '(') {
            n++;
        }
    }
    return n;
}

static int is_word_byte(unsigned char c) {
    return isalnum(c) || c == '_';
}

/* True when input[so..eo) has non-word bytes (or bounds) on both sides —
 * the same predicate the lib's \b wrap imposes. */
static int at_word_boundary(const char *s, size_t len, size_t so, size_t eo) {
    int left = (so == 0) || !is_word_byte((unsigned char)s[so - 1]);
    int right = (eo == len) || !is_word_byte((unsigned char)s[eo]);
    return left && right;
}

/* ---- replacement expansion ---- */

static int is_ascii_digit(char c) {
    return c >= '0' && c <= '9';
}

/* Apply JS String.replace $-substitution for one match.
 *   "$$" -> "$";  "$&" -> whole match;  "$1".."$99" -> capture group N
 *   (literal "$<digits>" when N is out of range, matching JS).
 * groups[0] is the whole match; unmatched groups are "". */
static int expand_replacement(fr_buf *out, const char *tpl,
                              char **groups, size_t num_groups) {
    size_t i = 0, n = strlen(tpl);
    while (i < n) {
        char c = tpl[i];
        if (c != '$') {
            if (fr_buf_push(out, &c, 1) != 0) {
                return -1;
            }
            i++;
            continue;
        }
        char nxt = (i + 1 < n) ? tpl[i + 1] : '\0';
        if (nxt == '$') {
            if (fr_buf_push(out, "$", 1) != 0) {
                return -1;
            }
            i += 2;
        } else if (nxt == '&') {
            if (fr_buf_push(out, groups[0], strlen(groups[0])) != 0) {
                return -1;
            }
            i += 2;
        } else if (is_ascii_digit(nxt)) {
            size_t d1 = (size_t)(nxt - '0');
            /* Greedily try a second digit ($nn), matching JS. */
            if (i + 2 < n && is_ascii_digit(tpl[i + 2])) {
                size_t d2 = d1 * 10 + (size_t)(tpl[i + 2] - '0');
                if (d2 >= 1 && d2 <= num_groups) {
                    if (fr_buf_push(out, groups[d2], strlen(groups[d2])) != 0) {
                        return -1;
                    }
                    i += 3;
                    continue;
                }
            }
            if (d1 >= 1 && d1 <= num_groups) {
                if (fr_buf_push(out, groups[d1], strlen(groups[d1])) != 0) {
                    return -1;
                }
                i += 2;
            } else {
                char lit[2] = {'$', nxt};
                if (fr_buf_push(out, lit, 2) != 0) {
                    return -1;
                }
                i += 2;
            }
        } else {
            if (fr_buf_push(out, "$", 1) != 0) {
                return -1;
            }
            i++;
        }
    }
    return 0;
}

/* ---- main entry ---- */

fr_result find_replace(const char *input, const char *find,
                       const char *replacement, fr_options o) {
    fr_result res = {NULL, 0, NULL};
    size_t in_len = strlen(input);

    if (find[0] == '\0') {
        res.result = strdup(input);
        return res; /* empty find is a no-op */
    }

    char *pattern = build_pattern(find, &o);
    if (pattern == NULL) {
        res.error = strdup("out of memory");
        return res;
    }

    int cflags = REG_EXTENDED;
    if (!o.case_sensitive) {
        cflags |= REG_ICASE;
    }
    if (o.is_regex && o.multiline) {
        cflags |= REG_NEWLINE;
    }

    regex_t re;
    int rc = regcomp(&re, pattern, cflags);
    if (rc != 0) {
        char msg[256];
        regerror(rc, &re, msg, sizeof msg);
        res.result = strdup(input);
        res.error = strdup(msg);
        free(pattern);
        return res;
    }

    size_t num_groups = count_ere_groups(pattern);
    size_t nmatch = num_groups + 1;
    regmatch_t *pm = calloc(nmatch, sizeof(regmatch_t));
    char **groups = calloc(nmatch, sizeof(char *));
    char *scratch = calloc(nmatch ? nmatch : 1, in_len + 1);
    if (pm == NULL || groups == NULL || scratch == NULL) {
        regfree(&re);
        free(pattern);
        free(pm);
        free(groups);
        free(scratch);
        res.error = strdup("out of memory");
        return res;
    }
    /* scratch[g] is a slot of in_len+1 bytes holding the g-th slice. */
    for (size_t g = 0; g < nmatch; g++) {
        groups[g] = scratch + g * (in_len + 1);
        groups[g][0] = '\0';
    }

    fr_buf out = {0};
    if (fr_buf_push(&out, "", 0) != 0) {
        res.error = strdup("out of memory");
        goto done;
    }
    out.len = 0;

    size_t last = 0;      /* copied up to here */
    size_t search = 0;    /* regexec scans input+search */
    size_t matches = 0;

    while (search <= in_len) {
        if (regexec(&re, input + search, nmatch, pm, 0) != 0) {
            break; /* REG_NOMATCH */
        }
        size_t so = search + (size_t)pm[0].rm_so;
        size_t eo = search + (size_t)pm[0].rm_eo;

        if (o.whole_word && !at_word_boundary(input, in_len, so, eo)) {
            /* Not at a word boundary: skip past this match's start and
             * keep searching (emulates the \b wrap's skip behavior). */
            search = so + 1;
            continue;
        }

        /* Gather [whole, g1..gN]; unmatched groups (-1) become "". */
        for (size_t g = 0; g < nmatch; g++) {
            if (pm[g].rm_so == -1) {
                groups[g][0] = '\0';
            } else {
                size_t gso = search + (size_t)pm[g].rm_so;
                size_t glen = (size_t)pm[g].rm_eo - (size_t)pm[g].rm_so;
                memcpy(groups[g], input + gso, glen);
                groups[g][glen] = '\0';
            }
        }

        if (fr_buf_push(&out, input + last, so - last) != 0 ||
            expand_replacement(&out, replacement, groups, num_groups) != 0) {
            res.error = strdup("out of memory");
            goto done;
        }
        last = eo;
        matches++;
        if (!o.global_) {
            break;
        }
        if (eo == so) { /* empty match: copy one byte and step, like JS */
            if (eo < in_len) {
                if (fr_buf_push(&out, input + eo, 1) != 0) {
                    res.error = strdup("out of memory");
                    goto done;
                }
                last = eo + 1;
            }
            search = last;
        } else {
            search = eo;
        }
    }

    if (fr_buf_push(&out, input + last, in_len - last) != 0) {
        res.error = strdup("out of memory");
        goto done;
    }

    if (matches != 0 && !o.global_) {
        matches = 1 + num_groups; /* JS String.match length quirk */
    }
    res.result = out.data ? out.data : strdup("");
    res.matches = matches;

done:
    regfree(&re);
    free(pattern);
    free(pm);
    free(groups);
    free(scratch);
    if (res.error != NULL) {
        free(out.data);
        res.result = strdup(input);
        res.matches = 0;
    }
    return res;
}

int main(void) {
    fr_options o = fr_options_default();
    o.case_sensitive = 0;
    fr_result r = find_replace("Hello World world", "world", "Universe", o);
    if (r.error != NULL) {
        printf("error: %s\n", r.error);
    } else {
        printf("%s  (%zu matches)\n", r.result, r.matches);
    }
    fr_result_free(&r);
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →