Skip to content

robots.txt Generator — Zig source

Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// robots-txt-generator — Zig port: standards-compliant robots.txt generator + parser.
//
// Display port of the CosmoDev Robots.txt Generator tool — same contract as
// cli/robots-txt-generator/robots-txt-generator.go (the live Go twin) and
// src/lib/robotsTxt.ts (canonical TypeScript). Pure + deterministic; stdlib
// only. A group carries a LIST of user-agents (the REP spec's stacked
// User-agent lines); parsing aggregates repeated blocks per agent, then
// re-merges agents whose accumulated rules are identical into one stacked
// group, exactly like the TS lib. Fixed-capacity demo storage: 8 groups,
// 12 agents, 16 rules/group, 8 KB body/scratch.
const std = @import("std");

const MAX_AGENTS = 12;
const MAX_RULES = 16;
const MAX_GROUPS = 8;

/// One REP group — mirrors `RuleGroup` in the Go twin. Slices point into
/// caller-owned storage (comptime literals for the demo, scratch for parsing).
pub const RuleGroup = struct {
    user_agents: []const []const u8,          // {"*"} or {"Googlebot", "Bingbot"}
    disallow: []const []const u8 = &.{},      // "" renders as a bare "Disallow:"
    allow: []const []const u8 = &.{},
    crawl_delay: ?f64 = null,                 // null = unset; non-finite never emitted
};

pub const RobotsConfig = struct {
    groups: []const RuleGroup,
    sitemaps: []const []const u8 = &.{},
};

/// Parse output: fixed-capacity storage so no allocator is needed. `Group`
/// doubles as the per-agent accumulator while parsing (one agent each) and as
/// the merged stacked-agent group afterwards.
const Group = struct {
    user_agents: [MAX_AGENTS][]const u8 = undefined,
    n_user_agents: usize = 0,
    disallow: [MAX_RULES][]const u8 = undefined,
    n_disallow: usize = 0,
    allow: [MAX_RULES][]const u8 = undefined,
    n_allow: usize = 0,
    crawl_delay: ?f64 = null,
};

pub const Parsed = struct {
    groups: [MAX_GROUPS]Group = undefined,
    n_groups: usize = 0,
    sitemaps: [MAX_RULES][]const u8 = undefined,
    n_sitemaps: usize = 0,
};

/// Growable-but-capped output body (fixed 8 KB; writes past the cap truncate).
const Buf = struct {
    data: [8192]u8 = undefined,
    len: usize = 0,

    fn add(self: *Buf, s: []const u8) void {
        if (self.len >= self.data.len) return;
        const n = @min(s.len, self.data.len - self.len);
        @memcpy(self.data[self.len..][0..n], s[0..n]);
        self.len += n;
    }

    fn fmt(self: *Buf, comptime f: []const u8, args: anytype) void {
        var tmp: [64]u8 = undefined;
        const s = std.fmt.bufPrint(&tmp, f, args) catch return;
        self.add(s);
    }

    fn slice(self: *const Buf) []const u8 {
        return self.data[0..self.len];
    }
};

fn trim(s: []const u8) []const u8 {
    return std.mem.trim(u8, s, " \t\r\n");
}

/// Normalize a Crawl-delay for the merge comparison: non-finite counts as
/// unset, exactly like the TS key (JSON.stringify turns NaN into null).
fn normCd(cd: ?f64) ?f64 {
    if (cd) |v| {
        if (!std.math.isFinite(v)) return null;
    }
    return cd;
}

fn rulesEq(a: *const Group, b: *const Group) bool {
    if (a.n_disallow != b.n_disallow or a.n_allow != b.n_allow) return false;
    for (a.disallow[0..a.n_disallow], b.disallow[0..b.n_disallow]) |x, y| {
        if (!std.mem.eql(u8, x, y)) return false;
    }
    for (a.allow[0..a.n_allow], b.allow[0..b.n_allow]) |x, y| {
        if (!std.mem.eql(u8, x, y)) return false;
    }
    return normCd(a.crawl_delay) == normCd(b.crawl_delay);
}

/// Move the pending User-agent stack into per-agent rule buckets (Google's
/// merge semantics: a rule line applies to every agent of the immediately
/// preceding consecutive User-agent stack).
fn flushStack(per_ua: *[MAX_AGENTS]Group, n_ua: *usize, stack: []const []const u8, cur: *[MAX_AGENTS]usize, n_cur: *usize) void {
    if (stack.len == 0) return;
    n_cur.* = 0;
    for (stack) |ua| {
        var i: usize = 0;
        while (i < n_ua.*) : (i += 1) {
            if (std.mem.eql(u8, per_ua[i].user_agents[0], ua)) break;
        }
        if (i == n_ua.*) {
            per_ua[n_ua.*] = .{ .n_user_agents = 1 };
            per_ua[n_ua.*].user_agents[0] = ua;
            n_ua.* += 1;
        }
        cur[n_cur.*] = i;
        n_cur.* += 1;
    }
}

/// Copy `s` into the scratch buffer and return the copy (fixed-capacity).
fn dupe(sc: []u8, cursor: *usize, s: []const u8) []const u8 {
    const n = @min(s.len, sc.len - cursor.*);
    @memcpy(sc[cursor.*..][0..n], s[0..n]);
    cursor.* += n;
    return sc[cursor.* - n .. cursor.*];
}

/// Fold 3+ consecutive newlines to exactly two, strip trailing whitespace and
/// guarantee a single terminating newline (mirrors TS /\n{3,}/g + trimEnd).
fn finish(out: *Buf) void {
    var w: usize = 0;
    var run: usize = 0;
    var r: usize = 0;
    while (r < out.len) : (r += 1) {
        if (out.data[r] == '\n') {
            run += 1;
            if (run <= 2) {
                out.data[w] = '\n';
                w += 1;
            }
        } else {
            run = 0;
            out.data[w] = out.data[r];
            w += 1;
        }
    }
    while (w > 0 and std.ascii.isWhitespace(out.data[w - 1])) w -= 1;
    out.data[w] = '\n';
    out.len = w + 1;
}

pub fn generateRobots(cfg: RobotsConfig, out: *Buf) void {
    for (cfg.groups) |g| {
        // cleanAgents in the TS lib: trim + drop blanks. A non-empty list
        // that trims away entirely degrades to {"*"}; an empty list means
        // "no group yet".
        var emitted: usize = 0;
        for (g.user_agents) |u| {
            const ua = trim(u);
            if (ua.len == 0) continue;
            out.add("User-agent: ");
            out.add(ua);
            out.add("\n");
            emitted += 1;
        }
        if (emitted == 0) {
            if (g.user_agents.len > 0) {
                out.add("User-agent: *\n");
            } else {
                continue; // no group yet
            }
        }
        for (g.allow) |a| { // blanks carry no meaning
            const p = trim(a);
            if (p.len > 0) {
                out.add("Allow: ");
                out.add(p);
                out.add("\n");
            }
        }
        if (g.disallow.len == 0) {
            out.add("Disallow:\n"); // no entries -> the allow-all marker
        } else {
            for (g.disallow) |d| { // "" kept verbatim
                out.add("Disallow: ");
                out.add(d);
                out.add("\n");
            }
        }
        if (g.crawl_delay) |cd| {
            if (std.math.isFinite(cd)) {
                out.add("Crawl-delay: ");
                if (cd == @trunc(cd)) { // JS-style number: 10.0 -> "10"
                    out.fmt("{d}", .{@as(i64, @intFromFloat(cd))});
                } else {
                    out.fmt("{d}", .{cd});
                }
                out.add("\n");
            }
        }
        out.add("\n"); // blank line separates groups
    }
    for (cfg.sitemaps) |s| {
        const url = trim(s);
        if (url.len > 0) {
            out.add("Sitemap: ");
            out.add(url);
            out.add("\n");
        }
    }
    finish(out);
}

/// ParseRobots in the Go twin: strip '#'-comments, split on the FIRST ':'
/// (values may contain colons), unknown directives ignored. Garbage
/// Crawl-delay values coerce to NaN like JS Number() — never re-emitted.
/// Rule values are copied into `scratch`; user-agent tokens stay as slices
/// into `text`, so both must outlive the `Parsed` result.
pub fn parseRobots(text: []const u8, scratch: []u8, out: *Parsed) void {
    var per_ua: [MAX_AGENTS]Group = undefined;
    var n_ua: usize = 0;
    var stack: [MAX_AGENTS][]const u8 = undefined;
    var n_stack: usize = 0;
    var cur: [MAX_AGENTS]usize = undefined;
    var n_cur: usize = 0;
    var sp: usize = 0; // scratch cursor
    out.n_groups = 0;
    out.n_sitemaps = 0;

    var it = std.mem.splitScalar(u8, text, '\n');
    while (it.next()) |raw| {
        const no_comment = if (std.mem.indexOfScalar(u8, raw, '#')) |h| raw[0..h] else raw;
        const line = trim(no_comment);
        if (line.len == 0) continue;
        const colon = std.mem.indexOfScalar(u8, line, ':') orelse continue;
        const field = trim(line[0..colon]);
        const value = trim(line[colon + 1 ..]);

        if (std.ascii.eqlIgnoreCase(field, "user-agent")) {
            const ua = if (value.len == 0) "*" else value;
            var dup = false;
            for (stack[0..n_stack]) |s| {
                if (std.mem.eql(u8, s, ua)) {
                    dup = true;
                    break;
                }
            }
            if (!dup and n_stack < MAX_AGENTS) {
                stack[n_stack] = ua;
                n_stack += 1;
            }
        } else if (std.ascii.eqlIgnoreCase(field, "disallow")) {
            flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
            n_stack = 0;
            const v = dupe(scratch, &sp, value);
            for (cur[0..n_cur]) |i| {
                if (per_ua[i].n_disallow < MAX_RULES) {
                    per_ua[i].disallow[per_ua[i].n_disallow] = v;
                    per_ua[i].n_disallow += 1;
                }
            }
        } else if (std.ascii.eqlIgnoreCase(field, "allow")) {
            flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
            n_stack = 0;
            const v = dupe(scratch, &sp, value);
            for (cur[0..n_cur]) |i| {
                if (per_ua[i].n_allow < MAX_RULES) {
                    per_ua[i].allow[per_ua[i].n_allow] = v;
                    per_ua[i].n_allow += 1;
                }
            }
        } else if (std.ascii.eqlIgnoreCase(field, "crawl-delay")) {
            flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
            n_stack = 0;
            const v: f64 = std.fmt.parseFloat(f64, value) catch std.math.nan(f64);
            for (cur[0..n_cur]) |i| per_ua[i].crawl_delay = v;
        } else if (std.ascii.eqlIgnoreCase(field, "sitemap")) {
            flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
            n_stack = 0;
            if (out.n_sitemaps < MAX_RULES) {
                out.sitemaps[out.n_sitemaps] = dupe(scratch, &sp, value);
                out.n_sitemaps += 1;
            }
        }
    }

    // Merge agents with identical rule sets into one stacked group.
    for (per_ua[0..n_ua]) |*acc| {
        var twin: usize = out.n_groups;
        for (out.groups[0..out.n_groups], 0..) |*g, gi| {
            if (rulesEq(g, acc)) {
                twin = gi;
                break;
            }
        }
        if (twin < out.n_groups) {
            const g = &out.groups[twin];
            if (g.n_user_agents < MAX_AGENTS) {
                g.user_agents[g.n_user_agents] = acc.user_agents[0];
                g.n_user_agents += 1;
            }
        } else if (out.n_groups < MAX_GROUPS) {
            const g = &out.groups[out.n_groups];
            g.* = .{ .n_user_agents = 1 };
            g.user_agents[0] = acc.user_agents[0];
            @memcpy(g.disallow[0..acc.n_disallow], acc.disallow[0..acc.n_disallow]);
            g.n_disallow = acc.n_disallow;
            @memcpy(g.allow[0..acc.n_allow], acc.allow[0..acc.n_allow]);
            g.n_allow = acc.n_allow;
            g.crawl_delay = acc.crawl_delay;
            out.n_groups += 1;
        }
    }
}

pub fn main() !void {
    const stdout = std.io.getStdOut().writer();
    // Sample config -> generated body -> parse round-trip check (tool page behavior).
    const config = RobotsConfig{
        .groups = &.{
            .{ .user_agents = &.{"*"}, .disallow = &.{ "/private/", "/admin/" }, .allow = &.{"/admin/public/"} },
            .{ .user_agents = &.{ "GPTBot", "CCBot" }, .disallow = &.{"/"}, .crawl_delay = 10 },
        },
        .sitemaps = &.{"https://example.com/sitemap.xml"},
    };
    var body: Buf = .{};
    generateRobots(config, &body);
    try stdout.print("{s}", .{body.slice()});

    var scratch: [8192]u8 = undefined;
    var parsed: Parsed = .{};
    parseRobots(body.slice(), &scratch, &parsed);
    var groups: [MAX_GROUPS]RuleGroup = undefined;
    for (parsed.groups[0..parsed.n_groups], 0..) |*pg, i| {
        groups[i] = .{
            .user_agents = pg.user_agents[0..pg.n_user_agents],
            .disallow = pg.disallow[0..pg.n_disallow],
            .allow = pg.allow[0..pg.n_allow],
            .crawl_delay = pg.crawl_delay,
        };
    }
    var again: Buf = .{};
    generateRobots(.{ .groups = groups[0..parsed.n_groups], .sitemaps = parsed.sitemaps[0..parsed.n_sitemaps] }, &again);
    try stdout.print("round-trip stable: {}\n", .{std.mem.eql(u8, again.slice(), body.slice())});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →