robots.txt Generator — Zig source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// robots-txt-generator — Zig port: standards-compliant robots.txt generator + parser.
//
// Display port of the CosmoDev Robots.txt Generator tool — same contract as
// cli/robots-txt-generator/robots-txt-generator.go (the live Go twin) and
// src/lib/robotsTxt.ts (canonical TypeScript). Pure + deterministic; stdlib
// only. A group carries a LIST of user-agents (the REP spec's stacked
// User-agent lines); parsing aggregates repeated blocks per agent, then
// re-merges agents whose accumulated rules are identical into one stacked
// group, exactly like the TS lib. Fixed-capacity demo storage: 8 groups,
// 12 agents, 16 rules/group, 8 KB body/scratch.
const std = @import("std");
const MAX_AGENTS = 12;
const MAX_RULES = 16;
const MAX_GROUPS = 8;
/// One REP group — mirrors `RuleGroup` in the Go twin. Slices point into
/// caller-owned storage (comptime literals for the demo, scratch for parsing).
pub const RuleGroup = struct {
user_agents: []const []const u8, // {"*"} or {"Googlebot", "Bingbot"}
disallow: []const []const u8 = &.{}, // "" renders as a bare "Disallow:"
allow: []const []const u8 = &.{},
crawl_delay: ?f64 = null, // null = unset; non-finite never emitted
};
pub const RobotsConfig = struct {
groups: []const RuleGroup,
sitemaps: []const []const u8 = &.{},
};
/// Parse output: fixed-capacity storage so no allocator is needed. `Group`
/// doubles as the per-agent accumulator while parsing (one agent each) and as
/// the merged stacked-agent group afterwards.
const Group = struct {
user_agents: [MAX_AGENTS][]const u8 = undefined,
n_user_agents: usize = 0,
disallow: [MAX_RULES][]const u8 = undefined,
n_disallow: usize = 0,
allow: [MAX_RULES][]const u8 = undefined,
n_allow: usize = 0,
crawl_delay: ?f64 = null,
};
pub const Parsed = struct {
groups: [MAX_GROUPS]Group = undefined,
n_groups: usize = 0,
sitemaps: [MAX_RULES][]const u8 = undefined,
n_sitemaps: usize = 0,
};
/// Growable-but-capped output body (fixed 8 KB; writes past the cap truncate).
const Buf = struct {
data: [8192]u8 = undefined,
len: usize = 0,
fn add(self: *Buf, s: []const u8) void {
if (self.len >= self.data.len) return;
const n = @min(s.len, self.data.len - self.len);
@memcpy(self.data[self.len..][0..n], s[0..n]);
self.len += n;
}
fn fmt(self: *Buf, comptime f: []const u8, args: anytype) void {
var tmp: [64]u8 = undefined;
const s = std.fmt.bufPrint(&tmp, f, args) catch return;
self.add(s);
}
fn slice(self: *const Buf) []const u8 {
return self.data[0..self.len];
}
};
fn trim(s: []const u8) []const u8 {
return std.mem.trim(u8, s, " \t\r\n");
}
/// Normalize a Crawl-delay for the merge comparison: non-finite counts as
/// unset, exactly like the TS key (JSON.stringify turns NaN into null).
fn normCd(cd: ?f64) ?f64 {
if (cd) |v| {
if (!std.math.isFinite(v)) return null;
}
return cd;
}
fn rulesEq(a: *const Group, b: *const Group) bool {
if (a.n_disallow != b.n_disallow or a.n_allow != b.n_allow) return false;
for (a.disallow[0..a.n_disallow], b.disallow[0..b.n_disallow]) |x, y| {
if (!std.mem.eql(u8, x, y)) return false;
}
for (a.allow[0..a.n_allow], b.allow[0..b.n_allow]) |x, y| {
if (!std.mem.eql(u8, x, y)) return false;
}
return normCd(a.crawl_delay) == normCd(b.crawl_delay);
}
/// Move the pending User-agent stack into per-agent rule buckets (Google's
/// merge semantics: a rule line applies to every agent of the immediately
/// preceding consecutive User-agent stack).
fn flushStack(per_ua: *[MAX_AGENTS]Group, n_ua: *usize, stack: []const []const u8, cur: *[MAX_AGENTS]usize, n_cur: *usize) void {
if (stack.len == 0) return;
n_cur.* = 0;
for (stack) |ua| {
var i: usize = 0;
while (i < n_ua.*) : (i += 1) {
if (std.mem.eql(u8, per_ua[i].user_agents[0], ua)) break;
}
if (i == n_ua.*) {
per_ua[n_ua.*] = .{ .n_user_agents = 1 };
per_ua[n_ua.*].user_agents[0] = ua;
n_ua.* += 1;
}
cur[n_cur.*] = i;
n_cur.* += 1;
}
}
/// Copy `s` into the scratch buffer and return the copy (fixed-capacity).
fn dupe(sc: []u8, cursor: *usize, s: []const u8) []const u8 {
const n = @min(s.len, sc.len - cursor.*);
@memcpy(sc[cursor.*..][0..n], s[0..n]);
cursor.* += n;
return sc[cursor.* - n .. cursor.*];
}
/// Fold 3+ consecutive newlines to exactly two, strip trailing whitespace and
/// guarantee a single terminating newline (mirrors TS /\n{3,}/g + trimEnd).
fn finish(out: *Buf) void {
var w: usize = 0;
var run: usize = 0;
var r: usize = 0;
while (r < out.len) : (r += 1) {
if (out.data[r] == '\n') {
run += 1;
if (run <= 2) {
out.data[w] = '\n';
w += 1;
}
} else {
run = 0;
out.data[w] = out.data[r];
w += 1;
}
}
while (w > 0 and std.ascii.isWhitespace(out.data[w - 1])) w -= 1;
out.data[w] = '\n';
out.len = w + 1;
}
pub fn generateRobots(cfg: RobotsConfig, out: *Buf) void {
for (cfg.groups) |g| {
// cleanAgents in the TS lib: trim + drop blanks. A non-empty list
// that trims away entirely degrades to {"*"}; an empty list means
// "no group yet".
var emitted: usize = 0;
for (g.user_agents) |u| {
const ua = trim(u);
if (ua.len == 0) continue;
out.add("User-agent: ");
out.add(ua);
out.add("\n");
emitted += 1;
}
if (emitted == 0) {
if (g.user_agents.len > 0) {
out.add("User-agent: *\n");
} else {
continue; // no group yet
}
}
for (g.allow) |a| { // blanks carry no meaning
const p = trim(a);
if (p.len > 0) {
out.add("Allow: ");
out.add(p);
out.add("\n");
}
}
if (g.disallow.len == 0) {
out.add("Disallow:\n"); // no entries -> the allow-all marker
} else {
for (g.disallow) |d| { // "" kept verbatim
out.add("Disallow: ");
out.add(d);
out.add("\n");
}
}
if (g.crawl_delay) |cd| {
if (std.math.isFinite(cd)) {
out.add("Crawl-delay: ");
if (cd == @trunc(cd)) { // JS-style number: 10.0 -> "10"
out.fmt("{d}", .{@as(i64, @intFromFloat(cd))});
} else {
out.fmt("{d}", .{cd});
}
out.add("\n");
}
}
out.add("\n"); // blank line separates groups
}
for (cfg.sitemaps) |s| {
const url = trim(s);
if (url.len > 0) {
out.add("Sitemap: ");
out.add(url);
out.add("\n");
}
}
finish(out);
}
/// ParseRobots in the Go twin: strip '#'-comments, split on the FIRST ':'
/// (values may contain colons), unknown directives ignored. Garbage
/// Crawl-delay values coerce to NaN like JS Number() — never re-emitted.
/// Rule values are copied into `scratch`; user-agent tokens stay as slices
/// into `text`, so both must outlive the `Parsed` result.
pub fn parseRobots(text: []const u8, scratch: []u8, out: *Parsed) void {
var per_ua: [MAX_AGENTS]Group = undefined;
var n_ua: usize = 0;
var stack: [MAX_AGENTS][]const u8 = undefined;
var n_stack: usize = 0;
var cur: [MAX_AGENTS]usize = undefined;
var n_cur: usize = 0;
var sp: usize = 0; // scratch cursor
out.n_groups = 0;
out.n_sitemaps = 0;
var it = std.mem.splitScalar(u8, text, '\n');
while (it.next()) |raw| {
const no_comment = if (std.mem.indexOfScalar(u8, raw, '#')) |h| raw[0..h] else raw;
const line = trim(no_comment);
if (line.len == 0) continue;
const colon = std.mem.indexOfScalar(u8, line, ':') orelse continue;
const field = trim(line[0..colon]);
const value = trim(line[colon + 1 ..]);
if (std.ascii.eqlIgnoreCase(field, "user-agent")) {
const ua = if (value.len == 0) "*" else value;
var dup = false;
for (stack[0..n_stack]) |s| {
if (std.mem.eql(u8, s, ua)) {
dup = true;
break;
}
}
if (!dup and n_stack < MAX_AGENTS) {
stack[n_stack] = ua;
n_stack += 1;
}
} else if (std.ascii.eqlIgnoreCase(field, "disallow")) {
flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
n_stack = 0;
const v = dupe(scratch, &sp, value);
for (cur[0..n_cur]) |i| {
if (per_ua[i].n_disallow < MAX_RULES) {
per_ua[i].disallow[per_ua[i].n_disallow] = v;
per_ua[i].n_disallow += 1;
}
}
} else if (std.ascii.eqlIgnoreCase(field, "allow")) {
flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
n_stack = 0;
const v = dupe(scratch, &sp, value);
for (cur[0..n_cur]) |i| {
if (per_ua[i].n_allow < MAX_RULES) {
per_ua[i].allow[per_ua[i].n_allow] = v;
per_ua[i].n_allow += 1;
}
}
} else if (std.ascii.eqlIgnoreCase(field, "crawl-delay")) {
flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
n_stack = 0;
const v: f64 = std.fmt.parseFloat(f64, value) catch std.math.nan(f64);
for (cur[0..n_cur]) |i| per_ua[i].crawl_delay = v;
} else if (std.ascii.eqlIgnoreCase(field, "sitemap")) {
flushStack(&per_ua, &n_ua, stack[0..n_stack], &cur, &n_cur);
n_stack = 0;
if (out.n_sitemaps < MAX_RULES) {
out.sitemaps[out.n_sitemaps] = dupe(scratch, &sp, value);
out.n_sitemaps += 1;
}
}
}
// Merge agents with identical rule sets into one stacked group.
for (per_ua[0..n_ua]) |*acc| {
var twin: usize = out.n_groups;
for (out.groups[0..out.n_groups], 0..) |*g, gi| {
if (rulesEq(g, acc)) {
twin = gi;
break;
}
}
if (twin < out.n_groups) {
const g = &out.groups[twin];
if (g.n_user_agents < MAX_AGENTS) {
g.user_agents[g.n_user_agents] = acc.user_agents[0];
g.n_user_agents += 1;
}
} else if (out.n_groups < MAX_GROUPS) {
const g = &out.groups[out.n_groups];
g.* = .{ .n_user_agents = 1 };
g.user_agents[0] = acc.user_agents[0];
@memcpy(g.disallow[0..acc.n_disallow], acc.disallow[0..acc.n_disallow]);
g.n_disallow = acc.n_disallow;
@memcpy(g.allow[0..acc.n_allow], acc.allow[0..acc.n_allow]);
g.n_allow = acc.n_allow;
g.crawl_delay = acc.crawl_delay;
out.n_groups += 1;
}
}
}
pub fn main() !void {
const stdout = std.io.getStdOut().writer();
// Sample config -> generated body -> parse round-trip check (tool page behavior).
const config = RobotsConfig{
.groups = &.{
.{ .user_agents = &.{"*"}, .disallow = &.{ "/private/", "/admin/" }, .allow = &.{"/admin/public/"} },
.{ .user_agents = &.{ "GPTBot", "CCBot" }, .disallow = &.{"/"}, .crawl_delay = 10 },
},
.sitemaps = &.{"https://example.com/sitemap.xml"},
};
var body: Buf = .{};
generateRobots(config, &body);
try stdout.print("{s}", .{body.slice()});
var scratch: [8192]u8 = undefined;
var parsed: Parsed = .{};
parseRobots(body.slice(), &scratch, &parsed);
var groups: [MAX_GROUPS]RuleGroup = undefined;
for (parsed.groups[0..parsed.n_groups], 0..) |*pg, i| {
groups[i] = .{
.user_agents = pg.user_agents[0..pg.n_user_agents],
.disallow = pg.disallow[0..pg.n_disallow],
.allow = pg.allow[0..pg.n_allow],
.crawl_delay = pg.crawl_delay,
};
}
var again: Buf = .{};
generateRobots(.{ .groups = groups[0..parsed.n_groups], .sitemaps = parsed.sitemaps[0..parsed.n_sitemaps] }, &again);
try stdout.print("round-trip stable: {}\n", .{std.mem.eql(u8, again.slice(), body.slice())});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →