Skip to content

robots.txt Generator — Rust source

Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! robots-txt-generator — standards-compliant robots.txt generator + parser.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the robots-txt-generator tool,
//!           ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (public API returns owned Strings).
//!   - Functionally equivalent to the TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no crates.io dependencies — no `regex`).
//!
//! The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
//! Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body
//! back into the same config shape, aggregating repeated User-agent blocks.

use std::collections::HashMap;

/// One set of directives scoped to a single user-agent token.
#[derive(Debug, Clone, Default, PartialEq)]
pub struct RuleGroup {
    /// `*` or a specific bot token ("Googlebot", ...).
    pub user_agent: String,
    /// Paths to disallow. An empty element renders as a bare `Disallow:`.
    pub disallow: Vec<String>,
    /// Paths to allow.
    pub allow: Vec<String>,
    /// Seconds between requests, if set. `None` means unset.
    pub crawl_delay: Option<f64>,
}

/// Full document: ordered rule groups + sitemap URLs.
#[derive(Debug, Clone, Default, PartialEq)]
pub struct RobotsConfig {
    pub groups: Vec<RuleGroup>,
    pub sitemaps: Vec<String>,
}

/// Reproduce ECMAScript `Number()` coercion for the values a robots.txt field
/// can hold: `""` -> 0.0, "Infinity"/"+Infinity"/"-Infinity" -> +/-inf, any
/// parseable float -> the value, anything else -> NaN. We need this so the
/// generator's `is_finite()` filter sees exactly what the TS would have stored;
/// Rust's `str::parse::<f64>()` returns `Err` (not NaN) on garbage.
fn js_number(s: &str) -> f64 {
    match s {
        "" => 0.0,
        "Infinity" | "+Infinity" => f64::INFINITY,
        "-Infinity" => f64::NEG_INFINITY,
        _ => s.parse::<f64>().unwrap_or(f64::NAN),
    }
}

/// Format an f64 the way an ECMAScript template literal would, so generated
/// 'Crawl-delay:' values match the TS byte-for-byte: 5.0 -> "5", 5.5 -> "5.5",
/// 0.0 -> "0". Rust's Display for f64 already strips a trailing ".0"; the
/// zero guard also normalizes -0.0 to "0" (String(-0) === "0" in JS).
fn num_to_string(f: f64) -> String {
    if f == 0.0 {
        // covers +0.0 and -0.0
        return "0".to_string();
    }
    f.to_string()
}

/// Remove an inline `#...` comment (everything from the first '#' to end of
/// line), mirroring the TS `/#.*$/` regex. Returns a slice into `line`.
fn strip_comment(line: &str) -> &str {
    match line.find('#') {
        Some(i) => &line[..i],
        None => line,
    }
}

/// Fold runs of 3+ consecutive '\n' down to exactly two — the TS source does
/// this with `/\n{3,}/g`, but Rust's stdlib has no regex (and the `regex`
/// crate is external), so we do a single byte pass. Operating on bytes is safe
/// here because '\n' is a single-byte ASCII char and never appears inside a
/// multi-byte UTF-8 sequence.
fn collapse_newlines(input: &str) -> String {
    let bytes = input.as_bytes();
    let mut out: Vec<u8> = Vec::with_capacity(bytes.len());
    let mut run = 0usize;
    for &b in bytes {
        if b == b'\n' {
            run += 1;
            if run <= 2 {
                out.push(b); // keep the first two newlines of any run
            }
            // else: already have two, drop the extras
        } else {
            run = 0;
            out.push(b);
        }
    }
    // Safe: we only dropped '\n' bytes, so no UTF-8 sequence was split.
    String::from_utf8(out).unwrap_or_else(|_| input.to_string())
}

/// Build a robots.txt body from a config. Never panics; it silently drops
/// malformed pieces (whitespace-only user-agent tokens, blank sitemaps).
pub fn generate_robots(cfg: &RobotsConfig) -> String {
    let mut out: Vec<String> = Vec::new();

    for g in &cfg.groups {
        // TS: `(g.userAgent || '*').trim()`. The empty String plays the role
        // of null/undefined/"" here, so we default THEN trim, then skip if
        // still empty (whitespace-only UA => dropped group).
        let raw = if g.user_agent.is_empty() {
            "*".to_string()
        } else {
            g.user_agent.clone()
        };
        let ua = raw.trim();
        if ua.is_empty() {
            continue;
        }
        out.push(format!("User-agent: {}", ua));

        // Allow: lines — skip blanks (a bare 'Allow:' carries no meaning).
        for a in &g.allow {
            let p = a.trim();
            if !p.is_empty() {
                out.push(format!("Allow: {}", p));
            }
        }

        // Disallow semantics:
        //  - no entries => single bare 'Disallow:' (the allow-all marker)
        //  - otherwise one line per entry, EMPTIES PRESERVED VERBATIM
        //    (an empty entry becomes 'Disallow: ' with a trailing space —
        //    matches the TS byte-for-byte; the path is not re-trimmed).
        if g.disallow.is_empty() {
            out.push("Disallow:".to_string());
        } else {
            for d in &g.disallow {
                out.push(format!("Disallow: {}", d));
            }
        }

        if let Some(cd) = g.crawl_delay {
            if cd.is_finite() {
                out.push(format!("Crawl-delay: {}", num_to_string(cd)));
            }
        }

        out.push(String::new()); // blank line separates groups
    }

    for s in &cfg.sitemaps {
        let url = s.trim();
        if !url.is_empty() {
            out.push(format!("Sitemap: {}", url));
        }
    }

    let joined = collapse_newlines(&out.join("\n"));
    // trimEnd(): strip trailing whitespace, guarantee a single terminating '\n'.
    format!("{}\n", joined.trim_end())
}

/// Parse a robots.txt body into a config. Unknown directives are ignored.
/// Repeated User-agent tokens aggregate into one group; groups appear in
/// first-seen order.
pub fn parse_robots(text: &str) -> RobotsConfig {
    let mut groups: Vec<RuleGroup> = Vec::new();
    // Map key is an owned String so it can outlive the borrow of `value` and
    // reference the stored group's user_agent. Cost is one allocation per UA.
    let mut by_ua: HashMap<String, usize> = HashMap::new();
    let mut sitemaps: Vec<String> = Vec::new();
    let mut current_idx: Option<usize> = None;

    for raw_line in text.split('\n') {
        // Strip an inline comment (# to end of line) and trim whitespace.
        let line = strip_comment(raw_line).trim();
        if line.is_empty() {
            continue;
        }

        // Find the FIRST ':' — values may themselves contain colons
        // (e.g. 'Disallow: http://...'), so we only split on the first one.
        let Some(colon) = line.find(':') else {
            continue;
        };
        let field = line[..colon].trim().to_ascii_lowercase();
        let value = line[colon + 1..].trim().to_string();

        match field.as_str() {
            "user-agent" => {
                let ua = if value.is_empty() { "*".to_string() } else { value };
                if let Some(&idx) = by_ua.get(&ua) {
                    current_idx = Some(idx);
                } else {
                    let idx = groups.len();
                    by_ua.insert(ua.clone(), idx);
                    groups.push(RuleGroup {
                        user_agent: ua,
                        disallow: Vec::new(),
                        allow: Vec::new(),
                        crawl_delay: None,
                    });
                    current_idx = Some(idx);
                }
            }
            "disallow" => {
                if let Some(i) = current_idx {
                    groups[i].disallow.push(value);
                }
            }
            "allow" => {
                if let Some(i) = current_idx {
                    groups[i].allow.push(value);
                }
            }
            "crawl-delay" => {
                if let Some(i) = current_idx {
                    groups[i].crawl_delay = Some(js_number(&value));
                }
            }
            "sitemap" => sitemaps.push(value),
            _ => {} // unknown directive — ignore, matching TS
        }
    }

    RobotsConfig { groups, sitemaps }
}

// ---------- showcase-only tests (the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn round_trip_simple() {
        let cfg = RobotsConfig {
            groups: vec![RuleGroup {
                user_agent: "*".into(),
                disallow: vec!["/private".into()],
                allow: vec!["/public".into()],
                crawl_delay: Some(5.0),
                ..Default::default()
            }],
            sitemaps: vec!["https://example.com/sitemap.xml".into()],
        };
        let txt = generate_robots(&cfg);
        assert!(txt.contains("User-agent: *"));
        assert!(txt.contains("Disallow: /private"));
        assert!(txt.contains("Allow: /public"));
        assert!(txt.contains("Crawl-delay: 5"));
        assert!(txt.contains("Sitemap: https://example.com/sitemap.xml"));
        assert!(txt.ends_with('\n'));
    }

    #[test]
    fn parse_aggregates_user_agents() {
        let txt = "User-agent: A\nDisallow: /a\nUser-agent: A\nDisallow: /b\n";
        let cfg = parse_robots(txt);
        assert_eq!(cfg.groups.len(), 1);
        assert_eq!(cfg.groups[0].disallow, vec!["/a".to_string(), "/b".into()]);
    }

    #[test]
    fn parse_strips_comments_and_case() {
        let txt = "User-agent: GoogleBot  # the bot\nCrawl-Delay: 10\n";
        let cfg = parse_robots(txt);
        assert_eq!(cfg.groups[0].user_agent, "GoogleBot");
        assert_eq!(cfg.groups[0].crawl_delay, Some(10.0));
    }

    #[test]
    fn empty_disallow_means_allow_all() {
        let cfg = RobotsConfig {
            groups: vec![RuleGroup {
                user_agent: "*".into(),
                ..Default::default()
            }],
            sitemaps: vec![],
        };
        assert!(generate_robots(&cfg).contains("Disallow:\n"));
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →