robots.txt Generator — Rust source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! robots-txt-generator — standards-compliant robots.txt generator + parser.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the robots-txt-generator tool,
//! ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (public API returns owned Strings).
//! - Functionally equivalent to the TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no crates.io dependencies — no `regex`).
//!
//! The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
//! Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body
//! back into the same config shape, aggregating repeated User-agent blocks.
use std::collections::HashMap;
/// One set of directives scoped to a single user-agent token.
#[derive(Debug, Clone, Default, PartialEq)]
pub struct RuleGroup {
/// `*` or a specific bot token ("Googlebot", ...).
pub user_agent: String,
/// Paths to disallow. An empty element renders as a bare `Disallow:`.
pub disallow: Vec<String>,
/// Paths to allow.
pub allow: Vec<String>,
/// Seconds between requests, if set. `None` means unset.
pub crawl_delay: Option<f64>,
}
/// Full document: ordered rule groups + sitemap URLs.
#[derive(Debug, Clone, Default, PartialEq)]
pub struct RobotsConfig {
pub groups: Vec<RuleGroup>,
pub sitemaps: Vec<String>,
}
/// Reproduce ECMAScript `Number()` coercion for the values a robots.txt field
/// can hold: `""` -> 0.0, "Infinity"/"+Infinity"/"-Infinity" -> +/-inf, any
/// parseable float -> the value, anything else -> NaN. We need this so the
/// generator's `is_finite()` filter sees exactly what the TS would have stored;
/// Rust's `str::parse::<f64>()` returns `Err` (not NaN) on garbage.
fn js_number(s: &str) -> f64 {
match s {
"" => 0.0,
"Infinity" | "+Infinity" => f64::INFINITY,
"-Infinity" => f64::NEG_INFINITY,
_ => s.parse::<f64>().unwrap_or(f64::NAN),
}
}
/// Format an f64 the way an ECMAScript template literal would, so generated
/// 'Crawl-delay:' values match the TS byte-for-byte: 5.0 -> "5", 5.5 -> "5.5",
/// 0.0 -> "0". Rust's Display for f64 already strips a trailing ".0"; the
/// zero guard also normalizes -0.0 to "0" (String(-0) === "0" in JS).
fn num_to_string(f: f64) -> String {
if f == 0.0 {
// covers +0.0 and -0.0
return "0".to_string();
}
f.to_string()
}
/// Remove an inline `#...` comment (everything from the first '#' to end of
/// line), mirroring the TS `/#.*$/` regex. Returns a slice into `line`.
fn strip_comment(line: &str) -> &str {
match line.find('#') {
Some(i) => &line[..i],
None => line,
}
}
/// Fold runs of 3+ consecutive '\n' down to exactly two — the TS source does
/// this with `/\n{3,}/g`, but Rust's stdlib has no regex (and the `regex`
/// crate is external), so we do a single byte pass. Operating on bytes is safe
/// here because '\n' is a single-byte ASCII char and never appears inside a
/// multi-byte UTF-8 sequence.
fn collapse_newlines(input: &str) -> String {
let bytes = input.as_bytes();
let mut out: Vec<u8> = Vec::with_capacity(bytes.len());
let mut run = 0usize;
for &b in bytes {
if b == b'\n' {
run += 1;
if run <= 2 {
out.push(b); // keep the first two newlines of any run
}
// else: already have two, drop the extras
} else {
run = 0;
out.push(b);
}
}
// Safe: we only dropped '\n' bytes, so no UTF-8 sequence was split.
String::from_utf8(out).unwrap_or_else(|_| input.to_string())
}
/// Build a robots.txt body from a config. Never panics; it silently drops
/// malformed pieces (whitespace-only user-agent tokens, blank sitemaps).
pub fn generate_robots(cfg: &RobotsConfig) -> String {
let mut out: Vec<String> = Vec::new();
for g in &cfg.groups {
// TS: `(g.userAgent || '*').trim()`. The empty String plays the role
// of null/undefined/"" here, so we default THEN trim, then skip if
// still empty (whitespace-only UA => dropped group).
let raw = if g.user_agent.is_empty() {
"*".to_string()
} else {
g.user_agent.clone()
};
let ua = raw.trim();
if ua.is_empty() {
continue;
}
out.push(format!("User-agent: {}", ua));
// Allow: lines — skip blanks (a bare 'Allow:' carries no meaning).
for a in &g.allow {
let p = a.trim();
if !p.is_empty() {
out.push(format!("Allow: {}", p));
}
}
// Disallow semantics:
// - no entries => single bare 'Disallow:' (the allow-all marker)
// - otherwise one line per entry, EMPTIES PRESERVED VERBATIM
// (an empty entry becomes 'Disallow: ' with a trailing space —
// matches the TS byte-for-byte; the path is not re-trimmed).
if g.disallow.is_empty() {
out.push("Disallow:".to_string());
} else {
for d in &g.disallow {
out.push(format!("Disallow: {}", d));
}
}
if let Some(cd) = g.crawl_delay {
if cd.is_finite() {
out.push(format!("Crawl-delay: {}", num_to_string(cd)));
}
}
out.push(String::new()); // blank line separates groups
}
for s in &cfg.sitemaps {
let url = s.trim();
if !url.is_empty() {
out.push(format!("Sitemap: {}", url));
}
}
let joined = collapse_newlines(&out.join("\n"));
// trimEnd(): strip trailing whitespace, guarantee a single terminating '\n'.
format!("{}\n", joined.trim_end())
}
/// Parse a robots.txt body into a config. Unknown directives are ignored.
/// Repeated User-agent tokens aggregate into one group; groups appear in
/// first-seen order.
pub fn parse_robots(text: &str) -> RobotsConfig {
let mut groups: Vec<RuleGroup> = Vec::new();
// Map key is an owned String so it can outlive the borrow of `value` and
// reference the stored group's user_agent. Cost is one allocation per UA.
let mut by_ua: HashMap<String, usize> = HashMap::new();
let mut sitemaps: Vec<String> = Vec::new();
let mut current_idx: Option<usize> = None;
for raw_line in text.split('\n') {
// Strip an inline comment (# to end of line) and trim whitespace.
let line = strip_comment(raw_line).trim();
if line.is_empty() {
continue;
}
// Find the FIRST ':' — values may themselves contain colons
// (e.g. 'Disallow: http://...'), so we only split on the first one.
let Some(colon) = line.find(':') else {
continue;
};
let field = line[..colon].trim().to_ascii_lowercase();
let value = line[colon + 1..].trim().to_string();
match field.as_str() {
"user-agent" => {
let ua = if value.is_empty() { "*".to_string() } else { value };
if let Some(&idx) = by_ua.get(&ua) {
current_idx = Some(idx);
} else {
let idx = groups.len();
by_ua.insert(ua.clone(), idx);
groups.push(RuleGroup {
user_agent: ua,
disallow: Vec::new(),
allow: Vec::new(),
crawl_delay: None,
});
current_idx = Some(idx);
}
}
"disallow" => {
if let Some(i) = current_idx {
groups[i].disallow.push(value);
}
}
"allow" => {
if let Some(i) = current_idx {
groups[i].allow.push(value);
}
}
"crawl-delay" => {
if let Some(i) = current_idx {
groups[i].crawl_delay = Some(js_number(&value));
}
}
"sitemap" => sitemaps.push(value),
_ => {} // unknown directive — ignore, matching TS
}
}
RobotsConfig { groups, sitemaps }
}
// ---------- showcase-only tests (the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn round_trip_simple() {
let cfg = RobotsConfig {
groups: vec![RuleGroup {
user_agent: "*".into(),
disallow: vec!["/private".into()],
allow: vec!["/public".into()],
crawl_delay: Some(5.0),
..Default::default()
}],
sitemaps: vec!["https://example.com/sitemap.xml".into()],
};
let txt = generate_robots(&cfg);
assert!(txt.contains("User-agent: *"));
assert!(txt.contains("Disallow: /private"));
assert!(txt.contains("Allow: /public"));
assert!(txt.contains("Crawl-delay: 5"));
assert!(txt.contains("Sitemap: https://example.com/sitemap.xml"));
assert!(txt.ends_with('\n'));
}
#[test]
fn parse_aggregates_user_agents() {
let txt = "User-agent: A\nDisallow: /a\nUser-agent: A\nDisallow: /b\n";
let cfg = parse_robots(txt);
assert_eq!(cfg.groups.len(), 1);
assert_eq!(cfg.groups[0].disallow, vec!["/a".to_string(), "/b".into()]);
}
#[test]
fn parse_strips_comments_and_case() {
let txt = "User-agent: GoogleBot # the bot\nCrawl-Delay: 10\n";
let cfg = parse_robots(txt);
assert_eq!(cfg.groups[0].user_agent, "GoogleBot");
assert_eq!(cfg.groups[0].crawl_delay, Some(10.0));
}
#[test]
fn empty_disallow_means_allow_all() {
let cfg = RobotsConfig {
groups: vec![RuleGroup {
user_agent: "*".into(),
..Default::default()
}],
sitemaps: vec![],
};
assert!(generate_robots(&cfg).contains("Disallow:\n"));
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →