robots.txt Generator — Python source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""robots-txt-generator — standards-compliant robots.txt generator + parser.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the robots-txt-generator tool,
ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never raises.
- Functionally equivalent to the TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no pip packages).
The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body back
into the same config shape, aggregating repeated User-agent blocks.
"""
from __future__ import annotations
import math
import re
from dataclasses import dataclass, field
from typing import List, Optional
__all__ = ["RuleGroup", "RobotsConfig", "generate_robots", "parse_robots"]
@dataclass
class RuleGroup:
"""One set of directives scoped to a single user-agent token.
``crawl_delay`` mirrors the TS optional: ``None`` means "unset" and emits
no Crawl-delay line.
"""
user_agent: str = ""
"""``'*'`` or a specific bot token ('Googlebot', ...)."""
disallow: List[str] = field(default_factory=list)
"""Paths to disallow. An empty element renders as a bare ``Disallow:``."""
allow: List[str] = field(default_factory=list)
"""Paths to allow."""
crawl_delay: Optional[float] = None
"""Seconds between requests, or None when unset."""
@dataclass
class RobotsConfig:
"""Full document: ordered rule groups + sitemap URLs."""
groups: List[RuleGroup] = field(default_factory=list)
sitemaps: List[str] = field(default_factory=list)
# --- Pre-compiled patterns ---------------------------------------------------
# Each mirrors a regex literal from the TypeScript reference. Compiling once at
# import time is the idiomatic Python way to avoid per-call recompilation.
# Inline comment: '#' to end of line (mirrors TS `/#.*$/`).
_COMMENT_RE = re.compile(r"#.*$")
# A run of 3+ newlines, folded to exactly two (mirrors TS `/\n{3,}/g`).
_BLANK_RUN_RE = re.compile(r"\n{3,}")
def _js_number(s: str) -> float:
"""Reproduce ECMAScript ``Number()`` coercion for the values a robots.txt
field can hold: ``''`` -> 0.0, ``'Infinity'``/``'+Infinity'``/``'-Infinity'``
-> +/-inf, a parseable float -> the value, anything else -> NaN.
We need this so the generator's ``isfinite()`` filter sees exactly what
the TS would have stored; Python's ``float('abc')`` raises rather than
returning NaN.
"""
if s == "":
return 0.0
lowered = s.lower()
if lowered in ("infinity", "+infinity"):
return math.inf
if lowered == "-infinity":
return -math.inf
try:
return float(s)
except ValueError:
return math.nan
def _num_to_string(f: float) -> str:
"""Format a float the way an ECMAScript template literal would, so
generated 'Crawl-delay:' values match the TS byte-for-byte: 5.0 -> '5',
5.5 -> '5.5', 0.0 -> '0'.
Whole floats lose the '.0' (via int conversion, which also normalizes
-0.0 to '0' the way JS ``String(-0)`` does); everything else uses
``repr`` for the shortest round-trip representation.
"""
if math.isinf(f) or math.isnan(f):
return repr(f)
if f == int(f):
return str(int(f))
return repr(f)
def generate_robots(config: RobotsConfig) -> str:
"""Build a robots.txt body from a config. Never raises; it silently drops
malformed pieces (whitespace-only user-agent tokens, blank sitemaps)."""
out: List[str] = []
for g in config.groups:
# TS: `(g.userAgent or '*').strip()`. A missing/empty UA defaults to
# the wildcard token; a UA still blank after strip drops the group.
raw = g.user_agent if g.user_agent else "*"
ua = raw.strip()
if not ua:
continue
out.append(f"User-agent: {ua}")
# Allow: lines — skip blanks (a bare 'Allow:' carries no meaning).
for a in g.allow:
p = a.strip()
if p:
out.append(f"Allow: {p}")
# Disallow semantics:
# - no entries -> single bare 'Disallow:' (the allow-all marker)
# - otherwise one line per entry, EMPTIES PRESERVED VERBATIM
# (an empty entry becomes 'Disallow: ' with a trailing space —
# matches the TS byte-for-byte; the path is not re-stripped).
if len(g.disallow) == 0:
out.append("Disallow:")
else:
for d in g.disallow:
out.append(f"Disallow: {d}")
# Crawl-delay only when explicitly set to a finite number (NaN/Infinity
# from a malformed parse round-trip are dropped, never re-emitted).
if g.crawl_delay is not None and math.isfinite(g.crawl_delay):
out.append(f"Crawl-delay: {_num_to_string(g.crawl_delay)}")
out.append("") # blank line separates groups
for s in config.sitemaps:
url = s.strip()
if url:
out.append(f"Sitemap: {url}")
# Collapse 3+ consecutive newlines to exactly two, strip trailing
# whitespace, and guarantee a single terminating newline.
joined = "\n".join(out)
joined = _BLANK_RUN_RE.sub("\n\n", joined)
return joined.rstrip() + "\n"
def parse_robots(text: Optional[str]) -> RobotsConfig:
"""Parse a robots.txt body into a config. Unknown directives are ignored.
Repeated User-agent tokens aggregate into one group; groups appear in
first-seen order (dict preserves insertion order since Python 3.7)."""
config = RobotsConfig()
by_ua: dict[str, int] = {} # user-agent -> index into config.groups
current_idx: Optional[int] = None
# `text or ''` coerces None to '' the way TS `(text ?? '')` does.
for raw_line in (text or "").split("\n"):
# Strip an inline comment (# to end of line) and trim whitespace.
line = _COMMENT_RE.sub("", raw_line).strip()
if not line:
continue
# Find the FIRST ':' — values may themselves contain colons
# (e.g. 'Disallow: http://...'), so we only split on the first one.
colon = line.find(":")
if colon == -1:
continue
field_name = line[:colon].strip().lower()
value = line[colon + 1:].strip()
if field_name == "user-agent":
ua = value or "*"
if ua in by_ua:
current_idx = by_ua[ua]
else:
# Switching the active group: subsequent Allow/Disallow/
# Crawl-delay lines attach to THIS user-agent, so two
# consecutive User-agent lines do NOT share their rules —
# they each get their own group.
current_idx = len(config.groups)
by_ua[ua] = current_idx
config.groups.append(RuleGroup(
user_agent=ua,
disallow=[],
allow=[],
))
elif field_name == "disallow" and current_idx is not None:
config.groups[current_idx].disallow.append(value)
elif field_name == "allow" and current_idx is not None:
config.groups[current_idx].allow.append(value)
elif field_name == "crawl-delay" and current_idx is not None:
config.groups[current_idx].crawl_delay = _js_number(value)
elif field_name == "sitemap":
config.sitemaps.append(value)
return config
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →