Skip to content

robots.txt Generator — Python source

Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""robots-txt-generator — standards-compliant robots.txt generator + parser.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the robots-txt-generator tool,
          ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises.
  - Functionally equivalent to the TS reference: same inputs -> same outputs.
  - Self-contained: stdlib only (no pip packages).

The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body back
into the same config shape, aggregating repeated User-agent blocks.
"""

from __future__ import annotations

import math
import re
from dataclasses import dataclass, field
from typing import List, Optional

__all__ = ["RuleGroup", "RobotsConfig", "generate_robots", "parse_robots"]


@dataclass
class RuleGroup:
    """One set of directives scoped to a single user-agent token.

    ``crawl_delay`` mirrors the TS optional: ``None`` means "unset" and emits
    no Crawl-delay line.
    """

    user_agent: str = ""
    """``'*'`` or a specific bot token ('Googlebot', ...)."""

    disallow: List[str] = field(default_factory=list)
    """Paths to disallow. An empty element renders as a bare ``Disallow:``."""

    allow: List[str] = field(default_factory=list)
    """Paths to allow."""

    crawl_delay: Optional[float] = None
    """Seconds between requests, or None when unset."""


@dataclass
class RobotsConfig:
    """Full document: ordered rule groups + sitemap URLs."""

    groups: List[RuleGroup] = field(default_factory=list)
    sitemaps: List[str] = field(default_factory=list)


# --- Pre-compiled patterns ---------------------------------------------------
# Each mirrors a regex literal from the TypeScript reference. Compiling once at
# import time is the idiomatic Python way to avoid per-call recompilation.

# Inline comment: '#' to end of line (mirrors TS `/#.*$/`).
_COMMENT_RE = re.compile(r"#.*$")
# A run of 3+ newlines, folded to exactly two (mirrors TS `/\n{3,}/g`).
_BLANK_RUN_RE = re.compile(r"\n{3,}")


def _js_number(s: str) -> float:
    """Reproduce ECMAScript ``Number()`` coercion for the values a robots.txt
    field can hold: ``''`` -> 0.0, ``'Infinity'``/``'+Infinity'``/``'-Infinity'``
    -> +/-inf, a parseable float -> the value, anything else -> NaN.

    We need this so the generator's ``isfinite()`` filter sees exactly what
    the TS would have stored; Python's ``float('abc')`` raises rather than
    returning NaN.
    """
    if s == "":
        return 0.0
    lowered = s.lower()
    if lowered in ("infinity", "+infinity"):
        return math.inf
    if lowered == "-infinity":
        return -math.inf
    try:
        return float(s)
    except ValueError:
        return math.nan


def _num_to_string(f: float) -> str:
    """Format a float the way an ECMAScript template literal would, so
    generated 'Crawl-delay:' values match the TS byte-for-byte: 5.0 -> '5',
    5.5 -> '5.5', 0.0 -> '0'.

    Whole floats lose the '.0' (via int conversion, which also normalizes
    -0.0 to '0' the way JS ``String(-0)`` does); everything else uses
    ``repr`` for the shortest round-trip representation.
    """
    if math.isinf(f) or math.isnan(f):
        return repr(f)
    if f == int(f):
        return str(int(f))
    return repr(f)


def generate_robots(config: RobotsConfig) -> str:
    """Build a robots.txt body from a config. Never raises; it silently drops
    malformed pieces (whitespace-only user-agent tokens, blank sitemaps)."""
    out: List[str] = []

    for g in config.groups:
        # TS: `(g.userAgent or '*').strip()`. A missing/empty UA defaults to
        # the wildcard token; a UA still blank after strip drops the group.
        raw = g.user_agent if g.user_agent else "*"
        ua = raw.strip()
        if not ua:
            continue
        out.append(f"User-agent: {ua}")

        # Allow: lines — skip blanks (a bare 'Allow:' carries no meaning).
        for a in g.allow:
            p = a.strip()
            if p:
                out.append(f"Allow: {p}")

        # Disallow semantics:
        #  - no entries -> single bare 'Disallow:' (the allow-all marker)
        #  - otherwise one line per entry, EMPTIES PRESERVED VERBATIM
        #    (an empty entry becomes 'Disallow: ' with a trailing space —
        #    matches the TS byte-for-byte; the path is not re-stripped).
        if len(g.disallow) == 0:
            out.append("Disallow:")
        else:
            for d in g.disallow:
                out.append(f"Disallow: {d}")

        # Crawl-delay only when explicitly set to a finite number (NaN/Infinity
        # from a malformed parse round-trip are dropped, never re-emitted).
        if g.crawl_delay is not None and math.isfinite(g.crawl_delay):
            out.append(f"Crawl-delay: {_num_to_string(g.crawl_delay)}")

        out.append("")  # blank line separates groups

    for s in config.sitemaps:
        url = s.strip()
        if url:
            out.append(f"Sitemap: {url}")

    # Collapse 3+ consecutive newlines to exactly two, strip trailing
    # whitespace, and guarantee a single terminating newline.
    joined = "\n".join(out)
    joined = _BLANK_RUN_RE.sub("\n\n", joined)
    return joined.rstrip() + "\n"


def parse_robots(text: Optional[str]) -> RobotsConfig:
    """Parse a robots.txt body into a config. Unknown directives are ignored.
    Repeated User-agent tokens aggregate into one group; groups appear in
    first-seen order (dict preserves insertion order since Python 3.7)."""
    config = RobotsConfig()
    by_ua: dict[str, int] = {}   # user-agent -> index into config.groups
    current_idx: Optional[int] = None

    # `text or ''` coerces None to '' the way TS `(text ?? '')` does.
    for raw_line in (text or "").split("\n"):
        # Strip an inline comment (# to end of line) and trim whitespace.
        line = _COMMENT_RE.sub("", raw_line).strip()
        if not line:
            continue

        # Find the FIRST ':' — values may themselves contain colons
        # (e.g. 'Disallow: http://...'), so we only split on the first one.
        colon = line.find(":")
        if colon == -1:
            continue

        field_name = line[:colon].strip().lower()
        value = line[colon + 1:].strip()

        if field_name == "user-agent":
            ua = value or "*"
            if ua in by_ua:
                current_idx = by_ua[ua]
            else:
                # Switching the active group: subsequent Allow/Disallow/
                # Crawl-delay lines attach to THIS user-agent, so two
                # consecutive User-agent lines do NOT share their rules —
                # they each get their own group.
                current_idx = len(config.groups)
                by_ua[ua] = current_idx
                config.groups.append(RuleGroup(
                    user_agent=ua,
                    disallow=[],
                    allow=[],
                ))
        elif field_name == "disallow" and current_idx is not None:
            config.groups[current_idx].disallow.append(value)
        elif field_name == "allow" and current_idx is not None:
            config.groups[current_idx].allow.append(value)
        elif field_name == "crawl-delay" and current_idx is not None:
            config.groups[current_idx].crawl_delay = _js_number(value)
        elif field_name == "sitemap":
            config.sitemaps.append(value)

    return config

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →