Skip to content

robots.txt Generator — JavaScript source

Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * robots-txt-generator - standards-compliant robots.txt generator + parser.
 *
 * Language:   JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
 * Source:     CosmoDev polyglot showcase port of the robots-txt-generator tool,
 *             ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
 * License:    display source - part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never throws on a well-formed config.
 *   - Functionally equivalent to the TS reference: same inputs -> same outputs.
 *   - Self-contained: stdlib only (no npm dependencies).
 *
 * The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
 * Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body back
 * into the same config shape, aggregating repeated User-agent blocks.
 */

'use strict';

/**
 * One rule group: directives scoped to a single user-agent token.
 * @typedef {Object} RuleGroup
 * @property {string} userAgent       '*' or a specific bot token ('Googlebot', ...).
 * @property {string[]} disallow      Paths to disallow. An '' entry => a bare 'Disallow:'.
 * @property {string[]} allow         Paths to allow.
 * @property {number|null} [crawlDelay] Seconds between requests, if set. null/undefined = unset.
 */

/**
 * @typedef {Object} RobotsConfig
 * @property {RuleGroup[]} groups
 * @property {string[]} sitemaps
 */

/**
 * Generate a robots.txt body from a config. Never throws on well-formed input.
 *
 * @param {RobotsConfig} config
 * @returns {string}
 */
function generateRobots(config) {
  const out = [];
  for (const g of config.groups) {
    // TS: `(g.userAgent || '*').trim()`. A missing/empty UA defaults to the
    // wildcard token; a UA that is still blank after trim drops the group.
    const ua = (g.userAgent || '*').trim();
    if (!ua) continue;
    out.push(`User-agent: ${ua}`);

    // Allow: lines - skip blanks (a bare 'Allow:' carries no meaning).
    for (const a of g.allow) {
      const p = a.trim();
      if (p) out.push(`Allow: ${p}`);
    }

    // Disallow handling:
    //  - no entries at all => emit a single bare 'Disallow:' (the allow-all marker)
    //  - otherwise emit one line per entry, preserving empties VERBATIM
    //    (an empty entry becomes 'Disallow: ' with a trailing space - this
    //    matches the TS byte-for-byte; the path is not re-trimmed).
    if (g.disallow.length === 0) {
      out.push('Disallow:');
    } else {
      for (const d of g.disallow) {
        out.push(`Disallow: ${d}`);
      }
    }

    // Crawl-delay only when explicitly set to a finite number (NaN/Infinity
    // from a malformed parse round-trip are dropped, never re-emitted).
    if (g.crawlDelay !== undefined && g.crawlDelay !== null && Number.isFinite(g.crawlDelay)) {
      out.push(`Crawl-delay: ${g.crawlDelay}`);
    }

    out.push(''); // blank line separates groups
  }

  for (const s of config.sitemaps) {
    const url = s.trim();
    if (url) out.push(`Sitemap: ${url}`);
  }

  // Collapse 3+ consecutive newlines to exactly two (defensive normalization
  // for blank-line runs), strip trailing whitespace, and guarantee a single
  // terminating newline.
  return out.join('\n').replace(/\n{3,}/g, '\n\n').trimEnd() + '\n';
}

/**
 * Parse a robots.txt body into a config. Unknown directives are ignored.
 * Repeated User-agent tokens aggregate into one group; groups are returned in
 * first-seen order (a Map preserves insertion order).
 *
 * @param {string} [text]
 * @returns {RobotsConfig}
 */
function parseRobots(text) {
  /** @type {RobotsConfig} */
  const config = { groups: [], sitemaps: [] };
  /** Insertion-ordered map of user-agent -> group. @type {Map<string, RuleGroup>} */
  const map = new Map();
  /** @type {RuleGroup | null} */
  let current = null;

  // `text ?? ''` defends against null/undefined the same way the TS source does.
  for (const rawLine of (text ?? '').split('\n')) {
    // Strip an inline comment (# to end of line) and surrounding whitespace.
    const line = rawLine.replace(/#.*$/, '').trim();
    if (!line) continue;

    const colon = line.indexOf(':');
    if (colon === -1) continue; // no field/value split -> ignore the line

    const field = line.slice(0, colon).trim().toLowerCase();
    const value = line.slice(colon + 1).trim();

    if (field === 'user-agent') {
      const ua = value || '*';
      if (!map.has(ua)) {
        map.set(ua, { userAgent: ua, disallow: [], allow: [] });
      }
      // Switching the active group: subsequent Allow/Disallow/Crawl-delay
      // lines attach to THIS user-agent, so two consecutive User-agent lines
      // do NOT share their rules - they each get their own group.
      current = map.get(ua);
    } else if (field === 'disallow' && current) {
      current.disallow.push(value);
    } else if (field === 'allow' && current) {
      current.allow.push(value);
    } else if (field === 'crawl-delay' && current) {
      // Number('') === 0, Number('abc') === NaN - both reproduced verbatim so
      // the generator's isFinite() filter sees exactly the value TS would.
      current.crawlDelay = Number(value);
    } else if (field === 'sitemap') {
      config.sitemaps.push(value);
    }
  }

  config.groups = [...map.values()];
  return config;
}

// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { generateRobots, parseRobots };

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →