robots.txt Generator — JavaScript source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* robots-txt-generator - standards-compliant robots.txt generator + parser.
*
* Language: JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
* Source: CosmoDev polyglot showcase port of the robots-txt-generator tool,
* ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
* License: display source - part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws on a well-formed config.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no npm dependencies).
*
* The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
* Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body back
* into the same config shape, aggregating repeated User-agent blocks.
*/
'use strict';
/**
* One rule group: directives scoped to a single user-agent token.
* @typedef {Object} RuleGroup
* @property {string} userAgent '*' or a specific bot token ('Googlebot', ...).
* @property {string[]} disallow Paths to disallow. An '' entry => a bare 'Disallow:'.
* @property {string[]} allow Paths to allow.
* @property {number|null} [crawlDelay] Seconds between requests, if set. null/undefined = unset.
*/
/**
* @typedef {Object} RobotsConfig
* @property {RuleGroup[]} groups
* @property {string[]} sitemaps
*/
/**
* Generate a robots.txt body from a config. Never throws on well-formed input.
*
* @param {RobotsConfig} config
* @returns {string}
*/
function generateRobots(config) {
const out = [];
for (const g of config.groups) {
// TS: `(g.userAgent || '*').trim()`. A missing/empty UA defaults to the
// wildcard token; a UA that is still blank after trim drops the group.
const ua = (g.userAgent || '*').trim();
if (!ua) continue;
out.push(`User-agent: ${ua}`);
// Allow: lines - skip blanks (a bare 'Allow:' carries no meaning).
for (const a of g.allow) {
const p = a.trim();
if (p) out.push(`Allow: ${p}`);
}
// Disallow handling:
// - no entries at all => emit a single bare 'Disallow:' (the allow-all marker)
// - otherwise emit one line per entry, preserving empties VERBATIM
// (an empty entry becomes 'Disallow: ' with a trailing space - this
// matches the TS byte-for-byte; the path is not re-trimmed).
if (g.disallow.length === 0) {
out.push('Disallow:');
} else {
for (const d of g.disallow) {
out.push(`Disallow: ${d}`);
}
}
// Crawl-delay only when explicitly set to a finite number (NaN/Infinity
// from a malformed parse round-trip are dropped, never re-emitted).
if (g.crawlDelay !== undefined && g.crawlDelay !== null && Number.isFinite(g.crawlDelay)) {
out.push(`Crawl-delay: ${g.crawlDelay}`);
}
out.push(''); // blank line separates groups
}
for (const s of config.sitemaps) {
const url = s.trim();
if (url) out.push(`Sitemap: ${url}`);
}
// Collapse 3+ consecutive newlines to exactly two (defensive normalization
// for blank-line runs), strip trailing whitespace, and guarantee a single
// terminating newline.
return out.join('\n').replace(/\n{3,}/g, '\n\n').trimEnd() + '\n';
}
/**
* Parse a robots.txt body into a config. Unknown directives are ignored.
* Repeated User-agent tokens aggregate into one group; groups are returned in
* first-seen order (a Map preserves insertion order).
*
* @param {string} [text]
* @returns {RobotsConfig}
*/
function parseRobots(text) {
/** @type {RobotsConfig} */
const config = { groups: [], sitemaps: [] };
/** Insertion-ordered map of user-agent -> group. @type {Map<string, RuleGroup>} */
const map = new Map();
/** @type {RuleGroup | null} */
let current = null;
// `text ?? ''` defends against null/undefined the same way the TS source does.
for (const rawLine of (text ?? '').split('\n')) {
// Strip an inline comment (# to end of line) and surrounding whitespace.
const line = rawLine.replace(/#.*$/, '').trim();
if (!line) continue;
const colon = line.indexOf(':');
if (colon === -1) continue; // no field/value split -> ignore the line
const field = line.slice(0, colon).trim().toLowerCase();
const value = line.slice(colon + 1).trim();
if (field === 'user-agent') {
const ua = value || '*';
if (!map.has(ua)) {
map.set(ua, { userAgent: ua, disallow: [], allow: [] });
}
// Switching the active group: subsequent Allow/Disallow/Crawl-delay
// lines attach to THIS user-agent, so two consecutive User-agent lines
// do NOT share their rules - they each get their own group.
current = map.get(ua);
} else if (field === 'disallow' && current) {
current.disallow.push(value);
} else if (field === 'allow' && current) {
current.allow.push(value);
} else if (field === 'crawl-delay' && current) {
// Number('') === 0, Number('abc') === NaN - both reproduced verbatim so
// the generator's isFinite() filter sees exactly the value TS would.
current.crawlDelay = Number(value);
} else if (field === 'sitemap') {
config.sitemaps.push(value);
}
}
config.groups = [...map.values()];
return config;
}
// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { generateRobots, parseRobots };
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →