robots.txt Generator — TypeScript source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure robots.txt generation/parsing - no React, no DOM, deterministic.
// Builds standards-compliant User-agent / Allow / Disallow / Crawl-delay groups
// plus Sitemap entries. A group carries a LIST of user-agents (the REP spec's
// stacked User-agent lines) so multi-agent groups round-trip in canonical form.
// Never throws.
export interface RuleGroup {
userAgents: string[]; // ['*'] or ['Googlebot', 'Bingbot'] - one REP group
disallow: string[]; // paths ('' means "Disallow:" → allow all)
allow: string[];
crawlDelay?: number | null;
}
export interface RobotsConfig {
groups: RuleGroup[];
sitemaps: string[];
}
export type RobotsIssueCode =
| 'duplicate-user-agent'
| 'path-missing-slash'
| 'crawl-delay-ignored'
| 'sitemap-not-absolute';
export interface RobotsIssue {
code: RobotsIssueCode;
severity: 'warn' | 'info';
groupIndex?: number;
value?: string;
}
// Trim the agent list; drop empties. A non-empty list that trims away entirely
// degrades to ['*'] (a group without a User-agent line is not valid robots.txt);
// an empty list means "no group yet" and skips the group.
function cleanAgents(list: string[]): string[] {
const trimmed = list.map((u) => u.trim()).filter((u) => u !== '');
if (trimmed.length === 0 && list.length > 0) return ['*'];
return trimmed;
}
export function generateRobots(config: RobotsConfig): string {
const out: string[] = [];
for (const g of config.groups) {
const uas = cleanAgents(g.userAgents);
if (uas.length === 0) continue;
for (const ua of uas) out.push(`User-agent: ${ua}`);
for (const a of g.allow) {
const p = a.trim();
if (p) out.push(`Allow: ${p}`);
}
if (g.disallow.length === 0) {
out.push('Disallow:');
} else {
for (const d of g.disallow) {
out.push(`Disallow: ${d}`); // keep '' → 'Disallow:'
}
}
if (g.crawlDelay !== undefined && g.crawlDelay !== null && Number.isFinite(g.crawlDelay)) {
out.push(`Crawl-delay: ${g.crawlDelay}`);
}
out.push('');
}
for (const s of config.sitemaps) {
const url = s.trim();
if (url) out.push(`Sitemap: ${url}`);
}
return out.join('\n').replace(/\n{3,}/g, '\n\n').trimEnd() + '\n';
}
export function parseRobots(text: string): RobotsConfig {
const config: RobotsConfig = { groups: [], sitemaps: [] };
// Rules attach PER AGENT (Google's merge semantics): a rule line applies to
// every agent of the immediately preceding consecutive User-agent stack. At
// the end, agents whose accumulated rule sets are identical are emitted as
// one stacked-agent group, so the common (A, B share everything) case stays
// compact while partially-overlapping blocks regroup losslessly.
interface Acc {
disallow: string[];
allow: string[];
crawlDelay?: number;
}
const perUA = new Map<string, Acc>();
const order: string[] = [];
let stack: string[] = [];
let current: string[] = [];
const flush = () => {
if (stack.length === 0) return;
for (const ua of stack) {
if (!perUA.has(ua)) {
perUA.set(ua, { disallow: [], allow: [] });
order.push(ua);
}
}
current = stack;
stack = [];
};
for (const rawLine of (text ?? '').split('\n')) {
const line = rawLine.replace(/#.*$/, '').trim();
if (!line) continue;
const colon = line.indexOf(':');
if (colon === -1) continue;
const field = line.slice(0, colon).trim().toLowerCase();
const value = line.slice(colon + 1).trim();
if (field === 'user-agent') {
const ua = value || '*';
if (!stack.includes(ua)) stack.push(ua);
} else if (field === 'disallow') {
flush();
for (const ua of current) perUA.get(ua)!.disallow.push(value);
} else if (field === 'allow') {
flush();
for (const ua of current) perUA.get(ua)!.allow.push(value);
} else if (field === 'crawl-delay') {
flush();
for (const ua of current) perUA.get(ua)!.crawlDelay = Number(value);
} else if (field === 'sitemap') {
flush(); // a Sitemap line is groupless; the last stack keeps receiving rules
config.sitemaps.push(value);
}
}
flush();
const buckets = new Map<string, RuleGroup>();
for (const ua of order) {
const acc = perUA.get(ua)!;
const key = JSON.stringify([acc.disallow, acc.allow, acc.crawlDelay ?? null]);
let g = buckets.get(key);
if (!g) {
g = {
userAgents: [],
disallow: acc.disallow,
allow: acc.allow,
crawlDelay: acc.crawlDelay ?? null,
};
buckets.set(key, g);
config.groups.push(g);
}
g.userAgents.push(ua);
}
return config;
}
export function validateRobots(config: RobotsConfig): RobotsIssue[] {
const issues: RobotsIssue[] = [];
const seen = new Set<string>();
config.groups.forEach((g, groupIndex) => {
for (const raw of g.userAgents) {
const ua = raw.trim();
if (!ua) continue;
if (seen.has(ua)) {
issues.push({ code: 'duplicate-user-agent', severity: 'warn', groupIndex, value: ua });
} else {
seen.add(ua);
}
}
for (const list of [g.disallow, g.allow]) {
for (const p of list) {
const t = p.trim();
if (t && !t.startsWith('/')) {
issues.push({ code: 'path-missing-slash', severity: 'warn', groupIndex, value: t });
}
}
}
if (g.crawlDelay != null && Number.isFinite(g.crawlDelay)) {
issues.push({ code: 'crawl-delay-ignored', severity: 'info', groupIndex });
}
});
config.sitemaps.forEach((s) => {
const url = s.trim();
if (url && !/^https?:\/\//i.test(url)) {
issues.push({ code: 'sitemap-not-absolute', severity: 'warn', value: url });
}
});
return issues;
}
// URL codecs for shareable state. Each component (agent, path, URL) is
// encodeURIComponent'd BEFORE the compact format is assembled, so the ',' '|'
// '=' delimiters below can never appear inside a component after decoding.
// Group param grammar: <agents>[|d=<disallow>][|a=<allow>][|cd=<seconds>]
// with comma-joined lists; absent sections are omitted.
const enc = (s: string) => encodeURIComponent(s);
const dec = (s: string): string => {
try {
return decodeURIComponent(s);
} catch {
return s; // malformed escape - keep verbatim rather than throw
}
};
export function toQuery(config: RobotsConfig): string {
const p = new URLSearchParams();
for (const g of config.groups) {
if (cleanAgents(g.userAgents).length === 0) continue;
const parts: string[] = [g.userAgents.map((u) => enc(u.trim())).join(',')];
if (g.disallow.length > 0) parts.push('d=' + g.disallow.map(enc).join(','));
if (g.allow.length > 0) parts.push('a=' + g.allow.map(enc).join(','));
if (g.crawlDelay != null && Number.isFinite(g.crawlDelay)) parts.push('cd=' + g.crawlDelay);
p.append('g', parts.join('|'));
}
for (const s of config.sitemaps) {
const url = s.trim();
if (url) p.append('sm', enc(url));
}
return p.toString();
}
export function fromQuery(params: URLSearchParams): RobotsConfig | null {
const gs = params.getAll('g');
const sms = params.getAll('sm');
if (gs.length === 0 && sms.length === 0) return null;
const groups: RuleGroup[] = [];
for (const raw of gs) {
const sections = raw.split('|');
const userAgents = sections[0]
.split(',')
.filter((s) => s !== '')
.map(dec);
if (userAgents.length === 0) continue;
const g: RuleGroup = { userAgents, disallow: [], allow: [] };
for (const sec of sections.slice(1)) {
if (sec.startsWith('d=')) {
g.disallow = sec.slice(2).split(',').map(dec);
} else if (sec.startsWith('a=')) {
g.allow = sec.slice(2).split(',').map(dec);
} else if (sec.startsWith('cd=')) {
const n = Number(sec.slice(3));
if (Number.isFinite(n)) g.crawlDelay = n;
}
}
groups.push(g);
}
const sitemaps = sms.map(dec).filter((s) => s.trim() !== '');
return { groups, sitemaps };
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →