robots.txt Generator — PHP source
Build a standards-compliant robots.txt with per-user-agent allow/disallow rules, crawl-delay, and sitemap entries.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* robots-txt-generator — standards-compliant robots.txt generator + parser.
*
* Language: PHP (8.1+, standard library only)
* Source: CosmoDev polyglot showcase port of the robots-txt-generator tool,
* ported from src/lib/robotsTxt.ts (the canonical TypeScript impl).
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never throws.
* - Functionally equivalent to the TS reference: same inputs -> same outputs.
* - Self-contained: stdlib only (no Composer packages).
*
* The generator builds RFC 9309-style groups (User-agent / Allow / Disallow /
* Crawl-delay) plus Sitemap entries. The parser inverts a robots.txt body back
* into the same config shape, aggregating repeated User-agent blocks.
*/
declare(strict_types=1);
/**
* Reproduce ECMAScript `Number()` coercion for the values a robots.txt field
* can hold: '' -> 0.0, 'Infinity'/'+Infinity'/'-Infinity' -> +/-INF, a
* parseable float -> the value, anything else -> NAN. We need this so the
* generator's is_finite() filter sees exactly what the TS would have stored;
* PHP's `(float) 'abc'` would silently yield 0.0, which is NOT equivalent.
*/
function robotstxt_js_number(string $s): float
{
if ($s === '') {
return 0.0;
}
if ($s === 'Infinity' || $s === '+Infinity') {
return INF;
}
if ($s === '-Infinity') {
return -INF;
}
$f = filter_var($s, FILTER_VALIDATE_FLOAT);
return $f === false ? NAN : (float) $f;
}
/**
* Format a float the way an ECMAScript template literal would, so generated
* 'Crawl-delay:' values match the TS byte-for-byte: 5.0 -> '5', 5.5 -> '5.5',
* 0.0 -> '0'. PHP's (string) cast on a whole float strips the '.0'; the
* zero guard also normalizes -0.0 to '0' (JS String(-0) === '0').
*/
function robotstxt_num_to_string(float $f): string
{
if ($f == 0.0) {
// covers +0.0 and -0.0
return '0';
}
return (string) $f;
}
/**
* Build a robots.txt body from a config. Never throws; it silently drops
* malformed pieces (whitespace-only user-agent tokens, blank sitemaps).
*
* The config shape mirrors the TS interfaces:
* groups: list of {
* userAgent: string,
* disallow: list<string>,
* allow: list<string>,
* crawlDelay: float|null (optional; null/absent = unset)
* }
* sitemaps: list<string>
*/
function generate_robots(array $config): string
{
$groups = $config['groups'] ?? [];
$sitemaps = $config['sitemaps'] ?? [];
$out = [];
foreach ($groups as $g) {
// TS: `(g.userAgent || '*').trim()`. A missing/empty UA defaults to
// the wildcard token; a UA still blank after trim drops the group.
$ua = trim((string) ($g['userAgent'] ?? '*'));
if ($ua === '') {
continue;
}
$out[] = "User-agent: {$ua}";
// Allow: lines — skip blanks (a bare 'Allow:' carries no meaning).
foreach ($g['allow'] ?? [] as $a) {
$p = trim((string) $a);
if ($p !== '') {
$out[] = "Allow: {$p}";
}
}
// Disallow semantics:
// - no entries => single bare 'Disallow:' (the allow-all marker)
// - otherwise one line per entry, EMPTIES PRESERVED VERBATIM
// (an empty entry becomes 'Disallow: ' with a trailing space —
// matches the TS byte-for-byte; the path is not re-trimmed).
$disallow = $g['disallow'] ?? [];
if (count($disallow) === 0) {
$out[] = 'Disallow:';
} else {
foreach ($disallow as $d) {
$out[] = "Disallow: {$d}";
}
}
// Crawl-delay only when explicitly set to a finite number.
if (array_key_exists('crawlDelay', $g)
&& $g['crawlDelay'] !== null
&& is_finite((float) $g['crawlDelay'])) {
$out[] = 'Crawl-delay: ' . robotstxt_num_to_string((float) $g['crawlDelay']);
}
$out[] = ''; // blank line separates groups
}
foreach ($sitemaps as $s) {
$url = trim((string) $s);
if ($url !== '') {
$out[] = "Sitemap: {$url}";
}
}
// Collapse 3+ consecutive newlines to exactly two, strip trailing
// whitespace, and guarantee a single terminating newline.
$joined = preg_replace('/\n{3,}/', "\n\n", implode("\n", $out));
return rtrim((string) $joined) . "\n";
}
/**
* Parse a robots.txt body into a config. Unknown directives are ignored.
* Repeated User-agent tokens aggregate into one group; groups appear in
* first-seen order (PHP arrays preserve insertion order).
*
* @return array{groups: array<int, array>, sitemaps: array<int, string>}
*/
function parse_robots(?string $text): array
{
$groups = []; // insertion-ordered list of group shapes
$byUA = []; // user-agent => index into $groups
$sitemaps = [];
$currentIdx = null; // index of the active group; null before any UA seen
// Coerce null to '' the way TS `(text ?? '')` does.
$text = $text ?? '';
foreach (explode("\n", $text) as $rawLine) {
// Strip an inline comment (# to end of line) and trim whitespace.
$line = trim(preg_replace('/#.*/', '', $rawLine));
if ($line === '') {
continue;
}
// Find the FIRST ':' — values may contain colons themselves.
$colon = strpos($line, ':');
if ($colon === false) {
continue;
}
$field = strtolower(trim(substr((string) $line, 0, $colon)));
$value = trim(substr((string) $line, $colon + 1));
switch ($field) {
case 'user-agent':
$ua = $value !== '' ? $value : '*';
if (isset($byUA[$ua])) {
$currentIdx = $byUA[$ua];
} else {
$currentIdx = count($groups);
$byUA[$ua] = $currentIdx;
$groups[] = [
'userAgent' => $ua,
'disallow' => [],
'allow' => [],
'crawlDelay' => null,
];
}
break;
case 'disallow':
if ($currentIdx !== null) {
$groups[$currentIdx]['disallow'][] = $value;
}
break;
case 'allow':
if ($currentIdx !== null) {
$groups[$currentIdx]['allow'][] = $value;
}
break;
case 'crawl-delay':
if ($currentIdx !== null) {
$groups[$currentIdx]['crawlDelay'] = robotstxt_js_number($value);
}
break;
case 'sitemap':
$sitemaps[] = $value;
break;
}
}
return ['groups' => $groups, 'sitemaps' => $sitemaps];
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →