diff --git a/docs/README.md b/docs/README.md index e8c464e14bd..5635878a6f3 100644 --- a/docs/README.md +++ b/docs/README.md @@ -38,11 +38,13 @@ All commands are run from the root of the project, from a terminal: ## ⚠️ Known Dev-Mode Limitations -### Sitemap not available in dev mode +### Sitemap behavior in dev and production -The sitemap (`/gh-aw/sitemap-index.xml`) is **only generated during a production build** (`npm run build`). It is not available when running the local development server (`npm run dev`). +The robots file references `/gh-aw/sitemap.xml`, which is the stable sitemap entrypoint for the docs site. -If a CI pipeline or automated tool checks for the sitemap URL during a local preview, it will receive a 404 response. To verify the sitemap, run `npm run build` followed by `npm run preview`. +During a production build (`npm run build`), Astro generates `/gh-aw/sitemap-index.xml`, and the static `/gh-aw/sitemap.xml` entrypoint points crawlers at that generated sitemap index. The generated sitemap index is not available when running the local development server (`npm run dev`). + +If a CI pipeline or automated tool checks the generated sitemap index URL during a local preview, it will receive a 404 response. To verify the production sitemap flow, run `npm run build` followed by `npm run preview`. ### Robots/AI discovery paths on GitHub Pages project sites diff --git a/docs/public/robots.txt b/docs/public/robots.txt index 42fabc7d7d5..d08c7312455 100644 --- a/docs/public/robots.txt +++ b/docs/public/robots.txt @@ -19,10 +19,10 @@ Allow: / User-agent: Claude-SearchBot Allow: / -User-agent: PerplexityBot +User-agent: claude-web Allow: / -User-agent: Googlebot +User-agent: PerplexityBot Allow: / User-agent: Perplexity-User @@ -37,13 +37,25 @@ Allow: / User-agent: Bingbot Allow: / +User-agent: Applebot-Extended +Allow: / + User-agent: cohere-ai Allow: / User-agent: DuckAssistBot Allow: / -User-agent: xAI-Bot +User-agent: Bytespider +Allow: / + +User-agent: meta-externalagent +Allow: / + +User-agent: Meta-ExternalFetcher +Allow: / + +User-agent: facebookexternalhit Allow: / User-agent: Amazonbot @@ -52,13 +64,22 @@ Allow: / User-agent: AI2Bot Allow: / -User-agent: YouBot +User-agent: AI2Bot-Dolma Allow: / -User-agent: CCBot +User-agent: xAI-Bot Allow: / -User-agent: Applebot-Extended +User-agent: Applebot +Allow: / + +User-agent: PetalBot +Allow: / + +User-agent: YouBot +Allow: / + +User-agent: CCBot Allow: / -Sitemap: https://github.github.com/gh-aw/sitemap-index.xml +Sitemap: https://github.github.com/gh-aw/sitemap.xml diff --git a/docs/tests/robots-txt.spec.ts b/docs/tests/robots-txt.spec.ts index 9cd0f15c6f1..19069634b08 100644 --- a/docs/tests/robots-txt.spec.ts +++ b/docs/tests/robots-txt.spec.ts @@ -22,10 +22,10 @@ const EXPECTED_ROBOTS_TXT = [ 'User-agent: Claude-SearchBot', 'Allow: /', '', - 'User-agent: PerplexityBot', + 'User-agent: claude-web', 'Allow: /', '', - 'User-agent: Googlebot', + 'User-agent: PerplexityBot', 'Allow: /', '', 'User-agent: Perplexity-User', @@ -40,13 +40,25 @@ const EXPECTED_ROBOTS_TXT = [ 'User-agent: Bingbot', 'Allow: /', '', + 'User-agent: Applebot-Extended', + 'Allow: /', + '', 'User-agent: cohere-ai', 'Allow: /', '', 'User-agent: DuckAssistBot', 'Allow: /', '', - 'User-agent: xAI-Bot', + 'User-agent: Bytespider', + 'Allow: /', + '', + 'User-agent: meta-externalagent', + 'Allow: /', + '', + 'User-agent: Meta-ExternalFetcher', + 'Allow: /', + '', + 'User-agent: facebookexternalhit', 'Allow: /', '', 'User-agent: Amazonbot', @@ -55,18 +67,30 @@ const EXPECTED_ROBOTS_TXT = [ 'User-agent: AI2Bot', 'Allow: /', '', + 'User-agent: AI2Bot-Dolma', + 'Allow: /', + '', + 'User-agent: xAI-Bot', + 'Allow: /', + '', + 'User-agent: Applebot', + 'Allow: /', + '', + 'User-agent: PetalBot', + 'Allow: /', + '', 'User-agent: YouBot', 'Allow: /', '', 'User-agent: CCBot', 'Allow: /', '', - 'Sitemap: https://github.github.com/gh-aw/sitemap-index.xml', + 'Sitemap: https://github.github.com/gh-aw/sitemap.xml', '', ].join('\n'); test.describe('robots.txt', () => { - test('should contain only the expected AI crawler directives and sitemap index', async ({ request }) => { + test('should contain the expected AI crawler directives and sitemap', async ({ request }) => { const response = await request.get('/gh-aw/robots.txt'); expect(response.ok()).toBeTruthy();