diff --git a/nextjs-app/__tests__/app/nextConfig.test.ts b/nextjs-app/__tests__/app/nextConfig.test.ts index a378151..9df25e8 100644 --- a/nextjs-app/__tests__/app/nextConfig.test.ts +++ b/nextjs-app/__tests__/app/nextConfig.test.ts @@ -47,3 +47,20 @@ describe('next.config.mjs', () => { expect(csp!.value).toContain('https://analytics.schoolcompare.co.uk'); }); }); + +describe('admin surface', () => { + it('serves noindex on the admin panel and the CMS API', async () => { + // robots.txt disallows these too, but a Disallow only blocks crawling — a + // URL found from an external link can still be indexed without ever being + // fetched. This header is what actually keeps them out. + const headers = await headerRules(); + for (const source of ['/admin/:path*', '/cms-api/:path*']) { + const rule = headers.find((entry) => entry.source === source); + expect(rule).toBeDefined(); + expect(rule!.headers).toContainEqual({ + key: 'X-Robots-Tag', + value: 'noindex, nofollow', + }); + } + }); +}); diff --git a/nextjs-app/__tests__/app/robots.test.ts b/nextjs-app/__tests__/app/robots.test.ts new file mode 100644 index 0000000..570e135 --- /dev/null +++ b/nextjs-app/__tests__/app/robots.test.ts @@ -0,0 +1,11 @@ +import robots from '@/app/robots'; + +describe('robots.txt', () => { + it('disallows the admin panel and the CMS API', () => { + const rules = robots().rules; + const rule = Array.isArray(rules) ? rules[0] : rules; + expect(rule.disallow).toEqual( + expect.arrayContaining(['/api/', '/_next/', '/admin/', '/cms-api/']), + ); + }); +}); diff --git a/nextjs-app/app/robots.ts b/nextjs-app/app/robots.ts index e2cb38e..e613020 100644 --- a/nextjs-app/app/robots.ts +++ b/nextjs-app/app/robots.ts @@ -12,7 +12,9 @@ export default function robots(): MetadataRoute.Robots { { userAgent: '*', allow: '/', - disallow: ['/api/', '/_next/'], + // /admin and /cms-api are also served X-Robots-Tag: noindex by + // next.config.mjs. A Disallow alone blocks crawling, not indexing. + disallow: ['/api/', '/_next/', '/admin/', '/cms-api/'], }, ], sitemap: absoluteUrl('/sitemap.xml'), diff --git a/nextjs-app/next.config.mjs b/nextjs-app/next.config.mjs index 0b5d9e1..e395893 100644 --- a/nextjs-app/next.config.mjs +++ b/nextjs-app/next.config.mjs @@ -88,6 +88,23 @@ const nextConfig = { }, ], }, + { + /* + * The admin panel and the CMS API must never be indexed. + * + * X-Robots-Tag, not just the robots.txt Disallow, for the same reason + * the staging rule above uses one: a Disallow blocks crawling, which + * is not indexing. A disallowed URL found from an external link can + * still be indexed without ever being fetched — and worse, blocking + * the crawl means the noindex is never seen. + */ + source: '/admin/:path*', + headers: [{ key: 'X-Robots-Tag', value: 'noindex, nofollow' }], + }, + { + source: '/cms-api/:path*', + headers: [{ key: 'X-Robots-Tag', value: 'noindex, nofollow' }], + }, { source: '/:path*', headers: [