fix(seo): keep staging out of the search index #111

Merged
tudor merged 2 commits from fix/staging-noindex into main 2026-08-20 22:39:38 +00:00
2 changed files with 62 additions and 0 deletions
Showing only changes of commit 1fc1e07d21 - Show all commits

No files matched your search

+31
View File
@@ -1600,3 +1600,34 @@ test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
});
/*
* Staging must not be indexable (spec 2026-08-20, W1 hygiene).
*
* These journeys only ever run against staging — deploy.yml passes
* STAGING_BASE_URL, and promote.yml only smoke-polls production without
* Playwright — so asserting the noindex header here is safe.
*/
test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => {
const res = await page.request.get('/');
expect(res.ok()).toBeTruthy();
const tag = res.headers()['x-robots-tag'];
expect(tag, 'staging must send X-Robots-Tag').toBeTruthy();
expect(tag).toContain('noindex');
// The other half, and the reason this is one test rather than two: a
// Disallow would stop Google fetching the page at all, so it would never
// see the noindex above. The two only work together.
const robots = await (await page.request.get('/robots.txt')).text();
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
});
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
const list = await page.request.get('/api/schools?search=primary&per_page=1');
const [first] = (await list.json()).schools ?? [];
expect(first, 'no school available').toBeTruthy();
const res = await page.request.get(`/school/${first.urn}-x`);
expect(res.headers()['x-robots-tag']).toContain('noindex');
});
+31
View File
@@ -55,6 +55,37 @@ const nextConfig = {
// Headers for caching and security
async headers() {
return [
{
/*
* Keep non-production hosts out of the index.
*
* Staging serves the same image as production off stx., so without
* this it is a full crawlable duplicate of the site.
*
* X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling,
* which is not the same as blocking indexing — a disallowed URL can
* still be indexed from external links, and worse, blocking the crawl
* means Google never fetches the page and never sees a noindex at all.
* Staging therefore stays crawlable and answers "noindex" when crawled.
*
* Matched on the staging host explicitly rather than "any host that is
* not production". The inverted form is tempting because it would cover
* future environments automatically, but its failure mode is
* deindexing production if the Host header ever arrives rewritten by a
* proxy. This form's failure mode is a new environment being indexable
* until someone adds it here — recoverable, where the other is not.
*
* Any new non-production hostname must be added to this list.
*/
source: '/:path*',
has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }],
headers: [
{
key: 'X-Robots-Tag',
value: 'noindex, nofollow',
},
],
},
{
source: '/:path*',
headers: [