From 1fc1e07d219686a1ab5027454c4e2ec15074ffcf Mon Sep 17 00:00:00 2001 From: Tudor Date: Thu, 20 Aug 2026 23:21:58 +0100 Subject: [PATCH] fix(seo): keep staging out of the search index MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Staging serves the same image as production off stx., with robots.txt saying Allow: / and no noindex — a fully crawlable duplicate of the site. Nothing appears indexed today, most likely because the pages canonicalise across to production, but that is a side effect rather than a control. X-Robots-Tag, not a robots.txt Disallow. Disallow blocks crawling, which is not the same as blocking indexing: a disallowed URL can still be indexed from external links, and blocking the crawl means Google never fetches the page and so never sees a noindex at all. Staging stays crawlable and answers noindex. Matched on the staging host explicitly rather than 'any host that is not production'. The inverted form would cover future environments automatically, but its failure mode is deindexing production if the Host header ever arrives rewritten by a proxy — which cannot be verified from here. This form's failure mode is a new environment being indexable until someone adds it, which is recoverable. Any new non-production hostname must be added. The journeys only ever run against staging (deploy.yml passes STAGING_BASE_URL; promote.yml smoke-polls production without Playwright), so asserting the header there is safe. The two assertions live in one test because the halves only work together. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- e2e/tests/journeys.spec.ts | 31 +++++++++++++++++++++++++++++++ nextjs-app/next.config.js | 31 +++++++++++++++++++++++++++++++ 2 files changed, 62 insertions(+) diff --git a/e2e/tests/journeys.spec.ts b/e2e/tests/journeys.spec.ts index 2909136..1051e21 100644 --- a/e2e/tests/journeys.spec.ts +++ b/e2e/tests/journeys.spec.ts @@ -1600,3 +1600,34 @@ test('the sitemap submits no Welsh or overseas school', async ({ page }) => { expect(xml).not.toContain('/school/401559'); expect(xml).not.toContain('/school/402426'); }); + +/* + * Staging must not be indexable (spec 2026-08-20, W1 hygiene). + * + * These journeys only ever run against staging — deploy.yml passes + * STAGING_BASE_URL, and promote.yml only smoke-polls production without + * Playwright — so asserting the noindex header here is safe. + */ +test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => { + const res = await page.request.get('/'); + expect(res.ok()).toBeTruthy(); + + const tag = res.headers()['x-robots-tag']; + expect(tag, 'staging must send X-Robots-Tag').toBeTruthy(); + expect(tag).toContain('noindex'); + + // The other half, and the reason this is one test rather than two: a + // Disallow would stop Google fetching the page at all, so it would never + // see the noindex above. The two only work together. + const robots = await (await page.request.get('/robots.txt')).text(); + expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi); +}); + +test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => { + const list = await page.request.get('/api/schools?search=primary&per_page=1'); + const [first] = (await list.json()).schools ?? []; + expect(first, 'no school available').toBeTruthy(); + + const res = await page.request.get(`/school/${first.urn}-x`); + expect(res.headers()['x-robots-tag']).toContain('noindex'); +}); diff --git a/nextjs-app/next.config.js b/nextjs-app/next.config.js index 54474c0..c282b97 100644 --- a/nextjs-app/next.config.js +++ b/nextjs-app/next.config.js @@ -55,6 +55,37 @@ const nextConfig = { // Headers for caching and security async headers() { return [ + { + /* + * Keep non-production hosts out of the index. + * + * Staging serves the same image as production off stx., so without + * this it is a full crawlable duplicate of the site. + * + * X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling, + * which is not the same as blocking indexing — a disallowed URL can + * still be indexed from external links, and worse, blocking the crawl + * means Google never fetches the page and never sees a noindex at all. + * Staging therefore stays crawlable and answers "noindex" when crawled. + * + * Matched on the staging host explicitly rather than "any host that is + * not production". The inverted form is tempting because it would cover + * future environments automatically, but its failure mode is + * deindexing production if the Host header ever arrives rewritten by a + * proxy. This form's failure mode is a new environment being indexable + * until someone adds it here — recoverable, where the other is not. + * + * Any new non-production hostname must be added to this list. + */ + source: '/:path*', + has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }], + headers: [ + { + key: 'X-Robots-Tag', + value: 'noindex, nofollow', + }, + ], + }, { source: '/:path*', headers: [