diff --git a/e2e/tests/journeys.spec.ts b/e2e/tests/journeys.spec.ts index 2909136..1051e21 100644 --- a/e2e/tests/journeys.spec.ts +++ b/e2e/tests/journeys.spec.ts @@ -1600,3 +1600,34 @@ test('the sitemap submits no Welsh or overseas school', async ({ page }) => { expect(xml).not.toContain('/school/401559'); expect(xml).not.toContain('/school/402426'); }); + +/* + * Staging must not be indexable (spec 2026-08-20, W1 hygiene). + * + * These journeys only ever run against staging — deploy.yml passes + * STAGING_BASE_URL, and promote.yml only smoke-polls production without + * Playwright — so asserting the noindex header here is safe. + */ +test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => { + const res = await page.request.get('/'); + expect(res.ok()).toBeTruthy(); + + const tag = res.headers()['x-robots-tag']; + expect(tag, 'staging must send X-Robots-Tag').toBeTruthy(); + expect(tag).toContain('noindex'); + + // The other half, and the reason this is one test rather than two: a + // Disallow would stop Google fetching the page at all, so it would never + // see the noindex above. The two only work together. + const robots = await (await page.request.get('/robots.txt')).text(); + expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi); +}); + +test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => { + const list = await page.request.get('/api/schools?search=primary&per_page=1'); + const [first] = (await list.json()).schools ?? []; + expect(first, 'no school available').toBeTruthy(); + + const res = await page.request.get(`/school/${first.urn}-x`); + expect(res.headers()['x-robots-tag']).toContain('noindex'); +}); diff --git a/nextjs-app/next.config.js b/nextjs-app/next.config.js index 54474c0..c282b97 100644 --- a/nextjs-app/next.config.js +++ b/nextjs-app/next.config.js @@ -55,6 +55,37 @@ const nextConfig = { // Headers for caching and security async headers() { return [ + { + /* + * Keep non-production hosts out of the index. + * + * Staging serves the same image as production off stx., so without + * this it is a full crawlable duplicate of the site. + * + * X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling, + * which is not the same as blocking indexing — a disallowed URL can + * still be indexed from external links, and worse, blocking the crawl + * means Google never fetches the page and never sees a noindex at all. + * Staging therefore stays crawlable and answers "noindex" when crawled. + * + * Matched on the staging host explicitly rather than "any host that is + * not production". The inverted form is tempting because it would cover + * future environments automatically, but its failure mode is + * deindexing production if the Host header ever arrives rewritten by a + * proxy. This form's failure mode is a new environment being indexable + * until someone adds it here — recoverable, where the other is not. + * + * Any new non-production hostname must be added to this list. + */ + source: '/:path*', + has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }], + headers: [ + { + key: 'X-Robots-Tag', + value: 'noindex, nofollow', + }, + ], + }, { source: '/:path*', headers: [