fix(seo): keep staging out of the search index #111
No files matched your search
@@ -1600,3 +1600,34 @@ test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
});
|
||||
|
||||
/*
|
||||
* Staging must not be indexable (spec 2026-08-20, W1 hygiene).
|
||||
*
|
||||
* These journeys only ever run against staging — deploy.yml passes
|
||||
* STAGING_BASE_URL, and promote.yml only smoke-polls production without
|
||||
* Playwright — so asserting the noindex header here is safe.
|
||||
*/
|
||||
test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => {
|
||||
const res = await page.request.get('/');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
|
||||
const tag = res.headers()['x-robots-tag'];
|
||||
expect(tag, 'staging must send X-Robots-Tag').toBeTruthy();
|
||||
expect(tag).toContain('noindex');
|
||||
|
||||
// The other half, and the reason this is one test rather than two: a
|
||||
// Disallow would stop Google fetching the page at all, so it would never
|
||||
// see the noindex above. The two only work together.
|
||||
const robots = await (await page.request.get('/robots.txt')).text();
|
||||
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
|
||||
});
|
||||
|
||||
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
|
||||
const list = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
const [first] = (await list.json()).schools ?? [];
|
||||
expect(first, 'no school available').toBeTruthy();
|
||||
|
||||
const res = await page.request.get(`/school/${first.urn}-x`);
|
||||
expect(res.headers()['x-robots-tag']).toContain('noindex');
|
||||
});
|
||||
@@ -55,6 +55,37 @@ const nextConfig = {
|
||||
// Headers for caching and security
|
||||
async headers() {
|
||||
return [
|
||||
{
|
||||
/*
|
||||
* Keep non-production hosts out of the index.
|
||||
*
|
||||
* Staging serves the same image as production off stx., so without
|
||||
* this it is a full crawlable duplicate of the site.
|
||||
*
|
||||
* X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling,
|
||||
* which is not the same as blocking indexing — a disallowed URL can
|
||||
* still be indexed from external links, and worse, blocking the crawl
|
||||
* means Google never fetches the page and never sees a noindex at all.
|
||||
* Staging therefore stays crawlable and answers "noindex" when crawled.
|
||||
*
|
||||
* Matched on the staging host explicitly rather than "any host that is
|
||||
* not production". The inverted form is tempting because it would cover
|
||||
* future environments automatically, but its failure mode is
|
||||
* deindexing production if the Host header ever arrives rewritten by a
|
||||
* proxy. This form's failure mode is a new environment being indexable
|
||||
* until someone adds it here — recoverable, where the other is not.
|
||||
*
|
||||
* Any new non-production hostname must be added to this list.
|
||||
*/
|
||||
source: '/:path*',
|
||||
has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }],
|
||||
headers: [
|
||||
{
|
||||
key: 'X-Robots-Tag',
|
||||
value: 'noindex, nofollow',
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
source: '/:path*',
|
||||
headers: [
|
||||
|
||||
Reference in new issue
Block a user