From 4e82e6c91636c75ae50dd57dbd65fcccc884bb89 Mon Sep 17 00:00:00 2001 From: Tudor Date: Fri, 21 Aug 2026 23:51:34 +0100 Subject: [PATCH] fix(e2e): three assertions that were wrong about correct behaviour MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The staging gate was red on three journeys. All three were faults in the tests; the site was behaving correctly in each case. Next normalises canonical URLs against trailingSlash:false, so the homepage ships "https://www.schoolcompare.co.uk" with no slash while every other route keeps its path. Both address the same document. The test hardcoded the slash and so failed only on the root — /rankings and /admissions passed throughout, which is what made it look like a homepage bug rather than a test bug. Compared with trailing slashes stripped from both sides. The robots.txt assertion matched "Disallow: /" anywhere in the file and tripped over the AI-crawler groups Cloudflare injects — ClaudeBot, GPTBot, Amazonbot and six others all carry a blanket disallow, deliberately, and none of them is Googlebot. It now parses the file into user-agent groups and checks only the "*" group, which is also the thing the test was always trying to say: Google may crawl the page, so it can see the noindex header. Both were the same mistake as the doubled brand: asserting a naive string rather than the semantics, and asserting against what the code assembles rather than what the page renders. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- e2e/tests/journeys.spec.ts | 46 +++++++++++++++++++++++++++++++++++--- 1 file changed, 43 insertions(+), 3 deletions(-) diff --git a/e2e/tests/journeys.spec.ts b/e2e/tests/journeys.spec.ts index 60b561d..0e2a778 100644 --- a/e2e/tests/journeys.spec.ts +++ b/e2e/tests/journeys.spec.ts @@ -1649,13 +1649,29 @@ const CANONICAL_ROUTES: Array<[string, string]> = [ ['/admissions', 'https://www.schoolcompare.co.uk/admissions'], ]; +/** + * Next normalises canonical URLs against `trailingSlash: false`, so the root + * ships as `https://www.schoolcompare.co.uk` with no slash while every other + * route keeps its path. Both forms address the same document, and which one + * Next emits is its business, not something worth pinning a test to. + * + * The first cut hardcoded the slash and failed only on the homepage — the + * same gap as the doubled brand: it asserted the metadata object rather than + * what the page actually renders. + */ +function sameUrl(a: string | null, b: string): boolean { + const strip = (u: string) => u.replace(/\/+$/, ''); + return strip(a ?? '') === strip(b); +} + for (const [path, expected] of CANONICAL_ROUTES) { test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => { await page.goto(path); const hrefs = await page.locator('link[rel="canonical"]').evaluateAll( (els) => els.map((e) => e.getAttribute('href'))); expect(hrefs, `${path} should declare one canonical`).toHaveLength(1); - expect(hrefs[0]).toBe(expected); + expect(sameUrl(hrefs[0], expected), + `${path} canonical was ${hrefs[0]}, expected ${expected}`).toBe(true); }); } @@ -1663,7 +1679,8 @@ test('a filtered homepage still canonicalises to the bare root', async ({ page } await page.goto('/?search=primary&phase=primary&sort=name&page=2'); const href = await page.locator('link[rel="canonical"]').first() .getAttribute('href'); - expect(href).toBe('https://www.schoolcompare.co.uk/'); + expect(sameUrl(href, 'https://www.schoolcompare.co.uk/'), + `filtered homepage canonical was ${href}`).toBe(true); }); test('a school page canonicalises to its own slug on the www host', async ({ page }) => { @@ -1714,10 +1731,33 @@ test('staging answers noindex, and stays crawlable so the noindex is seen', asyn // The other half, and the reason this is one test rather than two: a // Disallow would stop Google fetching the page at all, so it would never // see the noindex above. The two only work together. + // + // Scoped to the `*` group. The first cut matched `Disallow: /` anywhere in + // the file and tripped over the AI-crawler groups Cloudflare injects — + // ClaudeBot, GPTBot, Amazonbot and friends all carry a blanket disallow, + // deliberately, and none of them is Googlebot. const robots = await (await page.request.get('/robots.txt')).text(); - expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi); + expect(blocksEverything(robots, '*'), + 'the * group must not disallow the whole site, or the noindex is never seen') + .toBe(false); }); +/** True when `agent`'s group in a robots.txt disallows the entire site. */ +function blocksEverything(robots: string, agent: string): boolean { + let current: string | null = null; + let blocked = false; + for (const raw of robots.split('\n')) { + const line = raw.split('#')[0].trim(); + if (!line) continue; + const [key, ...rest] = line.split(':'); + const value = rest.join(':').trim(); + const k = key.trim().toLowerCase(); + if (k === 'user-agent') current = value; + else if (current === agent && k === 'disallow' && value === '/') blocked = true; + } + return blocked; +} + test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => { const list = await page.request.get('/api/schools?search=primary&per_page=1'); const [first] = (await list.json()).schools ?? [];