Merge pull request 'fix(e2e): three assertions that were wrong about correct behaviour' (#122) from fix/e2e-canonical-and-robots into main
Stage (build -> staging -> E2E gate) / Build Backend (FastAPI) (push) Successful in 13s
Stage (build -> staging -> E2E gate) / Build Frontend (Next.js) (push) Successful in 50s
Stage (build -> staging -> E2E gate) / Build Pipeline (Meltano + dbt + Airflow) (push) Successful in 13s
Stage (build -> staging -> E2E gate) / Deploy to Staging (push) Successful in 1s
Stage (build -> staging -> E2E gate) / E2E Journeys against Staging (push) Successful in 1m26s
Stage (build -> staging -> E2E gate) / Build Backend (FastAPI) (push) Successful in 13s
Stage (build -> staging -> E2E gate) / Build Frontend (Next.js) (push) Successful in 50s
Stage (build -> staging -> E2E gate) / Build Pipeline (Meltano + dbt + Airflow) (push) Successful in 13s
Stage (build -> staging -> E2E gate) / Deploy to Staging (push) Successful in 1s
Stage (build -> staging -> E2E gate) / E2E Journeys against Staging (push) Successful in 1m26s
Reviewed-on: #122
This commit was merged in pull request #122.
This commit is contained in:
commit
4a9a5c734b
1 file changed
+43
-3
@@ -1649,13 +1649,29 @@ const CANONICAL_ROUTES: Array<[string, string]> = [
|
||||
['/admissions', 'https://www.schoolcompare.co.uk/admissions'],
|
||||
];
|
||||
|
||||
/**
|
||||
* Next normalises canonical URLs against `trailingSlash: false`, so the root
|
||||
* ships as `https://www.schoolcompare.co.uk` with no slash while every other
|
||||
* route keeps its path. Both forms address the same document, and which one
|
||||
* Next emits is its business, not something worth pinning a test to.
|
||||
*
|
||||
* The first cut hardcoded the slash and failed only on the homepage — the
|
||||
* same gap as the doubled brand: it asserted the metadata object rather than
|
||||
* what the page actually renders.
|
||||
*/
|
||||
function sameUrl(a: string | null, b: string): boolean {
|
||||
const strip = (u: string) => u.replace(/\/+$/, '');
|
||||
return strip(a ?? '') === strip(b);
|
||||
}
|
||||
|
||||
for (const [path, expected] of CANONICAL_ROUTES) {
|
||||
test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => {
|
||||
await page.goto(path);
|
||||
const hrefs = await page.locator('link[rel="canonical"]').evaluateAll(
|
||||
(els) => els.map((e) => e.getAttribute('href')));
|
||||
expect(hrefs, `${path} should declare one canonical`).toHaveLength(1);
|
||||
expect(hrefs[0]).toBe(expected);
|
||||
expect(sameUrl(hrefs[0], expected),
|
||||
`${path} canonical was ${hrefs[0]}, expected ${expected}`).toBe(true);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1663,7 +1679,8 @@ test('a filtered homepage still canonicalises to the bare root', async ({ page }
|
||||
await page.goto('/?search=primary&phase=primary&sort=name&page=2');
|
||||
const href = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(href).toBe('https://www.schoolcompare.co.uk/');
|
||||
expect(sameUrl(href, 'https://www.schoolcompare.co.uk/'),
|
||||
`filtered homepage canonical was ${href}`).toBe(true);
|
||||
});
|
||||
|
||||
test('a school page canonicalises to its own slug on the www host', async ({ page }) => {
|
||||
@@ -1714,10 +1731,33 @@ test('staging answers noindex, and stays crawlable so the noindex is seen', asyn
|
||||
// The other half, and the reason this is one test rather than two: a
|
||||
// Disallow would stop Google fetching the page at all, so it would never
|
||||
// see the noindex above. The two only work together.
|
||||
//
|
||||
// Scoped to the `*` group. The first cut matched `Disallow: /` anywhere in
|
||||
// the file and tripped over the AI-crawler groups Cloudflare injects —
|
||||
// ClaudeBot, GPTBot, Amazonbot and friends all carry a blanket disallow,
|
||||
// deliberately, and none of them is Googlebot.
|
||||
const robots = await (await page.request.get('/robots.txt')).text();
|
||||
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
|
||||
expect(blocksEverything(robots, '*'),
|
||||
'the * group must not disallow the whole site, or the noindex is never seen')
|
||||
.toBe(false);
|
||||
});
|
||||
|
||||
/** True when `agent`'s group in a robots.txt disallows the entire site. */
|
||||
function blocksEverything(robots: string, agent: string): boolean {
|
||||
let current: string | null = null;
|
||||
let blocked = false;
|
||||
for (const raw of robots.split('\n')) {
|
||||
const line = raw.split('#')[0].trim();
|
||||
if (!line) continue;
|
||||
const [key, ...rest] = line.split(':');
|
||||
const value = rest.join(':').trim();
|
||||
const k = key.trim().toLowerCase();
|
||||
if (k === 'user-agent') current = value;
|
||||
else if (current === agent && k === 'disallow' && value === '/') blocked = true;
|
||||
}
|
||||
return blocked;
|
||||
}
|
||||
|
||||
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
|
||||
const list = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
const [first] = (await list.json()).schools ?? [];
|
||||
|
||||
Reference in new issue
Block a user