fix(seo): crawl hygiene and a per-family sitemap index (W1) #110

Merged
tudor merged 5 commits from feat/seo-crawl-hygiene-main into main 2026-08-20 22:23:10 +00:00
5 changed files with 76 additions and 0 deletions
Showing only changes of commit 51ce2d6373 - Show all commits

No files matched your search

+42
View File
@@ -1600,3 +1600,45 @@ test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
});
/*
* Canonical URLs (spec 2026-08-20, W1).
*
* Every indexable route declares exactly one canonical, on the www host, with
* no query string. The homepage's eleven search params filter a result set
* rather than making a new document, so they all collapse onto "/".
*/
const CANONICAL_ROUTES: Array<[string, string]> = [
['/', 'https://www.schoolcompare.co.uk/'],
['/rankings', 'https://www.schoolcompare.co.uk/rankings'],
['/admissions', 'https://www.schoolcompare.co.uk/admissions'],
];
for (const [path, expected] of CANONICAL_ROUTES) {
test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => {
await page.goto(path);
const hrefs = await page.locator('link[rel="canonical"]').evaluateAll(
(els) => els.map((e) => e.getAttribute('href')));
expect(hrefs, `${path} should declare one canonical`).toHaveLength(1);
expect(hrefs[0]).toBe(expected);
});
}
test('a filtered homepage still canonicalises to the bare root', async ({ page }) => {
await page.goto('/?search=primary&phase=primary&sort=name&page=2');
const href = await page.locator('link[rel="canonical"]').first()
.getAttribute('href');
expect(href).toBe('https://www.schoolcompare.co.uk/');
});
test('a school page canonicalises to its own slug on the www host', async ({ page }) => {
const res = await page.request.get('/api/schools?search=primary&per_page=1');
expect(res.ok()).toBeTruthy();
const [first] = (await res.json()).schools ?? [];
expect(first, 'no school available').toBeTruthy();
await page.goto(`/school/${first.urn}-x`);
const href = await page.locator('link[rel="canonical"]').first()
.getAttribute('href');
expect(href).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\/\d+-/);
});
+23
View File
@@ -0,0 +1,23 @@
import { metadata as homeMetadata } from '@/app/page';
import { metadata as rankingsMetadata } from '@/app/rankings/page';
import { metadata as admissionsMetadata } from '@/app/admissions/page';
describe('canonical URLs', () => {
it('the homepage canonicalises to the bare root', () => {
// page.tsx reads eleven search params. Without this, every filter
// combination is a crawlable near-duplicate of the one page we want to
// rank for "compare schools".
expect(homeMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/');
});
it('rankings canonicalises to the bare path', () => {
expect(rankingsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/rankings');
});
it('admissions canonicalises to the bare path', () => {
expect(admissionsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/admissions');
});
});
+2
View File
@@ -1,3 +1,4 @@
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
import { AdmissionsView } from '@/components/AdmissionsView';
@@ -7,6 +8,7 @@ export const metadata: Metadata = {
title: 'School Admissions Guide',
description:
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
alternates: { canonical: absoluteUrl('/admissions') },
};
export default function AdmissionsPage() {
+5
View File
@@ -3,6 +3,7 @@
* Main landing page with school search and browsing
*/
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
import { fetchSchools, fetchFilters, fetchDataInfo } from '@/lib/api';
import { formatAcademicYear } from '@/lib/utils';
@@ -35,6 +36,10 @@ interface HomePageProps {
export const metadata: Metadata = {
title: { absolute: 'schoolcompare | Compare every school in England' },
description: 'Search and compare school performance across England',
// This page reads eleven search params. They filter a result set; they do
// not make a new document. Collapsing every combination onto "/" stops the
// homepage competing with itself for its own head terms.
alternates: { canonical: absoluteUrl('/') },
};
// The page reads searchParams, which makes rendering dynamic by default.
+4
View File
@@ -5,6 +5,7 @@
import { fetchRankings, fetchFilters, fetchMetrics } from '@/lib/api';
import { RankingsView } from '@/components/RankingsView';
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
interface RankingsPageProps {
@@ -20,6 +21,9 @@ export const metadata: Metadata = {
title: 'School Rankings',
description: 'Top-ranked schools by SATs and GCSE performance across England',
keywords: 'school rankings, top schools, best schools, KS2 rankings, KS4 rankings, school league tables',
// Param forms (?metric=&local_authority=&year=&phase=) collapse here for
// now. W3 replaces them with real indexable paths.
alternates: { canonical: absoluteUrl('/rankings') },
};
// Dynamic via searchParams; remove force-dynamic so internal data fetches