feat(seo): split the sitemap into a per-family index
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m4s
PR Checks / Backend Smoke (pull_request) Successful in 7s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 44s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 2m2s

Search Console reports coverage per submitted sitemap, so one file per page
family is what will make W2's location pages measurable when they land. The
index's lastmod is generation time, which is the correct semantic there —
unlike on a <url>, where it would be a claim we cannot support.

Children sit under /sitemaps/ because Next only treats a whole bracketed path
segment as dynamic; a route folder named sitemap-[...parts] would be read as a
literal static segment and never match. Confirmed by the build output, which
lists /sitemaps/[...parts] as a dynamic route.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-20 23:15:00 +01:00
1 parent 24f3cb4c65
commit 2208ad93c1
6 files changed
+284 -73

No files matched your search

+43 -8
View File
@@ -1587,18 +1587,53 @@ test('a Welsh school URL 404s while an English one still resolves', async ({ pag
expect(welsh?.status(), 'a Welsh school should no longer resolve').toBe(404);
});
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
async function sitemapChildren(page: Page): Promise<string[]> {
const res = await page.request.get('/sitemap.xml');
expect(res.ok()).toBeTruthy();
const xml = await res.text();
const index = await res.text();
expect(index).toContain('<sitemapindex');
return [...index.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1]);
}
const urlCount = (xml.match(/<url>/g) ?? []).length;
expect(urlCount, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
test('the sitemap index names children that all resolve', async ({ page }) => {
const index = await (await page.request.get('/sitemap.xml')).text();
// An index holds <sitemap> entries only; mixing in <url> is invalid.
expect(index).not.toContain('<url>');
// 401559 (Cardiff) and 402426 (ACT Schools, Cardiff) were both submitted
// before the England-only filter landed.
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
const locs = await sitemapChildren(page);
expect(locs.length).toBeGreaterThanOrEqual(2);
for (const loc of locs) {
expect(loc.startsWith('https://www.schoolcompare.co.uk/sitemaps/')).toBeTruthy();
const child = await page.request.get(new URL(loc).pathname);
expect(child.ok(), `${loc} should resolve`).toBeTruthy();
expect(await child.text()).toContain('<urlset');
}
});
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
const locs = await sitemapChildren(page);
let total = 0;
for (const loc of locs) {
const xml = await (await page.request.get(new URL(loc).pathname)).text();
total += (xml.match(/<url>/g) ?? []).length;
// 401559 (Adamsdown, Cardiff) and 402426 (ACT Schools, Cardiff) were both
// submitted before the England-only filter landed.
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
}
expect(total, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
});
test('the sitemap invents no priority or changefreq', async ({ page }) => {
const [first] = await sitemapChildren(page);
expect(first).toBeTruthy();
const xml = await (await page.request.get(new URL(first).pathname)).text();
// Google ignores both. They were noise dressed as signal.
expect(xml).not.toContain('<priority>');
expect(xml).not.toContain('<changefreq>');
});
/*