feat(seo): split the sitemap into a per-family index
Search Console reports coverage per submitted sitemap, so one file per page family is what will make W2's location pages measurable when they land. The index's lastmod is generation time, which is the correct semantic there — unlike on a <url>, where it would be a claim we cannot support. Children sit under /sitemaps/ because Next only treats a whole bracketed path segment as dynamic; a route folder named sitemap-[...parts] would be read as a literal static segment and never match. Confirmed by the build output, which lists /sitemaps/[...parts] as a dynamic route. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
1 parent
b975f5186e
commit
786ec80dd4
6 files changed
+284
-73
No files matched your search
@@ -1537,18 +1537,53 @@ test('a Welsh school URL 404s while an English one still resolves', async ({ pag
|
||||
expect(welsh?.status(), 'a Welsh school should no longer resolve').toBe(404);
|
||||
});
|
||||
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
async function sitemapChildren(page: Page): Promise<string[]> {
|
||||
const res = await page.request.get('/sitemap.xml');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const xml = await res.text();
|
||||
const index = await res.text();
|
||||
expect(index).toContain('<sitemapindex');
|
||||
return [...index.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1]);
|
||||
}
|
||||
|
||||
const urlCount = (xml.match(/<url>/g) ?? []).length;
|
||||
expect(urlCount, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
test('the sitemap index names children that all resolve', async ({ page }) => {
|
||||
const index = await (await page.request.get('/sitemap.xml')).text();
|
||||
// An index holds <sitemap> entries only; mixing in <url> is invalid.
|
||||
expect(index).not.toContain('<url>');
|
||||
|
||||
// 401559 (Cardiff) and 402426 (ACT Schools, Cardiff) were both submitted
|
||||
// before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
const locs = await sitemapChildren(page);
|
||||
expect(locs.length).toBeGreaterThanOrEqual(2);
|
||||
|
||||
for (const loc of locs) {
|
||||
expect(loc.startsWith('https://www.schoolcompare.co.uk/sitemaps/')).toBeTruthy();
|
||||
const child = await page.request.get(new URL(loc).pathname);
|
||||
expect(child.ok(), `${loc} should resolve`).toBeTruthy();
|
||||
expect(await child.text()).toContain('<urlset');
|
||||
}
|
||||
});
|
||||
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
const locs = await sitemapChildren(page);
|
||||
|
||||
let total = 0;
|
||||
for (const loc of locs) {
|
||||
const xml = await (await page.request.get(new URL(loc).pathname)).text();
|
||||
total += (xml.match(/<url>/g) ?? []).length;
|
||||
// 401559 (Adamsdown, Cardiff) and 402426 (ACT Schools, Cardiff) were both
|
||||
// submitted before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
}
|
||||
expect(total, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
});
|
||||
|
||||
test('the sitemap invents no priority or changefreq', async ({ page }) => {
|
||||
const [first] = await sitemapChildren(page);
|
||||
expect(first).toBeTruthy();
|
||||
|
||||
const xml = await (await page.request.get(new URL(first).pathname)).text();
|
||||
// Google ignores both. They were noise dressed as signal.
|
||||
expect(xml).not.toContain('<priority>');
|
||||
expect(xml).not.toContain('<changefreq>');
|
||||
});
|
||||
|
||||
/*
|
||||
|
||||
Reference in new issue
Block a user