From c6079a68d86c2d54a25c0ea48d0d665cf8581942 Mon Sep 17 00:00:00 2001 From: Tudor Date: Thu, 20 Aug 2026 22:11:34 +0100 Subject: [PATCH] fix(seo): noindex parameterised comparisons, keep bare /compare 25,193 schools make ~317 million pairs. The bare page stays indexable as the landing page for the head term; the parameter space goes noindex, follow so its outbound links still count. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- e2e/tests/journeys.spec.ts | 18 ++++++++++++ nextjs-app/__tests__/app/metadata.test.ts | 30 +++++++++++++++++++ nextjs-app/app/compare/page.tsx | 35 ++++++++++++++++++----- 3 files changed, 76 insertions(+), 7 deletions(-) diff --git a/e2e/tests/journeys.spec.ts b/e2e/tests/journeys.spec.ts index a236a7e..2d090aa 100644 --- a/e2e/tests/journeys.spec.ts +++ b/e2e/tests/journeys.spec.ts @@ -1592,3 +1592,21 @@ test('a school page canonicalises to its own slug on the www host', async ({ pag .getAttribute('href'); expect(href).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\/\d+-/); }); + +test('a bare /compare is indexable, a parameterised one is not', async ({ page }) => { + await page.goto('/compare'); + await expect(page.locator('meta[name="robots"]')).toHaveCount(0); + + const [a, b] = await twoPrimaryUrns(page); + await page.goto(`/compare?urns=${a},${b}`); + const robots = await page.locator('meta[name="robots"]').first() + .getAttribute('content'); + expect(robots).toContain('noindex'); + expect(robots).toContain('follow'); + + // noindex but follow: the links out to each school page still count, so the + // canonical must still be present and point at the bare path. + const canonical = await page.locator('link[rel="canonical"]').first() + .getAttribute('href'); + expect(canonical).toBe('https://www.schoolcompare.co.uk/compare'); +}); diff --git a/nextjs-app/__tests__/app/metadata.test.ts b/nextjs-app/__tests__/app/metadata.test.ts index 7d283a3..e173201 100644 --- a/nextjs-app/__tests__/app/metadata.test.ts +++ b/nextjs-app/__tests__/app/metadata.test.ts @@ -1,6 +1,7 @@ import { metadata as homeMetadata } from '@/app/page'; import { metadata as rankingsMetadata } from '@/app/rankings/page'; import { metadata as admissionsMetadata } from '@/app/admissions/page'; +import { generateMetadata as compareMetadata } from '@/app/compare/page'; describe('canonical URLs', () => { it('the homepage canonicalises to the bare root', () => { @@ -21,3 +22,32 @@ describe('canonical URLs', () => { .toBe('https://www.schoolcompare.co.uk/admissions'); }); }); + +describe('/compare indexability', () => { + it('the bare compare page is indexable and canonical to itself', async () => { + // This is the landing page for the "compare schools" head term. + const meta = await compareMetadata({ searchParams: Promise.resolve({}) }); + expect(meta.alternates?.canonical) + .toBe('https://www.schoolcompare.co.uk/compare'); + expect(meta.robots).toBeUndefined(); + }); + + it('a comparison of specific schools is noindex, follow', async () => { + // ~317 million pairs before triples. Indexing the parameter space would + // swamp everything else in the corpus. + const meta = await compareMetadata({ + searchParams: Promise.resolve({ urns: '100001,100002' }), + }); + expect(meta.robots).toEqual({ index: false, follow: true }); + }); + + it('a parameterised comparison still canonicalises to the bare path', async () => { + // follow:true plus a canonical means the outbound links to each school + // page still pass value even though this URL is not indexed. + const meta = await compareMetadata({ + searchParams: Promise.resolve({ urns: '100001,100002' }), + }); + expect(meta.alternates?.canonical) + .toBe('https://www.schoolcompare.co.uk/compare'); + }); +}); diff --git a/nextjs-app/app/compare/page.tsx b/nextjs-app/app/compare/page.tsx index 0b2b4e4..2d8fb22 100644 --- a/nextjs-app/app/compare/page.tsx +++ b/nextjs-app/app/compare/page.tsx @@ -5,6 +5,7 @@ import { fetchComparison, fetchMetrics } from '@/lib/api'; import { ComparisonView } from '@/components/ComparisonView'; +import { absoluteUrl } from '@/lib/site'; import type { Metadata } from 'next'; interface ComparePageProps { @@ -14,13 +15,33 @@ interface ComparePageProps { }>; } -export const metadata: Metadata = { - title: 'Compare Schools', - description: - 'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.', - keywords: - 'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance', -}; +/** + * Indexability depends on the query string, so this cannot be a static export. + * + * Bare /compare is the landing page for the "compare schools" head term and + * stays indexable. /compare?urns=… is an unbounded parameter space — 25,193 + * schools make ~317 million pairs — so it goes noindex. It stays `follow` and + * keeps a canonical to the bare path, so the links out to each school page + * still count. + */ +export async function generateMetadata( + { searchParams }: ComparePageProps, +): Promise { + const { urns } = await searchParams; + + const base: Metadata = { + title: 'Compare Schools', + description: + 'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.', + keywords: + 'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance', + alternates: { canonical: absoluteUrl('/compare') }, + }; + + if (!urns) return base; + + return { ...base, robots: { index: false, follow: true } }; +} // Dynamic via searchParams; remove force-dynamic so internal data fetches // can still use Next.js's per-call revalidate cache.