From 0da8ab052b0fbb0eae359ee673d1dd2bdb6791df Mon Sep 17 00:00:00 2001 From: Tudor Date: Thu, 20 Aug 2026 22:10:07 +0100 Subject: [PATCH] fix(seo): canonicalise on the www host, which is the one that serves 200 The apex 301s to www at Cloudflare, but metadataBase, the school-page canonical, robots.txt's Sitemap: line and the sitemap's own entries all named the apex. Every one of those pointed Google at a redirect. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- backend/app.py | 4 ++- backend/tests/test_sitemap.py | 44 +++++++++++++++++++++++++++ nextjs-app/__tests__/lib/site.test.ts | 27 ++++++++++++++++ nextjs-app/app/layout.tsx | 5 +-- nextjs-app/app/robots.ts | 3 +- nextjs-app/app/school/[slug]/page.tsx | 5 +-- nextjs-app/lib/site.ts | 19 ++++++++++++ 7 files changed, 101 insertions(+), 6 deletions(-) create mode 100644 backend/tests/test_sitemap.py create mode 100644 nextjs-app/__tests__/lib/site.test.ts create mode 100644 nextjs-app/lib/site.ts diff --git a/backend/app.py b/backend/app.py index 20d9570..45a7772 100644 --- a/backend/app.py +++ b/backend/app.py @@ -48,7 +48,9 @@ PHASE_GROUPS: dict[str, set[str]] = { "all-through": {"all-through"}, } -BASE_URL = "https://schoolcompare.co.uk" +# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a +# sitemap that redirects wastes a crawl on every URL it lists. +BASE_URL = "https://www.schoolcompare.co.uk" MAX_SLUG_LENGTH = 60 # In-memory sitemap cache diff --git a/backend/tests/test_sitemap.py b/backend/tests/test_sitemap.py new file mode 100644 index 0000000..430db84 --- /dev/null +++ b/backend/tests/test_sitemap.py @@ -0,0 +1,44 @@ +"""Tests for sitemap generation (spec 2026-08-20, workstream W1). + +The sitemap is built from the in-memory school DataFrame, so these inject a +small frame via monkeypatch rather than touching a database. +""" + +import numpy as np +import pandas as pd +import pytest + + +def _schools_df() -> pd.DataFrame: + """Two schools: one with results, one with neither results nor Ofsted.""" + base = { + "local_authority": "Testshire", + "school_type": "Academy", + "phase": "Primary", + "year": 202425, + "ofsted_date": None, + } + return pd.DataFrame( + [ + {**base, "urn": 100001, "school_name": "Alpha Primary", + "rwm_expected_pct": 62.0, "attainment_8_score": np.nan, + "ofsted_grade": 2.0}, + {**base, "urn": 100002, "school_name": "Ghost Primary", + "rwm_expected_pct": np.nan, "attainment_8_score": np.nan, + "ofsted_grade": np.nan}, + ] + ) + + +@pytest.fixture() +def sitemap(monkeypatch) -> str: + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _schools_df) + return app_module.build_sitemap() + + +def test_every_loc_uses_the_www_host(sitemap): + # The apex 301s to www. A that redirects burns a crawl per URL. + assert "https://www.schoolcompare.co.uk" in sitemap + assert "https://schoolcompare.co.uk" not in sitemap diff --git a/nextjs-app/__tests__/lib/site.test.ts b/nextjs-app/__tests__/lib/site.test.ts new file mode 100644 index 0000000..2003959 --- /dev/null +++ b/nextjs-app/__tests__/lib/site.test.ts @@ -0,0 +1,27 @@ +import { SITE_URL, absoluteUrl } from '@/lib/site'; + +describe('SITE_URL', () => { + it('is the www host, which is the one that serves a 200', () => { + // The apex 301s to www at Cloudflare. A canonical pointing at a redirect + // is a wasted signal, so every absolute URL we emit must already be www. + expect(SITE_URL).toBe('https://www.schoolcompare.co.uk'); + }); + + it('has no trailing slash, so joins never double up', () => { + expect(SITE_URL.endsWith('/')).toBe(false); + }); +}); + +describe('absoluteUrl', () => { + it('joins a rooted path', () => { + expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings'); + }); + + it('joins a path missing its leading slash', () => { + expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings'); + }); + + it('maps the site root to a bare trailing slash', () => { + expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/'); + }); +}); diff --git a/nextjs-app/app/layout.tsx b/nextjs-app/app/layout.tsx index ebf9b67..20eb8d7 100644 --- a/nextjs-app/app/layout.tsx +++ b/nextjs-app/app/layout.tsx @@ -5,6 +5,7 @@ import { Navigation } from '@/components/Navigation'; import { Footer } from '@/components/Footer'; import { ComparisonToast } from '@/components/ComparisonToast'; import { ComparisonProvider } from '@/context/ComparisonProvider'; +import { SITE_URL } from '@/lib/site'; import './globals.css'; // Manrope carries headings and key messaging — the guideline's "friendly, @@ -57,12 +58,12 @@ export const metadata: Metadata = { // No `icons` key on purpose: setting it here would override the file // conventions. app/icon.svg and app/apple-icon.tsx are the source, and // app/opengraph-image.tsx supplies og:image and twitter:image. - metadataBase: new URL('https://schoolcompare.co.uk'), + metadataBase: new URL(SITE_URL), openGraph: { type: 'website', title: 'schoolcompare | Compare School Performance', description: 'Compare primary and secondary school SATs and GCSE performance across England', - url: 'https://schoolcompare.co.uk', + url: SITE_URL, siteName: 'schoolcompare', }, twitter: { diff --git a/nextjs-app/app/robots.ts b/nextjs-app/app/robots.ts index 8ab3bbc..e2cb38e 100644 --- a/nextjs-app/app/robots.ts +++ b/nextjs-app/app/robots.ts @@ -4,6 +4,7 @@ */ import { MetadataRoute } from 'next'; +import { absoluteUrl } from '@/lib/site'; export default function robots(): MetadataRoute.Robots { return { @@ -14,6 +15,6 @@ export default function robots(): MetadataRoute.Robots { disallow: ['/api/', '/_next/'], }, ], - sitemap: 'https://schoolcompare.co.uk/sitemap.xml', + sitemap: absoluteUrl('/sitemap.xml'), }; } diff --git a/nextjs-app/app/school/[slug]/page.tsx b/nextjs-app/app/school/[slug]/page.tsx index db2c229..4e9f6aa 100644 --- a/nextjs-app/app/school/[slug]/page.tsx +++ b/nextjs-app/app/school/[slug]/page.tsx @@ -15,6 +15,7 @@ import { } from '@/lib/schoolSections'; import { parseSchoolSlug, schoolUrl } from '@/lib/utils'; import type { NationalAverages } from '@/lib/types'; +import { absoluteUrl } from '@/lib/site'; import type { Metadata } from 'next'; /** @@ -97,7 +98,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise entries and the robots.txt + * Sitemap: line must all agree with it — a canonical pointing at a redirect + * makes Google resolve the hop before it can consolidate the signal. + * + * backend/app.py holds the same value as BASE_URL for the sitemap. The two are + * asserted against each other by the e2e journeys rather than shared at build + * time, because the backend and frontend ship as separate images. + */ +export const SITE_URL = 'https://www.schoolcompare.co.uk'; + +/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */ +export function absoluteUrl(path: string): string { + const rooted = path.startsWith('/') ? path : `/${path}`; + return `${SITE_URL}${rooted}`; +}