fix(seo): crawl hygiene and a per-family sitemap index (W1) #108

Merged
tudor merged 5 commits from feat/seo-crawl-hygiene into feat/england-only-corpus 2026-08-20 22:12:05 +00:00
7 changed files with 101 additions and 6 deletions
Showing only changes of commit 0da8ab052b - Show all commits

No files matched your search

+3 -1
View File
@@ -48,7 +48,9 @@ PHASE_GROUPS: dict[str, set[str]] = {
"all-through": {"all-through"},
}
BASE_URL = "https://schoolcompare.co.uk"
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
BASE_URL = "https://www.schoolcompare.co.uk"
MAX_SLUG_LENGTH = 60
# In-memory sitemap cache
+44
View File
@@ -0,0 +1,44 @@
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
The sitemap is built from the in-memory school DataFrame, so these inject a
small frame via monkeypatch rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
def _schools_df() -> pd.DataFrame:
"""Two schools: one with results, one with neither results nor Ofsted."""
base = {
"local_authority": "Testshire",
"school_type": "Academy",
"phase": "Primary",
"year": 202425,
"ofsted_date": None,
}
return pd.DataFrame(
[
{**base, "urn": 100001, "school_name": "Alpha Primary",
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
"ofsted_grade": 2.0},
{**base, "urn": 100002, "school_name": "Ghost Primary",
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
"ofsted_grade": np.nan},
]
)
@pytest.fixture()
def sitemap(monkeypatch) -> str:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemap()
def test_every_loc_uses_the_www_host(sitemap):
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
assert "https://www.schoolcompare.co.uk" in sitemap
assert "https://schoolcompare.co.uk" not in sitemap
+27
View File
@@ -0,0 +1,27 @@
import { SITE_URL, absoluteUrl } from '@/lib/site';
describe('SITE_URL', () => {
it('is the www host, which is the one that serves a 200', () => {
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
// is a wasted signal, so every absolute URL we emit must already be www.
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
});
it('has no trailing slash, so joins never double up', () => {
expect(SITE_URL.endsWith('/')).toBe(false);
});
});
describe('absoluteUrl', () => {
it('joins a rooted path', () => {
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('joins a path missing its leading slash', () => {
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('maps the site root to a bare trailing slash', () => {
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
});
});
+3 -2
View File
@@ -5,6 +5,7 @@ import { Navigation } from '@/components/Navigation';
import { Footer } from '@/components/Footer';
import { ComparisonToast } from '@/components/ComparisonToast';
import { ComparisonProvider } from '@/context/ComparisonProvider';
import { SITE_URL } from '@/lib/site';
import './globals.css';
// Manrope carries headings and key messaging — the guideline's "friendly,
@@ -57,12 +58,12 @@ export const metadata: Metadata = {
// No `icons` key on purpose: setting it here would override the file
// conventions. app/icon.svg and app/apple-icon.tsx are the source, and
// app/opengraph-image.tsx supplies og:image and twitter:image.
metadataBase: new URL('https://schoolcompare.co.uk'),
metadataBase: new URL(SITE_URL),
openGraph: {
type: 'website',
title: 'schoolcompare | Compare School Performance',
description: 'Compare primary and secondary school SATs and GCSE performance across England',
url: 'https://schoolcompare.co.uk',
url: SITE_URL,
siteName: 'schoolcompare',
},
twitter: {
+2 -1
View File
@@ -4,6 +4,7 @@
*/
import { MetadataRoute } from 'next';
import { absoluteUrl } from '@/lib/site';
export default function robots(): MetadataRoute.Robots {
return {
@@ -14,6 +15,6 @@ export default function robots(): MetadataRoute.Robots {
disallow: ['/api/', '/_next/'],
},
],
sitemap: 'https://schoolcompare.co.uk/sitemap.xml',
sitemap: absoluteUrl('/sitemap.xml'),
};
}
+3 -2
View File
@@ -15,6 +15,7 @@ import {
} from '@/lib/schoolSections';
import { parseSchoolSlug, schoolUrl } from '@/lib/utils';
import type { NationalAverages } from '@/lib/types';
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
/**
@@ -97,7 +98,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
title,
description,
type: 'website',
url: `https://schoolcompare.co.uk${canonicalPath}`,
url: absoluteUrl(canonicalPath),
siteName: 'schoolcompare',
},
twitter: {
@@ -106,7 +107,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
description,
},
alternates: {
canonical: `https://schoolcompare.co.uk${canonicalPath}`,
canonical: absoluteUrl(canonicalPath),
},
};
} catch {
+19
View File
@@ -0,0 +1,19 @@
/**
* The one place the site's absolute origin is written down.
*
* The apex domain 301s to www at Cloudflare, so www is the host that actually
* serves a 200. Canonicals, og:url, sitemap <loc> entries and the robots.txt
* Sitemap: line must all agree with it — a canonical pointing at a redirect
* makes Google resolve the hop before it can consolidate the signal.
*
* backend/app.py holds the same value as BASE_URL for the sitemap. The two are
* asserted against each other by the e2e journeys rather than shared at build
* time, because the backend and frontend ship as separate images.
*/
export const SITE_URL = 'https://www.schoolcompare.co.uk';
/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */
export function absoluteUrl(path: string): string {
const rooted = path.startsWith('/') ? path : `/${path}`;
return `${SITE_URL}${rooted}`;
}