Compare commits

..
Author SHA1 Message Date
TudorandClaude Opus 5 2208ad93c1 feat(seo): split the sitemap into a per-family index
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m4s
PR Checks / Backend Smoke (pull_request) Successful in 7s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 44s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 2m2s
Search Console reports coverage per submitted sitemap, so one file per page
family is what will make W2's location pages measurable when they land. The
index's lastmod is generation time, which is the correct semantic there —
unlike on a <url>, where it would be a claim we cannot support.

Children sit under /sitemaps/ because Next only treats a whole bracketed path
segment as dynamic; a route folder named sitemap-[...parts] would be read as a
literal static segment and never match. Confirmed by the build output, which
lists /sitemaps/[...parts] as a dynamic route.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-20 23:15:00 +01:00
TudorandClaude Opus 5 24f3cb4c65 fix(seo): submit only school pages that have something to show
Drops the schools with neither results nor an Ofsted grade, adds /admissions
which was never listed, replaces the invented priority and changefreq with a
lastmod taken from each school's Ofsted date.

lastmod is omitted where no date is known rather than defaulted to now. An
always-now lastmod is a claim Google learns to distrust; absent honestly
means unknown.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-20 23:15:00 +01:00
TudorandClaude Opus 5 3816b92d06 fix(seo): noindex parameterised comparisons, keep bare /compare
25,193 schools make ~317 million pairs. The bare page stays indexable as the
landing page for the head term; the parameter space goes noindex, follow so
its outbound links still count.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-20 23:15:00 +01:00
TudorandClaude Opus 5 51ce2d6373 fix(seo): declare a canonical on every route
The homepage read eleven search params and declared no canonical, so every
filter combination was a crawlable near-duplicate of the page we most want to
rank. Rankings and admissions declared none either.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-20 23:15:00 +01:00
TudorandClaude Opus 5 5ddc7314fd fix(seo): canonicalise on the www host, which is the one that serves 200
The apex 301s to www at Cloudflare, but metadataBase, the school-page
canonical, robots.txt's Sitemap: line and the sitemap's own <loc> entries all
named the apex. Every one of those pointed Google at a redirect.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-20 23:15:00 +01:00
18 changed files with 622 additions and 232 deletions

No files matched your search

+151 -50
View File
@@ -7,6 +7,7 @@ Uses real data from UK Government Compare School Performance downloads.
import hashlib
import re
from contextlib import asynccontextmanager
from datetime import datetime, timezone
from typing import Optional
import numpy as np
@@ -48,11 +49,14 @@ PHASE_GROUPS: dict[str, set[str]] = {
"all-through": {"all-through"},
}
BASE_URL = "https://schoolcompare.co.uk"
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
BASE_URL = "https://www.schoolcompare.co.uk"
MAX_SLUG_LENGTH = 60
# In-memory sitemap cache
_sitemap_xml: str | None = None
# In-memory sitemap cache: name -> XML. Populated on startup and by the admin
# regenerate endpoint after a pipeline run.
_sitemaps: dict[str, str] | None = None
def _slugify(text: str) -> str:
@@ -70,43 +74,128 @@ def _school_url(urn: int, school_name: str) -> str:
return f"/school/{urn}-{slug}"
def build_sitemap() -> str:
"""Generate sitemap XML from in-memory school data. Returns the XML string."""
# Routes worth submitting that are not a school page. /admissions was missing
# from the sitemap entirely despite being a static, indexable guide.
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
def _has_publishable_data(row) -> bool:
"""True when a school page has something a search result could state.
A school with no results in any year and no Ofsted grade renders an empty
page. Submitting it spends crawl budget and drags the corpus-wide quality
signal down, so it stays out of the sitemap. The page itself still resolves
for anyone who has the URL.
"""
for field in ("rwm_expected_pct", "attainment_8_score", "ofsted_grade"):
value = row.get(field)
if value is not None and not pd.isna(value):
return True
return False
def _url_element(loc: str, lastmod: str | None = None) -> str:
"""One <url> entry. No priority or changefreq — Google ignores both."""
body = f"<loc>{loc}</loc>"
if lastmod:
body += f"<lastmod>{lastmod}</lastmod>"
return f" <url>{body}</url>"
def _school_sitemap_rows(df) -> list[str]:
"""A <url> element per school that has something to show.
lastmod comes from the school's Ofsted date where there is one and is
omitted otherwise. An always-now lastmod is a claim Google learns to
distrust; an absent one honestly means "unknown".
"""
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
return []
rows: list[str] = []
seen: set[int] = set()
# Latest row per URN first, so a school's most recent Ofsted date wins.
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
for _, row in ordered.iterrows():
urn = int(row["urn"])
if urn in seen:
continue
seen.add(urn)
if not _has_publishable_data(row):
continue
lastmod = None
ofsted_date = row.get("ofsted_date")
if ofsted_date is not None and not pd.isna(ofsted_date):
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
rows.append(_url_element(
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
return rows
# Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to
# scan by eye in Search Console, which is the point of splitting at all:
# coverage is reported per submitted sitemap, so one file per page family is
# what makes an indexation problem attributable to a family.
SITEMAP_CHUNK_SIZE = 10_000
# Children are served under /sitemaps/ because Next.js only treats a whole
# bracketed path segment as dynamic — a route folder named "sitemap-[...parts]"
# is read as a literal static segment and never matches.
SITEMAP_CHILD_PREFIX = "/sitemaps"
def _urlset(rows: list[str]) -> str:
return "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
*rows,
"</urlset>",
])
def build_sitemaps() -> dict[str, str]:
"""Build the sitemap index and every child, keyed by name."""
df = load_school_data()
static_urls = [
(BASE_URL + "/", "daily", "1.0"),
(BASE_URL + "/rankings", "weekly", "0.8"),
(BASE_URL + "/compare", "weekly", "0.8"),
children: dict[str, str] = {
"static.xml": _urlset(
[_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]),
}
school_rows = _school_sitemap_rows(df)
# Always emit at least one school child, so the index shape is stable even
# on an empty database.
chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1):
children[f"schools-{n}.xml"] = _urlset(chunk)
# On a sitemap index, lastmod means "when this sitemap file last changed",
# so generation time is the correct value here — unlike on a <url>, where
# it would be a claim about content we cannot support.
generated = datetime.now(timezone.utc).date().isoformat()
index_rows = [
f" <sitemap><loc>{BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}</loc>"
f"<lastmod>{generated}</lastmod></sitemap>"
for name in children
]
index = "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
'<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
*index_rows,
"</sitemapindex>",
])
return {**children, "sitemap.xml": index}
lines = ['<?xml version="1.0" encoding="UTF-8"?>',
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
for url, freq, priority in static_urls:
lines.append(
f" <url><loc>{url}</loc>"
f"<changefreq>{freq}</changefreq>"
f"<priority>{priority}</priority></url>"
)
if not df.empty and "urn" in df.columns and "school_name" in df.columns:
seen = set()
for _, row in df[["urn", "school_name"]].drop_duplicates(subset="urn").iterrows():
urn = int(row["urn"])
name = str(row["school_name"])
if urn in seen:
continue
seen.add(urn)
path = _school_url(urn, name)
lines.append(
f" <url><loc>{BASE_URL}{path}</loc>"
f"<changefreq>monthly</changefreq>"
f"<priority>0.6</priority></url>"
)
lines.append("</urlset>")
return "\n".join(lines)
def build_sitemap() -> str:
"""The sitemap index. Kept for `lifespan` and the admin endpoint."""
return build_sitemaps()["sitemap.xml"]
def clean_filter_values(series: pd.Series) -> list[str]:
@@ -284,7 +373,7 @@ def validate_postcode(postcode: Optional[str]) -> Optional[str]:
@asynccontextmanager
async def lifespan(app: FastAPI):
"""Application lifespan - startup and shutdown events."""
global _sitemap_xml
global _sitemaps
print("Loading school data from marts...")
df = load_school_data()
if df.empty:
@@ -294,9 +383,9 @@ async def lifespan(app: FastAPI):
# Pre-compute the latest-year snapshot so the first search request is fast
await asyncio.to_thread(load_latest_school_data)
try:
_sitemap_xml = build_sitemap()
n = _sitemap_xml.count("<url>")
print(f"Sitemap built: {n} URLs.")
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.")
except Exception as e:
print(f"Warning: sitemap build failed on startup: {e}")
@@ -1087,16 +1176,28 @@ async def robots_txt():
return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain")
@app.get("/sitemap.xml")
async def sitemap_xml():
"""Serve sitemap.xml for search engine indexing."""
global _sitemap_xml
if _sitemap_xml is None:
def _serve_sitemap(name: str) -> Response:
global _sitemaps
if _sitemaps is None:
try:
_sitemap_xml = build_sitemap()
_sitemaps = build_sitemaps()
except Exception as e:
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
return Response(content=_sitemap_xml, media_type="application/xml")
if name not in _sitemaps:
raise HTTPException(status_code=404, detail="No such sitemap")
return Response(content=_sitemaps[name], media_type="application/xml")
@app.get("/sitemap.xml")
async def sitemap_xml():
"""Serve the sitemap index."""
return _serve_sitemap("sitemap.xml")
@app.get("/sitemaps/{name}")
async def sitemap_child(name: str):
"""Serve a child sitemap (static.xml, or schools-N.xml)."""
return _serve_sitemap(name)
@app.post("/api/admin/regenerate-sitemap")
@@ -1106,10 +1207,10 @@ async def regenerate_sitemap(
_: bool = Depends(verify_admin_api_key),
):
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
global _sitemap_xml
_sitemap_xml = build_sitemap()
n = _sitemap_xml.count("<url>")
return {"status": "ok", "urls": n}
global _sitemaps
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
# Mount static files directly (must be after all routes to avoid catching API calls)
+178
View File
@@ -0,0 +1,178 @@
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
The sitemap is built from the in-memory school DataFrame, so these inject a
small frame via monkeypatch rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
def _schools_df() -> pd.DataFrame:
"""Two schools: one with results, one with neither results nor Ofsted."""
base = {
"local_authority": "Testshire",
"school_type": "Academy",
"phase": "Primary",
"year": 202425,
"ofsted_date": None,
}
return pd.DataFrame(
[
{**base, "urn": 100001, "school_name": "Alpha Primary",
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
"ofsted_grade": 2.0},
{**base, "urn": 100002, "school_name": "Ghost Primary",
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
"ofsted_grade": np.nan},
]
)
@pytest.fixture()
def sitemap(monkeypatch) -> str:
"""The sitemap index."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemap()
@pytest.fixture()
def schools_child(monkeypatch) -> str:
"""The first school child sitemap, where school URLs actually live."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["schools-1.xml"]
@pytest.fixture()
def static_child(monkeypatch) -> str:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["static.xml"]
def test_every_loc_uses_the_www_host(sitemaps):
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
# Checked across every file, index included, not just one.
for name, xml in sitemaps.items():
assert "https://www.schoolcompare.co.uk" in xml, name
assert "https://schoolcompare.co.uk" not in xml, name
def test_school_with_results_is_listed(schools_child):
assert "/school/100001-alpha-primary" in schools_child
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
# Nothing for a search result to say about it. Submitting it spends crawl
# budget and drags the corpus-wide quality signal down.
#
# Asserted against the child, not the index: the index carries no school
# URLs at all, so it would pass this trivially and prove nothing.
assert "/school/100002" not in schools_child
def test_no_invented_priority_or_changefreq(sitemaps):
# Google ignores both. They were noise dressed as signal.
for name, xml in sitemaps.items():
assert "<priority>" not in xml, name
assert "<changefreq>" not in xml, name
def test_ofsted_date_becomes_lastmod(monkeypatch):
from backend import app as app_module
import datetime
def _df():
base = _schools_df()
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
return base
monkeypatch.setattr(app_module, "load_school_data", _df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>2024-03-14</lastmod>" in xml
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
# An always-now lastmod is a claim Google learns to distrust. Absent
# honestly means unknown.
from backend import app as app_module
def _df():
df = _schools_df()
df["ofsted_date"] = None
return df
monkeypatch.setattr(app_module, "load_school_data", _df)
# The child only. The index legitimately carries a lastmod, because there
# it means "when this sitemap file changed", which we do know.
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>" not in xml
def test_static_routes_are_listed(static_child):
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
@pytest.fixture()
def sitemaps(monkeypatch) -> dict:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()
def test_index_lists_each_child(sitemaps):
index = sitemaps["sitemap.xml"]
assert "<sitemapindex" in index
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
def test_index_carries_no_url_elements(sitemaps):
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
assert "<url>" not in sitemaps["sitemap.xml"]
def test_index_does_not_list_itself(sitemaps):
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
def test_static_child_holds_the_static_routes(sitemaps):
static = sitemaps["static.xml"]
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
def test_school_child_holds_the_schools(sitemaps):
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
def test_children_are_chunked_under_the_limit(monkeypatch):
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
# small enough to eyeball in Search Console.
from backend import app as app_module
import pandas as _pd
rows = [
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": None,
"ofsted_grade": 2.0, "ofsted_date": None}
for i in range(10_001)
]
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
maps = app_module.build_sitemaps()
assert maps["schools-1.xml"].count("<url>") == 10_000
assert maps["schools-2.xml"].count("<url>") == 1
def test_build_sitemap_still_returns_the_index(sitemap):
# lifespan and the admin endpoint call build_sitemap(); keep it working.
assert "<sitemapindex" in sitemap
-85
View File
@@ -1,85 +0,0 @@
# Remote branch cleanup, 2026-08-20
# Restore any branch with: git push origin <sha>:refs/heads/<name>
## Deleted: fully merged into main (content is in main)
75677f4252b759ef895e7d5f7c19f8f1745bdb59 add-contact-form-footer
fa1abff642683dfd26ba88a295a0a6710d77147a chore/byline-removal-and-audit-figure
95081d38bdf87764ef5d298676c25fae4cd3b792 chore/remove-parent-view
6877abedebfc1c7d95f1f6ebe68945c62be528ab chore/staged-prod-promotion
090d5f7bec824e083d3252e2c6e636686016304d ci/frontend-checks-speedup
8c3a5cc4e9f551f0190d85357ce7741ad87f3a4c design/cohort-identity
955659580067ca8b81bd77e31e4e2554103f094d feat/allow-analytics-iframe-embed
6828f6cd4417284ea3eb6f088fa20945b8b40ed3 feat/compare-chips-two-per-row
d5cd0abfee226885119665da5d2aa8288b59217f feat/compare-data-foundation
6dd9b04b50bee146682da87efad8fc8b526251c5 feat/compare-frontend-rebuild
96d5fcf5b07b6f175b48e9b20fcb765a320a907f feat/detail-header-details-reveal
f1388ff5bd0af1409823a1e047b7ba84246a0f70 feat/gias-sixth-form-flag
eddf74745f86c9c6d9eb07d867246ff9cc90dc20 feat/hero-artwork-v2
3015c37bac6dc28db58c80fdb9235942c25b83d7 feat/hero-byline
8e763e39d17a4964cf558e51c03f044186371f6d feat/info-popover-tooltip
88c653215d520ab6e902c9de55bf27d86eeb90c3 feat/last-distance-offered
a72323874f7aebdb2e64b6d64a5febd61152d09d feat/last-distance-offered-full
c9a1892bfb0370e0672e5849cba294ddfabe3c65 feat/latest-cutoff-only
4e8df006d75d8be2a1d8529ddf855c445854cba1 feat/near-me-by-search
45ab479062c6a1639facad636fc0cc0cf0fd9155 feat/proposed-to-close-schools
3bf2e8f262cbe058fda6de6f8ea3e050224a51b9 feat/school-detail-visualisations
1f80571b1ff217dc92b660a936c02b5f49d07f0b feat/umami-heatmap-recorder
609bb923d96aa5730131463ea35d5efdd404bf96 feature/ingest-independent-schools
94151c58ea38a9256d66d15161293486a500c7a2 fix/admissions-section-height
79246edc22961c2beb3520437e9d064d3b809d10 fix/annual-dag-ks4-national-selector
59ac9c10b97e0ef1143f57fea06b324e72ac3d4a fix/chart-marker-contrast
9f8dba227c95706ca3527bd48d381e7622cc0a5e fix/compare-chart-refetch-resilience
e74d3882ce78a141fa1a57daa3102d7a58852dc3 fix/compare-expert-fixes
80176cac4db4820e76ea2a156c7a2974bee2f203 fix/compare-final-review-mustfix
f579630fab6c456e26a6984a3e8eebdbe3184202 fix/compare-mockup-drift
dc85254ad2ddf134b4434065d4762cef374d2220 fix/compare-null-year-blanks-chart
43a2c4a6bc539b621f31655aec05ef319a25f343 fix/compare-refresh-and-fetch
d677b5453365b72c81d6df2de62b1fa0d05d684d fix/daily-dag-cache-invalidation
f6bb037c471553e8195b5a8b147467ce0d07a688 fix/detail-all-through
17bd4d5a5eb0b12ca79b97db587f14f7d671e85b fix/detail-chart-truthfulness
4e6be0ce65647410b4ff74f8b763207c28920c26 fix/detail-inclusion-admissions
fdda52ff0af3fae03a4b059a655973cdbb269f91 fix/detail-ofsted-correctness
e36125b24aba254a8d15c5c33b4d2a296e691995 fix/detail-provenance-anchoring
b31e71ac884df6507f567adc046c9fc52d9d310a fix/detail-report-card-render-date
32f8a02862be6a1d49f4c3928b17fcf15d4c94cc fix/detail-trend-chart-taller
e65688d600a86818fe21ae4c61ba27e5b6ec8d7c fix/e2e-brand-assertions
3adea73ee04cdedfab54b0351878b297f72756ad fix/e2e-compare-chips-phase
06e4898c30feaedc471f97aba28ddb0d61379f4d fix/e2e-compare-samephase
9abd020967670a855e80fe5a908c8048a3aa9f14 fix/e2e-distance-locator
acec8135e1ec7c9c3c255e5b23733a0ef862b590 fix/e2e-rankings-year-pick
5944d88f0b1517ef1ef56af1b62270d2cf28e717 fix/expert-signoff-mustfixes
2433101fa08be5df6f170d41790512ffe823d33e fix/font-cascade-and-map-palette
74ca76d150deec6725259d9637ea86d7bb90c683 fix/gias-legacy-fallback
bdaa05cd542f563ef74c45307cd8f7fc465193c9 fix/hero-fallback-and-sharp
d52d384cf23d282b44e9251176f8f3402d808600 fix/hero-map-ios-fullscreen
4043270a77bbe4edb18207fa5f1d94d4747fe12f fix/hero-mobile-and-wording
22e9eb2d48b0d6623e88fd67cb6ca8e5e4074583 fix/homepage-education-accuracy
8d50afef1e8a2b621b7344eadf475b0d609ad7a2 fix/leaflet-specificity-and-font-assertion
b2b2cad5acf534ae7a667d3fb2be15efff37c4a7 fix/list-map-report-card-signal
dc21e80a5e9eebd13aaab84735642eb77cef35e6 fix/mobile-cell-name-size
e5f7f4c959f024c073472122d858333a0f24866c fix/mobile-compare-polish
a00cbe916182d1e04661c750a1ce2e7ad27a68ae fix/mobile-sort-select-overflow
2fd997bfe640c419a6713e85df008463de7f56e8 fix/modal-keyboard-viewport
3e7705756776a0c44d27966dfc972023f1f38b69 fix/ofsted-link-text
ce422e64363e2b03c186ef316832b6de15502d67 fix/promote-status-token
4522cbf64560db1e1cd519e119aa466b42cd16a1 fix/proposed-to-close-copy
15da060e4af37fbae919e0edf25f108266de2585 fix/rankings-admissions-accuracy
6c872ce726f210433354ca38dc5314bc6a467534 fix/rankings-year-validation
1c1df7796194af3d47f8e5ac0a0fbe6f323700e7 fix/report-card-chip-alignment
b2dc4d0779ced02709429afaa5985acf4abda794 fix/results-map-ios-fullscreen
95a5783da1fc994df76cb97238b55596dce4cd8f fix/runtime-api-proxy
8a9ba30cc24e29653a72c03c0e817684b7db7c07 fix/sats-per-level-national
536832a524fc4f9ed858f047e00b9a4d9d429c46 fix/school-detail-nan-500
fef83b3bf244a9bf3cb4afa75dbc433d8725014f fix/schoolbar-sticky-offset
b0c5b6bb57c879477da12c37b15cff950c29ebd3 fix/secondary-anchors-button-affordance
261403bcd2b01aa4f26ee212e26302fc0f769bf9 fix/special-note-full-width
ea5249a2ea6faf6bfa5a1522644387ba59770ec8 fix/standardise-distance-units
e4565e9f158721d4df6b918f2b065de82851f8d9 fix/trends-chart-height
3aad5101a842539105022f9059e85c224fc5973a fix/welsh-establishment-leak
315f1feede70bdf3101d2fdd305b3d06e037fdac perf/batch-supplementary
d2dc78aeb599e16b7ef5019be2b360b08df463bc perf/compare-loading
e098ad4bd1130152705788d4773837b7d1e7112e perf/server-client-split
## Deleted: superseded by PR #110 (content preserved on feat/seo-crawl-hygiene-main)
786ec80dd4de4a3cb674a89e35b3b5e461639289 feat/seo-crawl-hygiene
a5ac0bcd1b37bc10dcbce88f8601d01bf7b3eaaf feat/england-only-corpus
+92 -28
View File
@@ -1587,47 +1587,111 @@ test('a Welsh school URL 404s while an English one still resolves', async ({ pag
expect(welsh?.status(), 'a Welsh school should no longer resolve').toBe(404);
});
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
async function sitemapChildren(page: Page): Promise<string[]> {
const res = await page.request.get('/sitemap.xml');
expect(res.ok()).toBeTruthy();
const xml = await res.text();
const index = await res.text();
expect(index).toContain('<sitemapindex');
return [...index.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1]);
}
const urlCount = (xml.match(/<url>/g) ?? []).length;
expect(urlCount, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
test('the sitemap index names children that all resolve', async ({ page }) => {
const index = await (await page.request.get('/sitemap.xml')).text();
// An index holds <sitemap> entries only; mixing in <url> is invalid.
expect(index).not.toContain('<url>');
// 401559 (Cardiff) and 402426 (ACT Schools, Cardiff) were both submitted
// before the England-only filter landed.
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
const locs = await sitemapChildren(page);
expect(locs.length).toBeGreaterThanOrEqual(2);
for (const loc of locs) {
expect(loc.startsWith('https://www.schoolcompare.co.uk/sitemaps/')).toBeTruthy();
const child = await page.request.get(new URL(loc).pathname);
expect(child.ok(), `${loc} should resolve`).toBeTruthy();
expect(await child.text()).toContain('<urlset');
}
});
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
const locs = await sitemapChildren(page);
let total = 0;
for (const loc of locs) {
const xml = await (await page.request.get(new URL(loc).pathname)).text();
total += (xml.match(/<url>/g) ?? []).length;
// 401559 (Adamsdown, Cardiff) and 402426 (ACT Schools, Cardiff) were both
// submitted before the England-only filter landed.
expect(xml).not.toContain('/school/401559');
expect(xml).not.toContain('/school/402426');
}
expect(total, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
});
test('the sitemap invents no priority or changefreq', async ({ page }) => {
const [first] = await sitemapChildren(page);
expect(first).toBeTruthy();
const xml = await (await page.request.get(new URL(first).pathname)).text();
// Google ignores both. They were noise dressed as signal.
expect(xml).not.toContain('<priority>');
expect(xml).not.toContain('<changefreq>');
});
/*
* Staging must not be indexable (spec 2026-08-20, W1 hygiene).
* Canonical URLs (spec 2026-08-20, W1).
*
* These journeys only ever run against staging — deploy.yml passes
* STAGING_BASE_URL, and promote.yml only smoke-polls production without
* Playwright — so asserting the noindex header here is safe.
* Every indexable route declares exactly one canonical, on the www host, with
* no query string. The homepage's eleven search params filter a result set
* rather than making a new document, so they all collapse onto "/".
*/
test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => {
const res = await page.request.get('/');
expect(res.ok()).toBeTruthy();
const CANONICAL_ROUTES: Array<[string, string]> = [
['/', 'https://www.schoolcompare.co.uk/'],
['/rankings', 'https://www.schoolcompare.co.uk/rankings'],
['/admissions', 'https://www.schoolcompare.co.uk/admissions'],
];
const tag = res.headers()['x-robots-tag'];
expect(tag, 'staging must send X-Robots-Tag').toBeTruthy();
expect(tag).toContain('noindex');
for (const [path, expected] of CANONICAL_ROUTES) {
test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => {
await page.goto(path);
const hrefs = await page.locator('link[rel="canonical"]').evaluateAll(
(els) => els.map((e) => e.getAttribute('href')));
expect(hrefs, `${path} should declare one canonical`).toHaveLength(1);
expect(hrefs[0]).toBe(expected);
});
}
// The other half, and the reason this is one test rather than two: a
// Disallow would stop Google fetching the page at all, so it would never
// see the noindex above. The two only work together.
const robots = await (await page.request.get('/robots.txt')).text();
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
test('a filtered homepage still canonicalises to the bare root', async ({ page }) => {
await page.goto('/?search=primary&phase=primary&sort=name&page=2');
const href = await page.locator('link[rel="canonical"]').first()
.getAttribute('href');
expect(href).toBe('https://www.schoolcompare.co.uk/');
});
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
const list = await page.request.get('/api/schools?search=primary&per_page=1');
const [first] = (await list.json()).schools ?? [];
test('a school page canonicalises to its own slug on the www host', async ({ page }) => {
const res = await page.request.get('/api/schools?search=primary&per_page=1');
expect(res.ok()).toBeTruthy();
const [first] = (await res.json()).schools ?? [];
expect(first, 'no school available').toBeTruthy();
const res = await page.request.get(`/school/${first.urn}-x`);
expect(res.headers()['x-robots-tag']).toContain('noindex');
await page.goto(`/school/${first.urn}-x`);
const href = await page.locator('link[rel="canonical"]').first()
.getAttribute('href');
expect(href).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\/\d+-/);
});
test('a bare /compare is indexable, a parameterised one is not', async ({ page }) => {
await page.goto('/compare');
await expect(page.locator('meta[name="robots"]')).toHaveCount(0);
const [a, b] = await twoPrimaryUrns(page);
await page.goto(`/compare?urns=${a},${b}`);
const robots = await page.locator('meta[name="robots"]').first()
.getAttribute('content');
expect(robots).toContain('noindex');
expect(robots).toContain('follow');
// noindex but follow: the links out to each school page still count, so the
// canonical must still be present and point at the bare path.
const canonical = await page.locator('link[rel="canonical"]').first()
.getAttribute('href');
expect(canonical).toBe('https://www.schoolcompare.co.uk/compare');
});
+53
View File
@@ -0,0 +1,53 @@
import { metadata as homeMetadata } from '@/app/page';
import { metadata as rankingsMetadata } from '@/app/rankings/page';
import { metadata as admissionsMetadata } from '@/app/admissions/page';
import { generateMetadata as compareMetadata } from '@/app/compare/page';
describe('canonical URLs', () => {
it('the homepage canonicalises to the bare root', () => {
// page.tsx reads eleven search params. Without this, every filter
// combination is a crawlable near-duplicate of the one page we want to
// rank for "compare schools".
expect(homeMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/');
});
it('rankings canonicalises to the bare path', () => {
expect(rankingsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/rankings');
});
it('admissions canonicalises to the bare path', () => {
expect(admissionsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/admissions');
});
});
describe('/compare indexability', () => {
it('the bare compare page is indexable and canonical to itself', async () => {
// This is the landing page for the "compare schools" head term.
const meta = await compareMetadata({ searchParams: Promise.resolve({}) });
expect(meta.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/compare');
expect(meta.robots).toBeUndefined();
});
it('a comparison of specific schools is noindex, follow', async () => {
// ~317 million pairs before triples. Indexing the parameter space would
// swamp everything else in the corpus.
const meta = await compareMetadata({
searchParams: Promise.resolve({ urns: '100001,100002' }),
});
expect(meta.robots).toEqual({ index: false, follow: true });
});
it('a parameterised comparison still canonicalises to the bare path', async () => {
// follow:true plus a canonical means the outbound links to each school
// page still pass value even though this URL is not indexed.
const meta = await compareMetadata({
searchParams: Promise.resolve({ urns: '100001,100002' }),
});
expect(meta.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/compare');
});
});
+27
View File
@@ -0,0 +1,27 @@
import { SITE_URL, absoluteUrl } from '@/lib/site';
describe('SITE_URL', () => {
it('is the www host, which is the one that serves a 200', () => {
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
// is a wasted signal, so every absolute URL we emit must already be www.
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
});
it('has no trailing slash, so joins never double up', () => {
expect(SITE_URL.endsWith('/')).toBe(false);
});
});
describe('absoluteUrl', () => {
it('joins a rooted path', () => {
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('joins a path missing its leading slash', () => {
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('maps the site root to a bare trailing slash', () => {
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
});
});
+2
View File
@@ -1,3 +1,4 @@
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
import { AdmissionsView } from '@/components/AdmissionsView';
@@ -7,6 +8,7 @@ export const metadata: Metadata = {
title: 'School Admissions Guide',
description:
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
alternates: { canonical: absoluteUrl('/admissions') },
};
export default function AdmissionsPage() {
+28 -7
View File
@@ -5,6 +5,7 @@
import { fetchComparison, fetchMetrics } from '@/lib/api';
import { ComparisonView } from '@/components/ComparisonView';
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
interface ComparePageProps {
@@ -14,13 +15,33 @@ interface ComparePageProps {
}>;
}
export const metadata: Metadata = {
title: 'Compare Schools',
description:
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
keywords:
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
};
/**
* Indexability depends on the query string, so this cannot be a static export.
*
* Bare /compare is the landing page for the "compare schools" head term and
* stays indexable. /compare?urns=… is an unbounded parameter space — 25,193
* schools make ~317 million pairs — so it goes noindex. It stays `follow` and
* keeps a canonical to the bare path, so the links out to each school page
* still count.
*/
export async function generateMetadata(
{ searchParams }: ComparePageProps,
): Promise<Metadata> {
const { urns } = await searchParams;
const base: Metadata = {
title: 'Compare Schools',
description:
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
keywords:
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
alternates: { canonical: absoluteUrl('/compare') },
};
if (!urns) return base;
return { ...base, robots: { index: false, follow: true } };
}
// Dynamic via searchParams; remove force-dynamic so internal data fetches
// can still use Next.js's per-call revalidate cache.
+3 -2
View File
@@ -5,6 +5,7 @@ import { Navigation } from '@/components/Navigation';
import { Footer } from '@/components/Footer';
import { ComparisonToast } from '@/components/ComparisonToast';
import { ComparisonProvider } from '@/context/ComparisonProvider';
import { SITE_URL } from '@/lib/site';
import './globals.css';
// Manrope carries headings and key messaging — the guideline's "friendly,
@@ -57,12 +58,12 @@ export const metadata: Metadata = {
// No `icons` key on purpose: setting it here would override the file
// conventions. app/icon.svg and app/apple-icon.tsx are the source, and
// app/opengraph-image.tsx supplies og:image and twitter:image.
metadataBase: new URL('https://schoolcompare.co.uk'),
metadataBase: new URL(SITE_URL),
openGraph: {
type: 'website',
title: 'schoolcompare | Compare School Performance',
description: 'Compare primary and secondary school SATs and GCSE performance across England',
url: 'https://schoolcompare.co.uk',
url: SITE_URL,
siteName: 'schoolcompare',
},
twitter: {
+5
View File
@@ -3,6 +3,7 @@
* Main landing page with school search and browsing
*/
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
import { fetchSchools, fetchFilters, fetchDataInfo } from '@/lib/api';
import { formatAcademicYear } from '@/lib/utils';
@@ -35,6 +36,10 @@ interface HomePageProps {
export const metadata: Metadata = {
title: { absolute: 'schoolcompare | Compare every school in England' },
description: 'Search and compare school performance across England',
// This page reads eleven search params. They filter a result set; they do
// not make a new document. Collapsing every combination onto "/" stops the
// homepage competing with itself for its own head terms.
alternates: { canonical: absoluteUrl('/') },
};
// The page reads searchParams, which makes rendering dynamic by default.
+4
View File
@@ -5,6 +5,7 @@
import { fetchRankings, fetchFilters, fetchMetrics } from '@/lib/api';
import { RankingsView } from '@/components/RankingsView';
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
interface RankingsPageProps {
@@ -20,6 +21,9 @@ export const metadata: Metadata = {
title: 'School Rankings',
description: 'Top-ranked schools by SATs and GCSE performance across England',
keywords: 'school rankings, top schools, best schools, KS2 rankings, KS4 rankings, school league tables',
// Param forms (?metric=&local_authority=&year=&phase=) collapse here for
// now. W3 replaces them with real indexable paths.
alternates: { canonical: absoluteUrl('/rankings') },
};
// Dynamic via searchParams; remove force-dynamic so internal data fetches
+2 -1
View File
@@ -4,6 +4,7 @@
*/
import { MetadataRoute } from 'next';
import { absoluteUrl } from '@/lib/site';
export default function robots(): MetadataRoute.Robots {
return {
@@ -14,6 +15,6 @@ export default function robots(): MetadataRoute.Robots {
disallow: ['/api/', '/_next/'],
},
],
sitemap: 'https://schoolcompare.co.uk/sitemap.xml',
sitemap: absoluteUrl('/sitemap.xml'),
};
}
+3 -2
View File
@@ -15,6 +15,7 @@ import {
} from '@/lib/schoolSections';
import { parseSchoolSlug, schoolUrl } from '@/lib/utils';
import type { NationalAverages } from '@/lib/types';
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
/**
@@ -97,7 +98,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
title,
description,
type: 'website',
url: `https://schoolcompare.co.uk${canonicalPath}`,
url: absoluteUrl(canonicalPath),
siteName: 'schoolcompare',
},
twitter: {
@@ -106,7 +107,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
description,
},
alternates: {
canonical: `https://schoolcompare.co.uk${canonicalPath}`,
canonical: absoluteUrl(canonicalPath),
},
};
} catch {
+2 -26
View File
@@ -1,32 +1,8 @@
/**
* Runtime proxy for /sitemap.xml → the FastAPI backend's generated sitemap.
*
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
* baking the backend host into the build, so one image works in every
* environment. robots.ts points crawlers here.
*/
import { NextResponse } from 'next/server';
import { proxySitemap } from '@/lib/sitemapProxy';
export const dynamic = 'force-dynamic';
export const runtime = 'nodejs';
function backendOrigin(): string {
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
return base.replace(/\/api$/, '');
}
export async function GET() {
let upstream: Response;
try {
upstream = await fetch(`${backendOrigin()}/sitemap.xml`, { cache: 'no-store' });
} catch {
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
}
const body = await upstream.text();
return new NextResponse(body, {
status: upstream.status,
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
});
return proxySitemap('/sitemap.xml');
}
@@ -0,0 +1,24 @@
import { NextResponse } from 'next/server';
import { proxySitemap } from '@/lib/sitemapProxy';
export const dynamic = 'force-dynamic';
export const runtime = 'nodejs';
/**
* Children are /sitemaps/static.xml and /sitemaps/schools-{n}.xml. The name is
* validated here rather than passed through, so this route cannot be used to
* reach arbitrary backend paths.
*/
const CHILD = /^(static|schools-\d+)\.xml$/;
export async function GET(
_request: Request,
{ params }: { params: Promise<{ parts: string[] }> },
) {
const { parts } = await params;
const name = parts.join('/');
if (!CHILD.test(name)) {
return new NextResponse('Not found', { status: 404 });
}
return proxySitemap(`/sitemaps/${name}`);
}
+19
View File
@@ -0,0 +1,19 @@
/**
* The one place the site's absolute origin is written down.
*
* The apex domain 301s to www at Cloudflare, so www is the host that actually
* serves a 200. Canonicals, og:url, sitemap <loc> entries and the robots.txt
* Sitemap: line must all agree with it — a canonical pointing at a redirect
* makes Google resolve the hop before it can consolidate the signal.
*
* backend/app.py holds the same value as BASE_URL for the sitemap. The two are
* asserted against each other by the e2e journeys rather than shared at build
* time, because the backend and frontend ship as separate images.
*/
export const SITE_URL = 'https://www.schoolcompare.co.uk';
/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */
export function absoluteUrl(path: string): string {
const rooted = path.startsWith('/') ? path : `/${path}`;
return `${SITE_URL}${rooted}`;
}
+29
View File
@@ -0,0 +1,29 @@
/**
* Runtime proxy for the sitemap family → the FastAPI backend.
*
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
* baking the backend host into the build, so one image works in every
* environment. robots.ts points crawlers at /sitemap.xml, which is the index;
* the index names children under /sitemaps/, which land on the same proxy.
*/
import { NextResponse } from 'next/server';
function backendOrigin(): string {
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
return base.replace(/\/api$/, '');
}
export async function proxySitemap(path: string): Promise<NextResponse> {
let upstream: Response;
try {
upstream = await fetch(`${backendOrigin()}${path}`, { cache: 'no-store' });
} catch {
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
}
const body = await upstream.text();
return new NextResponse(body, {
status: upstream.status,
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
});
}
-31
View File
@@ -55,37 +55,6 @@ const nextConfig = {
// Headers for caching and security
async headers() {
return [
{
/*
* Keep non-production hosts out of the index.
*
* Staging serves the same image as production off stx., so without
* this it is a full crawlable duplicate of the site.
*
* X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling,
* which is not the same as blocking indexing — a disallowed URL can
* still be indexed from external links, and worse, blocking the crawl
* means Google never fetches the page and never sees a noindex at all.
* Staging therefore stays crawlable and answers "noindex" when crawled.
*
* Matched on the staging host explicitly rather than "any host that is
* not production". The inverted form is tempting because it would cover
* future environments automatically, but its failure mode is
* deindexing production if the Host header ever arrives rewritten by a
* proxy. This form's failure mode is a new environment being indexable
* until someone adds it here — recoverable, where the other is not.
*
* Any new non-production hostname must be added to this list.
*/
source: '/:path*',
has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }],
headers: [
{
key: 'X-Robots-Tag',
value: 'noindex, nofollow',
},
],
},
{
source: '/:path*',
headers: [