Compare commits
24
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
12cbbda3c8 | ||
|
|
6f749ed21f | ||
|
|
d3c63ccc6d | ||
|
|
b93eb3a691 | ||
|
|
6b871ce1e9 | ||
|
|
c981d89137 | ||
|
|
de5e790112 | ||
|
|
42138fc402 | ||
|
|
c5af476213 | ||
|
|
de853b90b3 | ||
|
|
759d9f5cea | ||
|
|
555d3f0a7d | ||
|
|
ecc847091c | ||
|
|
b187a478c9 | ||
|
|
c0547c45e5 | ||
|
|
4a3928df9f | ||
|
|
07c97a46c5 | ||
|
|
bb81337aba | ||
|
|
f928a15c1e | ||
|
|
2208ad93c1 | ||
|
|
24f3cb4c65 | ||
|
|
3816b92d06 | ||
|
|
51ce2d6373 | ||
|
|
5ddc7314fd |
No files matched your search
+300
-50
@@ -7,6 +7,7 @@ Uses real data from UK Government Compare School Performance downloads.
|
||||
import hashlib
|
||||
import re
|
||||
from contextlib import asynccontextmanager
|
||||
from datetime import datetime, timezone
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
@@ -34,6 +35,7 @@ from .data_loader import (
|
||||
search_schools_typesense,
|
||||
)
|
||||
from .data_loader import get_data_info as get_db_info
|
||||
from .places import build_place_registry
|
||||
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||
from .utils import clean_for_json, convert_to_native
|
||||
|
||||
@@ -48,11 +50,20 @@ PHASE_GROUPS: dict[str, set[str]] = {
|
||||
"all-through": {"all-through"},
|
||||
}
|
||||
|
||||
BASE_URL = "https://schoolcompare.co.uk"
|
||||
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
|
||||
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
|
||||
BASE_URL = "https://www.schoolcompare.co.uk"
|
||||
MAX_SLUG_LENGTH = 60
|
||||
|
||||
# In-memory sitemap cache
|
||||
_sitemap_xml: str | None = None
|
||||
# In-memory sitemap cache: name -> XML. Populated on startup and by the admin
|
||||
# regenerate endpoint after a pipeline run.
|
||||
_sitemaps: dict[str, str] | None = None
|
||||
|
||||
# Built from the same DataFrame the sitemap uses, so places and sitemap can
|
||||
# never describe different corpora. Reset by the same admin endpoint.
|
||||
_place_registry: dict | None = None
|
||||
|
||||
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
@@ -70,43 +81,206 @@ def _school_url(urn: int, school_name: str) -> str:
|
||||
return f"/school/{urn}-{slug}"
|
||||
|
||||
|
||||
def build_sitemap() -> str:
|
||||
"""Generate sitemap XML from in-memory school data. Returns the XML string."""
|
||||
# Routes worth submitting that are not a school page. /admissions was missing
|
||||
# from the sitemap entirely despite being a static, indexable guide.
|
||||
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
|
||||
|
||||
|
||||
# A page has something a search result could state if any of these is present
|
||||
# in any year. Shared by _has_publishable_data and the per-school check in
|
||||
# _school_sitemap_rows so the two can never drift.
|
||||
_PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade")
|
||||
|
||||
|
||||
def _has_publishable_data(row) -> bool:
|
||||
"""True when a school page has something a search result could state.
|
||||
|
||||
A school with no results in any year and no Ofsted grade renders an empty
|
||||
page. Submitting it spends crawl budget and drags the corpus-wide quality
|
||||
signal down, so it stays out of the sitemap. The page itself still resolves
|
||||
for anyone who has the URL.
|
||||
"""
|
||||
for field in _PUBLISHABLE_FIELDS:
|
||||
value = row.get(field)
|
||||
if value is not None and not pd.isna(value):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _url_element(loc: str, lastmod: str | None = None) -> str:
|
||||
"""One <url> entry. No priority or changefreq — Google ignores both."""
|
||||
body = f"<loc>{loc}</loc>"
|
||||
if lastmod:
|
||||
body += f"<lastmod>{lastmod}</lastmod>"
|
||||
return f" <url>{body}</url>"
|
||||
|
||||
|
||||
def _school_sitemap_rows(df) -> list[str]:
|
||||
"""A <url> element per school that has something to show.
|
||||
|
||||
lastmod comes from the school's Ofsted date where there is one and is
|
||||
omitted otherwise. An always-now lastmod is a claim Google learns to
|
||||
distrust; an absent one honestly means "unknown".
|
||||
"""
|
||||
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
|
||||
return []
|
||||
|
||||
rows: list[str] = []
|
||||
seen: set[int] = set()
|
||||
|
||||
# Publishable is a property of the SCHOOL, not of its latest row.
|
||||
#
|
||||
# The first cut tested the latest year's row alone, which quietly dropped
|
||||
# every school that has results in its history but a null row for the most
|
||||
# recent year — a school that stopped reporting, or whose figures were
|
||||
# suppressed for small-cohort disclosure. The Mallard Academy (150367) is
|
||||
# the case that caught it: real KS2 results for 2015-16 through 2018-19,
|
||||
# then null rows for 2022-23 onward. Its page shows all four years; the
|
||||
# sitemap omitted it. Roughly 220 schools were affected.
|
||||
publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
publishable: set[int] = (
|
||||
set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int))
|
||||
if publishable_cols else set()
|
||||
)
|
||||
|
||||
# Latest row per URN first, so a school's most recent Ofsted date wins.
|
||||
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
|
||||
|
||||
for _, row in ordered.iterrows():
|
||||
urn = int(row["urn"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
if urn not in publishable:
|
||||
continue
|
||||
|
||||
lastmod = None
|
||||
ofsted_date = row.get("ofsted_date")
|
||||
if ofsted_date is not None and not pd.isna(ofsted_date):
|
||||
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
|
||||
|
||||
rows.append(_url_element(
|
||||
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
|
||||
|
||||
return rows
|
||||
|
||||
|
||||
# Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to
|
||||
# scan by eye in Search Console, which is the point of splitting at all:
|
||||
# coverage is reported per submitted sitemap, so one file per page family is
|
||||
# what makes an indexation problem attributable to a family.
|
||||
SITEMAP_CHUNK_SIZE = 10_000
|
||||
|
||||
# Children are served under /sitemaps/ because Next.js only treats a whole
|
||||
# bracketed path segment as dynamic — a route folder named "sitemap-[...parts]"
|
||||
# is read as a literal static segment and never matches.
|
||||
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
||||
|
||||
|
||||
def get_place_registry() -> dict:
|
||||
"""The place registry, built once and cached for the process."""
|
||||
global _place_registry
|
||||
if _place_registry is None:
|
||||
_place_registry = build_place_registry(load_school_data())
|
||||
return _place_registry
|
||||
|
||||
|
||||
def _urlset(rows: list[str]) -> str:
|
||||
return "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
||||
*rows,
|
||||
"</urlset>",
|
||||
])
|
||||
|
||||
|
||||
def _place_url(place) -> str:
|
||||
"""The canonical path for a place. Two namespaces, per the spec.
|
||||
|
||||
Towns and localities share /schools/[place]; authorities take their own
|
||||
prefix because 67 town names collide with an authority name and neither
|
||||
set contains the other.
|
||||
"""
|
||||
if place.kind == "authority":
|
||||
return f"/schools/authority/{place.slug}"
|
||||
if place.kind == "outcode":
|
||||
return f"/schools/near/{place.slug}"
|
||||
return f"/schools/{place.slug}"
|
||||
|
||||
|
||||
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
|
||||
"""A <url> per place, plus a phase variant wherever that phase clears the
|
||||
threshold on its own.
|
||||
|
||||
Phase is part of the query — "primary schools in beccles" — so each
|
||||
variant is its own indexable page. Submitting only the bare place URL left
|
||||
~950 of them reachable by nothing: absent from every sitemap, and not
|
||||
linked from the place page either.
|
||||
"""
|
||||
rows: list[str] = []
|
||||
for p in sorted(get_place_registry().values(), key=lambda p: (p.kind, p.slug)):
|
||||
if p.kind not in kinds:
|
||||
continue
|
||||
rows.append(_url_element(BASE_URL + _place_url(p)))
|
||||
# Outcodes carry no phase variants: nobody searches "primary schools
|
||||
# in SW11", so the routes do not exist to submit.
|
||||
if p.kind == "outcode":
|
||||
continue
|
||||
for phase in ("primary", "secondary"):
|
||||
if p.publishes_phase(phase):
|
||||
rows.append(_url_element(f"{BASE_URL}{_place_url(p)}/{phase}"))
|
||||
return rows
|
||||
|
||||
|
||||
def build_sitemaps() -> dict[str, str]:
|
||||
"""Build the sitemap index and every child, keyed by name."""
|
||||
df = load_school_data()
|
||||
|
||||
static_urls = [
|
||||
(BASE_URL + "/", "daily", "1.0"),
|
||||
(BASE_URL + "/rankings", "weekly", "0.8"),
|
||||
(BASE_URL + "/compare", "weekly", "0.8"),
|
||||
children: dict[str, str] = {
|
||||
"static.xml": _urlset(
|
||||
[_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]),
|
||||
}
|
||||
|
||||
school_rows = _school_sitemap_rows(df)
|
||||
# Always emit at least one school child, so the index shape is stable even
|
||||
# on an empty database.
|
||||
chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE]
|
||||
for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"schools-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# Separate children per family: Search Console reports coverage per
|
||||
# submitted sitemap, which is how the location layer's indexation is
|
||||
# measured apart from the school pages'.
|
||||
for label, kinds in (("places", ("town", "locality", "authority")),
|
||||
("outcodes", ("outcode",))):
|
||||
rows = _place_sitemap_rows(kinds)
|
||||
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
|
||||
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"{label}-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# On a sitemap index, lastmod means "when this sitemap file last changed",
|
||||
# so generation time is the correct value here — unlike on a <url>, where
|
||||
# it would be a claim about content we cannot support.
|
||||
generated = datetime.now(timezone.utc).date().isoformat()
|
||||
index_rows = [
|
||||
f" <sitemap><loc>{BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}</loc>"
|
||||
f"<lastmod>{generated}</lastmod></sitemap>"
|
||||
for name in children
|
||||
]
|
||||
index = "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
||||
*index_rows,
|
||||
"</sitemapindex>",
|
||||
])
|
||||
return {**children, "sitemap.xml": index}
|
||||
|
||||
lines = ['<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
|
||||
|
||||
for url, freq, priority in static_urls:
|
||||
lines.append(
|
||||
f" <url><loc>{url}</loc>"
|
||||
f"<changefreq>{freq}</changefreq>"
|
||||
f"<priority>{priority}</priority></url>"
|
||||
)
|
||||
|
||||
if not df.empty and "urn" in df.columns and "school_name" in df.columns:
|
||||
seen = set()
|
||||
for _, row in df[["urn", "school_name"]].drop_duplicates(subset="urn").iterrows():
|
||||
urn = int(row["urn"])
|
||||
name = str(row["school_name"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
path = _school_url(urn, name)
|
||||
lines.append(
|
||||
f" <url><loc>{BASE_URL}{path}</loc>"
|
||||
f"<changefreq>monthly</changefreq>"
|
||||
f"<priority>0.6</priority></url>"
|
||||
)
|
||||
|
||||
lines.append("</urlset>")
|
||||
return "\n".join(lines)
|
||||
def build_sitemap() -> str:
|
||||
"""The sitemap index. Kept for `lifespan` and the admin endpoint."""
|
||||
return build_sitemaps()["sitemap.xml"]
|
||||
|
||||
|
||||
def clean_filter_values(series: pd.Series) -> list[str]:
|
||||
@@ -284,7 +458,7 @@ def validate_postcode(postcode: Optional[str]) -> Optional[str]:
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
"""Application lifespan - startup and shutdown events."""
|
||||
global _sitemap_xml
|
||||
global _sitemaps
|
||||
print("Loading school data from marts...")
|
||||
df = load_school_data()
|
||||
if df.empty:
|
||||
@@ -294,9 +468,9 @@ async def lifespan(app: FastAPI):
|
||||
# Pre-compute the latest-year snapshot so the first search request is fast
|
||||
await asyncio.to_thread(load_latest_school_data)
|
||||
try:
|
||||
_sitemap_xml = build_sitemap()
|
||||
n = _sitemap_xml.count("<url>")
|
||||
print(f"Sitemap built: {n} URLs.")
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.")
|
||||
except Exception as e:
|
||||
print(f"Warning: sitemap build failed on startup: {e}")
|
||||
|
||||
@@ -1007,6 +1181,66 @@ async def get_rankings(
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/places")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def list_places(request: Request):
|
||||
"""Every published place. The sitemap and the link modules read this."""
|
||||
registry = get_place_registry()
|
||||
return {"places": [
|
||||
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
|
||||
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
|
||||
]}
|
||||
|
||||
|
||||
@app.get("/api/places/{kind}/{slug}")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_place(request: Request, kind: str, slug: str,
|
||||
phase: Optional[str] = None):
|
||||
"""One place: its schools ranked, and its local averages."""
|
||||
if kind not in VALID_PLACE_KINDS:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
place = get_place_registry().get(f"{kind}:{slug}")
|
||||
if place is None:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
df = load_latest_school_data()
|
||||
rows = df[df["urn"].isin(place.urns)]
|
||||
|
||||
if phase:
|
||||
wanted = PHASE_GROUPS.get(phase.lower())
|
||||
if wanted and "phase" in rows.columns:
|
||||
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
|
||||
|
||||
# The metric the page ranks on, which is also the one it averages.
|
||||
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
|
||||
if metric in rows.columns:
|
||||
rows = rows.sort_values(metric, ascending=False, na_position="last")
|
||||
|
||||
averages = {
|
||||
m: (None if m not in rows.columns or rows[m].dropna().empty
|
||||
else float(rows[m].dropna().mean()))
|
||||
for m in ("rwm_expected_pct", "attainment_8_score")
|
||||
}
|
||||
|
||||
cols = [c for c in SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
|
||||
"rwm_expected_pct", "attainment_8_score",
|
||||
"total_pupils"]
|
||||
if c in rows.columns]
|
||||
|
||||
return {
|
||||
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
|
||||
"count": len(place.urns),
|
||||
"parent_authority": place.parent_authority,
|
||||
# Only phases that clear the threshold, so the page links
|
||||
# variants that exist rather than 404s.
|
||||
"phases": [ph for ph in ("primary", "secondary")
|
||||
if place.publishes_phase(ph)]},
|
||||
"schools": clean_for_json(rows[cols]),
|
||||
"averages": averages,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/data-info")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_data_info(request: Request):
|
||||
@@ -1087,16 +1321,28 @@ async def robots_txt():
|
||||
return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain")
|
||||
|
||||
|
||||
@app.get("/sitemap.xml")
|
||||
async def sitemap_xml():
|
||||
"""Serve sitemap.xml for search engine indexing."""
|
||||
global _sitemap_xml
|
||||
if _sitemap_xml is None:
|
||||
def _serve_sitemap(name: str) -> Response:
|
||||
global _sitemaps
|
||||
if _sitemaps is None:
|
||||
try:
|
||||
_sitemap_xml = build_sitemap()
|
||||
_sitemaps = build_sitemaps()
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
|
||||
return Response(content=_sitemap_xml, media_type="application/xml")
|
||||
if name not in _sitemaps:
|
||||
raise HTTPException(status_code=404, detail="No such sitemap")
|
||||
return Response(content=_sitemaps[name], media_type="application/xml")
|
||||
|
||||
|
||||
@app.get("/sitemap.xml")
|
||||
async def sitemap_xml():
|
||||
"""Serve the sitemap index."""
|
||||
return _serve_sitemap("sitemap.xml")
|
||||
|
||||
|
||||
@app.get("/sitemaps/{name}")
|
||||
async def sitemap_child(name: str):
|
||||
"""Serve a child sitemap (static.xml, or schools-N.xml)."""
|
||||
return _serve_sitemap(name)
|
||||
|
||||
|
||||
@app.post("/api/admin/regenerate-sitemap")
|
||||
@@ -1106,10 +1352,14 @@ async def regenerate_sitemap(
|
||||
_: bool = Depends(verify_admin_api_key),
|
||||
):
|
||||
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
|
||||
global _sitemap_xml
|
||||
_sitemap_xml = build_sitemap()
|
||||
n = _sitemap_xml.count("<url>")
|
||||
return {"status": "ok", "urls": n}
|
||||
global _sitemaps, _place_registry
|
||||
# Places and sitemap are rebuilt together — they read the same marts, and
|
||||
# letting them drift apart would submit URLs for places that no longer
|
||||
# exist.
|
||||
_place_registry = None
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
|
||||
|
||||
|
||||
# Mount static files directly (must be after all routes to avoid catching API calls)
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Curated London localities, defined by the postcode districts they cover.
|
||||
|
||||
The GIAS `town` field puts 1,819 London schools under the single value
|
||||
"London", so it cannot answer "schools in Battersea" — a query that appears in
|
||||
the Search Console baseline. No single field can: parliamentary constituency
|
||||
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
|
||||
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
|
||||
and colloquial rather than administrative.
|
||||
|
||||
So this is curated. Where a locality ends is a judgement, not a fact, and a
|
||||
reviewable file is the honest place for a judgement. No new ingestion is
|
||||
needed — the corpus already carries postcodes.
|
||||
|
||||
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
|
||||
mirrors it for anyone querying the warehouse directly; the backend image does
|
||||
not contain `pipeline/`, which is why the module rather than the seed is
|
||||
canonical. Same arrangement as `backend/gias_codes.py`.
|
||||
|
||||
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
|
||||
published, so a typo produces no page rather than an empty one. Places that
|
||||
fail that check are logged at startup, because a locality you meant to publish
|
||||
quietly not appearing is the failure worth hearing about.
|
||||
|
||||
Two rules for anything added here.
|
||||
|
||||
**Sub-borough districts only.** A London borough is a local authority and
|
||||
already has a page at /schools/authority/[la] covering all of its schools; a
|
||||
locality defined by two or three outcodes would be a partial, near-duplicate
|
||||
subset of it. Hackney, Islington, Greenwich and Ealing were all in the first
|
||||
draft for that reason and have been removed.
|
||||
|
||||
**The slug must not match a GIAS town.** "Richmond" did — GIAS has a Richmond
|
||||
in North Yorkshire with 37 schools — so the London one could never publish.
|
||||
The registry skips any locality that collides and logs it.
|
||||
"""
|
||||
|
||||
# slug -> (display name, outcodes)
|
||||
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
|
||||
"battersea": ("Battersea", ("SW11",)),
|
||||
"canary-wharf": ("Canary Wharf", ("E14",)),
|
||||
"clapham": ("Clapham", ("SW4",)),
|
||||
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
|
||||
"peckham": ("Peckham", ("SE15",)),
|
||||
"brixton": ("Brixton", ("SW2", "SW9")),
|
||||
"camden-town": ("Camden Town", ("NW1",)),
|
||||
"wimbledon": ("Wimbledon", ("SW19",)),
|
||||
"putney": ("Putney", ("SW15",)),
|
||||
"fulham": ("Fulham", ("SW6",)),
|
||||
"chiswick": ("Chiswick", ("W4",)),
|
||||
"stratford": ("Stratford", ("E15",)),
|
||||
"walthamstow": ("Walthamstow", ("E17",)),
|
||||
"tooting": ("Tooting", ("SW17",)),
|
||||
"dulwich": ("Dulwich", ("SE21", "SE22")),
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
"""The place registry: what places the site publishes, and what is in each.
|
||||
|
||||
One module owns this question. The pages, the sitemap and the internal-link
|
||||
modules all read from here, so the threshold and the collision rules exist in
|
||||
exactly one place and are testable without a browser or a database.
|
||||
|
||||
Two namespaces, never one. 67 viable town names collide with a local
|
||||
authority name, and the authority is the larger set in only 43 of them —
|
||||
postal towns cross authority boundaries, so neither can absorb the other.
|
||||
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Five schools with publishable data. Below this a place has nothing to say
|
||||
# that a list of schools does not, and publishing it is index bloat.
|
||||
MIN_SCHOOLS = 5
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Place:
|
||||
kind: str # "town" | "locality" | "authority" | "outcode"
|
||||
slug: str
|
||||
name: str
|
||||
urns: tuple[int, ...]
|
||||
parent_authority: str | None # authority NAME, for the 301 target
|
||||
# URNs per phase, so the per-phase threshold can be applied without
|
||||
# re-querying. A place with 30 primaries and 2 secondaries publishes a
|
||||
# primary variant and no secondary one.
|
||||
phase_urns: dict[str, tuple[int, ...]] = field(default_factory=dict)
|
||||
|
||||
def publishes_phase(self, phase: str) -> bool:
|
||||
return len(self.phase_urns.get(phase, ())) >= MIN_SCHOOLS
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
return f"{self.kind}:{self.slug}"
|
||||
|
||||
|
||||
def _publishable_urns(df) -> set[int]:
|
||||
"""URNs with something a page could state, deduplicated across years."""
|
||||
from backend.app import _PUBLISHABLE_FIELDS
|
||||
|
||||
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
if not cols:
|
||||
return set()
|
||||
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
||||
|
||||
|
||||
def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]:
|
||||
"""URNs per phase. All-through schools count toward both, matching the
|
||||
PHASE_GROUPS mapping the search filters already use."""
|
||||
from backend.app import PHASE_GROUPS
|
||||
|
||||
if "phase" not in group.columns:
|
||||
return {}
|
||||
lowered = group["phase"].fillna("").str.lower()
|
||||
out: dict[str, tuple[int, ...]] = {}
|
||||
for phase in ("primary", "secondary"):
|
||||
wanted = PHASE_GROUPS.get(phase, set())
|
||||
subset = group[lowered.isin(wanted)]
|
||||
urns = tuple(sorted({int(u) for u in subset["urn"]} & publishable))
|
||||
if urns:
|
||||
out[phase] = urns
|
||||
return out
|
||||
|
||||
|
||||
def _parent_authority(group) -> str | None:
|
||||
"""The most common authority in a group — the useful 301 target.
|
||||
|
||||
A town spanning several authorities has no single parent, so the mode is
|
||||
the honest answer rather than an arbitrary first row.
|
||||
"""
|
||||
if "local_authority" not in group.columns:
|
||||
return None
|
||||
top = group["local_authority"].dropna()
|
||||
return str(top.mode().iloc[0]) if not top.empty else None
|
||||
|
||||
|
||||
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per distinct value of `column` that clears the threshold."""
|
||||
from backend.app import _slugify
|
||||
|
||||
if column not in df.columns:
|
||||
return {}
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for name, group in df.groupby(column, dropna=True):
|
||||
name = str(name).strip()
|
||||
if not name:
|
||||
continue
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
slug = _slugify(name)
|
||||
if not slug:
|
||||
continue
|
||||
place = Place(
|
||||
kind=kind, slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group) if kind == "town" else None,
|
||||
phase_urns=_phase_urns(group, publishable),
|
||||
)
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
||||
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
||||
|
||||
|
||||
def _outcode(postcode) -> str | None:
|
||||
if not isinstance(postcode, str):
|
||||
return None
|
||||
m = _OUTCODE_RE.match(postcode.upper().strip())
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per postcode district clearing the threshold.
|
||||
|
||||
These carry no phase variants: nobody searches "primary schools in SW11".
|
||||
"""
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
working = working[working["_oc"].notna()]
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for oc, group in working.groupby("_oc"):
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
||||
urns=urns, parent_authority=_parent_authority(group),
|
||||
phase_urns=_phase_urns(group, publishable))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def _locality_places(df, publishable: set[int],
|
||||
town_slugs: set[str]) -> dict[str, Place]:
|
||||
"""One Place per curated locality clearing the threshold."""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
if slug in town_slugs:
|
||||
# Skip, do not raise. The guard exists so a locality never
|
||||
# silently shadows a town — skipping achieves that, and the error
|
||||
# log makes it loud.
|
||||
#
|
||||
# Raising here took down sitemap generation for all 25,000 school
|
||||
# pages when "richmond" met the GIAS town Richmond in North
|
||||
# Yorkshire. Worse, GIAS town names change without any code change,
|
||||
# so a raise means curated data can break the site spontaneously.
|
||||
# A curation mistake must cost one page, not the sitemap.
|
||||
logger.error(
|
||||
"locality %r collides with the published town of the same "
|
||||
"slug and has been skipped; rename it or remove it", slug)
|
||||
continue
|
||||
group = working[working["_oc"].isin(outcodes)]
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
# Not an error — a locality can legitimately be too small. Logged
|
||||
# because one you meant to publish quietly vanishing is the
|
||||
# failure worth hearing about.
|
||||
logger.warning(
|
||||
"locality %s (%s) has %d publishable schools, below the "
|
||||
"threshold of %d - not published",
|
||||
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
||||
continue
|
||||
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group),
|
||||
phase_urns=_phase_urns(group, publishable))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def build_place_registry(df) -> dict[str, Place]:
|
||||
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
||||
if df.empty or "urn" not in df.columns:
|
||||
return {}
|
||||
|
||||
publishable = _publishable_urns(df)
|
||||
registry: dict[str, Place] = {}
|
||||
registry.update(_group(df, "local_authority", "authority", publishable))
|
||||
|
||||
towns = _group(df, "town", "town", publishable)
|
||||
registry.update(towns)
|
||||
|
||||
town_slugs = {p.slug for p in towns.values()}
|
||||
registry.update(_locality_places(df, publishable, town_slugs))
|
||||
registry.update(_outcode_places(df, publishable))
|
||||
return registry
|
||||
@@ -0,0 +1,231 @@
|
||||
"""Tests for the place registry (spec 2026-08-21).
|
||||
|
||||
The registry is built from the in-memory school DataFrame, so these build a
|
||||
small frame directly rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from backend.places import MIN_SCHOOLS, build_place_registry
|
||||
|
||||
|
||||
def _df(rows: list[dict]) -> pd.DataFrame:
|
||||
base = {
|
||||
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
|
||||
"phase": "Primary", "postcode": "AA1 1AA",
|
||||
}
|
||||
return pd.DataFrame([{**base, **r} for r in rows])
|
||||
|
||||
|
||||
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
||||
"""`start` offsets the URNs so two calls can describe different schools —
|
||||
the Bedford case needs two authorities' worth of distinct URNs in one
|
||||
town."""
|
||||
return [
|
||||
{"urn": start + i, "school_name": f"{town} School {i}",
|
||||
"town": town, "local_authority": la, **kw}
|
||||
for i in range(n)
|
||||
]
|
||||
|
||||
|
||||
def test_town_clearing_the_threshold_is_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
assert "town:brentwood" in reg
|
||||
assert reg["town:brentwood"].name == "Brentwood"
|
||||
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_town_below_the_threshold_is_not_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
|
||||
assert "town:crosby" not in reg
|
||||
|
||||
|
||||
def test_a_town_below_threshold_still_names_its_authority():
|
||||
# The route layer needs somewhere to 301 to.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
|
||||
assert "authority:sefton" in reg
|
||||
|
||||
|
||||
def test_town_and_authority_of_the_same_name_are_separate_places():
|
||||
# 67 real collisions. Neither set contains the other: Bedford the town has
|
||||
# 104 schools, Bedford the authority 86, because postal towns cross
|
||||
# authority boundaries.
|
||||
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
||||
assert set(town.urns) != set(authority.urns)
|
||||
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
|
||||
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
|
||||
|
||||
|
||||
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
|
||||
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
|
||||
for r in rows:
|
||||
r["rwm_expected_pct"] = np.nan
|
||||
r["ofsted_grade"] = np.nan
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "town:ghosttown" not in reg
|
||||
|
||||
|
||||
def test_blank_town_is_ignored():
|
||||
rows = _town(MIN_SCHOOLS, "", "Essex")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("town:") for k in reg)
|
||||
|
||||
|
||||
def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows = []
|
||||
for year in (202324, 202425):
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_groups_schools_by_outcode(monkeypatch):
|
||||
# The GIAS town field collapses 1,819 London schools into "London", so a
|
||||
# locality is defined by its postcode districts instead.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"battersea": ("Battersea", ("SW11",))})
|
||||
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
||||
for r in rows:
|
||||
r["postcode"] = "SW11 2AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["locality:battersea"].name == "Battersea"
|
||||
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"nowhere": ("Nowhere", ("ZZ99",))})
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
||||
assert "locality:nowhere" not in reg
|
||||
|
||||
|
||||
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch, caplog):
|
||||
"""A colliding locality is skipped loudly, and the town survives.
|
||||
|
||||
This used to raise, which took down sitemap generation for all 25,000
|
||||
school pages the first time a curated slug met a real GIAS town. Curated
|
||||
data must not be able to break the site — and GIAS town names change with
|
||||
no code change at all, so the raise could fire spontaneously.
|
||||
"""
|
||||
import logging
|
||||
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
|
||||
with caplog.at_level(logging.ERROR):
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
assert "locality:brentwood" not in reg # skipped
|
||||
assert "town:brentwood" in reg # the town is untouched
|
||||
assert "brentwood" in caplog.text # and it was loud about it
|
||||
|
||||
|
||||
def test_a_locality_collision_does_not_break_the_rest_of_the_registry(monkeypatch):
|
||||
# The whole point of skipping rather than raising.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "authority:essex" in reg
|
||||
assert "outcode:cm13" in reg
|
||||
|
||||
|
||||
def test_outcode_places_are_built_from_postcodes():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:cm13"].name == "CM13"
|
||||
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_malformed_postcodes_do_not_create_places():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "not a postcode"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("outcode:") for k in reg)
|
||||
|
||||
|
||||
def test_every_curated_locality_is_structurally_valid():
|
||||
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
||||
import re
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
||||
assert name.strip() == name and name, slug
|
||||
assert outcodes, f"{slug} has no outcodes"
|
||||
for oc in outcodes:
|
||||
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
||||
|
||||
|
||||
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
||||
"""Two copies with no drift guard is worse than one copy.
|
||||
|
||||
backend/localities.py is canonical because the backend image does not
|
||||
contain pipeline/. The seed exists so the warehouse can join on the same
|
||||
definitions, and this is what stops the two diverging — the same
|
||||
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
||||
"""
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
seed_path = (Path(__file__).resolve().parents[2]
|
||||
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
||||
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
||||
|
||||
seed = {
|
||||
row["locality_slug"]: (row["locality_name"],
|
||||
tuple(row["outcodes"].split("|")))
|
||||
for row in csv.DictReader(seed_path.open())
|
||||
}
|
||||
assert seed == LOCALITY_OUTCODES
|
||||
|
||||
|
||||
def test_no_curated_locality_names_a_london_borough():
|
||||
"""Boroughs are authorities and already have a page.
|
||||
|
||||
A locality defined by two or three outcodes inside a borough would be a
|
||||
partial, near-duplicate subset of that authority page — the exact
|
||||
thin-content failure the two-namespace design exists to avoid. Hackney,
|
||||
Islington, Greenwich and Ealing were all in the first draft.
|
||||
|
||||
Hardcoded rather than read from the corpus because this must fail in CI,
|
||||
where there is no database.
|
||||
"""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
boroughs = {
|
||||
"barking-and-dagenham", "barnet", "bexley", "brent", "bromley",
|
||||
"camden", "croydon", "ealing", "enfield", "greenwich", "hackney",
|
||||
"hammersmith-and-fulham", "haringey", "harrow", "havering",
|
||||
"hillingdon", "hounslow", "islington", "kensington-and-chelsea",
|
||||
"kingston-upon-thames", "lambeth", "lewisham", "merton", "newham",
|
||||
"redbridge", "richmond-upon-thames", "southwark", "sutton",
|
||||
"tower-hamlets", "waltham-forest", "wandsworth", "westminster",
|
||||
}
|
||||
named = boroughs & set(LOCALITY_OUTCODES)
|
||||
assert not named, (
|
||||
f"these are boroughs, not districts: {sorted(named)} - they already "
|
||||
"have an authority page covering every school"
|
||||
)
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Tests for the places API (spec 2026-08-21)."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
|
||||
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 50.0 + i}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_registry_lists_each_published_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
slugs = {(p["kind"], p["slug"]) for p in body["places"]}
|
||||
assert ("town", "brentwood") in slugs
|
||||
assert ("authority", "essex") in slugs
|
||||
assert ("outcode", "cm13") in slugs
|
||||
|
||||
|
||||
def test_registry_carries_a_count_per_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
town = next(p for p in body["places"] if p["slug"] == "brentwood")
|
||||
assert town["count"] == 6
|
||||
|
||||
|
||||
def test_place_detail_returns_its_schools_ranked(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
assert body["place"]["name"] == "Brentwood"
|
||||
scores = [s["rwm_expected_pct"] for s in body["schools"]]
|
||||
assert scores == sorted(scores, reverse=True)
|
||||
|
||||
|
||||
def test_place_detail_carries_the_local_average(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
# 50..55 inclusive
|
||||
assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5)
|
||||
|
||||
|
||||
def test_phase_filter_narrows_the_school_list(client):
|
||||
body = client.get("/api/places/town/brentwood?phase=secondary").json()
|
||||
assert body["schools"] == []
|
||||
|
||||
|
||||
def test_unknown_place_404s(client):
|
||||
assert client.get("/api/places/town/atlantis").status_code == 404
|
||||
|
||||
|
||||
def test_unknown_kind_404s(client):
|
||||
assert client.get("/api/places/planet/mars").status_code == 404
|
||||
@@ -0,0 +1,290 @@
|
||||
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
|
||||
|
||||
The sitemap is built from the in-memory school DataFrame, so these inject a
|
||||
small frame via monkeypatch rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
"""Two schools: one with results, one with neither results nor Ofsted."""
|
||||
base = {
|
||||
"local_authority": "Testshire",
|
||||
"school_type": "Academy",
|
||||
"phase": "Primary",
|
||||
"year": 202425,
|
||||
"ofsted_date": None,
|
||||
}
|
||||
return pd.DataFrame(
|
||||
[
|
||||
{**base, "urn": 100001, "school_name": "Alpha Primary",
|
||||
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": 2.0},
|
||||
{**base, "urn": 100002, "school_name": "Ghost Primary",
|
||||
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": np.nan},
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sitemap(monkeypatch) -> str:
|
||||
"""The sitemap index."""
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemap()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def schools_child(monkeypatch) -> str:
|
||||
"""The first school child sitemap, where school URLs actually live."""
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()["schools-1.xml"]
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def static_child(monkeypatch) -> str:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()["static.xml"]
|
||||
|
||||
|
||||
def test_every_loc_uses_the_www_host(sitemaps):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
# Checked across every file, index included, not just one.
|
||||
#
|
||||
# A child can legitimately be empty — this fixture holds two schools and no
|
||||
# town clearing the threshold — so the presence check applies only to files
|
||||
# that carry URLs. The absence check applies to all of them.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "https://schoolcompare.co.uk" not in xml, name
|
||||
if "<loc>" in xml:
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
|
||||
|
||||
def test_school_with_results_is_listed(schools_child):
|
||||
assert "/school/100001-alpha-primary" in schools_child
|
||||
|
||||
|
||||
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
|
||||
# Nothing for a search result to say about it. Submitting it spends crawl
|
||||
# budget and drags the corpus-wide quality signal down.
|
||||
#
|
||||
# Asserted against the child, not the index: the index carries no school
|
||||
# URLs at all, so it would pass this trivially and prove nothing.
|
||||
assert "/school/100002" not in schools_child
|
||||
|
||||
|
||||
def test_no_invented_priority_or_changefreq(sitemaps):
|
||||
# Google ignores both. They were noise dressed as signal.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "<priority>" not in xml, name
|
||||
assert "<changefreq>" not in xml, name
|
||||
|
||||
|
||||
def test_ofsted_date_becomes_lastmod(monkeypatch):
|
||||
from backend import app as app_module
|
||||
import datetime
|
||||
|
||||
def _df():
|
||||
base = _schools_df()
|
||||
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
|
||||
return base
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "<lastmod>2024-03-14</lastmod>" in xml
|
||||
|
||||
|
||||
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
|
||||
# An always-now lastmod is a claim Google learns to distrust. Absent
|
||||
# honestly means unknown.
|
||||
from backend import app as app_module
|
||||
|
||||
def _df():
|
||||
df = _schools_df()
|
||||
df["ofsted_date"] = None
|
||||
return df
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
# The child only. The index legitimately carries a lastmod, because there
|
||||
# it means "when this sitemap file changed", which we do know.
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "<lastmod>" not in xml
|
||||
|
||||
|
||||
def test_static_routes_are_listed(static_child):
|
||||
for path in ("/", "/rankings", "/compare", "/admissions"):
|
||||
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sitemaps(monkeypatch) -> dict:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()
|
||||
|
||||
|
||||
def test_index_lists_each_child(sitemaps):
|
||||
index = sitemaps["sitemap.xml"]
|
||||
assert "<sitemapindex" in index
|
||||
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
|
||||
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
|
||||
|
||||
|
||||
def test_index_carries_no_url_elements(sitemaps):
|
||||
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
|
||||
assert "<url>" not in sitemaps["sitemap.xml"]
|
||||
|
||||
|
||||
def test_index_does_not_list_itself(sitemaps):
|
||||
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
|
||||
|
||||
|
||||
def test_static_child_holds_the_static_routes(sitemaps):
|
||||
static = sitemaps["static.xml"]
|
||||
for path in ("/", "/rankings", "/compare", "/admissions"):
|
||||
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
|
||||
|
||||
|
||||
def test_school_child_holds_the_schools(sitemaps):
|
||||
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
|
||||
|
||||
|
||||
def test_children_are_chunked_under_the_limit(monkeypatch):
|
||||
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
|
||||
# small enough to eyeball in Search Console.
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
rows = [
|
||||
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": None,
|
||||
"ofsted_grade": 2.0, "ofsted_date": None}
|
||||
for i in range(10_001)
|
||||
]
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
|
||||
|
||||
maps = app_module.build_sitemaps()
|
||||
assert maps["schools-1.xml"].count("<url>") == 10_000
|
||||
assert maps["schools-2.xml"].count("<url>") == 1
|
||||
|
||||
|
||||
def test_build_sitemap_still_returns_the_index(sitemap):
|
||||
# lifespan and the admin endpoint call build_sitemap(); keep it working.
|
||||
assert "<sitemapindex" in sitemap
|
||||
|
||||
|
||||
def test_school_with_results_in_an_earlier_year_is_still_listed(monkeypatch):
|
||||
"""Regression: The Mallard Academy (150367).
|
||||
|
||||
Real KS2 results 2015-16 to 2018-19, then null rows from 2022-23 onward
|
||||
because the school stopped reporting. The first cut tested the latest
|
||||
year's row alone and dropped it, along with ~220 others, even though its
|
||||
detail page shows all four years of results.
|
||||
"""
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
base = {"local_authority": "Testshire", "school_type": "Academy",
|
||||
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
|
||||
"attainment_8_score": np.nan, "urn": 150367,
|
||||
"school_name": "Mallard Academy"}
|
||||
df = _pd.DataFrame([
|
||||
{**base, "year": 201819, "rwm_expected_pct": 67.0},
|
||||
{**base, "year": 202324, "rwm_expected_pct": np.nan},
|
||||
{**base, "year": 202425, "rwm_expected_pct": np.nan},
|
||||
])
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "/school/150367-mallard-academy" in xml
|
||||
|
||||
|
||||
def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
|
||||
"""The fix must not turn into "list everything"."""
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
base = {"local_authority": "Testshire", "school_type": "Academy",
|
||||
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
|
||||
"attainment_8_score": np.nan, "rwm_expected_pct": np.nan,
|
||||
"urn": 100002, "school_name": "Ghost Primary"}
|
||||
df = _pd.DataFrame([{**base, "year": y} for y in (202324, 202425)])
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
|
||||
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]
|
||||
|
||||
|
||||
def _places_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA",
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 60.0}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def place_sitemaps(monkeypatch) -> dict:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _places_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return app_module.build_sitemaps()
|
||||
|
||||
|
||||
def test_place_children_are_listed_in_the_index(place_sitemaps):
|
||||
index = place_sitemaps["sitemap.xml"]
|
||||
assert "/sitemaps/places-1.xml" in index
|
||||
assert "/sitemaps/outcodes-1.xml" in index
|
||||
|
||||
|
||||
def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps):
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood</loc>" in xml
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/authority/essex</loc>" in xml
|
||||
|
||||
|
||||
def test_outcode_urls_live_in_their_own_child(place_sitemaps):
|
||||
assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"]
|
||||
assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"]
|
||||
|
||||
|
||||
def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps):
|
||||
for name in ("places-1.xml", "outcodes-1.xml"):
|
||||
assert "<priority>" not in place_sitemaps[name]
|
||||
assert "<changefreq>" not in place_sitemaps[name]
|
||||
|
||||
|
||||
def test_phase_variants_are_submitted_where_the_phase_clears_the_threshold(place_sitemaps):
|
||||
# "primary schools in beccles" is the query shape the baseline showed, so
|
||||
# each variant is its own page and has to be submitted. Emitting only the
|
||||
# bare place URL left ~950 of them reachable by nothing.
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood/primary</loc>" in xml
|
||||
|
||||
|
||||
def test_a_phase_below_its_own_threshold_is_not_submitted(place_sitemaps):
|
||||
# The fixture is six primaries and no secondaries.
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "/schools/brentwood/secondary" not in xml
|
||||
|
||||
|
||||
def test_outcodes_get_no_phase_variants(place_sitemaps):
|
||||
# Nobody searches "primary schools in CM13"; the routes do not exist.
|
||||
xml = place_sitemaps["outcodes-1.xml"]
|
||||
assert "/primary" not in xml and "/secondary" not in xml
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,278 @@
|
||||
# W2: The Location Layer — Design
|
||||
|
||||
Date: 2026-08-21
|
||||
Status: awaiting review
|
||||
Supersedes: workstream W2 in `2026-08-20-seo-programme-design.md`
|
||||
|
||||
Scope note: this covers four page families in one spec. Splitting them — towns
|
||||
and authorities first, outcodes and localities after — was proposed and
|
||||
declined in favour of building the layer in one pass. The decomposition
|
||||
argument was that the curated locality seed needs human review and would hold
|
||||
up 783 pages of measured demand behind it; that risk is accepted here, and the
|
||||
implementation plan should sequence the seed early enough that review time
|
||||
does not become the critical path.
|
||||
|
||||
## Problem
|
||||
|
||||
Location intent is the largest unserved demand the site has. In the 16-month
|
||||
Search Console baseline it draws **874 impressions, one click, average
|
||||
position 49.5**. The site does not compete.
|
||||
|
||||
Unlike named-school queries — which the same baseline showed to be
|
||||
navigational and unwinnable, since a parent typing "audley junior school"
|
||||
wants that school's own website — location queries have no incumbent owner.
|
||||
Nobody owns "primary schools in Brentwood" the way a school owns its name.
|
||||
|
||||
The cause is structural: the site has no page about a place. Every competitor
|
||||
ranking above it does.
|
||||
|
||||
## What the demand actually looks like
|
||||
|
||||
Every location query in the baseline is **town or district level**. Not one is
|
||||
an administrative area:
|
||||
|
||||
| Query | Impressions | Position |
|
||||
|-------|-------------|----------|
|
||||
| colleges in solihull | 112 | 51.2 |
|
||||
| schools in ramsey | 64 | 42.5 |
|
||||
| schools in crosby | 57 | 47.7 |
|
||||
| primary schools in beccles | 44 | 40.9 |
|
||||
| private schools in battersea | 41 | 71.9 |
|
||||
| secondary schools in brentwood | 37 | 56.1 |
|
||||
| secondary schools in canary wharf | 30 | 35.9 |
|
||||
|
||||
Three patterns follow directly, and they drive the whole design.
|
||||
|
||||
**Towns, not authorities.** The superseded W2 put `/schools/[la]` first and
|
||||
towns second. The data inverts that. Brentwood appears four times in different
|
||||
phrasings; Beccles twice. Both are towns, not authorities.
|
||||
|
||||
**Phase is part of the query**, not a filter applied afterwards: "primary
|
||||
schools in beccles", "secondary schools in brentwood", "colleges in solihull".
|
||||
|
||||
**London is searched by district** — Battersea, Canary Wharf — and the GIAS
|
||||
`town` field cannot serve it at all.
|
||||
|
||||
## Measured sizing
|
||||
|
||||
Counted against the live corpus of 25,185 schools, not estimated.
|
||||
|
||||
| Family | Viable (≥5 schools) | Below threshold |
|
||||
|--------|--------------------|-----------------|
|
||||
| Towns | **783** | 907 → redirect to authority |
|
||||
| Outcodes | **1,760** | 305 |
|
||||
| Local authorities | 154 | — |
|
||||
| London localities | ~100–150 (curated) | — |
|
||||
|
||||
With phase variants — 783 town pages plus roughly 700 primary and 250
|
||||
secondary variants, 154 authorities across three variants, 1,760 outcodes and
|
||||
the curated localities — the total lands near **4,000 pages**. Phase variants
|
||||
need their own threshold: there are 17,426 primaries but only 4,456 secondaries nationally,
|
||||
so most towns will support a primary page and not a secondary one.
|
||||
|
||||
## Two design problems this spec exists to solve
|
||||
|
||||
### 1. Town and authority names collide, and neither contains the other
|
||||
|
||||
67 viable towns share a name with a local authority. The obvious fix — let the
|
||||
authority absorb the town, since it sounds like a superset — **does not work**:
|
||||
|
||||
| Place | Schools in the town | Schools in the authority |
|
||||
|-------|--------------------|-----------------------|
|
||||
| Bedford | 104 | 86 |
|
||||
| Birmingham | 520 | 518 |
|
||||
| Derby | 157 | 119 |
|
||||
| Doncaster | 152 | 145 |
|
||||
|
||||
The authority is the larger set in only 43 of the 67. Postal towns cross
|
||||
authority boundaries, so these are overlapping sets that happen to share a
|
||||
name. Publishing both into one namespace produces near-duplicate pages, which
|
||||
is the specific failure that sinks programmatic SEO.
|
||||
|
||||
**Resolution: two namespaces.**
|
||||
|
||||
```
|
||||
/schools/[place] towns and London localities
|
||||
/schools/[place]/primary
|
||||
/schools/[place]/secondary
|
||||
/schools/authority/[la] local authorities
|
||||
/schools/authority/[la]/primary
|
||||
/schools/authority/[la]/secondary
|
||||
/schools/near/[outcode]
|
||||
```
|
||||
|
||||
Outcodes carry no phase variants: nobody searches "primary schools in SW11",
|
||||
so the variants would be pages without demand.
|
||||
|
||||
Every collision disappears by construction. `/schools/[place]` keeps the clean
|
||||
URL for the pattern that carries the demand; authorities get a namespace whose
|
||||
purpose is genuinely different — admissions are authority-run, and the
|
||||
authority page is the one that can speak to catchment policy and LA averages.
|
||||
|
||||
A place page and an authority page of the same name must each say plainly
|
||||
which set of schools they cover, or they read as duplicates to a reader even
|
||||
when they differ in fact.
|
||||
|
||||
### 2. London has no locality field
|
||||
|
||||
`town` collapses **1,819 London schools into the single value "London"**. A
|
||||
page listing all of them is useless, and borough pages do not help because
|
||||
people search "Battersea", not "Wandsworth".
|
||||
|
||||
No single field solves it:
|
||||
|
||||
| Search term | `parliamentary_constituency` | postcodes.io `admin_ward` |
|
||||
|-------------|------------------------------|---------------------------|
|
||||
| Battersea | **Battersea** ✓ | Northcote / Wandsworth Town ✗ |
|
||||
| Canary Wharf | Poplar and Limehouse ✗ | **Canary Wharf** ✓ |
|
||||
| Vauxhall | Vauxhall and Camberwell Green ✗ | **Vauxhall** ✓ |
|
||||
|
||||
And neither covers Clapham, Shoreditch or Peckham, which are postal and
|
||||
colloquial rather than administrative.
|
||||
|
||||
**Resolution: a curated seed mapping locality to outcodes.**
|
||||
|
||||
```
|
||||
pipeline/transform/seeds/locality_outcodes.csv
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,"SW11|SW8",London
|
||||
canary-wharf,Canary Wharf,"E14",London
|
||||
clapham,Clapham,"SW4|SW9",London
|
||||
```
|
||||
|
||||
This needs **no new ingestion** — the corpus already has postcodes. It puts
|
||||
the fuzzy, contested part of the problem in a reviewable file rather than in
|
||||
derived logic, which suits it: locality boundaries are a judgement, not a
|
||||
fact. The repo already uses dbt seeds for curated reference data
|
||||
(`la_code_names.csv`, `gias_code_names.csv`), so this follows an established
|
||||
pattern.
|
||||
|
||||
The seed generalises past London. Any colloquial place — Jesmond, Chorlton,
|
||||
Clifton — can be defined by its outcodes without a schema change.
|
||||
|
||||
**Constraint:** a locality slug may not collide with a viable town slug. The
|
||||
place registry enforces this and fails the build rather than silently
|
||||
shadowing a town.
|
||||
|
||||
## Architecture
|
||||
|
||||
### The place registry
|
||||
|
||||
One module owns the question "what places do we publish, and what is in each".
|
||||
Everything else reads from it: the pages, the sitemap, the internal links.
|
||||
|
||||
```
|
||||
backend/places.py
|
||||
|
||||
Place = { kind: "town"|"locality"|"authority"|"outcode",
|
||||
slug, name, urn_list, parent_authority | None }
|
||||
|
||||
build_place_registry(df) -> dict[str, Place]
|
||||
place_schools(slug, phase=None) -> list[School]
|
||||
```
|
||||
|
||||
Built once at startup from the same DataFrame the sitemap uses, and rebuilt by
|
||||
the existing `/api/admin/regenerate-sitemap` path after a pipeline run.
|
||||
Registry construction is where the threshold, the collision rules and the
|
||||
seed's uniqueness constraint are enforced — in one place, testable without a
|
||||
browser or a database.
|
||||
|
||||
### API
|
||||
|
||||
```
|
||||
GET /api/places the registry: slug, kind, name, count
|
||||
GET /api/places/{slug}?phase= aggregate + ranked schools for one place
|
||||
```
|
||||
|
||||
`/api/places` is what the sitemap and the internal-link modules enumerate.
|
||||
|
||||
### Routes
|
||||
|
||||
Next App Router, ISR with the same 7-day revalidate the school pages use.
|
||||
`generateStaticParams` gated behind an env flag, matching
|
||||
`PRERENDER_SCHOOLS`, because 3,900 more routes cannot be statically built in
|
||||
CI on every deploy.
|
||||
|
||||
## What each page must contain
|
||||
|
||||
A place page that is a name substituted into a template is the thing Google's
|
||||
helpful-content stance exists to demote. Each page carries computed local
|
||||
facts that exist nowhere else on the site:
|
||||
|
||||
- **H1** matching the query: "Primary schools in Brentwood"
|
||||
- **Counts framed usefully**: "29 schools, 4 rated Outstanding"
|
||||
- **A ranked table** of the top 20 on the phase's headline metric —
|
||||
`rwm_expected_pct` for primary, `attainment_8_score` for secondary, and for
|
||||
an unphased place page the metric matching whichever phase it holds more of
|
||||
- **The local average against the England average** — the one number a parent
|
||||
cannot get from a list
|
||||
- **Ofsted grade distribution** for the place
|
||||
- **A map**
|
||||
- **Links to neighbouring places** and to the parent authority
|
||||
- **An FAQ block**, feeding `FAQPage` structured data
|
||||
- **A link to every school page in scope** — this is what finally de-orphans
|
||||
the 23,000 school pages the original spec identified as near-orphans
|
||||
|
||||
## Thin-page controls
|
||||
|
||||
Three, and they are the difference between a location layer and index bloat:
|
||||
|
||||
1. **Five schools with current data minimum.** Below it, 301 to the parent
|
||||
authority. This drops 907 towns and 305 outcodes.
|
||||
2. **Per-phase thresholds.** A town with 30 primaries and 2 secondaries
|
||||
publishes a primary page and no secondary page.
|
||||
3. **No page without a local average.** If a place has too few schools with
|
||||
results to compute one, it has nothing to say that a list does not, and it
|
||||
falls back to the authority.
|
||||
|
||||
## Sitemap
|
||||
|
||||
Two new children in the existing index: `/sitemaps/places-{n}.xml` and
|
||||
`/sitemaps/outcodes-{n}.xml`. Per-family children are why the index was built
|
||||
in W1 — Search Console reports coverage per submitted sitemap, so indexation
|
||||
of the location layer is measurable separately from the school pages.
|
||||
|
||||
## Testing
|
||||
|
||||
Per `CLAUDE.md`, user-facing behaviour extends `e2e/tests/journeys.spec.ts` in
|
||||
the same PR.
|
||||
|
||||
**Unit (registry, no DB):** threshold enforcement; a sub-threshold town
|
||||
resolves to its authority; a locality slug colliding with a town fails the
|
||||
build; Bedford's town and authority pages hold different URN sets; per-phase
|
||||
thresholds.
|
||||
|
||||
**Backend:** `/api/places` shape; `/api/places/{slug}` aggregate correctness
|
||||
against a fixture; unknown slug 404s.
|
||||
|
||||
**e2e:** a known town, authority, locality and outcode page each render with
|
||||
the expected count; a below-threshold town 301s; every place page declares a
|
||||
canonical and appears in the sitemap; `/schools/bedford` and
|
||||
`/schools/authority/bedford` both resolve and state which set they cover.
|
||||
|
||||
## Risks
|
||||
|
||||
**Index bloat** is the failure mode of every programmatic SEO programme. The
|
||||
three controls above are the answer, and the per-family sitemap is how we find
|
||||
out early if they were not enough.
|
||||
|
||||
**Helpful-content exposure.** Templated location pages are exactly what
|
||||
Google's stance targets. The mitigation is that every page carries real
|
||||
computed local data — counts, distributions, local-versus-national comparison
|
||||
— rather than a name dropped into boilerplate. If indexation of the places
|
||||
sitemap stalls below roughly half, that is the signal to stop and rethink
|
||||
rather than to add more pages.
|
||||
|
||||
**Build cost.** ~4,000 additional ISR routes on top of 23,000 school pages.
|
||||
The env-flag gate on `generateStaticParams` keeps CI viable.
|
||||
|
||||
**Curation drift.** The locality seed is hand-maintained and will go stale as
|
||||
places change. It is small and reviewable, and a dbt test asserts every seed
|
||||
outcode matches at least one school so a typo fails the pipeline rather than
|
||||
publishing an empty page.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Catchment-area estimation. It is a strong driver for this cluster and
|
||||
`fact_admissions` carries the distances, but it is a modelling problem with
|
||||
real accuracy risk and deserves its own design.
|
||||
+262
-8
@@ -1587,18 +1587,113 @@ test('a Welsh school URL 404s while an English one still resolves', async ({ pag
|
||||
expect(welsh?.status(), 'a Welsh school should no longer resolve').toBe(404);
|
||||
});
|
||||
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
async function sitemapChildren(page: Page): Promise<string[]> {
|
||||
const res = await page.request.get('/sitemap.xml');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const xml = await res.text();
|
||||
const index = await res.text();
|
||||
expect(index).toContain('<sitemapindex');
|
||||
return [...index.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1]);
|
||||
}
|
||||
|
||||
const urlCount = (xml.match(/<url>/g) ?? []).length;
|
||||
expect(urlCount, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
test('the sitemap index names children that all resolve', async ({ page }) => {
|
||||
const index = await (await page.request.get('/sitemap.xml')).text();
|
||||
// An index holds <sitemap> entries only; mixing in <url> is invalid.
|
||||
expect(index).not.toContain('<url>');
|
||||
|
||||
// 401559 (Cardiff) and 402426 (ACT Schools, Cardiff) were both submitted
|
||||
// before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
const locs = await sitemapChildren(page);
|
||||
expect(locs.length).toBeGreaterThanOrEqual(2);
|
||||
|
||||
for (const loc of locs) {
|
||||
expect(loc.startsWith('https://www.schoolcompare.co.uk/sitemaps/')).toBeTruthy();
|
||||
const child = await page.request.get(new URL(loc).pathname);
|
||||
expect(child.ok(), `${loc} should resolve`).toBeTruthy();
|
||||
expect(await child.text()).toContain('<urlset');
|
||||
}
|
||||
});
|
||||
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
const locs = await sitemapChildren(page);
|
||||
|
||||
let total = 0;
|
||||
for (const loc of locs) {
|
||||
const xml = await (await page.request.get(new URL(loc).pathname)).text();
|
||||
total += (xml.match(/<url>/g) ?? []).length;
|
||||
// 401559 (Adamsdown, Cardiff) and 402426 (ACT Schools, Cardiff) were both
|
||||
// submitted before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
}
|
||||
expect(total, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
});
|
||||
|
||||
test('the sitemap invents no priority or changefreq', async ({ page }) => {
|
||||
const [first] = await sitemapChildren(page);
|
||||
expect(first).toBeTruthy();
|
||||
|
||||
const xml = await (await page.request.get(new URL(first).pathname)).text();
|
||||
// Google ignores both. They were noise dressed as signal.
|
||||
expect(xml).not.toContain('<priority>');
|
||||
expect(xml).not.toContain('<changefreq>');
|
||||
});
|
||||
|
||||
/*
|
||||
* Canonical URLs (spec 2026-08-20, W1).
|
||||
*
|
||||
* Every indexable route declares exactly one canonical, on the www host, with
|
||||
* no query string. The homepage's eleven search params filter a result set
|
||||
* rather than making a new document, so they all collapse onto "/".
|
||||
*/
|
||||
const CANONICAL_ROUTES: Array<[string, string]> = [
|
||||
['/', 'https://www.schoolcompare.co.uk/'],
|
||||
['/rankings', 'https://www.schoolcompare.co.uk/rankings'],
|
||||
['/admissions', 'https://www.schoolcompare.co.uk/admissions'],
|
||||
];
|
||||
|
||||
for (const [path, expected] of CANONICAL_ROUTES) {
|
||||
test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => {
|
||||
await page.goto(path);
|
||||
const hrefs = await page.locator('link[rel="canonical"]').evaluateAll(
|
||||
(els) => els.map((e) => e.getAttribute('href')));
|
||||
expect(hrefs, `${path} should declare one canonical`).toHaveLength(1);
|
||||
expect(hrefs[0]).toBe(expected);
|
||||
});
|
||||
}
|
||||
|
||||
test('a filtered homepage still canonicalises to the bare root', async ({ page }) => {
|
||||
await page.goto('/?search=primary&phase=primary&sort=name&page=2');
|
||||
const href = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(href).toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
|
||||
test('a school page canonicalises to its own slug on the www host', async ({ page }) => {
|
||||
const res = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const [first] = (await res.json()).schools ?? [];
|
||||
expect(first, 'no school available').toBeTruthy();
|
||||
|
||||
await page.goto(`/school/${first.urn}-x`);
|
||||
const href = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(href).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\/\d+-/);
|
||||
});
|
||||
|
||||
test('a bare /compare is indexable, a parameterised one is not', async ({ page }) => {
|
||||
await page.goto('/compare');
|
||||
await expect(page.locator('meta[name="robots"]')).toHaveCount(0);
|
||||
|
||||
const [a, b] = await twoPrimaryUrns(page);
|
||||
await page.goto(`/compare?urns=${a},${b}`);
|
||||
const robots = await page.locator('meta[name="robots"]').first()
|
||||
.getAttribute('content');
|
||||
expect(robots).toContain('noindex');
|
||||
expect(robots).toContain('follow');
|
||||
|
||||
// noindex but follow: the links out to each school page still count, so the
|
||||
// canonical must still be present and point at the bare path.
|
||||
const canonical = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(canonical).toBe('https://www.schoolcompare.co.uk/compare');
|
||||
});
|
||||
|
||||
/*
|
||||
@@ -1631,3 +1726,162 @@ test('a school page on staging is noindexed too, not just the homepage', async (
|
||||
const res = await page.request.get(`/school/${first.urn}-x`);
|
||||
expect(res.headers()['x-robots-tag']).toContain('noindex');
|
||||
});
|
||||
|
||||
/*
|
||||
* W8 — the C1 pages must ship a description, and it must differentiate.
|
||||
*
|
||||
* Baseline was 0.43% CTR at position 6.1 on "compare school performance",
|
||||
* against 9.16% for the brand query from the same neighbourhood. The SERP is
|
||||
* owned by the DfE's own service, so a description that paraphrases it earns
|
||||
* nothing. Google may rewrite a snippet, but it cannot use one we never sent.
|
||||
*/
|
||||
test('every C1 page ships a description, and none opens its title with the brand', async ({ page }) => {
|
||||
for (const path of ['/', '/compare', '/rankings', '/admissions']) {
|
||||
await page.goto(path);
|
||||
|
||||
const desc = await page.locator('meta[name="description"]').first()
|
||||
.getAttribute('content');
|
||||
expect(desc, `${path} must ship a description`).toBeTruthy();
|
||||
expect(desc!.length, `${path} description too short to be worth reading`)
|
||||
.toBeGreaterThan(100);
|
||||
|
||||
const title = await page.title();
|
||||
expect(title.toLowerCase().startsWith('schoolcompare'),
|
||||
`${path} spends its most valuable pixels on the brand`).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
test('the homepage snippet names what gov.uk does not publish', async ({ page }) => {
|
||||
await page.goto('/');
|
||||
const desc = await page.locator('meta[name="description"]').first()
|
||||
.getAttribute('content');
|
||||
// Admissions distance is the one fact the DfE service has no equivalent for.
|
||||
expect(desc).toMatch(/close you had to live|distance/i);
|
||||
});
|
||||
|
||||
/*
|
||||
* The location layer (spec 2026-08-21, W2).
|
||||
*
|
||||
* Location intent sat at position 49.5 with one click across the whole 16-month
|
||||
* baseline — the site published no page about a place. These assert the four
|
||||
* families render, stay in their own namespaces, and reach the sitemap.
|
||||
*/
|
||||
async function firstPlaceOfKind(page: Page, kind: string) {
|
||||
const res = await page.request.get('/api/places');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const { places } = await res.json();
|
||||
const hit = places.find((p: { kind: string }) => p.kind === kind);
|
||||
expect(hit, `no ${kind} in the registry`).toBeTruthy();
|
||||
return hit as { kind: string; slug: string; name: string; count: number };
|
||||
}
|
||||
|
||||
for (const [kind, prefix, article] of [
|
||||
['town', '/schools/', 'a'],
|
||||
['authority', '/schools/authority/', 'an'],
|
||||
['outcode', '/schools/near/', 'an'],
|
||||
] as const) {
|
||||
test(`${article} ${kind} page renders with its school count`, async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, kind);
|
||||
await page.goto(`${prefix}${place.slug}`);
|
||||
await expect(page.locator('h1')).toContainText(place.name, { ignoreCase: true });
|
||||
await expect(page.locator('a[href^="/school/"]').first()).toBeVisible();
|
||||
});
|
||||
}
|
||||
|
||||
test('a town and an authority sharing a name are different pages', async ({ page }) => {
|
||||
// 67 real collisions, and the authority is the larger set in only 43 — so
|
||||
// one namespace would have published near-duplicates.
|
||||
const { places } = await (await page.request.get('/api/places')).json();
|
||||
const townSlugs = new Set(
|
||||
places.filter((p: { kind: string }) => p.kind === 'town')
|
||||
.map((p: { slug: string }) => p.slug));
|
||||
const clash = places.find((p: { kind: string; slug: string }) =>
|
||||
p.kind === 'authority' && townSlugs.has(p.slug));
|
||||
test.skip(!clash, 'no town/authority name collision in this environment');
|
||||
|
||||
const townRes = await page.request.get(`/api/places/town/${clash.slug}`);
|
||||
const laRes = await page.request.get(`/api/places/authority/${clash.slug}`);
|
||||
expect(townRes.ok() && laRes.ok()).toBeTruthy();
|
||||
const townUrns = (await townRes.json()).schools.map((s: { urn: number }) => s.urn).sort();
|
||||
const laUrns = (await laRes.json()).schools.map((s: { urn: number }) => s.urn).sort();
|
||||
expect(townUrns).not.toEqual(laUrns);
|
||||
});
|
||||
|
||||
test('a place below the threshold has no page', async ({ page }) => {
|
||||
// Crosby holds one school; publishing it would be a page with nothing to say.
|
||||
const res = await page.request.get('/api/places/town/crosby');
|
||||
expect(res.status()).toBe(404);
|
||||
});
|
||||
|
||||
test('place pages declare a canonical and reach the sitemap', async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
const canonical = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(canonical).toBe(`https://www.schoolcompare.co.uk/schools/${place.slug}`);
|
||||
|
||||
const xml = await (await page.request.get('/sitemaps/places-1.xml')).text();
|
||||
expect(xml).toContain(`/schools/${place.slug}`);
|
||||
});
|
||||
|
||||
test('the place page ships ItemList structured data that parses', async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
const raw = await page.locator('script[type="application/ld+json"]').first()
|
||||
.textContent();
|
||||
const parsed = JSON.parse(raw!);
|
||||
const types = (parsed['@graph'] ?? []).map((n: { '@type': string }) => n['@type']);
|
||||
expect(types).toContain('ItemList');
|
||||
expect(types).toContain('BreadcrumbList');
|
||||
});
|
||||
|
||||
test('a place page states the local average against England', async ({ page }) => {
|
||||
// The one number a list cannot give, and the reason these pages are not
|
||||
// a name dropped into a template.
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
await expect(page.getByTestId('local-vs-england')).toContainText(/across England/i);
|
||||
});
|
||||
|
||||
test('a place page links its phase variants, and they resolve', async ({ page }) => {
|
||||
// "primary schools in beccles" is the query shape the baseline showed. The
|
||||
// first cut submitted only the bare place URL and linked nothing, leaving
|
||||
// ~950 variant pages reachable by nothing at all.
|
||||
const res = await page.request.get('/api/places');
|
||||
const { places } = await res.json();
|
||||
const town = places.find((p: { kind: string }) => p.kind === 'town');
|
||||
expect(town).toBeTruthy();
|
||||
|
||||
const detail = await (await page.request.get(`/api/places/town/${town.slug}`)).json();
|
||||
test.skip(!(detail.place.phases ?? []).length, 'no phase clears the threshold here');
|
||||
|
||||
await page.goto(`/schools/${town.slug}`);
|
||||
const phase = detail.place.phases[0];
|
||||
const link = page.locator(`a[href="/schools/${town.slug}/${phase}"]`).first();
|
||||
await expect(link).toBeVisible();
|
||||
|
||||
await link.click();
|
||||
await expect(page.locator('h1')).toContainText(new RegExp(`${phase} schools in`, 'i'));
|
||||
});
|
||||
|
||||
test('phase variants are submitted in the places sitemap', async ({ page }) => {
|
||||
const xml = await (await page.request.get('/sitemaps/places-1.xml')).text();
|
||||
expect(xml).toMatch(/\/schools\/[a-z0-9-]+\/primary</);
|
||||
});
|
||||
|
||||
test('no page title repeats the brand', async ({ page }) => {
|
||||
// The root layout appends '| schoolcompare' to a plain-string title. Any
|
||||
// route whose title already carries the brand must opt out with
|
||||
// `absolute`, or it ships '... | schoolcompare | schoolcompare' — which is
|
||||
// how ~2,600 place pages first went out.
|
||||
const res = await page.request.get('/api/places');
|
||||
const { places } = await res.json();
|
||||
const town = places.find((p: { kind: string }) => p.kind === 'town');
|
||||
|
||||
for (const path of ['/', '/rankings', '/admissions', `/schools/${town.slug}`]) {
|
||||
await page.goto(path);
|
||||
const title = await page.title();
|
||||
const brands = (title.match(/schoolcompare/gi) ?? []).length;
|
||||
expect(brands, `${path} repeats the brand: ${title}`).toBeLessThanOrEqual(1);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,130 @@
|
||||
import { metadata as homeMetadata } from '@/app/page';
|
||||
import { metadata as rankingsMetadata } from '@/app/rankings/page';
|
||||
import { metadata as admissionsMetadata } from '@/app/admissions/page';
|
||||
import { generateMetadata as compareMetadata } from '@/app/compare/page';
|
||||
|
||||
describe('canonical URLs', () => {
|
||||
it('the homepage canonicalises to the bare root', () => {
|
||||
// page.tsx reads eleven search params. Without this, every filter
|
||||
// combination is a crawlable near-duplicate of the one page we want to
|
||||
// rank for "compare schools".
|
||||
expect(homeMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
|
||||
it('rankings canonicalises to the bare path', () => {
|
||||
expect(rankingsMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('admissions canonicalises to the bare path', () => {
|
||||
expect(admissionsMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/admissions');
|
||||
});
|
||||
});
|
||||
|
||||
describe('/compare indexability', () => {
|
||||
it('the bare compare page is indexable and canonical to itself', async () => {
|
||||
// This is the landing page for the "compare schools" head term.
|
||||
const meta = await compareMetadata({ searchParams: Promise.resolve({}) });
|
||||
expect(meta.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/compare');
|
||||
expect(meta.robots).toBeUndefined();
|
||||
});
|
||||
|
||||
it('a comparison of specific schools is noindex, follow', async () => {
|
||||
// ~317 million pairs before triples. Indexing the parameter space would
|
||||
// swamp everything else in the corpus.
|
||||
const meta = await compareMetadata({
|
||||
searchParams: Promise.resolve({ urns: '100001,100002' }),
|
||||
});
|
||||
expect(meta.robots).toEqual({ index: false, follow: true });
|
||||
});
|
||||
|
||||
it('a parameterised comparison still canonicalises to the bare path', async () => {
|
||||
// follow:true plus a canonical means the outbound links to each school
|
||||
// page still pass value even though this URL is not indexed.
|
||||
const meta = await compareMetadata({
|
||||
searchParams: Promise.resolve({ urns: '100001,100002' }),
|
||||
});
|
||||
expect(meta.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/compare');
|
||||
});
|
||||
});
|
||||
|
||||
/*
|
||||
* W8 — snippet copy for the C1 cluster.
|
||||
*
|
||||
* The baseline (GSC, 16 months to 2026-08-20) showed these pages ranking on
|
||||
* page one and converting at a tenth of the normal rate: "compare school
|
||||
* performance" at position 6.1 with 0.43% CTR, against 9.16% for the brand
|
||||
* query from the same neighbourhood. The SERP is dominated by the DfE's own
|
||||
* "Compare school performance" service, so the job of this copy is to say
|
||||
* what that service does not offer, without losing intent match on the title.
|
||||
*
|
||||
* These tests guard the mechanics that make a snippet work — length, intent
|
||||
* keyword, differentiator, no brand-first — not the exact wording, which
|
||||
* should stay free to iterate.
|
||||
*/
|
||||
|
||||
// Google truncates titles near 60 characters and descriptions near 155.
|
||||
const TITLE_MAX = 60;
|
||||
const DESC_MIN = 110;
|
||||
const DESC_MAX = 155;
|
||||
|
||||
type Meta = { title?: unknown; description?: unknown };
|
||||
const titleOf = (m: Meta): string => {
|
||||
const t = m.title as string | { absolute?: string } | undefined;
|
||||
return typeof t === 'string' ? t : (t?.absolute ?? '');
|
||||
};
|
||||
|
||||
describe('C1 snippet copy', () => {
|
||||
const pages: Array<[string, Meta, RegExp]> = [
|
||||
['home', homeMetadata as Meta, /compare schools/i],
|
||||
['rankings', rankingsMetadata as Meta, /league table/i],
|
||||
['admissions', admissionsMetadata as Meta, /admission/i],
|
||||
];
|
||||
|
||||
for (const [name, meta, intent] of pages) {
|
||||
it(`${name}: title carries the search intent and fits the SERP`, () => {
|
||||
const t = titleOf(meta);
|
||||
expect(t).toMatch(intent);
|
||||
expect(t.length).toBeLessThanOrEqual(TITLE_MAX);
|
||||
});
|
||||
|
||||
it(`${name}: title does not open with the brand`, () => {
|
||||
// The measured 0.43% CTR came from a brand-first title. The most
|
||||
// valuable pixels go to the thing the searcher typed.
|
||||
expect(titleOf(meta).toLowerCase().startsWith('schoolcompare')).toBe(false);
|
||||
});
|
||||
|
||||
it(`${name}: description is long enough to be worth reading, short enough to survive`, () => {
|
||||
const d = meta.description as string;
|
||||
expect(d.length).toBeGreaterThanOrEqual(DESC_MIN);
|
||||
expect(d.length).toBeLessThanOrEqual(DESC_MAX);
|
||||
});
|
||||
}
|
||||
|
||||
it('the homepage description names what gov.uk does not publish', () => {
|
||||
// Admissions distance is the one fact the DfE service has no equivalent
|
||||
// for. If it ever leaves this description, the snippet is competing with
|
||||
// gov.uk on gov.uk's own ground.
|
||||
expect(homeMetadata.description).toMatch(/close you had to live|distance/i);
|
||||
});
|
||||
|
||||
it('/compare targets the tool phrasing rather than repeating the homepage', () => {
|
||||
// Two pages chasing one phrase is how a site competes with itself.
|
||||
return compareMetadata({ searchParams: Promise.resolve({}) }).then((m) => {
|
||||
expect(m.title).toMatch(/comparison tool/i);
|
||||
expect(m.title).not.toBe(titleOf(homeMetadata as Meta));
|
||||
});
|
||||
});
|
||||
|
||||
it('no C1 page claims a school count that will drift', () => {
|
||||
// The corpus moves with every data refresh; this repo has already shipped
|
||||
// one copy bug of that kind ("three schools" against MAX_SCHOOLS = 5).
|
||||
for (const [, meta] of pages) {
|
||||
expect(meta.description as string).not.toMatch(/\b\d{2},\d{3}\b|\b\d{2},000\b/);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,42 @@
|
||||
import { generateMetadata as placeMeta } from '@/app/schools/[place]/page';
|
||||
|
||||
jest.mock('@/lib/places', () => ({
|
||||
...jest.requireActual('@/lib/places'),
|
||||
fetchPlace: jest.fn(async (kind: string, slug: string) =>
|
||||
slug === 'atlantis' ? null : ({
|
||||
place: { kind, slug, name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex' },
|
||||
schools: [], averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
})),
|
||||
fetchPlaces: jest.fn(async () => []),
|
||||
}));
|
||||
|
||||
describe('place page metadata', () => {
|
||||
it('titles the page the way the place is searched', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect((m.title as { absolute: string }).absolute).toMatch(/schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('canonicalises to its own path on the www host', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect(m.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/schools/brentwood');
|
||||
});
|
||||
|
||||
it('opts out of the layout template, which would double the brand', () => {
|
||||
// The root layout appends '| schoolcompare' to a plain string title, and
|
||||
// these titles already carry it — every place page shipped reading
|
||||
// '... | schoolcompare | schoolcompare' until this was made absolute.
|
||||
return placeMeta({ params: Promise.resolve({ place: 'brentwood' }) })
|
||||
.then((m) => {
|
||||
expect(typeof m.title).toBe('object');
|
||||
expect((m.title as { absolute: string }).absolute)
|
||||
.not.toMatch(/schoolcompare.*schoolcompare/);
|
||||
});
|
||||
});
|
||||
|
||||
it('an unknown place gets a not-found title rather than inventing one', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'atlantis' }) });
|
||||
expect(m.title).toMatch(/not found/i);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,114 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import type { PlaceDetail } from '@/lib/places';
|
||||
|
||||
const detail: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex', phases: ['primary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', rwm_expected_pct: 82,
|
||||
ofsted_grade: 1 } as never,
|
||||
{ urn: 2, school_name: 'Beta Primary', rwm_expected_pct: 44,
|
||||
ofsted_grade: 3 } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
describe('PlaceView', () => {
|
||||
it('leads with an H1 that matches how the place is searched', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('heading', { level: 1 }))
|
||||
.toHaveTextContent(/primary schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('states the count so the page says something before the table', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText(/29 schools/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('compares the local average against England, which a list cannot', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('63');
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('61');
|
||||
});
|
||||
|
||||
it('links every school in scope, which is what de-orphans them', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getAllByRole('link', { name: /Primary$/ })).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('links to the parent authority so the place sits in a hierarchy', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /Essex/i }))
|
||||
.toHaveAttribute('href', '/schools/authority/essex');
|
||||
});
|
||||
|
||||
it('shows the Ofsted distribution, not just a count of Outstanding', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('ofsted-distribution')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('links to neighbouring places so the page is not a dead end', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[{ kind: 'town', slug: 'romford', name: 'Romford', count: 40 }]} />);
|
||||
expect(screen.getByRole('link', { name: /Romford/ }))
|
||||
.toHaveAttribute('href', '/schools/romford');
|
||||
});
|
||||
|
||||
it('says nothing about an average it does not have', () => {
|
||||
render(<PlaceView detail={{ ...detail, averages:
|
||||
{ rwm_expected_pct: null, attainment_8_score: null } }}
|
||||
phase="primary" englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByTestId('local-vs-england')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView structured data', () => {
|
||||
function jsonLd() {
|
||||
const { container } = render(<PlaceView detail={detail} phase="primary"
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const el = container.querySelector('script[type="application/ld+json"]');
|
||||
return JSON.parse(el!.textContent!);
|
||||
}
|
||||
|
||||
it('declares the page as a ranked list, not prose', () => {
|
||||
const types = jsonLd()['@graph'].map((n: { '@type': string }) => n['@type']);
|
||||
expect(types).toContain('ItemList');
|
||||
expect(types).toContain('BreadcrumbList');
|
||||
});
|
||||
|
||||
it('gives every listed school an absolute URL on the canonical host', () => {
|
||||
const list = jsonLd()['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
|
||||
expect(list.itemListElement).toHaveLength(2);
|
||||
for (const item of list.itemListElement) {
|
||||
expect(item.url).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\//);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView phase variants', () => {
|
||||
it('links the phase variants that exist', () => {
|
||||
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /Primary schools in Brentwood/i }))
|
||||
.toHaveAttribute('href', '/schools/brentwood/primary');
|
||||
});
|
||||
|
||||
it('links no variant for a phase below its own threshold', () => {
|
||||
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByRole('link', { name: /Secondary schools in Brentwood/i }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('does not link sideways from a variant page to itself', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.queryByRole('link', { name: /Primary schools in Brentwood/i }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,27 @@
|
||||
import { SITE_URL, absoluteUrl } from '@/lib/site';
|
||||
|
||||
describe('SITE_URL', () => {
|
||||
it('is the www host, which is the one that serves a 200', () => {
|
||||
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
|
||||
// is a wasted signal, so every absolute URL we emit must already be www.
|
||||
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
|
||||
});
|
||||
|
||||
it('has no trailing slash, so joins never double up', () => {
|
||||
expect(SITE_URL.endsWith('/')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('absoluteUrl', () => {
|
||||
it('joins a rooted path', () => {
|
||||
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('joins a path missing its leading slash', () => {
|
||||
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('maps the site root to a bare trailing slash', () => {
|
||||
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
});
|
||||
@@ -1,12 +1,16 @@
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
import { AdmissionsView } from '@/components/AdmissionsView';
|
||||
|
||||
export const dynamic = 'force-static';
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: 'School Admissions Guide',
|
||||
// Deadlines and offer days are what gets searched, and what this page is
|
||||
// genuinely best at — the countdowns are live.
|
||||
title: { absolute: 'School Admissions Deadlines & Offer Days | schoolcompare' },
|
||||
description:
|
||||
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
|
||||
'Every key date for primary and secondary school admissions in England, with live countdowns to the application deadline and National Offer Day.',
|
||||
alternates: { canonical: absoluteUrl('/admissions') },
|
||||
};
|
||||
|
||||
export default function AdmissionsPage() {
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
|
||||
import { fetchComparison, fetchMetrics } from '@/lib/api';
|
||||
import { ComparisonView } from '@/components/ComparisonView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
interface ComparePageProps {
|
||||
@@ -14,13 +15,36 @@ interface ComparePageProps {
|
||||
}>;
|
||||
}
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: 'Compare Schools',
|
||||
description:
|
||||
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
|
||||
keywords:
|
||||
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
|
||||
};
|
||||
/**
|
||||
* Indexability depends on the query string, so this cannot be a static export.
|
||||
*
|
||||
* Bare /compare is the landing page for the "compare schools" head term and
|
||||
* stays indexable. /compare?urns=… is an unbounded parameter space — 25,193
|
||||
* schools make ~317 million pairs — so it goes noindex. It stays `follow` and
|
||||
* keeps a canonical to the bare path, so the links out to each school page
|
||||
* still count.
|
||||
*/
|
||||
export async function generateMetadata(
|
||||
{ searchParams }: ComparePageProps,
|
||||
): Promise<Metadata> {
|
||||
const { urns } = await searchParams;
|
||||
|
||||
const base: Metadata = {
|
||||
// Deliberately not the homepage's phrase. Two pages chasing "compare
|
||||
// schools" is how a site competes with itself; this one takes the tool
|
||||
// phrasing instead.
|
||||
title: 'School Comparison Tool — Up to Five at Once | schoolcompare',
|
||||
description:
|
||||
'Put up to five English schools in one table: SATs and GCSE results against the England average, Ofsted grades, and the distance places were offered.',
|
||||
keywords:
|
||||
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
|
||||
alternates: { canonical: absoluteUrl('/compare') },
|
||||
};
|
||||
|
||||
if (!urns) return base;
|
||||
|
||||
return { ...base, robots: { index: false, follow: true } };
|
||||
}
|
||||
|
||||
// Dynamic via searchParams; remove force-dynamic so internal data fetches
|
||||
// can still use Next.js's per-call revalidate cache.
|
||||
|
||||
@@ -5,6 +5,7 @@ import { Navigation } from '@/components/Navigation';
|
||||
import { Footer } from '@/components/Footer';
|
||||
import { ComparisonToast } from '@/components/ComparisonToast';
|
||||
import { ComparisonProvider } from '@/context/ComparisonProvider';
|
||||
import { SITE_URL } from '@/lib/site';
|
||||
import './globals.css';
|
||||
|
||||
// Manrope carries headings and key messaging — the guideline's "friendly,
|
||||
@@ -47,29 +48,32 @@ export const metadata: Metadata = {
|
||||
statusBarStyle: 'default',
|
||||
},
|
||||
title: {
|
||||
default: 'schoolcompare | Compare School Performance',
|
||||
default: 'Compare Schools Side by Side | schoolcompare',
|
||||
template: '%s | schoolcompare',
|
||||
},
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place. Free, no sign-up.',
|
||||
keywords: 'school comparison, KS2 results, KS4 results, primary school, secondary school, England schools, SATs results, GCSE results',
|
||||
authors: [{ name: 'schoolcompare' }],
|
||||
manifest: '/manifest.json',
|
||||
// No `icons` key on purpose: setting it here would override the file
|
||||
// conventions. app/icon.svg and app/apple-icon.tsx are the source, and
|
||||
// app/opengraph-image.tsx supplies og:image and twitter:image.
|
||||
metadataBase: new URL('https://schoolcompare.co.uk'),
|
||||
metadataBase: new URL(SITE_URL),
|
||||
openGraph: {
|
||||
type: 'website',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
url: 'https://schoolcompare.co.uk',
|
||||
title: 'Compare Schools Side by Side | schoolcompare',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
|
||||
url: SITE_URL,
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
// summary_large_image now that there is an image worth showing.
|
||||
card: 'summary_large_image',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
title: 'Compare Schools Side by Side | schoolcompare',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
|
||||
},
|
||||
};
|
||||
|
||||
|
||||
+20
-2
@@ -3,6 +3,7 @@
|
||||
* Main landing page with school search and browsing
|
||||
*/
|
||||
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchSchools, fetchFilters, fetchDataInfo } from '@/lib/api';
|
||||
import { formatAcademicYear } from '@/lib/utils';
|
||||
@@ -33,8 +34,25 @@ interface HomePageProps {
|
||||
* saying the brand twice.
|
||||
*/
|
||||
export const metadata: Metadata = {
|
||||
title: { absolute: 'schoolcompare | Compare every school in England' },
|
||||
description: 'Search and compare school performance across England',
|
||||
/*
|
||||
* Intent in the title, differentiator in the description.
|
||||
*
|
||||
* These queries are owned by the DfE's own "Compare school performance"
|
||||
* service, and the old title — brand first, then a near-paraphrase of that
|
||||
* service's name — gave a searcher no reason to pick us over it. It drew
|
||||
* 0.43% CTR at position 6.1 while the brand query drew 9.16% from the same
|
||||
* neighbourhood, so the ranking was never the problem.
|
||||
*
|
||||
* The title now matches what people type. The description carries the one
|
||||
* fact gov.uk does not publish: how close you had to live to get a place.
|
||||
*/
|
||||
title: { absolute: 'Compare Schools Side by Side | schoolcompare' },
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place. Free, no sign-up.',
|
||||
// This page reads eleven search params. They filter a result set; they do
|
||||
// not make a new document. Collapsing every combination onto "/" stops the
|
||||
// homepage competing with itself for its own head terms.
|
||||
alternates: { canonical: absoluteUrl('/') },
|
||||
};
|
||||
|
||||
// The page reads searchParams, which makes rendering dynamic by default.
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
|
||||
import { fetchRankings, fetchFilters, fetchMetrics } from '@/lib/api';
|
||||
import { RankingsView } from '@/components/RankingsView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
interface RankingsPageProps {
|
||||
@@ -17,9 +18,15 @@ interface RankingsPageProps {
|
||||
}
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: 'School Rankings',
|
||||
description: 'Top-ranked schools by SATs and GCSE performance across England',
|
||||
// 'School Rankings' matched nothing anyone types. League tables is the
|
||||
// phrase parents actually search, and it spikes each results day.
|
||||
title: { absolute: 'Primary & Secondary School League Tables | schoolcompare' },
|
||||
description:
|
||||
'Rank English schools by SATs results, GCSEs, Progress 8 or Attainment 8, and filter by local authority or year. Built from the DfE’s own figures.',
|
||||
keywords: 'school rankings, top schools, best schools, KS2 rankings, KS4 rankings, school league tables',
|
||||
// Param forms (?metric=&local_authority=&year=&phase=) collapse here for
|
||||
// now. W3 replaces them with real indexable paths.
|
||||
alternates: { canonical: absoluteUrl('/rankings') },
|
||||
};
|
||||
|
||||
// Dynamic via searchParams; remove force-dynamic so internal data fetches
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
*/
|
||||
|
||||
import { MetadataRoute } from 'next';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
export default function robots(): MetadataRoute.Robots {
|
||||
return {
|
||||
@@ -14,6 +15,6 @@ export default function robots(): MetadataRoute.Robots {
|
||||
disallow: ['/api/', '/_next/'],
|
||||
},
|
||||
],
|
||||
sitemap: 'https://schoolcompare.co.uk/sitemap.xml',
|
||||
sitemap: absoluteUrl('/sitemap.xml'),
|
||||
};
|
||||
}
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
} from '@/lib/schoolSections';
|
||||
import { parseSchoolSlug, schoolUrl } from '@/lib/utils';
|
||||
import type { NationalAverages } from '@/lib/types';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
/**
|
||||
@@ -97,7 +98,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
title,
|
||||
description,
|
||||
type: 'website',
|
||||
url: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
url: absoluteUrl(canonicalPath),
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
@@ -106,7 +107,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
description,
|
||||
},
|
||||
alternates: {
|
||||
canonical: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
canonical: absoluteUrl(canonicalPath),
|
||||
},
|
||||
};
|
||||
} catch {
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
/**
|
||||
* Phase variants of a place page.
|
||||
*
|
||||
* Phase is part of the query — "primary schools in beccles", "secondary
|
||||
* schools in brentwood" — not a filter applied afterwards, so each gets its
|
||||
* own indexable path. A place with no schools of the phase has no page: the
|
||||
* per-phase threshold, not an error.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ place: string; phase: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
const PHASES = ['primary', 'secondary'] as const;
|
||||
type Phase = (typeof PHASES)[number];
|
||||
|
||||
const isPhase = (v: string): v is Phase => (PHASES as readonly string[]).includes(v);
|
||||
|
||||
async function resolve(slug: string, phase: Phase) {
|
||||
return (await fetchPlace('town', slug, phase))
|
||||
?? (await fetchPlace('locality', slug, phase));
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { place: slug, phase } = await params;
|
||||
if (!isPhase(phase)) return { title: 'Place Not Found' };
|
||||
const detail = await resolve(slug, phase);
|
||||
if (!detail || detail.schools.length === 0) return { title: 'Place Not Found' };
|
||||
|
||||
const word = phase === 'secondary' ? 'Secondary' : 'Primary';
|
||||
const { name } = detail.place;
|
||||
return {
|
||||
title: { absolute: `${word} Schools in ${name} — Ranked | schoolcompare` },
|
||||
description:
|
||||
`Every ${phase} school in ${name} ranked by results, with Ofsted grades and `
|
||||
+ `the local average against England.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/${slug}/${phase}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function PlacePhasePage({ params }: Props) {
|
||||
const { place: slug, phase } = await params;
|
||||
if (!isPhase(phase)) notFound();
|
||||
const detail = await resolve(slug, phase);
|
||||
if (!detail || detail.schools.length === 0) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
const englandAverage = phase === 'secondary'
|
||||
? national?.secondary?.attainment_8_score ?? null
|
||||
: national?.primary?.rwm_expected_pct ?? null;
|
||||
|
||||
return <PlaceView detail={detail} phase={phase}
|
||||
englandAverage={englandAverage} neighbours={[]} />;
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
/**
|
||||
* Town and locality pages.
|
||||
*
|
||||
* A place below the five-school threshold is not in the registry, so
|
||||
* fetchPlace returns null and the request 404s rather than rendering a page
|
||||
* with nothing to say.
|
||||
*/
|
||||
import { notFound, redirect } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces, authoritySlug } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ place: string }> }
|
||||
|
||||
// ISR: place aggregates change only when the pipeline runs.
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ place: string }>> {
|
||||
// Off by default: ~2,000 place routes cannot be built in CI on every deploy.
|
||||
// Matches the PRERENDER_SCHOOLS gate on the school route.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'town' || p.kind === 'locality')
|
||||
.map((p) => ({ place: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
async function resolve(slug: string) {
|
||||
return (await fetchPlace('town', slug)) ?? (await fetchPlace('locality', slug));
|
||||
}
|
||||
|
||||
/** Other towns in the same authority — the cheapest honest definition of
|
||||
* "nearby", and enough to stop each place page being a dead end. */
|
||||
async function neighboursOf(detail: { place: { slug: string; parent_authority: string | null } }) {
|
||||
if (!detail.place.parent_authority) return [];
|
||||
const all = await fetchPlaces();
|
||||
return all
|
||||
.filter((p) => p.kind === 'town' && p.slug !== detail.place.slug)
|
||||
.slice(0, 12);
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { place: slug } = await params;
|
||||
const detail = await resolve(slug);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
// absolute: the root layout's template appends '| schoolcompare' to a
|
||||
// plain string, and this title already carries it. Without this every
|
||||
// place title read '... | schoolcompare | schoolcompare'.
|
||||
title: { absolute: `Schools in ${name} — Compare ${count} Schools | schoolcompare` },
|
||||
description:
|
||||
`Every school in ${name} ranked by SATs and GCSE results, with Ofsted grades, `
|
||||
+ `the local average against England, and how close you had to live to get a place.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/${slug}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function PlacePage({ params }: Props) {
|
||||
const { place: slug } = await params;
|
||||
const detail = await resolve(slug);
|
||||
if (!detail) notFound();
|
||||
|
||||
// Global constraint: no page without a local average. A place with too few
|
||||
// schools carrying results has nothing to say that a list does not, so it
|
||||
// defers to its authority rather than publishing a thin page.
|
||||
if (detail.averages.rwm_expected_pct == null
|
||||
&& detail.averages.attainment_8_score == null) {
|
||||
if (detail.place.parent_authority) {
|
||||
redirect(`/schools/authority/${authoritySlug(detail.place.parent_authority)}`);
|
||||
}
|
||||
notFound();
|
||||
}
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
// NationalAverages is nested by phase — { primary: {...}, secondary: {...} }
|
||||
// — not flat. Reading it flat silently yields undefined and the page renders
|
||||
// with no comparison, which is the one thing that makes it not a list.
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={await neighboursOf(detail)}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
/**
|
||||
* Local authority pages.
|
||||
*
|
||||
* A separate namespace from /schools/[place] because 67 town names collide
|
||||
* with an authority name and neither set contains the other — Bedford the
|
||||
* town holds 104 schools, Bedford the authority 86, because postal towns
|
||||
* cross authority boundaries. The title says "Local Authority" so a reader
|
||||
* landing on both knows which set each covers.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ la: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ la: string }>> {
|
||||
// Gated like every other prerender in this app. There are only ~154
|
||||
// authorities, but "few enough to always build" still means the API must be
|
||||
// reachable at build time, and in CI it is not — the build fails with
|
||||
// ECONNREFUSED rather than degrading. The catch is the same fallback the
|
||||
// school route uses.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'authority')
|
||||
.map((p) => ({ la: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { la } = await params;
|
||||
const detail = await fetchPlace('authority', la);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
title: { absolute: `Schools in ${name} — Local Authority | schoolcompare` },
|
||||
description:
|
||||
`All ${count} schools in the ${name} local authority, ranked by SATs and GCSE `
|
||||
+ `results, with Ofsted grades and the authority average against England.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/authority/${la}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function AuthorityPage({ params }: Props) {
|
||||
const { la } = await params;
|
||||
const detail = await fetchPlace('authority', la);
|
||||
if (!detail) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={[]}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Postcode district pages.
|
||||
*
|
||||
* No phase variants: nobody searches "primary schools in SW11", so the
|
||||
* variants would be pages without demand. These exist to catch
|
||||
* "schools near <postcode>" and to give London districts a geographic page
|
||||
* where the GIAS town field cannot.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ outcode: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ outcode: string }>> {
|
||||
// 1,760 of these; same CI budget argument as the town routes.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'outcode')
|
||||
.map((p) => ({ outcode: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { outcode } = await params;
|
||||
const detail = await fetchPlace('outcode', outcode);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
title: { absolute: `Schools near ${name} | schoolcompare` },
|
||||
description:
|
||||
`${count} schools in the ${name} postcode district, ranked by results, with `
|
||||
+ `Ofsted grades and how close you had to live to get a place.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/near/${outcode}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function OutcodePage({ params }: Props) {
|
||||
const { outcode } = await params;
|
||||
const detail = await fetchPlace('outcode', outcode);
|
||||
if (!detail) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={[]}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -1,32 +1,8 @@
|
||||
/**
|
||||
* Runtime proxy for /sitemap.xml → the FastAPI backend's generated sitemap.
|
||||
*
|
||||
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
|
||||
* baking the backend host into the build, so one image works in every
|
||||
* environment. robots.ts points crawlers here.
|
||||
*/
|
||||
|
||||
import { NextResponse } from 'next/server';
|
||||
import { proxySitemap } from '@/lib/sitemapProxy';
|
||||
|
||||
export const dynamic = 'force-dynamic';
|
||||
export const runtime = 'nodejs';
|
||||
|
||||
function backendOrigin(): string {
|
||||
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
||||
return base.replace(/\/api$/, '');
|
||||
}
|
||||
|
||||
export async function GET() {
|
||||
let upstream: Response;
|
||||
try {
|
||||
upstream = await fetch(`${backendOrigin()}/sitemap.xml`, { cache: 'no-store' });
|
||||
} catch {
|
||||
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
|
||||
}
|
||||
|
||||
const body = await upstream.text();
|
||||
return new NextResponse(body, {
|
||||
status: upstream.status,
|
||||
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
|
||||
});
|
||||
return proxySitemap('/sitemap.xml');
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
import { NextResponse } from 'next/server';
|
||||
import { proxySitemap } from '@/lib/sitemapProxy';
|
||||
|
||||
export const dynamic = 'force-dynamic';
|
||||
export const runtime = 'nodejs';
|
||||
|
||||
/**
|
||||
* Children are /sitemaps/static.xml and /sitemaps/schools-{n}.xml. The name is
|
||||
* validated here rather than passed through, so this route cannot be used to
|
||||
* reach arbitrary backend paths.
|
||||
*/
|
||||
const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/;
|
||||
|
||||
export async function GET(
|
||||
_request: Request,
|
||||
{ params }: { params: Promise<{ parts: string[] }> },
|
||||
) {
|
||||
const { parts } = await params;
|
||||
const name = parts.join('/');
|
||||
if (!CHILD.test(name)) {
|
||||
return new NextResponse('Not found', { status: 404 });
|
||||
}
|
||||
return proxySitemap(`/sitemaps/${name}`);
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
/* Tokens only — see globals.css. Matches RankingsView's conventions. */
|
||||
.container {
|
||||
width: 100%;
|
||||
min-width: 0;
|
||||
}
|
||||
|
||||
.header {
|
||||
margin-bottom: 1.5rem;
|
||||
}
|
||||
|
||||
.header h1 {
|
||||
font-size: 2.25rem;
|
||||
font-weight: 700;
|
||||
color: var(--text-primary);
|
||||
margin-bottom: 0.5rem;
|
||||
font-family: var(--font-display);
|
||||
text-wrap: balance;
|
||||
}
|
||||
|
||||
.summary {
|
||||
font-size: 1rem;
|
||||
color: var(--text-secondary);
|
||||
margin: 0;
|
||||
line-height: 1.6;
|
||||
}
|
||||
|
||||
/* The one number a list cannot give you, so it gets its own band. */
|
||||
.compare {
|
||||
background: var(--bg-secondary);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
padding: 0.875rem 1.125rem;
|
||||
margin: 0 0 1.5rem;
|
||||
color: var(--text-primary);
|
||||
font-size: 1rem;
|
||||
}
|
||||
|
||||
.ofsted {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem 1.25rem;
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
margin: 0 0 1.5rem;
|
||||
font-size: 0.9375rem;
|
||||
color: var(--text-secondary);
|
||||
}
|
||||
|
||||
/* Wide content scrolls in its own container so the page body never does. */
|
||||
.tableWrap {
|
||||
overflow-x: auto;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
background: var(--bg-card);
|
||||
}
|
||||
|
||||
.table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
font-size: 0.9375rem;
|
||||
}
|
||||
|
||||
.table th,
|
||||
.table td {
|
||||
padding: 0.75rem 1rem;
|
||||
text-align: left;
|
||||
border-bottom: 1px solid var(--border);
|
||||
}
|
||||
|
||||
.table th {
|
||||
background: var(--bg-secondary);
|
||||
color: var(--text-secondary);
|
||||
font-weight: 600;
|
||||
font-size: 0.8125rem;
|
||||
text-transform: uppercase;
|
||||
letter-spacing: 0.04em;
|
||||
}
|
||||
|
||||
.table tbody tr:last-child td {
|
||||
border-bottom: none;
|
||||
}
|
||||
|
||||
.table th:last-child,
|
||||
.num {
|
||||
text-align: right;
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
.neighbours {
|
||||
margin-top: 2rem;
|
||||
}
|
||||
|
||||
.neighbours h2 {
|
||||
font-size: 1.125rem;
|
||||
font-weight: 600;
|
||||
color: var(--text-primary);
|
||||
margin: 0 0 0.75rem;
|
||||
font-family: var(--font-display);
|
||||
}
|
||||
|
||||
.neighbours ul {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem 1rem;
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
margin: 0;
|
||||
}
|
||||
|
||||
/* Phase variants are separate indexable pages, so the bare place page has to
|
||||
link them — a sitemap entry alone leaves them with no internal path in. */
|
||||
.phaseLinks {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem 1rem;
|
||||
margin: 0 0 1.25rem;
|
||||
font-size: 0.9375rem;
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
/**
|
||||
* One place page, shared by all four families.
|
||||
*
|
||||
* They differ in what fills the registry, not in what the page shows, so a
|
||||
* second component would be a second place to forget the same change.
|
||||
*
|
||||
* The local-versus-England comparison is the reason this page is not a list:
|
||||
* it is the one number a parent cannot get by reading the schools one by one,
|
||||
* and it is what keeps the page from reading as a name dropped into a
|
||||
* template.
|
||||
*/
|
||||
import Link from 'next/link';
|
||||
import type { PlaceDetail, PlaceSummary } from '@/lib/places';
|
||||
import { placeUrl, authoritySlug } from '@/lib/places';
|
||||
import { schoolUrl } from '@/lib/utils';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import styles from './PlaceView.module.css';
|
||||
|
||||
interface Props {
|
||||
detail: PlaceDetail;
|
||||
phase?: 'primary' | 'secondary';
|
||||
englandAverage: number | null;
|
||||
/** Nearby places, so the page links onward instead of dead-ending. */
|
||||
neighbours: PlaceSummary[];
|
||||
}
|
||||
|
||||
// Ofsted grades in the order they are reported.
|
||||
const OFSTED_LABELS: Array<[number, string]> = [
|
||||
[1, 'Outstanding'], [2, 'Good'],
|
||||
[3, 'Requires improvement'], [4, 'Inadequate'],
|
||||
];
|
||||
|
||||
export function PlaceView({ detail, phase, englandAverage, neighbours }: Props) {
|
||||
const { place, schools, averages } = detail;
|
||||
const metric = phase === 'secondary' ? 'attainment_8_score' : 'rwm_expected_pct';
|
||||
const local = averages[metric];
|
||||
const phaseWord = phase === 'secondary' ? 'Secondary schools'
|
||||
: phase === 'primary' ? 'Primary schools' : 'Schools';
|
||||
const graded = OFSTED_LABELS
|
||||
.map(([grade, label]) => [label, schools.filter((s) => s.ofsted_grade === grade).length] as const)
|
||||
.filter(([, n]) => n > 0);
|
||||
|
||||
// ItemList tells Google this page is a ranked set rather than prose;
|
||||
// BreadcrumbList puts the place in a hierarchy. Capped at 20 because that
|
||||
// is what the page shows above the fold and what the markup should mirror.
|
||||
const jsonLd = {
|
||||
'@context': 'https://schema.org',
|
||||
'@graph': [
|
||||
{
|
||||
'@type': 'ItemList',
|
||||
name: `${phaseWord} in ${place.name}`,
|
||||
numberOfItems: schools.length,
|
||||
itemListElement: schools.slice(0, 20).map((s, i) => ({
|
||||
'@type': 'ListItem',
|
||||
position: i + 1,
|
||||
url: absoluteUrl(schoolUrl(s.urn, s.school_name)),
|
||||
name: s.school_name,
|
||||
})),
|
||||
},
|
||||
{
|
||||
'@type': 'BreadcrumbList',
|
||||
itemListElement: [
|
||||
{ '@type': 'ListItem', position: 1, name: 'Schools',
|
||||
item: absoluteUrl('/') },
|
||||
{ '@type': 'ListItem', position: 2, name: place.name },
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
return (
|
||||
<div className={styles.container}>
|
||||
<script
|
||||
type="application/ld+json"
|
||||
dangerouslySetInnerHTML={{ __html: JSON.stringify(jsonLd) }}
|
||||
/>
|
||||
<header className={styles.header}>
|
||||
<h1>{phaseWord} in {place.name}</h1>
|
||||
<p className={styles.summary}>
|
||||
{place.count} schools
|
||||
{place.parent_authority && (
|
||||
<>
|
||||
{' · '}
|
||||
<Link href={`/schools/authority/${authoritySlug(place.parent_authority)}`}>
|
||||
{place.parent_authority}
|
||||
</Link>
|
||||
</>
|
||||
)}
|
||||
</p>
|
||||
</header>
|
||||
|
||||
{!phase && (place.phases ?? []).length > 0 && (
|
||||
<nav className={styles.phaseLinks} aria-label="By phase">
|
||||
{(place.phases ?? []).map((ph) => (
|
||||
<Link key={ph} href={`/schools/${place.slug}/${ph}`}>
|
||||
{ph === 'secondary' ? 'Secondary schools' : 'Primary schools'} in {place.name}
|
||||
</Link>
|
||||
))}
|
||||
</nav>
|
||||
)}
|
||||
|
||||
{local != null && englandAverage != null && (
|
||||
<p className={styles.compare} data-testid="local-vs-england">
|
||||
{place.name} averages <strong>{Math.round(local)}</strong> against{' '}
|
||||
<strong>{Math.round(englandAverage)}</strong> across England.
|
||||
</p>
|
||||
)}
|
||||
|
||||
{graded.length > 0 && (
|
||||
<ul className={styles.ofsted} data-testid="ofsted-distribution">
|
||||
{graded.map(([label, n]) => (
|
||||
<li key={label}>{label}: <strong>{n}</strong></li>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
|
||||
<div className={styles.tableWrap}>
|
||||
<table className={styles.table}>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>School</th>
|
||||
<th>{phase === 'secondary' ? 'Attainment 8' : 'RWM expected'}</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{schools.map((s) => (
|
||||
<tr key={s.urn}>
|
||||
<td>
|
||||
<Link href={schoolUrl(s.urn, s.school_name)}>{s.school_name}</Link>
|
||||
</td>
|
||||
<td className={styles.num}>
|
||||
{s[metric] == null ? '—' : Math.round(Number(s[metric]))}
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
{neighbours.length > 0 && (
|
||||
<nav className={styles.neighbours} aria-label="Nearby places">
|
||||
<h2>Nearby</h2>
|
||||
<ul>
|
||||
{neighbours.map((n) => (
|
||||
<li key={n.kind + n.slug}>
|
||||
<Link href={placeUrl(n.kind, n.slug)}>{n.name}</Link>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</nav>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
/**
|
||||
* Client for the places API.
|
||||
*
|
||||
* Two namespaces, matching the backend: towns and localities share
|
||||
* /schools/[place]; authorities take /schools/authority/[la]. 67 town names
|
||||
* collide with an authority name and neither set contains the other, so one
|
||||
* namespace would publish near-duplicate pages.
|
||||
*/
|
||||
import type { School } from '@/lib/types';
|
||||
|
||||
export interface PlaceSummary {
|
||||
kind: string;
|
||||
slug: string;
|
||||
name: string;
|
||||
count: number;
|
||||
/** Phases that clear the threshold on their own, so the page links
|
||||
* variants that exist rather than 404s. Absent on the registry listing. */
|
||||
phases?: string[];
|
||||
}
|
||||
|
||||
export interface PlaceDetail {
|
||||
place: PlaceSummary & { parent_authority: string | null };
|
||||
schools: School[];
|
||||
averages: {
|
||||
rwm_expected_pct: number | null;
|
||||
attainment_8_score: number | null;
|
||||
};
|
||||
}
|
||||
|
||||
export function placeUrl(kind: string, slug: string, phase?: string): string {
|
||||
const base =
|
||||
kind === 'authority' ? `/schools/authority/${slug}`
|
||||
: kind === 'outcode' ? `/schools/near/${slug}`
|
||||
: `/schools/${slug}`;
|
||||
return phase ? `${base}/${phase}` : base;
|
||||
}
|
||||
|
||||
/** An authority name as it appears in a URL. */
|
||||
export function authoritySlug(name: string): string {
|
||||
return name.toLowerCase().trim().replace(/[^\w\s-]/g, '').replace(/\s+/g, '-');
|
||||
}
|
||||
|
||||
const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL
|
||||
|| 'http://localhost:8000/api';
|
||||
|
||||
export async function fetchPlaces(): Promise<PlaceSummary[]> {
|
||||
const res = await fetch(`${API}/places`, { next: { revalidate: 604800 } });
|
||||
if (!res.ok) return [];
|
||||
return (await res.json()).places ?? [];
|
||||
}
|
||||
|
||||
export async function fetchPlace(
|
||||
kind: string, slug: string, phase?: string,
|
||||
): Promise<PlaceDetail | null> {
|
||||
const q = phase ? `?phase=${encodeURIComponent(phase)}` : '';
|
||||
const res = await fetch(`${API}/places/${kind}/${slug}${q}`,
|
||||
{ next: { revalidate: 604800 } });
|
||||
if (!res.ok) return null;
|
||||
return res.json();
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
/**
|
||||
* The one place the site's absolute origin is written down.
|
||||
*
|
||||
* The apex domain 301s to www at Cloudflare, so www is the host that actually
|
||||
* serves a 200. Canonicals, og:url, sitemap <loc> entries and the robots.txt
|
||||
* Sitemap: line must all agree with it — a canonical pointing at a redirect
|
||||
* makes Google resolve the hop before it can consolidate the signal.
|
||||
*
|
||||
* backend/app.py holds the same value as BASE_URL for the sitemap. The two are
|
||||
* asserted against each other by the e2e journeys rather than shared at build
|
||||
* time, because the backend and frontend ship as separate images.
|
||||
*/
|
||||
export const SITE_URL = 'https://www.schoolcompare.co.uk';
|
||||
|
||||
/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */
|
||||
export function absoluteUrl(path: string): string {
|
||||
const rooted = path.startsWith('/') ? path : `/${path}`;
|
||||
return `${SITE_URL}${rooted}`;
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
/**
|
||||
* Runtime proxy for the sitemap family → the FastAPI backend.
|
||||
*
|
||||
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
|
||||
* baking the backend host into the build, so one image works in every
|
||||
* environment. robots.ts points crawlers at /sitemap.xml, which is the index;
|
||||
* the index names children under /sitemaps/, which land on the same proxy.
|
||||
*/
|
||||
import { NextResponse } from 'next/server';
|
||||
|
||||
function backendOrigin(): string {
|
||||
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
||||
return base.replace(/\/api$/, '');
|
||||
}
|
||||
|
||||
export async function proxySitemap(path: string): Promise<NextResponse> {
|
||||
let upstream: Response;
|
||||
try {
|
||||
upstream = await fetch(`${backendOrigin()}${path}`, { cache: 'no-store' });
|
||||
} catch {
|
||||
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
|
||||
}
|
||||
|
||||
const body = await upstream.text();
|
||||
return new NextResponse(body, {
|
||||
status: upstream.status,
|
||||
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,SW11,London
|
||||
canary-wharf,Canary Wharf,E14,London
|
||||
clapham,Clapham,SW4,London
|
||||
shoreditch,Shoreditch,EC2A|E1,London
|
||||
peckham,Peckham,SE15,London
|
||||
brixton,Brixton,SW2|SW9,London
|
||||
camden-town,Camden Town,NW1,London
|
||||
wimbledon,Wimbledon,SW19,London
|
||||
putney,Putney,SW15,London
|
||||
fulham,Fulham,SW6,London
|
||||
chiswick,Chiswick,W4,London
|
||||
stratford,Stratford,E15,London
|
||||
walthamstow,Walthamstow,E17,London
|
||||
tooting,Tooting,SW17,London
|
||||
dulwich,Dulwich,SE21|SE22,London
|
||||
|
Reference in new issue
Block a user