Compare commits
18
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c4b7868cbf | ||
|
|
6b871ce1e9 | ||
|
|
c981d89137 | ||
|
|
de5e790112 | ||
|
|
42138fc402 | ||
|
|
c5af476213 | ||
|
|
de853b90b3 | ||
|
|
759d9f5cea | ||
|
|
555d3f0a7d | ||
|
|
ecc847091c | ||
|
|
b187a478c9 | ||
|
|
c0547c45e5 | ||
|
|
4a3928df9f | ||
|
|
07c97a46c5 | ||
|
|
bb81337aba | ||
|
|
b34511e459 | ||
|
|
f928a15c1e | ||
|
|
1fc1e07d21 |
No files matched your search
+133
-3
@@ -35,6 +35,7 @@ from .data_loader import (
|
||||
search_schools_typesense,
|
||||
)
|
||||
from .data_loader import get_data_info as get_db_info
|
||||
from .places import build_place_registry
|
||||
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||
from .utils import clean_for_json, convert_to_native
|
||||
|
||||
@@ -58,6 +59,12 @@ MAX_SLUG_LENGTH = 60
|
||||
# regenerate endpoint after a pipeline run.
|
||||
_sitemaps: dict[str, str] | None = None
|
||||
|
||||
# Built from the same DataFrame the sitemap uses, so places and sitemap can
|
||||
# never describe different corpora. Reset by the same admin endpoint.
|
||||
_place_registry: dict | None = None
|
||||
|
||||
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
text = text.lower()
|
||||
@@ -79,6 +86,12 @@ def _school_url(urn: int, school_name: str) -> str:
|
||||
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
|
||||
|
||||
|
||||
# A page has something a search result could state if any of these is present
|
||||
# in any year. Shared by _has_publishable_data and the per-school check in
|
||||
# _school_sitemap_rows so the two can never drift.
|
||||
_PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade")
|
||||
|
||||
|
||||
def _has_publishable_data(row) -> bool:
|
||||
"""True when a school page has something a search result could state.
|
||||
|
||||
@@ -87,7 +100,7 @@ def _has_publishable_data(row) -> bool:
|
||||
signal down, so it stays out of the sitemap. The page itself still resolves
|
||||
for anyone who has the URL.
|
||||
"""
|
||||
for field in ("rwm_expected_pct", "attainment_8_score", "ofsted_grade"):
|
||||
for field in _PUBLISHABLE_FIELDS:
|
||||
value = row.get(field)
|
||||
if value is not None and not pd.isna(value):
|
||||
return True
|
||||
@@ -115,6 +128,21 @@ def _school_sitemap_rows(df) -> list[str]:
|
||||
rows: list[str] = []
|
||||
seen: set[int] = set()
|
||||
|
||||
# Publishable is a property of the SCHOOL, not of its latest row.
|
||||
#
|
||||
# The first cut tested the latest year's row alone, which quietly dropped
|
||||
# every school that has results in its history but a null row for the most
|
||||
# recent year — a school that stopped reporting, or whose figures were
|
||||
# suppressed for small-cohort disclosure. The Mallard Academy (150367) is
|
||||
# the case that caught it: real KS2 results for 2015-16 through 2018-19,
|
||||
# then null rows for 2022-23 onward. Its page shows all four years; the
|
||||
# sitemap omitted it. Roughly 220 schools were affected.
|
||||
publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
publishable: set[int] = (
|
||||
set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int))
|
||||
if publishable_cols else set()
|
||||
)
|
||||
|
||||
# Latest row per URN first, so a school's most recent Ofsted date wins.
|
||||
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
|
||||
|
||||
@@ -123,7 +151,7 @@ def _school_sitemap_rows(df) -> list[str]:
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
if not _has_publishable_data(row):
|
||||
if urn not in publishable:
|
||||
continue
|
||||
|
||||
lastmod = None
|
||||
@@ -149,6 +177,14 @@ SITEMAP_CHUNK_SIZE = 10_000
|
||||
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
||||
|
||||
|
||||
def get_place_registry() -> dict:
|
||||
"""The place registry, built once and cached for the process."""
|
||||
global _place_registry
|
||||
if _place_registry is None:
|
||||
_place_registry = build_place_registry(load_school_data())
|
||||
return _place_registry
|
||||
|
||||
|
||||
def _urlset(rows: list[str]) -> str:
|
||||
return "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
@@ -158,6 +194,29 @@ def _urlset(rows: list[str]) -> str:
|
||||
])
|
||||
|
||||
|
||||
def _place_url(place) -> str:
|
||||
"""The canonical path for a place. Two namespaces, per the spec.
|
||||
|
||||
Towns and localities share /schools/[place]; authorities take their own
|
||||
prefix because 67 town names collide with an authority name and neither
|
||||
set contains the other.
|
||||
"""
|
||||
if place.kind == "authority":
|
||||
return f"/schools/authority/{place.slug}"
|
||||
if place.kind == "outcode":
|
||||
return f"/schools/near/{place.slug}"
|
||||
return f"/schools/{place.slug}"
|
||||
|
||||
|
||||
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
|
||||
return [
|
||||
_url_element(BASE_URL + _place_url(p))
|
||||
for p in sorted(get_place_registry().values(),
|
||||
key=lambda p: (p.kind, p.slug))
|
||||
if p.kind in kinds
|
||||
]
|
||||
|
||||
|
||||
def build_sitemaps() -> dict[str, str]:
|
||||
"""Build the sitemap index and every child, keyed by name."""
|
||||
df = load_school_data()
|
||||
@@ -175,6 +234,17 @@ def build_sitemaps() -> dict[str, str]:
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"schools-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# Separate children per family: Search Console reports coverage per
|
||||
# submitted sitemap, which is how the location layer's indexation is
|
||||
# measured apart from the school pages'.
|
||||
for label, kinds in (("places", ("town", "locality", "authority")),
|
||||
("outcodes", ("outcode",))):
|
||||
rows = _place_sitemap_rows(kinds)
|
||||
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
|
||||
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"{label}-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# On a sitemap index, lastmod means "when this sitemap file last changed",
|
||||
# so generation time is the correct value here — unlike on a <url>, where
|
||||
# it would be a claim about content we cannot support.
|
||||
@@ -1096,6 +1166,62 @@ async def get_rankings(
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/places")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def list_places(request: Request):
|
||||
"""Every published place. The sitemap and the link modules read this."""
|
||||
registry = get_place_registry()
|
||||
return {"places": [
|
||||
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
|
||||
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
|
||||
]}
|
||||
|
||||
|
||||
@app.get("/api/places/{kind}/{slug}")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_place(request: Request, kind: str, slug: str,
|
||||
phase: Optional[str] = None):
|
||||
"""One place: its schools ranked, and its local averages."""
|
||||
if kind not in VALID_PLACE_KINDS:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
place = get_place_registry().get(f"{kind}:{slug}")
|
||||
if place is None:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
df = load_latest_school_data()
|
||||
rows = df[df["urn"].isin(place.urns)]
|
||||
|
||||
if phase:
|
||||
wanted = PHASE_GROUPS.get(phase.lower())
|
||||
if wanted and "phase" in rows.columns:
|
||||
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
|
||||
|
||||
# The metric the page ranks on, which is also the one it averages.
|
||||
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
|
||||
if metric in rows.columns:
|
||||
rows = rows.sort_values(metric, ascending=False, na_position="last")
|
||||
|
||||
averages = {
|
||||
m: (None if m not in rows.columns or rows[m].dropna().empty
|
||||
else float(rows[m].dropna().mean()))
|
||||
for m in ("rwm_expected_pct", "attainment_8_score")
|
||||
}
|
||||
|
||||
cols = [c for c in SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
|
||||
"rwm_expected_pct", "attainment_8_score",
|
||||
"total_pupils"]
|
||||
if c in rows.columns]
|
||||
|
||||
return {
|
||||
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
|
||||
"count": len(place.urns),
|
||||
"parent_authority": place.parent_authority},
|
||||
"schools": clean_for_json(rows[cols]),
|
||||
"averages": averages,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/data-info")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_data_info(request: Request):
|
||||
@@ -1207,7 +1333,11 @@ async def regenerate_sitemap(
|
||||
_: bool = Depends(verify_admin_api_key),
|
||||
):
|
||||
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
|
||||
global _sitemaps
|
||||
global _sitemaps, _place_registry
|
||||
# Places and sitemap are rebuilt together — they read the same marts, and
|
||||
# letting them drift apart would submit URLs for places that no longer
|
||||
# exist.
|
||||
_place_registry = None
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Curated London localities, defined by the postcode districts they cover.
|
||||
|
||||
The GIAS `town` field puts 1,819 London schools under the single value
|
||||
"London", so it cannot answer "schools in Battersea" — a query that appears in
|
||||
the Search Console baseline. No single field can: parliamentary constituency
|
||||
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
|
||||
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
|
||||
and colloquial rather than administrative.
|
||||
|
||||
So this is curated. Where a locality ends is a judgement, not a fact, and a
|
||||
reviewable file is the honest place for a judgement. No new ingestion is
|
||||
needed — the corpus already carries postcodes.
|
||||
|
||||
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
|
||||
mirrors it for anyone querying the warehouse directly; the backend image does
|
||||
not contain `pipeline/`, which is why the module rather than the seed is
|
||||
canonical. Same arrangement as `backend/gias_codes.py`.
|
||||
|
||||
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
|
||||
published, so a typo produces no page rather than an empty one. Places that
|
||||
fail that check are logged at startup, because a locality you meant to publish
|
||||
quietly not appearing is the failure worth hearing about.
|
||||
|
||||
Two rules for anything added here.
|
||||
|
||||
**Sub-borough districts only.** A London borough is a local authority and
|
||||
already has a page at /schools/authority/[la] covering all of its schools; a
|
||||
locality defined by two or three outcodes would be a partial, near-duplicate
|
||||
subset of it. Hackney, Islington, Greenwich and Ealing were all in the first
|
||||
draft for that reason and have been removed.
|
||||
|
||||
**The slug must not match a GIAS town.** "Richmond" did — GIAS has a Richmond
|
||||
in North Yorkshire with 37 schools — so the London one could never publish.
|
||||
The registry skips any locality that collides and logs it.
|
||||
"""
|
||||
|
||||
# slug -> (display name, outcodes)
|
||||
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
|
||||
"battersea": ("Battersea", ("SW11",)),
|
||||
"canary-wharf": ("Canary Wharf", ("E14",)),
|
||||
"clapham": ("Clapham", ("SW4",)),
|
||||
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
|
||||
"peckham": ("Peckham", ("SE15",)),
|
||||
"brixton": ("Brixton", ("SW2", "SW9")),
|
||||
"camden-town": ("Camden Town", ("NW1",)),
|
||||
"wimbledon": ("Wimbledon", ("SW19",)),
|
||||
"putney": ("Putney", ("SW15",)),
|
||||
"fulham": ("Fulham", ("SW6",)),
|
||||
"chiswick": ("Chiswick", ("W4",)),
|
||||
"stratford": ("Stratford", ("E15",)),
|
||||
"walthamstow": ("Walthamstow", ("E17",)),
|
||||
"tooting": ("Tooting", ("SW17",)),
|
||||
"dulwich": ("Dulwich", ("SE21", "SE22")),
|
||||
}
|
||||
@@ -0,0 +1,176 @@
|
||||
"""The place registry: what places the site publishes, and what is in each.
|
||||
|
||||
One module owns this question. The pages, the sitemap and the internal-link
|
||||
modules all read from here, so the threshold and the collision rules exist in
|
||||
exactly one place and are testable without a browser or a database.
|
||||
|
||||
Two namespaces, never one. 67 viable town names collide with a local
|
||||
authority name, and the authority is the larger set in only 43 of them —
|
||||
postal towns cross authority boundaries, so neither can absorb the other.
|
||||
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Five schools with publishable data. Below this a place has nothing to say
|
||||
# that a list of schools does not, and publishing it is index bloat.
|
||||
MIN_SCHOOLS = 5
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Place:
|
||||
kind: str # "town" | "locality" | "authority" | "outcode"
|
||||
slug: str
|
||||
name: str
|
||||
urns: tuple[int, ...]
|
||||
parent_authority: str | None # authority NAME, for the 301 target
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
return f"{self.kind}:{self.slug}"
|
||||
|
||||
|
||||
def _publishable_urns(df) -> set[int]:
|
||||
"""URNs with something a page could state, deduplicated across years."""
|
||||
from backend.app import _PUBLISHABLE_FIELDS
|
||||
|
||||
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
if not cols:
|
||||
return set()
|
||||
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
||||
|
||||
|
||||
def _parent_authority(group) -> str | None:
|
||||
"""The most common authority in a group — the useful 301 target.
|
||||
|
||||
A town spanning several authorities has no single parent, so the mode is
|
||||
the honest answer rather than an arbitrary first row.
|
||||
"""
|
||||
if "local_authority" not in group.columns:
|
||||
return None
|
||||
top = group["local_authority"].dropna()
|
||||
return str(top.mode().iloc[0]) if not top.empty else None
|
||||
|
||||
|
||||
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per distinct value of `column` that clears the threshold."""
|
||||
from backend.app import _slugify
|
||||
|
||||
if column not in df.columns:
|
||||
return {}
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for name, group in df.groupby(column, dropna=True):
|
||||
name = str(name).strip()
|
||||
if not name:
|
||||
continue
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
slug = _slugify(name)
|
||||
if not slug:
|
||||
continue
|
||||
place = Place(
|
||||
kind=kind, slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group) if kind == "town" else None,
|
||||
)
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
||||
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
||||
|
||||
|
||||
def _outcode(postcode) -> str | None:
|
||||
if not isinstance(postcode, str):
|
||||
return None
|
||||
m = _OUTCODE_RE.match(postcode.upper().strip())
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per postcode district clearing the threshold.
|
||||
|
||||
These carry no phase variants: nobody searches "primary schools in SW11".
|
||||
"""
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
working = working[working["_oc"].notna()]
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for oc, group in working.groupby("_oc"):
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
||||
urns=urns, parent_authority=_parent_authority(group))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def _locality_places(df, publishable: set[int],
|
||||
town_slugs: set[str]) -> dict[str, Place]:
|
||||
"""One Place per curated locality clearing the threshold."""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
if slug in town_slugs:
|
||||
# Skip, do not raise. The guard exists so a locality never
|
||||
# silently shadows a town — skipping achieves that, and the error
|
||||
# log makes it loud.
|
||||
#
|
||||
# Raising here took down sitemap generation for all 25,000 school
|
||||
# pages when "richmond" met the GIAS town Richmond in North
|
||||
# Yorkshire. Worse, GIAS town names change without any code change,
|
||||
# so a raise means curated data can break the site spontaneously.
|
||||
# A curation mistake must cost one page, not the sitemap.
|
||||
logger.error(
|
||||
"locality %r collides with the published town of the same "
|
||||
"slug and has been skipped; rename it or remove it", slug)
|
||||
continue
|
||||
group = working[working["_oc"].isin(outcodes)]
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
# Not an error — a locality can legitimately be too small. Logged
|
||||
# because one you meant to publish quietly vanishing is the
|
||||
# failure worth hearing about.
|
||||
logger.warning(
|
||||
"locality %s (%s) has %d publishable schools, below the "
|
||||
"threshold of %d - not published",
|
||||
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
||||
continue
|
||||
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def build_place_registry(df) -> dict[str, Place]:
|
||||
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
||||
if df.empty or "urn" not in df.columns:
|
||||
return {}
|
||||
|
||||
publishable = _publishable_urns(df)
|
||||
registry: dict[str, Place] = {}
|
||||
registry.update(_group(df, "local_authority", "authority", publishable))
|
||||
|
||||
towns = _group(df, "town", "town", publishable)
|
||||
registry.update(towns)
|
||||
|
||||
town_slugs = {p.slug for p in towns.values()}
|
||||
registry.update(_locality_places(df, publishable, town_slugs))
|
||||
registry.update(_outcode_places(df, publishable))
|
||||
return registry
|
||||
@@ -0,0 +1,231 @@
|
||||
"""Tests for the place registry (spec 2026-08-21).
|
||||
|
||||
The registry is built from the in-memory school DataFrame, so these build a
|
||||
small frame directly rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from backend.places import MIN_SCHOOLS, build_place_registry
|
||||
|
||||
|
||||
def _df(rows: list[dict]) -> pd.DataFrame:
|
||||
base = {
|
||||
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
|
||||
"phase": "Primary", "postcode": "AA1 1AA",
|
||||
}
|
||||
return pd.DataFrame([{**base, **r} for r in rows])
|
||||
|
||||
|
||||
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
||||
"""`start` offsets the URNs so two calls can describe different schools —
|
||||
the Bedford case needs two authorities' worth of distinct URNs in one
|
||||
town."""
|
||||
return [
|
||||
{"urn": start + i, "school_name": f"{town} School {i}",
|
||||
"town": town, "local_authority": la, **kw}
|
||||
for i in range(n)
|
||||
]
|
||||
|
||||
|
||||
def test_town_clearing_the_threshold_is_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
assert "town:brentwood" in reg
|
||||
assert reg["town:brentwood"].name == "Brentwood"
|
||||
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_town_below_the_threshold_is_not_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
|
||||
assert "town:crosby" not in reg
|
||||
|
||||
|
||||
def test_a_town_below_threshold_still_names_its_authority():
|
||||
# The route layer needs somewhere to 301 to.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
|
||||
assert "authority:sefton" in reg
|
||||
|
||||
|
||||
def test_town_and_authority_of_the_same_name_are_separate_places():
|
||||
# 67 real collisions. Neither set contains the other: Bedford the town has
|
||||
# 104 schools, Bedford the authority 86, because postal towns cross
|
||||
# authority boundaries.
|
||||
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
||||
assert set(town.urns) != set(authority.urns)
|
||||
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
|
||||
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
|
||||
|
||||
|
||||
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
|
||||
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
|
||||
for r in rows:
|
||||
r["rwm_expected_pct"] = np.nan
|
||||
r["ofsted_grade"] = np.nan
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "town:ghosttown" not in reg
|
||||
|
||||
|
||||
def test_blank_town_is_ignored():
|
||||
rows = _town(MIN_SCHOOLS, "", "Essex")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("town:") for k in reg)
|
||||
|
||||
|
||||
def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows = []
|
||||
for year in (202324, 202425):
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_groups_schools_by_outcode(monkeypatch):
|
||||
# The GIAS town field collapses 1,819 London schools into "London", so a
|
||||
# locality is defined by its postcode districts instead.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"battersea": ("Battersea", ("SW11",))})
|
||||
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
||||
for r in rows:
|
||||
r["postcode"] = "SW11 2AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["locality:battersea"].name == "Battersea"
|
||||
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"nowhere": ("Nowhere", ("ZZ99",))})
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
||||
assert "locality:nowhere" not in reg
|
||||
|
||||
|
||||
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch, caplog):
|
||||
"""A colliding locality is skipped loudly, and the town survives.
|
||||
|
||||
This used to raise, which took down sitemap generation for all 25,000
|
||||
school pages the first time a curated slug met a real GIAS town. Curated
|
||||
data must not be able to break the site — and GIAS town names change with
|
||||
no code change at all, so the raise could fire spontaneously.
|
||||
"""
|
||||
import logging
|
||||
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
|
||||
with caplog.at_level(logging.ERROR):
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
assert "locality:brentwood" not in reg # skipped
|
||||
assert "town:brentwood" in reg # the town is untouched
|
||||
assert "brentwood" in caplog.text # and it was loud about it
|
||||
|
||||
|
||||
def test_a_locality_collision_does_not_break_the_rest_of_the_registry(monkeypatch):
|
||||
# The whole point of skipping rather than raising.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "authority:essex" in reg
|
||||
assert "outcode:cm13" in reg
|
||||
|
||||
|
||||
def test_outcode_places_are_built_from_postcodes():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:cm13"].name == "CM13"
|
||||
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_malformed_postcodes_do_not_create_places():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "not a postcode"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("outcode:") for k in reg)
|
||||
|
||||
|
||||
def test_every_curated_locality_is_structurally_valid():
|
||||
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
||||
import re
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
||||
assert name.strip() == name and name, slug
|
||||
assert outcodes, f"{slug} has no outcodes"
|
||||
for oc in outcodes:
|
||||
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
||||
|
||||
|
||||
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
||||
"""Two copies with no drift guard is worse than one copy.
|
||||
|
||||
backend/localities.py is canonical because the backend image does not
|
||||
contain pipeline/. The seed exists so the warehouse can join on the same
|
||||
definitions, and this is what stops the two diverging — the same
|
||||
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
||||
"""
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
seed_path = (Path(__file__).resolve().parents[2]
|
||||
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
||||
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
||||
|
||||
seed = {
|
||||
row["locality_slug"]: (row["locality_name"],
|
||||
tuple(row["outcodes"].split("|")))
|
||||
for row in csv.DictReader(seed_path.open())
|
||||
}
|
||||
assert seed == LOCALITY_OUTCODES
|
||||
|
||||
|
||||
def test_no_curated_locality_names_a_london_borough():
|
||||
"""Boroughs are authorities and already have a page.
|
||||
|
||||
A locality defined by two or three outcodes inside a borough would be a
|
||||
partial, near-duplicate subset of that authority page — the exact
|
||||
thin-content failure the two-namespace design exists to avoid. Hackney,
|
||||
Islington, Greenwich and Ealing were all in the first draft.
|
||||
|
||||
Hardcoded rather than read from the corpus because this must fail in CI,
|
||||
where there is no database.
|
||||
"""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
boroughs = {
|
||||
"barking-and-dagenham", "barnet", "bexley", "brent", "bromley",
|
||||
"camden", "croydon", "ealing", "enfield", "greenwich", "hackney",
|
||||
"hammersmith-and-fulham", "haringey", "harrow", "havering",
|
||||
"hillingdon", "hounslow", "islington", "kensington-and-chelsea",
|
||||
"kingston-upon-thames", "lambeth", "lewisham", "merton", "newham",
|
||||
"redbridge", "richmond-upon-thames", "southwark", "sutton",
|
||||
"tower-hamlets", "waltham-forest", "wandsworth", "westminster",
|
||||
}
|
||||
named = boroughs & set(LOCALITY_OUTCODES)
|
||||
assert not named, (
|
||||
f"these are boroughs, not districts: {sorted(named)} - they already "
|
||||
"have an authority page covering every school"
|
||||
)
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Tests for the places API (spec 2026-08-21)."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
|
||||
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 50.0 + i}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_registry_lists_each_published_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
slugs = {(p["kind"], p["slug"]) for p in body["places"]}
|
||||
assert ("town", "brentwood") in slugs
|
||||
assert ("authority", "essex") in slugs
|
||||
assert ("outcode", "cm13") in slugs
|
||||
|
||||
|
||||
def test_registry_carries_a_count_per_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
town = next(p for p in body["places"] if p["slug"] == "brentwood")
|
||||
assert town["count"] == 6
|
||||
|
||||
|
||||
def test_place_detail_returns_its_schools_ranked(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
assert body["place"]["name"] == "Brentwood"
|
||||
scores = [s["rwm_expected_pct"] for s in body["schools"]]
|
||||
assert scores == sorted(scores, reverse=True)
|
||||
|
||||
|
||||
def test_place_detail_carries_the_local_average(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
# 50..55 inclusive
|
||||
assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5)
|
||||
|
||||
|
||||
def test_phase_filter_narrows_the_school_list(client):
|
||||
body = client.get("/api/places/town/brentwood?phase=secondary").json()
|
||||
assert body["schools"] == []
|
||||
|
||||
|
||||
def test_unknown_place_404s(client):
|
||||
assert client.get("/api/places/town/atlantis").status_code == 404
|
||||
|
||||
|
||||
def test_unknown_kind_404s(client):
|
||||
assert client.get("/api/places/planet/mars").status_code == 404
|
||||
@@ -59,9 +59,14 @@ def static_child(monkeypatch) -> str:
|
||||
def test_every_loc_uses_the_www_host(sitemaps):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
# Checked across every file, index included, not just one.
|
||||
#
|
||||
# A child can legitimately be empty — this fixture holds two schools and no
|
||||
# town clearing the threshold — so the presence check applies only to files
|
||||
# that carry URLs. The absence check applies to all of them.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
assert "https://schoolcompare.co.uk" not in xml, name
|
||||
if "<loc>" in xml:
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
|
||||
|
||||
def test_school_with_results_is_listed(schools_child):
|
||||
@@ -176,3 +181,90 @@ def test_children_are_chunked_under_the_limit(monkeypatch):
|
||||
def test_build_sitemap_still_returns_the_index(sitemap):
|
||||
# lifespan and the admin endpoint call build_sitemap(); keep it working.
|
||||
assert "<sitemapindex" in sitemap
|
||||
|
||||
|
||||
def test_school_with_results_in_an_earlier_year_is_still_listed(monkeypatch):
|
||||
"""Regression: The Mallard Academy (150367).
|
||||
|
||||
Real KS2 results 2015-16 to 2018-19, then null rows from 2022-23 onward
|
||||
because the school stopped reporting. The first cut tested the latest
|
||||
year's row alone and dropped it, along with ~220 others, even though its
|
||||
detail page shows all four years of results.
|
||||
"""
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
base = {"local_authority": "Testshire", "school_type": "Academy",
|
||||
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
|
||||
"attainment_8_score": np.nan, "urn": 150367,
|
||||
"school_name": "Mallard Academy"}
|
||||
df = _pd.DataFrame([
|
||||
{**base, "year": 201819, "rwm_expected_pct": 67.0},
|
||||
{**base, "year": 202324, "rwm_expected_pct": np.nan},
|
||||
{**base, "year": 202425, "rwm_expected_pct": np.nan},
|
||||
])
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "/school/150367-mallard-academy" in xml
|
||||
|
||||
|
||||
def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
|
||||
"""The fix must not turn into "list everything"."""
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
base = {"local_authority": "Testshire", "school_type": "Academy",
|
||||
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
|
||||
"attainment_8_score": np.nan, "rwm_expected_pct": np.nan,
|
||||
"urn": 100002, "school_name": "Ghost Primary"}
|
||||
df = _pd.DataFrame([{**base, "year": y} for y in (202324, 202425)])
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
|
||||
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]
|
||||
|
||||
|
||||
def _places_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA",
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 60.0}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def place_sitemaps(monkeypatch) -> dict:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _places_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return app_module.build_sitemaps()
|
||||
|
||||
|
||||
def test_place_children_are_listed_in_the_index(place_sitemaps):
|
||||
index = place_sitemaps["sitemap.xml"]
|
||||
assert "/sitemaps/places-1.xml" in index
|
||||
assert "/sitemaps/outcodes-1.xml" in index
|
||||
|
||||
|
||||
def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps):
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood</loc>" in xml
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/authority/essex</loc>" in xml
|
||||
|
||||
|
||||
def test_outcode_urls_live_in_their_own_child(place_sitemaps):
|
||||
assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"]
|
||||
assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"]
|
||||
|
||||
|
||||
def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps):
|
||||
for name in ("places-1.xml", "outcodes-1.xml"):
|
||||
assert "<priority>" not in place_sitemaps[name]
|
||||
assert "<changefreq>" not in place_sitemaps[name]
|
||||
@@ -0,0 +1,85 @@
|
||||
# Remote branch cleanup, 2026-08-20
|
||||
# Restore any branch with: git push origin <sha>:refs/heads/<name>
|
||||
|
||||
## Deleted: fully merged into main (content is in main)
|
||||
75677f4252b759ef895e7d5f7c19f8f1745bdb59 add-contact-form-footer
|
||||
fa1abff642683dfd26ba88a295a0a6710d77147a chore/byline-removal-and-audit-figure
|
||||
95081d38bdf87764ef5d298676c25fae4cd3b792 chore/remove-parent-view
|
||||
6877abedebfc1c7d95f1f6ebe68945c62be528ab chore/staged-prod-promotion
|
||||
090d5f7bec824e083d3252e2c6e636686016304d ci/frontend-checks-speedup
|
||||
8c3a5cc4e9f551f0190d85357ce7741ad87f3a4c design/cohort-identity
|
||||
955659580067ca8b81bd77e31e4e2554103f094d feat/allow-analytics-iframe-embed
|
||||
6828f6cd4417284ea3eb6f088fa20945b8b40ed3 feat/compare-chips-two-per-row
|
||||
d5cd0abfee226885119665da5d2aa8288b59217f feat/compare-data-foundation
|
||||
6dd9b04b50bee146682da87efad8fc8b526251c5 feat/compare-frontend-rebuild
|
||||
96d5fcf5b07b6f175b48e9b20fcb765a320a907f feat/detail-header-details-reveal
|
||||
f1388ff5bd0af1409823a1e047b7ba84246a0f70 feat/gias-sixth-form-flag
|
||||
eddf74745f86c9c6d9eb07d867246ff9cc90dc20 feat/hero-artwork-v2
|
||||
3015c37bac6dc28db58c80fdb9235942c25b83d7 feat/hero-byline
|
||||
8e763e39d17a4964cf558e51c03f044186371f6d feat/info-popover-tooltip
|
||||
88c653215d520ab6e902c9de55bf27d86eeb90c3 feat/last-distance-offered
|
||||
a72323874f7aebdb2e64b6d64a5febd61152d09d feat/last-distance-offered-full
|
||||
c9a1892bfb0370e0672e5849cba294ddfabe3c65 feat/latest-cutoff-only
|
||||
4e8df006d75d8be2a1d8529ddf855c445854cba1 feat/near-me-by-search
|
||||
45ab479062c6a1639facad636fc0cc0cf0fd9155 feat/proposed-to-close-schools
|
||||
3bf2e8f262cbe058fda6de6f8ea3e050224a51b9 feat/school-detail-visualisations
|
||||
1f80571b1ff217dc92b660a936c02b5f49d07f0b feat/umami-heatmap-recorder
|
||||
609bb923d96aa5730131463ea35d5efdd404bf96 feature/ingest-independent-schools
|
||||
94151c58ea38a9256d66d15161293486a500c7a2 fix/admissions-section-height
|
||||
79246edc22961c2beb3520437e9d064d3b809d10 fix/annual-dag-ks4-national-selector
|
||||
59ac9c10b97e0ef1143f57fea06b324e72ac3d4a fix/chart-marker-contrast
|
||||
9f8dba227c95706ca3527bd48d381e7622cc0a5e fix/compare-chart-refetch-resilience
|
||||
e74d3882ce78a141fa1a57daa3102d7a58852dc3 fix/compare-expert-fixes
|
||||
80176cac4db4820e76ea2a156c7a2974bee2f203 fix/compare-final-review-mustfix
|
||||
f579630fab6c456e26a6984a3e8eebdbe3184202 fix/compare-mockup-drift
|
||||
dc85254ad2ddf134b4434065d4762cef374d2220 fix/compare-null-year-blanks-chart
|
||||
43a2c4a6bc539b621f31655aec05ef319a25f343 fix/compare-refresh-and-fetch
|
||||
d677b5453365b72c81d6df2de62b1fa0d05d684d fix/daily-dag-cache-invalidation
|
||||
f6bb037c471553e8195b5a8b147467ce0d07a688 fix/detail-all-through
|
||||
17bd4d5a5eb0b12ca79b97db587f14f7d671e85b fix/detail-chart-truthfulness
|
||||
4e6be0ce65647410b4ff74f8b763207c28920c26 fix/detail-inclusion-admissions
|
||||
fdda52ff0af3fae03a4b059a655973cdbb269f91 fix/detail-ofsted-correctness
|
||||
e36125b24aba254a8d15c5c33b4d2a296e691995 fix/detail-provenance-anchoring
|
||||
b31e71ac884df6507f567adc046c9fc52d9d310a fix/detail-report-card-render-date
|
||||
32f8a02862be6a1d49f4c3928b17fcf15d4c94cc fix/detail-trend-chart-taller
|
||||
e65688d600a86818fe21ae4c61ba27e5b6ec8d7c fix/e2e-brand-assertions
|
||||
3adea73ee04cdedfab54b0351878b297f72756ad fix/e2e-compare-chips-phase
|
||||
06e4898c30feaedc471f97aba28ddb0d61379f4d fix/e2e-compare-samephase
|
||||
9abd020967670a855e80fe5a908c8048a3aa9f14 fix/e2e-distance-locator
|
||||
acec8135e1ec7c9c3c255e5b23733a0ef862b590 fix/e2e-rankings-year-pick
|
||||
5944d88f0b1517ef1ef56af1b62270d2cf28e717 fix/expert-signoff-mustfixes
|
||||
2433101fa08be5df6f170d41790512ffe823d33e fix/font-cascade-and-map-palette
|
||||
74ca76d150deec6725259d9637ea86d7bb90c683 fix/gias-legacy-fallback
|
||||
bdaa05cd542f563ef74c45307cd8f7fc465193c9 fix/hero-fallback-and-sharp
|
||||
d52d384cf23d282b44e9251176f8f3402d808600 fix/hero-map-ios-fullscreen
|
||||
4043270a77bbe4edb18207fa5f1d94d4747fe12f fix/hero-mobile-and-wording
|
||||
22e9eb2d48b0d6623e88fd67cb6ca8e5e4074583 fix/homepage-education-accuracy
|
||||
8d50afef1e8a2b621b7344eadf475b0d609ad7a2 fix/leaflet-specificity-and-font-assertion
|
||||
b2b2cad5acf534ae7a667d3fb2be15efff37c4a7 fix/list-map-report-card-signal
|
||||
dc21e80a5e9eebd13aaab84735642eb77cef35e6 fix/mobile-cell-name-size
|
||||
e5f7f4c959f024c073472122d858333a0f24866c fix/mobile-compare-polish
|
||||
a00cbe916182d1e04661c750a1ce2e7ad27a68ae fix/mobile-sort-select-overflow
|
||||
2fd997bfe640c419a6713e85df008463de7f56e8 fix/modal-keyboard-viewport
|
||||
3e7705756776a0c44d27966dfc972023f1f38b69 fix/ofsted-link-text
|
||||
ce422e64363e2b03c186ef316832b6de15502d67 fix/promote-status-token
|
||||
4522cbf64560db1e1cd519e119aa466b42cd16a1 fix/proposed-to-close-copy
|
||||
15da060e4af37fbae919e0edf25f108266de2585 fix/rankings-admissions-accuracy
|
||||
6c872ce726f210433354ca38dc5314bc6a467534 fix/rankings-year-validation
|
||||
1c1df7796194af3d47f8e5ac0a0fbe6f323700e7 fix/report-card-chip-alignment
|
||||
b2dc4d0779ced02709429afaa5985acf4abda794 fix/results-map-ios-fullscreen
|
||||
95a5783da1fc994df76cb97238b55596dce4cd8f fix/runtime-api-proxy
|
||||
8a9ba30cc24e29653a72c03c0e817684b7db7c07 fix/sats-per-level-national
|
||||
536832a524fc4f9ed858f047e00b9a4d9d429c46 fix/school-detail-nan-500
|
||||
fef83b3bf244a9bf3cb4afa75dbc433d8725014f fix/schoolbar-sticky-offset
|
||||
b0c5b6bb57c879477da12c37b15cff950c29ebd3 fix/secondary-anchors-button-affordance
|
||||
261403bcd2b01aa4f26ee212e26302fc0f769bf9 fix/special-note-full-width
|
||||
ea5249a2ea6faf6bfa5a1522644387ba59770ec8 fix/standardise-distance-units
|
||||
e4565e9f158721d4df6b918f2b065de82851f8d9 fix/trends-chart-height
|
||||
3aad5101a842539105022f9059e85c224fc5973a fix/welsh-establishment-leak
|
||||
315f1feede70bdf3101d2fdd305b3d06e037fdac perf/batch-supplementary
|
||||
d2dc78aeb599e16b7ef5019be2b360b08df463bc perf/compare-loading
|
||||
e098ad4bd1130152705788d4773837b7d1e7112e perf/server-client-split
|
||||
|
||||
## Deleted: superseded by PR #110 (content preserved on feat/seo-crawl-hygiene-main)
|
||||
786ec80dd4de4a3cb674a89e35b3b5e461639289 feat/seo-crawl-hygiene
|
||||
a5ac0bcd1b37bc10dcbce88f8601d01bf7b3eaaf feat/england-only-corpus
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,278 @@
|
||||
# W2: The Location Layer — Design
|
||||
|
||||
Date: 2026-08-21
|
||||
Status: awaiting review
|
||||
Supersedes: workstream W2 in `2026-08-20-seo-programme-design.md`
|
||||
|
||||
Scope note: this covers four page families in one spec. Splitting them — towns
|
||||
and authorities first, outcodes and localities after — was proposed and
|
||||
declined in favour of building the layer in one pass. The decomposition
|
||||
argument was that the curated locality seed needs human review and would hold
|
||||
up 783 pages of measured demand behind it; that risk is accepted here, and the
|
||||
implementation plan should sequence the seed early enough that review time
|
||||
does not become the critical path.
|
||||
|
||||
## Problem
|
||||
|
||||
Location intent is the largest unserved demand the site has. In the 16-month
|
||||
Search Console baseline it draws **874 impressions, one click, average
|
||||
position 49.5**. The site does not compete.
|
||||
|
||||
Unlike named-school queries — which the same baseline showed to be
|
||||
navigational and unwinnable, since a parent typing "audley junior school"
|
||||
wants that school's own website — location queries have no incumbent owner.
|
||||
Nobody owns "primary schools in Brentwood" the way a school owns its name.
|
||||
|
||||
The cause is structural: the site has no page about a place. Every competitor
|
||||
ranking above it does.
|
||||
|
||||
## What the demand actually looks like
|
||||
|
||||
Every location query in the baseline is **town or district level**. Not one is
|
||||
an administrative area:
|
||||
|
||||
| Query | Impressions | Position |
|
||||
|-------|-------------|----------|
|
||||
| colleges in solihull | 112 | 51.2 |
|
||||
| schools in ramsey | 64 | 42.5 |
|
||||
| schools in crosby | 57 | 47.7 |
|
||||
| primary schools in beccles | 44 | 40.9 |
|
||||
| private schools in battersea | 41 | 71.9 |
|
||||
| secondary schools in brentwood | 37 | 56.1 |
|
||||
| secondary schools in canary wharf | 30 | 35.9 |
|
||||
|
||||
Three patterns follow directly, and they drive the whole design.
|
||||
|
||||
**Towns, not authorities.** The superseded W2 put `/schools/[la]` first and
|
||||
towns second. The data inverts that. Brentwood appears four times in different
|
||||
phrasings; Beccles twice. Both are towns, not authorities.
|
||||
|
||||
**Phase is part of the query**, not a filter applied afterwards: "primary
|
||||
schools in beccles", "secondary schools in brentwood", "colleges in solihull".
|
||||
|
||||
**London is searched by district** — Battersea, Canary Wharf — and the GIAS
|
||||
`town` field cannot serve it at all.
|
||||
|
||||
## Measured sizing
|
||||
|
||||
Counted against the live corpus of 25,185 schools, not estimated.
|
||||
|
||||
| Family | Viable (≥5 schools) | Below threshold |
|
||||
|--------|--------------------|-----------------|
|
||||
| Towns | **783** | 907 → redirect to authority |
|
||||
| Outcodes | **1,760** | 305 |
|
||||
| Local authorities | 154 | — |
|
||||
| London localities | ~100–150 (curated) | — |
|
||||
|
||||
With phase variants — 783 town pages plus roughly 700 primary and 250
|
||||
secondary variants, 154 authorities across three variants, 1,760 outcodes and
|
||||
the curated localities — the total lands near **4,000 pages**. Phase variants
|
||||
need their own threshold: there are 17,426 primaries but only 4,456 secondaries nationally,
|
||||
so most towns will support a primary page and not a secondary one.
|
||||
|
||||
## Two design problems this spec exists to solve
|
||||
|
||||
### 1. Town and authority names collide, and neither contains the other
|
||||
|
||||
67 viable towns share a name with a local authority. The obvious fix — let the
|
||||
authority absorb the town, since it sounds like a superset — **does not work**:
|
||||
|
||||
| Place | Schools in the town | Schools in the authority |
|
||||
|-------|--------------------|-----------------------|
|
||||
| Bedford | 104 | 86 |
|
||||
| Birmingham | 520 | 518 |
|
||||
| Derby | 157 | 119 |
|
||||
| Doncaster | 152 | 145 |
|
||||
|
||||
The authority is the larger set in only 43 of the 67. Postal towns cross
|
||||
authority boundaries, so these are overlapping sets that happen to share a
|
||||
name. Publishing both into one namespace produces near-duplicate pages, which
|
||||
is the specific failure that sinks programmatic SEO.
|
||||
|
||||
**Resolution: two namespaces.**
|
||||
|
||||
```
|
||||
/schools/[place] towns and London localities
|
||||
/schools/[place]/primary
|
||||
/schools/[place]/secondary
|
||||
/schools/authority/[la] local authorities
|
||||
/schools/authority/[la]/primary
|
||||
/schools/authority/[la]/secondary
|
||||
/schools/near/[outcode]
|
||||
```
|
||||
|
||||
Outcodes carry no phase variants: nobody searches "primary schools in SW11",
|
||||
so the variants would be pages without demand.
|
||||
|
||||
Every collision disappears by construction. `/schools/[place]` keeps the clean
|
||||
URL for the pattern that carries the demand; authorities get a namespace whose
|
||||
purpose is genuinely different — admissions are authority-run, and the
|
||||
authority page is the one that can speak to catchment policy and LA averages.
|
||||
|
||||
A place page and an authority page of the same name must each say plainly
|
||||
which set of schools they cover, or they read as duplicates to a reader even
|
||||
when they differ in fact.
|
||||
|
||||
### 2. London has no locality field
|
||||
|
||||
`town` collapses **1,819 London schools into the single value "London"**. A
|
||||
page listing all of them is useless, and borough pages do not help because
|
||||
people search "Battersea", not "Wandsworth".
|
||||
|
||||
No single field solves it:
|
||||
|
||||
| Search term | `parliamentary_constituency` | postcodes.io `admin_ward` |
|
||||
|-------------|------------------------------|---------------------------|
|
||||
| Battersea | **Battersea** ✓ | Northcote / Wandsworth Town ✗ |
|
||||
| Canary Wharf | Poplar and Limehouse ✗ | **Canary Wharf** ✓ |
|
||||
| Vauxhall | Vauxhall and Camberwell Green ✗ | **Vauxhall** ✓ |
|
||||
|
||||
And neither covers Clapham, Shoreditch or Peckham, which are postal and
|
||||
colloquial rather than administrative.
|
||||
|
||||
**Resolution: a curated seed mapping locality to outcodes.**
|
||||
|
||||
```
|
||||
pipeline/transform/seeds/locality_outcodes.csv
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,"SW11|SW8",London
|
||||
canary-wharf,Canary Wharf,"E14",London
|
||||
clapham,Clapham,"SW4|SW9",London
|
||||
```
|
||||
|
||||
This needs **no new ingestion** — the corpus already has postcodes. It puts
|
||||
the fuzzy, contested part of the problem in a reviewable file rather than in
|
||||
derived logic, which suits it: locality boundaries are a judgement, not a
|
||||
fact. The repo already uses dbt seeds for curated reference data
|
||||
(`la_code_names.csv`, `gias_code_names.csv`), so this follows an established
|
||||
pattern.
|
||||
|
||||
The seed generalises past London. Any colloquial place — Jesmond, Chorlton,
|
||||
Clifton — can be defined by its outcodes without a schema change.
|
||||
|
||||
**Constraint:** a locality slug may not collide with a viable town slug. The
|
||||
place registry enforces this and fails the build rather than silently
|
||||
shadowing a town.
|
||||
|
||||
## Architecture
|
||||
|
||||
### The place registry
|
||||
|
||||
One module owns the question "what places do we publish, and what is in each".
|
||||
Everything else reads from it: the pages, the sitemap, the internal links.
|
||||
|
||||
```
|
||||
backend/places.py
|
||||
|
||||
Place = { kind: "town"|"locality"|"authority"|"outcode",
|
||||
slug, name, urn_list, parent_authority | None }
|
||||
|
||||
build_place_registry(df) -> dict[str, Place]
|
||||
place_schools(slug, phase=None) -> list[School]
|
||||
```
|
||||
|
||||
Built once at startup from the same DataFrame the sitemap uses, and rebuilt by
|
||||
the existing `/api/admin/regenerate-sitemap` path after a pipeline run.
|
||||
Registry construction is where the threshold, the collision rules and the
|
||||
seed's uniqueness constraint are enforced — in one place, testable without a
|
||||
browser or a database.
|
||||
|
||||
### API
|
||||
|
||||
```
|
||||
GET /api/places the registry: slug, kind, name, count
|
||||
GET /api/places/{slug}?phase= aggregate + ranked schools for one place
|
||||
```
|
||||
|
||||
`/api/places` is what the sitemap and the internal-link modules enumerate.
|
||||
|
||||
### Routes
|
||||
|
||||
Next App Router, ISR with the same 7-day revalidate the school pages use.
|
||||
`generateStaticParams` gated behind an env flag, matching
|
||||
`PRERENDER_SCHOOLS`, because 3,900 more routes cannot be statically built in
|
||||
CI on every deploy.
|
||||
|
||||
## What each page must contain
|
||||
|
||||
A place page that is a name substituted into a template is the thing Google's
|
||||
helpful-content stance exists to demote. Each page carries computed local
|
||||
facts that exist nowhere else on the site:
|
||||
|
||||
- **H1** matching the query: "Primary schools in Brentwood"
|
||||
- **Counts framed usefully**: "29 schools, 4 rated Outstanding"
|
||||
- **A ranked table** of the top 20 on the phase's headline metric —
|
||||
`rwm_expected_pct` for primary, `attainment_8_score` for secondary, and for
|
||||
an unphased place page the metric matching whichever phase it holds more of
|
||||
- **The local average against the England average** — the one number a parent
|
||||
cannot get from a list
|
||||
- **Ofsted grade distribution** for the place
|
||||
- **A map**
|
||||
- **Links to neighbouring places** and to the parent authority
|
||||
- **An FAQ block**, feeding `FAQPage` structured data
|
||||
- **A link to every school page in scope** — this is what finally de-orphans
|
||||
the 23,000 school pages the original spec identified as near-orphans
|
||||
|
||||
## Thin-page controls
|
||||
|
||||
Three, and they are the difference between a location layer and index bloat:
|
||||
|
||||
1. **Five schools with current data minimum.** Below it, 301 to the parent
|
||||
authority. This drops 907 towns and 305 outcodes.
|
||||
2. **Per-phase thresholds.** A town with 30 primaries and 2 secondaries
|
||||
publishes a primary page and no secondary page.
|
||||
3. **No page without a local average.** If a place has too few schools with
|
||||
results to compute one, it has nothing to say that a list does not, and it
|
||||
falls back to the authority.
|
||||
|
||||
## Sitemap
|
||||
|
||||
Two new children in the existing index: `/sitemaps/places-{n}.xml` and
|
||||
`/sitemaps/outcodes-{n}.xml`. Per-family children are why the index was built
|
||||
in W1 — Search Console reports coverage per submitted sitemap, so indexation
|
||||
of the location layer is measurable separately from the school pages.
|
||||
|
||||
## Testing
|
||||
|
||||
Per `CLAUDE.md`, user-facing behaviour extends `e2e/tests/journeys.spec.ts` in
|
||||
the same PR.
|
||||
|
||||
**Unit (registry, no DB):** threshold enforcement; a sub-threshold town
|
||||
resolves to its authority; a locality slug colliding with a town fails the
|
||||
build; Bedford's town and authority pages hold different URN sets; per-phase
|
||||
thresholds.
|
||||
|
||||
**Backend:** `/api/places` shape; `/api/places/{slug}` aggregate correctness
|
||||
against a fixture; unknown slug 404s.
|
||||
|
||||
**e2e:** a known town, authority, locality and outcode page each render with
|
||||
the expected count; a below-threshold town 301s; every place page declares a
|
||||
canonical and appears in the sitemap; `/schools/bedford` and
|
||||
`/schools/authority/bedford` both resolve and state which set they cover.
|
||||
|
||||
## Risks
|
||||
|
||||
**Index bloat** is the failure mode of every programmatic SEO programme. The
|
||||
three controls above are the answer, and the per-family sitemap is how we find
|
||||
out early if they were not enough.
|
||||
|
||||
**Helpful-content exposure.** Templated location pages are exactly what
|
||||
Google's stance targets. The mitigation is that every page carries real
|
||||
computed local data — counts, distributions, local-versus-national comparison
|
||||
— rather than a name dropped into boilerplate. If indexation of the places
|
||||
sitemap stalls below roughly half, that is the signal to stop and rethink
|
||||
rather than to add more pages.
|
||||
|
||||
**Build cost.** ~4,000 additional ISR routes on top of 23,000 school pages.
|
||||
The env-flag gate on `generateStaticParams` keeps CI viable.
|
||||
|
||||
**Curation drift.** The locality seed is hand-maintained and will go stale as
|
||||
places change. It is small and reviewable, and a dbt test asserts every seed
|
||||
outcode matches at least one school so a typo fails the pipeline rather than
|
||||
publishing an empty page.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Catchment-area estimation. It is a strong driver for this cluster and
|
||||
`fact_admissions` carries the distances, but it is a modelling problem with
|
||||
real accuracy risk and deserves its own design.
|
||||
@@ -1695,3 +1695,150 @@ test('a bare /compare is indexable, a parameterised one is not', async ({ page }
|
||||
.getAttribute('href');
|
||||
expect(canonical).toBe('https://www.schoolcompare.co.uk/compare');
|
||||
});
|
||||
|
||||
/*
|
||||
* Staging must not be indexable (spec 2026-08-20, W1 hygiene).
|
||||
*
|
||||
* These journeys only ever run against staging — deploy.yml passes
|
||||
* STAGING_BASE_URL, and promote.yml only smoke-polls production without
|
||||
* Playwright — so asserting the noindex header here is safe.
|
||||
*/
|
||||
test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => {
|
||||
const res = await page.request.get('/');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
|
||||
const tag = res.headers()['x-robots-tag'];
|
||||
expect(tag, 'staging must send X-Robots-Tag').toBeTruthy();
|
||||
expect(tag).toContain('noindex');
|
||||
|
||||
// The other half, and the reason this is one test rather than two: a
|
||||
// Disallow would stop Google fetching the page at all, so it would never
|
||||
// see the noindex above. The two only work together.
|
||||
const robots = await (await page.request.get('/robots.txt')).text();
|
||||
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
|
||||
});
|
||||
|
||||
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
|
||||
const list = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
const [first] = (await list.json()).schools ?? [];
|
||||
expect(first, 'no school available').toBeTruthy();
|
||||
|
||||
const res = await page.request.get(`/school/${first.urn}-x`);
|
||||
expect(res.headers()['x-robots-tag']).toContain('noindex');
|
||||
});
|
||||
|
||||
/*
|
||||
* W8 — the C1 pages must ship a description, and it must differentiate.
|
||||
*
|
||||
* Baseline was 0.43% CTR at position 6.1 on "compare school performance",
|
||||
* against 9.16% for the brand query from the same neighbourhood. The SERP is
|
||||
* owned by the DfE's own service, so a description that paraphrases it earns
|
||||
* nothing. Google may rewrite a snippet, but it cannot use one we never sent.
|
||||
*/
|
||||
test('every C1 page ships a description, and none opens its title with the brand', async ({ page }) => {
|
||||
for (const path of ['/', '/compare', '/rankings', '/admissions']) {
|
||||
await page.goto(path);
|
||||
|
||||
const desc = await page.locator('meta[name="description"]').first()
|
||||
.getAttribute('content');
|
||||
expect(desc, `${path} must ship a description`).toBeTruthy();
|
||||
expect(desc!.length, `${path} description too short to be worth reading`)
|
||||
.toBeGreaterThan(100);
|
||||
|
||||
const title = await page.title();
|
||||
expect(title.toLowerCase().startsWith('schoolcompare'),
|
||||
`${path} spends its most valuable pixels on the brand`).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
test('the homepage snippet names what gov.uk does not publish', async ({ page }) => {
|
||||
await page.goto('/');
|
||||
const desc = await page.locator('meta[name="description"]').first()
|
||||
.getAttribute('content');
|
||||
// Admissions distance is the one fact the DfE service has no equivalent for.
|
||||
expect(desc).toMatch(/close you had to live|distance/i);
|
||||
});
|
||||
|
||||
/*
|
||||
* The location layer (spec 2026-08-21, W2).
|
||||
*
|
||||
* Location intent sat at position 49.5 with one click across the whole 16-month
|
||||
* baseline — the site published no page about a place. These assert the four
|
||||
* families render, stay in their own namespaces, and reach the sitemap.
|
||||
*/
|
||||
async function firstPlaceOfKind(page: Page, kind: string) {
|
||||
const res = await page.request.get('/api/places');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const { places } = await res.json();
|
||||
const hit = places.find((p: { kind: string }) => p.kind === kind);
|
||||
expect(hit, `no ${kind} in the registry`).toBeTruthy();
|
||||
return hit as { kind: string; slug: string; name: string; count: number };
|
||||
}
|
||||
|
||||
for (const [kind, prefix, article] of [
|
||||
['town', '/schools/', 'a'],
|
||||
['authority', '/schools/authority/', 'an'],
|
||||
['outcode', '/schools/near/', 'an'],
|
||||
] as const) {
|
||||
test(`${article} ${kind} page renders with its school count`, async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, kind);
|
||||
await page.goto(`${prefix}${place.slug}`);
|
||||
await expect(page.locator('h1')).toContainText(place.name, { ignoreCase: true });
|
||||
await expect(page.locator('a[href^="/school/"]').first()).toBeVisible();
|
||||
});
|
||||
}
|
||||
|
||||
test('a town and an authority sharing a name are different pages', async ({ page }) => {
|
||||
// 67 real collisions, and the authority is the larger set in only 43 — so
|
||||
// one namespace would have published near-duplicates.
|
||||
const { places } = await (await page.request.get('/api/places')).json();
|
||||
const townSlugs = new Set(
|
||||
places.filter((p: { kind: string }) => p.kind === 'town')
|
||||
.map((p: { slug: string }) => p.slug));
|
||||
const clash = places.find((p: { kind: string; slug: string }) =>
|
||||
p.kind === 'authority' && townSlugs.has(p.slug));
|
||||
test.skip(!clash, 'no town/authority name collision in this environment');
|
||||
|
||||
const townRes = await page.request.get(`/api/places/town/${clash.slug}`);
|
||||
const laRes = await page.request.get(`/api/places/authority/${clash.slug}`);
|
||||
expect(townRes.ok() && laRes.ok()).toBeTruthy();
|
||||
const townUrns = (await townRes.json()).schools.map((s: { urn: number }) => s.urn).sort();
|
||||
const laUrns = (await laRes.json()).schools.map((s: { urn: number }) => s.urn).sort();
|
||||
expect(townUrns).not.toEqual(laUrns);
|
||||
});
|
||||
|
||||
test('a place below the threshold has no page', async ({ page }) => {
|
||||
// Crosby holds one school; publishing it would be a page with nothing to say.
|
||||
const res = await page.request.get('/api/places/town/crosby');
|
||||
expect(res.status()).toBe(404);
|
||||
});
|
||||
|
||||
test('place pages declare a canonical and reach the sitemap', async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
const canonical = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(canonical).toBe(`https://www.schoolcompare.co.uk/schools/${place.slug}`);
|
||||
|
||||
const xml = await (await page.request.get('/sitemaps/places-1.xml')).text();
|
||||
expect(xml).toContain(`/schools/${place.slug}`);
|
||||
});
|
||||
|
||||
test('the place page ships ItemList structured data that parses', async ({ page }) => {
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
const raw = await page.locator('script[type="application/ld+json"]').first()
|
||||
.textContent();
|
||||
const parsed = JSON.parse(raw!);
|
||||
const types = (parsed['@graph'] ?? []).map((n: { '@type': string }) => n['@type']);
|
||||
expect(types).toContain('ItemList');
|
||||
expect(types).toContain('BreadcrumbList');
|
||||
});
|
||||
|
||||
test('a place page states the local average against England', async ({ page }) => {
|
||||
// The one number a list cannot give, and the reason these pages are not
|
||||
// a name dropped into a template.
|
||||
const place = await firstPlaceOfKind(page, 'town');
|
||||
await page.goto(`/schools/${place.slug}`);
|
||||
await expect(page.getByTestId('local-vs-england')).toContainText(/across England/i);
|
||||
});
|
||||
@@ -51,3 +51,80 @@ describe('/compare indexability', () => {
|
||||
.toBe('https://www.schoolcompare.co.uk/compare');
|
||||
});
|
||||
});
|
||||
|
||||
/*
|
||||
* W8 — snippet copy for the C1 cluster.
|
||||
*
|
||||
* The baseline (GSC, 16 months to 2026-08-20) showed these pages ranking on
|
||||
* page one and converting at a tenth of the normal rate: "compare school
|
||||
* performance" at position 6.1 with 0.43% CTR, against 9.16% for the brand
|
||||
* query from the same neighbourhood. The SERP is dominated by the DfE's own
|
||||
* "Compare school performance" service, so the job of this copy is to say
|
||||
* what that service does not offer, without losing intent match on the title.
|
||||
*
|
||||
* These tests guard the mechanics that make a snippet work — length, intent
|
||||
* keyword, differentiator, no brand-first — not the exact wording, which
|
||||
* should stay free to iterate.
|
||||
*/
|
||||
|
||||
// Google truncates titles near 60 characters and descriptions near 155.
|
||||
const TITLE_MAX = 60;
|
||||
const DESC_MIN = 110;
|
||||
const DESC_MAX = 155;
|
||||
|
||||
type Meta = { title?: unknown; description?: unknown };
|
||||
const titleOf = (m: Meta): string => {
|
||||
const t = m.title as string | { absolute?: string } | undefined;
|
||||
return typeof t === 'string' ? t : (t?.absolute ?? '');
|
||||
};
|
||||
|
||||
describe('C1 snippet copy', () => {
|
||||
const pages: Array<[string, Meta, RegExp]> = [
|
||||
['home', homeMetadata as Meta, /compare schools/i],
|
||||
['rankings', rankingsMetadata as Meta, /league table/i],
|
||||
['admissions', admissionsMetadata as Meta, /admission/i],
|
||||
];
|
||||
|
||||
for (const [name, meta, intent] of pages) {
|
||||
it(`${name}: title carries the search intent and fits the SERP`, () => {
|
||||
const t = titleOf(meta);
|
||||
expect(t).toMatch(intent);
|
||||
expect(t.length).toBeLessThanOrEqual(TITLE_MAX);
|
||||
});
|
||||
|
||||
it(`${name}: title does not open with the brand`, () => {
|
||||
// The measured 0.43% CTR came from a brand-first title. The most
|
||||
// valuable pixels go to the thing the searcher typed.
|
||||
expect(titleOf(meta).toLowerCase().startsWith('schoolcompare')).toBe(false);
|
||||
});
|
||||
|
||||
it(`${name}: description is long enough to be worth reading, short enough to survive`, () => {
|
||||
const d = meta.description as string;
|
||||
expect(d.length).toBeGreaterThanOrEqual(DESC_MIN);
|
||||
expect(d.length).toBeLessThanOrEqual(DESC_MAX);
|
||||
});
|
||||
}
|
||||
|
||||
it('the homepage description names what gov.uk does not publish', () => {
|
||||
// Admissions distance is the one fact the DfE service has no equivalent
|
||||
// for. If it ever leaves this description, the snippet is competing with
|
||||
// gov.uk on gov.uk's own ground.
|
||||
expect(homeMetadata.description).toMatch(/close you had to live|distance/i);
|
||||
});
|
||||
|
||||
it('/compare targets the tool phrasing rather than repeating the homepage', () => {
|
||||
// Two pages chasing one phrase is how a site competes with itself.
|
||||
return compareMetadata({ searchParams: Promise.resolve({}) }).then((m) => {
|
||||
expect(m.title).toMatch(/comparison tool/i);
|
||||
expect(m.title).not.toBe(titleOf(homeMetadata as Meta));
|
||||
});
|
||||
});
|
||||
|
||||
it('no C1 page claims a school count that will drift', () => {
|
||||
// The corpus moves with every data refresh; this repo has already shipped
|
||||
// one copy bug of that kind ("three schools" against MAX_SCHOOLS = 5).
|
||||
for (const [, meta] of pages) {
|
||||
expect(meta.description as string).not.toMatch(/\b\d{2},\d{3}\b|\b\d{2},000\b/);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,30 @@
|
||||
import { generateMetadata as placeMeta } from '@/app/schools/[place]/page';
|
||||
|
||||
jest.mock('@/lib/places', () => ({
|
||||
...jest.requireActual('@/lib/places'),
|
||||
fetchPlace: jest.fn(async (kind: string, slug: string) =>
|
||||
slug === 'atlantis' ? null : ({
|
||||
place: { kind, slug, name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex' },
|
||||
schools: [], averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
})),
|
||||
fetchPlaces: jest.fn(async () => []),
|
||||
}));
|
||||
|
||||
describe('place page metadata', () => {
|
||||
it('titles the page the way the place is searched', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect(m.title).toMatch(/schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('canonicalises to its own path on the www host', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect(m.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/schools/brentwood');
|
||||
});
|
||||
|
||||
it('an unknown place gets a not-found title rather than inventing one', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'atlantis' }) });
|
||||
expect(m.title).toMatch(/not found/i);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,93 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import type { PlaceDetail } from '@/lib/places';
|
||||
|
||||
const detail: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex' },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', rwm_expected_pct: 82,
|
||||
ofsted_grade: 1 } as never,
|
||||
{ urn: 2, school_name: 'Beta Primary', rwm_expected_pct: 44,
|
||||
ofsted_grade: 3 } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
describe('PlaceView', () => {
|
||||
it('leads with an H1 that matches how the place is searched', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('heading', { level: 1 }))
|
||||
.toHaveTextContent(/primary schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('states the count so the page says something before the table', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText(/29 schools/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('compares the local average against England, which a list cannot', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('63');
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('61');
|
||||
});
|
||||
|
||||
it('links every school in scope, which is what de-orphans them', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getAllByRole('link', { name: /Primary$/ })).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('links to the parent authority so the place sits in a hierarchy', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /Essex/i }))
|
||||
.toHaveAttribute('href', '/schools/authority/essex');
|
||||
});
|
||||
|
||||
it('shows the Ofsted distribution, not just a count of Outstanding', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('ofsted-distribution')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('links to neighbouring places so the page is not a dead end', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[{ kind: 'town', slug: 'romford', name: 'Romford', count: 40 }]} />);
|
||||
expect(screen.getByRole('link', { name: /Romford/ }))
|
||||
.toHaveAttribute('href', '/schools/romford');
|
||||
});
|
||||
|
||||
it('says nothing about an average it does not have', () => {
|
||||
render(<PlaceView detail={{ ...detail, averages:
|
||||
{ rwm_expected_pct: null, attainment_8_score: null } }}
|
||||
phase="primary" englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByTestId('local-vs-england')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView structured data', () => {
|
||||
function jsonLd() {
|
||||
const { container } = render(<PlaceView detail={detail} phase="primary"
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const el = container.querySelector('script[type="application/ld+json"]');
|
||||
return JSON.parse(el!.textContent!);
|
||||
}
|
||||
|
||||
it('declares the page as a ranked list, not prose', () => {
|
||||
const types = jsonLd()['@graph'].map((n: { '@type': string }) => n['@type']);
|
||||
expect(types).toContain('ItemList');
|
||||
expect(types).toContain('BreadcrumbList');
|
||||
});
|
||||
|
||||
it('gives every listed school an absolute URL on the canonical host', () => {
|
||||
const list = jsonLd()['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
|
||||
expect(list.itemListElement).toHaveLength(2);
|
||||
for (const item of list.itemListElement) {
|
||||
expect(item.url).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\//);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -5,9 +5,11 @@ import { AdmissionsView } from '@/components/AdmissionsView';
|
||||
export const dynamic = 'force-static';
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: 'School Admissions Guide',
|
||||
// Deadlines and offer days are what gets searched, and what this page is
|
||||
// genuinely best at — the countdowns are live.
|
||||
title: { absolute: 'School Admissions Deadlines & Offer Days | schoolcompare' },
|
||||
description:
|
||||
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
|
||||
'Every key date for primary and secondary school admissions in England, with live countdowns to the application deadline and National Offer Day.',
|
||||
alternates: { canonical: absoluteUrl('/admissions') },
|
||||
};
|
||||
|
||||
|
||||
@@ -30,9 +30,12 @@ export async function generateMetadata(
|
||||
const { urns } = await searchParams;
|
||||
|
||||
const base: Metadata = {
|
||||
title: 'Compare Schools',
|
||||
// Deliberately not the homepage's phrase. Two pages chasing "compare
|
||||
// schools" is how a site competes with itself; this one takes the tool
|
||||
// phrasing instead.
|
||||
title: 'School Comparison Tool — Up to Five at Once | schoolcompare',
|
||||
description:
|
||||
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
|
||||
'Put up to five English schools in one table: SATs and GCSE results against the England average, Ofsted grades, and the distance places were offered.',
|
||||
keywords:
|
||||
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
|
||||
alternates: { canonical: absoluteUrl('/compare') },
|
||||
|
||||
@@ -48,10 +48,11 @@ export const metadata: Metadata = {
|
||||
statusBarStyle: 'default',
|
||||
},
|
||||
title: {
|
||||
default: 'schoolcompare | Compare School Performance',
|
||||
default: 'Compare Schools Side by Side | schoolcompare',
|
||||
template: '%s | schoolcompare',
|
||||
},
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place. Free, no sign-up.',
|
||||
keywords: 'school comparison, KS2 results, KS4 results, primary school, secondary school, England schools, SATs results, GCSE results',
|
||||
authors: [{ name: 'schoolcompare' }],
|
||||
manifest: '/manifest.json',
|
||||
@@ -61,16 +62,18 @@ export const metadata: Metadata = {
|
||||
metadataBase: new URL(SITE_URL),
|
||||
openGraph: {
|
||||
type: 'website',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
title: 'Compare Schools Side by Side | schoolcompare',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
|
||||
url: SITE_URL,
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
// summary_large_image now that there is an image worth showing.
|
||||
card: 'summary_large_image',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
title: 'Compare Schools Side by Side | schoolcompare',
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
|
||||
},
|
||||
};
|
||||
|
||||
|
||||
+15
-2
@@ -34,8 +34,21 @@ interface HomePageProps {
|
||||
* saying the brand twice.
|
||||
*/
|
||||
export const metadata: Metadata = {
|
||||
title: { absolute: 'schoolcompare | Compare every school in England' },
|
||||
description: 'Search and compare school performance across England',
|
||||
/*
|
||||
* Intent in the title, differentiator in the description.
|
||||
*
|
||||
* These queries are owned by the DfE's own "Compare school performance"
|
||||
* service, and the old title — brand first, then a near-paraphrase of that
|
||||
* service's name — gave a searcher no reason to pick us over it. It drew
|
||||
* 0.43% CTR at position 6.1 while the brand query drew 9.16% from the same
|
||||
* neighbourhood, so the ranking was never the problem.
|
||||
*
|
||||
* The title now matches what people type. The description carries the one
|
||||
* fact gov.uk does not publish: how close you had to live to get a place.
|
||||
*/
|
||||
title: { absolute: 'Compare Schools Side by Side | schoolcompare' },
|
||||
description:
|
||||
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place. Free, no sign-up.',
|
||||
// This page reads eleven search params. They filter a result set; they do
|
||||
// not make a new document. Collapsing every combination onto "/" stops the
|
||||
// homepage competing with itself for its own head terms.
|
||||
|
||||
@@ -18,8 +18,11 @@ interface RankingsPageProps {
|
||||
}
|
||||
|
||||
export const metadata: Metadata = {
|
||||
title: 'School Rankings',
|
||||
description: 'Top-ranked schools by SATs and GCSE performance across England',
|
||||
// 'School Rankings' matched nothing anyone types. League tables is the
|
||||
// phrase parents actually search, and it spikes each results day.
|
||||
title: { absolute: 'Primary & Secondary School League Tables | schoolcompare' },
|
||||
description:
|
||||
'Rank English schools by SATs results, GCSEs, Progress 8 or Attainment 8, and filter by local authority or year. Built from the DfE’s own figures.',
|
||||
keywords: 'school rankings, top schools, best schools, KS2 rankings, KS4 rankings, school league tables',
|
||||
// Param forms (?metric=&local_authority=&year=&phase=) collapse here for
|
||||
// now. W3 replaces them with real indexable paths.
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
/**
|
||||
* Phase variants of a place page.
|
||||
*
|
||||
* Phase is part of the query — "primary schools in beccles", "secondary
|
||||
* schools in brentwood" — not a filter applied afterwards, so each gets its
|
||||
* own indexable path. A place with no schools of the phase has no page: the
|
||||
* per-phase threshold, not an error.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ place: string; phase: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
const PHASES = ['primary', 'secondary'] as const;
|
||||
type Phase = (typeof PHASES)[number];
|
||||
|
||||
const isPhase = (v: string): v is Phase => (PHASES as readonly string[]).includes(v);
|
||||
|
||||
async function resolve(slug: string, phase: Phase) {
|
||||
return (await fetchPlace('town', slug, phase))
|
||||
?? (await fetchPlace('locality', slug, phase));
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { place: slug, phase } = await params;
|
||||
if (!isPhase(phase)) return { title: 'Place Not Found' };
|
||||
const detail = await resolve(slug, phase);
|
||||
if (!detail || detail.schools.length === 0) return { title: 'Place Not Found' };
|
||||
|
||||
const word = phase === 'secondary' ? 'Secondary' : 'Primary';
|
||||
const { name } = detail.place;
|
||||
return {
|
||||
title: `${word} Schools in ${name} — Ranked | schoolcompare`,
|
||||
description:
|
||||
`Every ${phase} school in ${name} ranked by results, with Ofsted grades and `
|
||||
+ `the local average against England.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/${slug}/${phase}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function PlacePhasePage({ params }: Props) {
|
||||
const { place: slug, phase } = await params;
|
||||
if (!isPhase(phase)) notFound();
|
||||
const detail = await resolve(slug, phase);
|
||||
if (!detail || detail.schools.length === 0) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
const englandAverage = phase === 'secondary'
|
||||
? national?.secondary?.attainment_8_score ?? null
|
||||
: national?.primary?.rwm_expected_pct ?? null;
|
||||
|
||||
return <PlaceView detail={detail} phase={phase}
|
||||
englandAverage={englandAverage} neighbours={[]} />;
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
/**
|
||||
* Town and locality pages.
|
||||
*
|
||||
* A place below the five-school threshold is not in the registry, so
|
||||
* fetchPlace returns null and the request 404s rather than rendering a page
|
||||
* with nothing to say.
|
||||
*/
|
||||
import { notFound, redirect } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces, authoritySlug } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ place: string }> }
|
||||
|
||||
// ISR: place aggregates change only when the pipeline runs.
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ place: string }>> {
|
||||
// Off by default: ~2,000 place routes cannot be built in CI on every deploy.
|
||||
// Matches the PRERENDER_SCHOOLS gate on the school route.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'town' || p.kind === 'locality')
|
||||
.map((p) => ({ place: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
async function resolve(slug: string) {
|
||||
return (await fetchPlace('town', slug)) ?? (await fetchPlace('locality', slug));
|
||||
}
|
||||
|
||||
/** Other towns in the same authority — the cheapest honest definition of
|
||||
* "nearby", and enough to stop each place page being a dead end. */
|
||||
async function neighboursOf(detail: { place: { slug: string; parent_authority: string | null } }) {
|
||||
if (!detail.place.parent_authority) return [];
|
||||
const all = await fetchPlaces();
|
||||
return all
|
||||
.filter((p) => p.kind === 'town' && p.slug !== detail.place.slug)
|
||||
.slice(0, 12);
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { place: slug } = await params;
|
||||
const detail = await resolve(slug);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
title: `Schools in ${name} — Compare ${count} Schools | schoolcompare`,
|
||||
description:
|
||||
`Every school in ${name} ranked by SATs and GCSE results, with Ofsted grades, `
|
||||
+ `the local average against England, and how close you had to live to get a place.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/${slug}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function PlacePage({ params }: Props) {
|
||||
const { place: slug } = await params;
|
||||
const detail = await resolve(slug);
|
||||
if (!detail) notFound();
|
||||
|
||||
// Global constraint: no page without a local average. A place with too few
|
||||
// schools carrying results has nothing to say that a list does not, so it
|
||||
// defers to its authority rather than publishing a thin page.
|
||||
if (detail.averages.rwm_expected_pct == null
|
||||
&& detail.averages.attainment_8_score == null) {
|
||||
if (detail.place.parent_authority) {
|
||||
redirect(`/schools/authority/${authoritySlug(detail.place.parent_authority)}`);
|
||||
}
|
||||
notFound();
|
||||
}
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
// NationalAverages is nested by phase — { primary: {...}, secondary: {...} }
|
||||
// — not flat. Reading it flat silently yields undefined and the page renders
|
||||
// with no comparison, which is the one thing that makes it not a list.
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={await neighboursOf(detail)}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
/**
|
||||
* Local authority pages.
|
||||
*
|
||||
* A separate namespace from /schools/[place] because 67 town names collide
|
||||
* with an authority name and neither set contains the other — Bedford the
|
||||
* town holds 104 schools, Bedford the authority 86, because postal towns
|
||||
* cross authority boundaries. The title says "Local Authority" so a reader
|
||||
* landing on both knows which set each covers.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ la: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ la: string }>> {
|
||||
// Gated like every other prerender in this app. There are only ~154
|
||||
// authorities, but "few enough to always build" still means the API must be
|
||||
// reachable at build time, and in CI it is not — the build fails with
|
||||
// ECONNREFUSED rather than degrading. The catch is the same fallback the
|
||||
// school route uses.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'authority')
|
||||
.map((p) => ({ la: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { la } = await params;
|
||||
const detail = await fetchPlace('authority', la);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
title: `Schools in ${name} — Local Authority | schoolcompare`,
|
||||
description:
|
||||
`All ${count} schools in the ${name} local authority, ranked by SATs and GCSE `
|
||||
+ `results, with Ofsted grades and the authority average against England.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/authority/${la}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function AuthorityPage({ params }: Props) {
|
||||
const { la } = await params;
|
||||
const detail = await fetchPlace('authority', la);
|
||||
if (!detail) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={[]}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/**
|
||||
* Postcode district pages.
|
||||
*
|
||||
* No phase variants: nobody searches "primary schools in SW11", so the
|
||||
* variants would be pages without demand. These exist to catch
|
||||
* "schools near <postcode>" and to give London districts a geographic page
|
||||
* where the GIAS town field cannot.
|
||||
*/
|
||||
import { notFound } from 'next/navigation';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchPlace, fetchPlaces } from '@/lib/places';
|
||||
import { fetchNationalAverages } from '@/lib/api';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
interface Props { params: Promise<{ outcode: string }> }
|
||||
|
||||
export const revalidate = 604800;
|
||||
export const dynamicParams = true;
|
||||
|
||||
export async function generateStaticParams(): Promise<Array<{ outcode: string }>> {
|
||||
// 1,760 of these; same CI budget argument as the town routes.
|
||||
if (process.env.PRERENDER_PLACES !== '1') return [];
|
||||
try {
|
||||
return (await fetchPlaces())
|
||||
.filter((p) => p.kind === 'outcode')
|
||||
.map((p) => ({ outcode: p.slug }));
|
||||
} catch (error) {
|
||||
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function generateMetadata({ params }: Props): Promise<Metadata> {
|
||||
const { outcode } = await params;
|
||||
const detail = await fetchPlace('outcode', outcode);
|
||||
if (!detail) return { title: 'Place Not Found' };
|
||||
|
||||
const { name, count } = detail.place;
|
||||
return {
|
||||
title: `Schools near ${name} | schoolcompare`,
|
||||
description:
|
||||
`${count} schools in the ${name} postcode district, ranked by results, with `
|
||||
+ `Ofsted grades and how close you had to live to get a place.`,
|
||||
alternates: { canonical: absoluteUrl(`/schools/near/${outcode}`) },
|
||||
};
|
||||
}
|
||||
|
||||
export default async function OutcodePage({ params }: Props) {
|
||||
const { outcode } = await params;
|
||||
const detail = await fetchPlace('outcode', outcode);
|
||||
if (!detail) notFound();
|
||||
|
||||
const national = await fetchNationalAverages().catch(() => null);
|
||||
return (
|
||||
<PlaceView
|
||||
detail={detail}
|
||||
englandAverage={national?.primary?.rwm_expected_pct ?? null}
|
||||
neighbours={[]}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -9,7 +9,7 @@ export const runtime = 'nodejs';
|
||||
* validated here rather than passed through, so this route cannot be used to
|
||||
* reach arbitrary backend paths.
|
||||
*/
|
||||
const CHILD = /^(static|schools-\d+)\.xml$/;
|
||||
const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/;
|
||||
|
||||
export async function GET(
|
||||
_request: Request,
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
/* Tokens only — see globals.css. Matches RankingsView's conventions. */
|
||||
.container {
|
||||
width: 100%;
|
||||
min-width: 0;
|
||||
}
|
||||
|
||||
.header {
|
||||
margin-bottom: 1.5rem;
|
||||
}
|
||||
|
||||
.header h1 {
|
||||
font-size: 2.25rem;
|
||||
font-weight: 700;
|
||||
color: var(--text-primary);
|
||||
margin-bottom: 0.5rem;
|
||||
font-family: var(--font-display);
|
||||
text-wrap: balance;
|
||||
}
|
||||
|
||||
.summary {
|
||||
font-size: 1rem;
|
||||
color: var(--text-secondary);
|
||||
margin: 0;
|
||||
line-height: 1.6;
|
||||
}
|
||||
|
||||
/* The one number a list cannot give you, so it gets its own band. */
|
||||
.compare {
|
||||
background: var(--bg-secondary);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
padding: 0.875rem 1.125rem;
|
||||
margin: 0 0 1.5rem;
|
||||
color: var(--text-primary);
|
||||
font-size: 1rem;
|
||||
}
|
||||
|
||||
.ofsted {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem 1.25rem;
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
margin: 0 0 1.5rem;
|
||||
font-size: 0.9375rem;
|
||||
color: var(--text-secondary);
|
||||
}
|
||||
|
||||
/* Wide content scrolls in its own container so the page body never does. */
|
||||
.tableWrap {
|
||||
overflow-x: auto;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
background: var(--bg-card);
|
||||
}
|
||||
|
||||
.table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
font-size: 0.9375rem;
|
||||
}
|
||||
|
||||
.table th,
|
||||
.table td {
|
||||
padding: 0.75rem 1rem;
|
||||
text-align: left;
|
||||
border-bottom: 1px solid var(--border);
|
||||
}
|
||||
|
||||
.table th {
|
||||
background: var(--bg-secondary);
|
||||
color: var(--text-secondary);
|
||||
font-weight: 600;
|
||||
font-size: 0.8125rem;
|
||||
text-transform: uppercase;
|
||||
letter-spacing: 0.04em;
|
||||
}
|
||||
|
||||
.table tbody tr:last-child td {
|
||||
border-bottom: none;
|
||||
}
|
||||
|
||||
.table th:last-child,
|
||||
.num {
|
||||
text-align: right;
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
.neighbours {
|
||||
margin-top: 2rem;
|
||||
}
|
||||
|
||||
.neighbours h2 {
|
||||
font-size: 1.125rem;
|
||||
font-weight: 600;
|
||||
color: var(--text-primary);
|
||||
margin: 0 0 0.75rem;
|
||||
font-family: var(--font-display);
|
||||
}
|
||||
|
||||
.neighbours ul {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem 1rem;
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
margin: 0;
|
||||
}
|
||||
@@ -0,0 +1,144 @@
|
||||
/**
|
||||
* One place page, shared by all four families.
|
||||
*
|
||||
* They differ in what fills the registry, not in what the page shows, so a
|
||||
* second component would be a second place to forget the same change.
|
||||
*
|
||||
* The local-versus-England comparison is the reason this page is not a list:
|
||||
* it is the one number a parent cannot get by reading the schools one by one,
|
||||
* and it is what keeps the page from reading as a name dropped into a
|
||||
* template.
|
||||
*/
|
||||
import Link from 'next/link';
|
||||
import type { PlaceDetail, PlaceSummary } from '@/lib/places';
|
||||
import { placeUrl, authoritySlug } from '@/lib/places';
|
||||
import { schoolUrl } from '@/lib/utils';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import styles from './PlaceView.module.css';
|
||||
|
||||
interface Props {
|
||||
detail: PlaceDetail;
|
||||
phase?: 'primary' | 'secondary';
|
||||
englandAverage: number | null;
|
||||
/** Nearby places, so the page links onward instead of dead-ending. */
|
||||
neighbours: PlaceSummary[];
|
||||
}
|
||||
|
||||
// Ofsted grades in the order they are reported.
|
||||
const OFSTED_LABELS: Array<[number, string]> = [
|
||||
[1, 'Outstanding'], [2, 'Good'],
|
||||
[3, 'Requires improvement'], [4, 'Inadequate'],
|
||||
];
|
||||
|
||||
export function PlaceView({ detail, phase, englandAverage, neighbours }: Props) {
|
||||
const { place, schools, averages } = detail;
|
||||
const metric = phase === 'secondary' ? 'attainment_8_score' : 'rwm_expected_pct';
|
||||
const local = averages[metric];
|
||||
const phaseWord = phase === 'secondary' ? 'Secondary schools'
|
||||
: phase === 'primary' ? 'Primary schools' : 'Schools';
|
||||
const graded = OFSTED_LABELS
|
||||
.map(([grade, label]) => [label, schools.filter((s) => s.ofsted_grade === grade).length] as const)
|
||||
.filter(([, n]) => n > 0);
|
||||
|
||||
// ItemList tells Google this page is a ranked set rather than prose;
|
||||
// BreadcrumbList puts the place in a hierarchy. Capped at 20 because that
|
||||
// is what the page shows above the fold and what the markup should mirror.
|
||||
const jsonLd = {
|
||||
'@context': 'https://schema.org',
|
||||
'@graph': [
|
||||
{
|
||||
'@type': 'ItemList',
|
||||
name: `${phaseWord} in ${place.name}`,
|
||||
numberOfItems: schools.length,
|
||||
itemListElement: schools.slice(0, 20).map((s, i) => ({
|
||||
'@type': 'ListItem',
|
||||
position: i + 1,
|
||||
url: absoluteUrl(schoolUrl(s.urn, s.school_name)),
|
||||
name: s.school_name,
|
||||
})),
|
||||
},
|
||||
{
|
||||
'@type': 'BreadcrumbList',
|
||||
itemListElement: [
|
||||
{ '@type': 'ListItem', position: 1, name: 'Schools',
|
||||
item: absoluteUrl('/') },
|
||||
{ '@type': 'ListItem', position: 2, name: place.name },
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
return (
|
||||
<div className={styles.container}>
|
||||
<script
|
||||
type="application/ld+json"
|
||||
dangerouslySetInnerHTML={{ __html: JSON.stringify(jsonLd) }}
|
||||
/>
|
||||
<header className={styles.header}>
|
||||
<h1>{phaseWord} in {place.name}</h1>
|
||||
<p className={styles.summary}>
|
||||
{place.count} schools
|
||||
{place.parent_authority && (
|
||||
<>
|
||||
{' · '}
|
||||
<Link href={`/schools/authority/${authoritySlug(place.parent_authority)}`}>
|
||||
{place.parent_authority}
|
||||
</Link>
|
||||
</>
|
||||
)}
|
||||
</p>
|
||||
</header>
|
||||
|
||||
{local != null && englandAverage != null && (
|
||||
<p className={styles.compare} data-testid="local-vs-england">
|
||||
{place.name} averages <strong>{Math.round(local)}</strong> against{' '}
|
||||
<strong>{Math.round(englandAverage)}</strong> across England.
|
||||
</p>
|
||||
)}
|
||||
|
||||
{graded.length > 0 && (
|
||||
<ul className={styles.ofsted} data-testid="ofsted-distribution">
|
||||
{graded.map(([label, n]) => (
|
||||
<li key={label}>{label}: <strong>{n}</strong></li>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
|
||||
<div className={styles.tableWrap}>
|
||||
<table className={styles.table}>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>School</th>
|
||||
<th>{phase === 'secondary' ? 'Attainment 8' : 'RWM expected'}</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{schools.map((s) => (
|
||||
<tr key={s.urn}>
|
||||
<td>
|
||||
<Link href={schoolUrl(s.urn, s.school_name)}>{s.school_name}</Link>
|
||||
</td>
|
||||
<td className={styles.num}>
|
||||
{s[metric] == null ? '—' : Math.round(Number(s[metric]))}
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
{neighbours.length > 0 && (
|
||||
<nav className={styles.neighbours} aria-label="Nearby places">
|
||||
<h2>Nearby</h2>
|
||||
<ul>
|
||||
{neighbours.map((n) => (
|
||||
<li key={n.kind + n.slug}>
|
||||
<Link href={placeUrl(n.kind, n.slug)}>{n.name}</Link>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</nav>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
/**
|
||||
* Client for the places API.
|
||||
*
|
||||
* Two namespaces, matching the backend: towns and localities share
|
||||
* /schools/[place]; authorities take /schools/authority/[la]. 67 town names
|
||||
* collide with an authority name and neither set contains the other, so one
|
||||
* namespace would publish near-duplicate pages.
|
||||
*/
|
||||
import type { School } from '@/lib/types';
|
||||
|
||||
export interface PlaceSummary {
|
||||
kind: string;
|
||||
slug: string;
|
||||
name: string;
|
||||
count: number;
|
||||
}
|
||||
|
||||
export interface PlaceDetail {
|
||||
place: PlaceSummary & { parent_authority: string | null };
|
||||
schools: School[];
|
||||
averages: {
|
||||
rwm_expected_pct: number | null;
|
||||
attainment_8_score: number | null;
|
||||
};
|
||||
}
|
||||
|
||||
export function placeUrl(kind: string, slug: string, phase?: string): string {
|
||||
const base =
|
||||
kind === 'authority' ? `/schools/authority/${slug}`
|
||||
: kind === 'outcode' ? `/schools/near/${slug}`
|
||||
: `/schools/${slug}`;
|
||||
return phase ? `${base}/${phase}` : base;
|
||||
}
|
||||
|
||||
/** An authority name as it appears in a URL. */
|
||||
export function authoritySlug(name: string): string {
|
||||
return name.toLowerCase().trim().replace(/[^\w\s-]/g, '').replace(/\s+/g, '-');
|
||||
}
|
||||
|
||||
const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL
|
||||
|| 'http://localhost:8000/api';
|
||||
|
||||
export async function fetchPlaces(): Promise<PlaceSummary[]> {
|
||||
const res = await fetch(`${API}/places`, { next: { revalidate: 604800 } });
|
||||
if (!res.ok) return [];
|
||||
return (await res.json()).places ?? [];
|
||||
}
|
||||
|
||||
export async function fetchPlace(
|
||||
kind: string, slug: string, phase?: string,
|
||||
): Promise<PlaceDetail | null> {
|
||||
const q = phase ? `?phase=${encodeURIComponent(phase)}` : '';
|
||||
const res = await fetch(`${API}/places/${kind}/${slug}${q}`,
|
||||
{ next: { revalidate: 604800 } });
|
||||
if (!res.ok) return null;
|
||||
return res.json();
|
||||
}
|
||||
@@ -55,6 +55,37 @@ const nextConfig = {
|
||||
// Headers for caching and security
|
||||
async headers() {
|
||||
return [
|
||||
{
|
||||
/*
|
||||
* Keep non-production hosts out of the index.
|
||||
*
|
||||
* Staging serves the same image as production off stx., so without
|
||||
* this it is a full crawlable duplicate of the site.
|
||||
*
|
||||
* X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling,
|
||||
* which is not the same as blocking indexing — a disallowed URL can
|
||||
* still be indexed from external links, and worse, blocking the crawl
|
||||
* means Google never fetches the page and never sees a noindex at all.
|
||||
* Staging therefore stays crawlable and answers "noindex" when crawled.
|
||||
*
|
||||
* Matched on the staging host explicitly rather than "any host that is
|
||||
* not production". The inverted form is tempting because it would cover
|
||||
* future environments automatically, but its failure mode is
|
||||
* deindexing production if the Host header ever arrives rewritten by a
|
||||
* proxy. This form's failure mode is a new environment being indexable
|
||||
* until someone adds it here — recoverable, where the other is not.
|
||||
*
|
||||
* Any new non-production hostname must be added to this list.
|
||||
*/
|
||||
source: '/:path*',
|
||||
has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }],
|
||||
headers: [
|
||||
{
|
||||
key: 'X-Robots-Tag',
|
||||
value: 'noindex, nofollow',
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
source: '/:path*',
|
||||
headers: [
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,SW11,London
|
||||
canary-wharf,Canary Wharf,E14,London
|
||||
clapham,Clapham,SW4,London
|
||||
shoreditch,Shoreditch,EC2A|E1,London
|
||||
peckham,Peckham,SE15,London
|
||||
brixton,Brixton,SW2|SW9,London
|
||||
camden-town,Camden Town,NW1,London
|
||||
wimbledon,Wimbledon,SW19,London
|
||||
putney,Putney,SW15,London
|
||||
fulham,Fulham,SW6,London
|
||||
chiswick,Chiswick,W4,London
|
||||
stratford,Stratford,E15,London
|
||||
walthamstow,Walthamstow,E17,London
|
||||
tooting,Tooting,SW17,London
|
||||
dulwich,Dulwich,SE21|SE22,London
|
||||
|
Reference in new issue
Block a user