diff --git a/backend/app.py b/backend/app.py index a3bfc4f..934f0e4 100644 --- a/backend/app.py +++ b/backend/app.py @@ -35,6 +35,7 @@ from .data_loader import ( search_schools_typesense, ) from .data_loader import get_data_info as get_db_info +from .places import build_place_registry from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS from .utils import clean_for_json, convert_to_native @@ -58,6 +59,12 @@ MAX_SLUG_LENGTH = 60 # regenerate endpoint after a pipeline run. _sitemaps: dict[str, str] | None = None +# Built from the same DataFrame the sitemap uses, so places and sitemap can +# never describe different corpora. Reset by the same admin endpoint. +_place_registry: dict | None = None + +VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode") + def _slugify(text: str) -> str: text = text.lower() @@ -170,6 +177,14 @@ SITEMAP_CHUNK_SIZE = 10_000 SITEMAP_CHILD_PREFIX = "/sitemaps" +def get_place_registry() -> dict: + """The place registry, built once and cached for the process.""" + global _place_registry + if _place_registry is None: + _place_registry = build_place_registry(load_school_data()) + return _place_registry + + def _urlset(rows: list[str]) -> str: return "\n".join([ '', @@ -179,6 +194,29 @@ def _urlset(rows: list[str]) -> str: ]) +def _place_url(place) -> str: + """The canonical path for a place. Two namespaces, per the spec. + + Towns and localities share /schools/[place]; authorities take their own + prefix because 67 town names collide with an authority name and neither + set contains the other. + """ + if place.kind == "authority": + return f"/schools/authority/{place.slug}" + if place.kind == "outcode": + return f"/schools/near/{place.slug}" + return f"/schools/{place.slug}" + + +def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]: + return [ + _url_element(BASE_URL + _place_url(p)) + for p in sorted(get_place_registry().values(), + key=lambda p: (p.kind, p.slug)) + if p.kind in kinds + ] + + def build_sitemaps() -> dict[str, str]: """Build the sitemap index and every child, keyed by name.""" df = load_school_data() @@ -196,6 +234,17 @@ def build_sitemaps() -> dict[str, str]: for n, chunk in enumerate(chunks, start=1): children[f"schools-{n}.xml"] = _urlset(chunk) + # Separate children per family: Search Console reports coverage per + # submitted sitemap, which is how the location layer's indexation is + # measured apart from the school pages'. + for label, kinds in (("places", ("town", "locality", "authority")), + ("outcodes", ("outcode",))): + rows = _place_sitemap_rows(kinds) + chunks = [rows[i:i + SITEMAP_CHUNK_SIZE] + for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]] + for n, chunk in enumerate(chunks, start=1): + children[f"{label}-{n}.xml"] = _urlset(chunk) + # On a sitemap index, lastmod means "when this sitemap file last changed", # so generation time is the correct value here — unlike on a , where # it would be a claim about content we cannot support. @@ -1117,6 +1166,62 @@ async def get_rankings( } +@app.get("/api/places") +@limiter.limit(f"{settings.rate_limit_per_minute}/minute") +async def list_places(request: Request): + """Every published place. The sitemap and the link modules read this.""" + registry = get_place_registry() + return {"places": [ + {"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)} + for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)) + ]} + + +@app.get("/api/places/{kind}/{slug}") +@limiter.limit(f"{settings.rate_limit_per_minute}/minute") +async def get_place(request: Request, kind: str, slug: str, + phase: Optional[str] = None): + """One place: its schools ranked, and its local averages.""" + if kind not in VALID_PLACE_KINDS: + raise HTTPException(status_code=404, detail="No such place") + + place = get_place_registry().get(f"{kind}:{slug}") + if place is None: + raise HTTPException(status_code=404, detail="No such place") + + df = load_latest_school_data() + rows = df[df["urn"].isin(place.urns)] + + if phase: + wanted = PHASE_GROUPS.get(phase.lower()) + if wanted and "phase" in rows.columns: + rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)] + + # The metric the page ranks on, which is also the one it averages. + metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct" + if metric in rows.columns: + rows = rows.sort_values(metric, ascending=False, na_position="last") + + averages = { + m: (None if m not in rows.columns or rows[m].dropna().empty + else float(rows[m].dropna().mean())) + for m in ("rwm_expected_pct", "attainment_8_score") + } + + cols = [c for c in SCHOOL_COLUMNS + ["latitude", "longitude", "phase", + "rwm_expected_pct", "attainment_8_score", + "total_pupils"] + if c in rows.columns] + + return { + "place": {"kind": place.kind, "slug": place.slug, "name": place.name, + "count": len(place.urns), + "parent_authority": place.parent_authority}, + "schools": clean_for_json(rows[cols]), + "averages": averages, + } + + @app.get("/api/data-info") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_data_info(request: Request): @@ -1228,7 +1333,11 @@ async def regenerate_sitemap( _: bool = Depends(verify_admin_api_key), ): """Rebuild and cache the sitemap from current school data. Called by Airflow after data updates.""" - global _sitemaps + global _sitemaps, _place_registry + # Places and sitemap are rebuilt together — they read the same marts, and + # letting them drift apart would submit URLs for places that no longer + # exist. + _place_registry = None _sitemaps = build_sitemaps() n = sum(x.count("") for x in _sitemaps.values()) return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)} diff --git a/backend/localities.py b/backend/localities.py new file mode 100644 index 0000000..ecd3077 --- /dev/null +++ b/backend/localities.py @@ -0,0 +1,47 @@ +"""Curated London localities, defined by the postcode districts they cover. + +The GIAS `town` field puts 1,819 London schools under the single value +"London", so it cannot answer "schools in Battersea" — a query that appears in +the Search Console baseline. No single field can: parliamentary constituency +gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary +Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal +and colloquial rather than administrative. + +So this is curated. Where a locality ends is a judgement, not a fact, and a +reviewable file is the honest place for a judgement. No new ingestion is +needed — the corpus already carries postcodes. + +This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv` +mirrors it for anyone querying the warehouse directly; the backend image does +not contain `pipeline/`, which is why the module rather than the seed is +canonical. Same arrangement as `backend/gias_codes.py`. + +A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not +published, so a typo produces no page rather than an empty one. Places that +fail that check are logged at startup, because a locality you meant to publish +quietly not appearing is the failure worth hearing about. +""" + +# slug -> (display name, outcodes) +LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = { + "battersea": ("Battersea", ("SW11",)), + "canary-wharf": ("Canary Wharf", ("E14",)), + "clapham": ("Clapham", ("SW4",)), + "shoreditch": ("Shoreditch", ("EC2A", "E1")), + "peckham": ("Peckham", ("SE15",)), + "brixton": ("Brixton", ("SW2", "SW9")), + "hackney": ("Hackney", ("E5", "E8", "E9")), + "islington": ("Islington", ("N1", "N5", "N7")), + "camden-town": ("Camden Town", ("NW1",)), + "greenwich": ("Greenwich", ("SE10",)), + "wimbledon": ("Wimbledon", ("SW19",)), + "putney": ("Putney", ("SW15",)), + "fulham": ("Fulham", ("SW6",)), + "chiswick": ("Chiswick", ("W4",)), + "ealing": ("Ealing", ("W5", "W13")), + "richmond": ("Richmond", ("TW9", "TW10")), + "stratford": ("Stratford", ("E15",)), + "walthamstow": ("Walthamstow", ("E17",)), + "tooting": ("Tooting", ("SW17",)), + "dulwich": ("Dulwich", ("SE21", "SE22")), +} diff --git a/backend/places.py b/backend/places.py new file mode 100644 index 0000000..12eb84a --- /dev/null +++ b/backend/places.py @@ -0,0 +1,167 @@ +"""The place registry: what places the site publishes, and what is in each. + +One module owns this question. The pages, the sitemap and the internal-link +modules all read from here, so the threshold and the collision rules exist in +exactly one place and are testable without a browser or a database. + +Two namespaces, never one. 67 viable town names collide with a local +authority name, and the authority is the larger set in only 43 of them — +postal towns cross authority boundaries, so neither can absorb the other. +Keys are ":" so the collision cannot reappear in the dict. +""" + +from __future__ import annotations + +import logging +import re +from dataclasses import dataclass + +logger = logging.getLogger(__name__) + +# Five schools with publishable data. Below this a place has nothing to say +# that a list of schools does not, and publishing it is index bloat. +MIN_SCHOOLS = 5 + + +@dataclass(frozen=True) +class Place: + kind: str # "town" | "locality" | "authority" | "outcode" + slug: str + name: str + urns: tuple[int, ...] + parent_authority: str | None # authority NAME, for the 301 target + + @property + def key(self) -> str: + return f"{self.kind}:{self.slug}" + + +def _publishable_urns(df) -> set[int]: + """URNs with something a page could state, deduplicated across years.""" + from backend.app import _PUBLISHABLE_FIELDS + + cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns] + if not cols: + return set() + return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int)) + + +def _parent_authority(group) -> str | None: + """The most common authority in a group — the useful 301 target. + + A town spanning several authorities has no single parent, so the mode is + the honest answer rather than an arbitrary first row. + """ + if "local_authority" not in group.columns: + return None + top = group["local_authority"].dropna() + return str(top.mode().iloc[0]) if not top.empty else None + + +def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]: + """One Place per distinct value of `column` that clears the threshold.""" + from backend.app import _slugify + + if column not in df.columns: + return {} + + out: dict[str, Place] = {} + for name, group in df.groupby(column, dropna=True): + name = str(name).strip() + if not name: + continue + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + slug = _slugify(name) + if not slug: + continue + place = Place( + kind=kind, slug=slug, name=name, urns=urns, + parent_authority=_parent_authority(group) if kind == "town" else None, + ) + out[place.key] = place + return out + + +# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter. +_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s") + + +def _outcode(postcode) -> str | None: + if not isinstance(postcode, str): + return None + m = _OUTCODE_RE.match(postcode.upper().strip()) + return m.group(1) if m else None + + +def _outcode_places(df, publishable: set[int]) -> dict[str, Place]: + """One Place per postcode district clearing the threshold. + + These carry no phase variants: nobody searches "primary schools in SW11". + """ + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + working = working[working["_oc"].notna()] + + out: dict[str, Place] = {} + for oc, group in working.groupby("_oc"): + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc), + urns=urns, parent_authority=_parent_authority(group)) + out[place.key] = place + return out + + +def _locality_places(df, publishable: set[int], + town_slugs: set[str]) -> dict[str, Place]: + """One Place per curated locality clearing the threshold.""" + from backend.localities import LOCALITY_OUTCODES + + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + + out: dict[str, Place] = {} + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + if slug in town_slugs: + raise ValueError( + f"locality {slug!r} collides with a published town of the same " + "slug; publishing both would shadow the town silently" + ) + group = working[working["_oc"].isin(outcodes)] + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + # Not an error — a locality can legitimately be too small. Logged + # because one you meant to publish quietly vanishing is the + # failure worth hearing about. + logger.warning( + "locality %s (%s) has %d publishable schools, below the " + "threshold of %d - not published", + slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS) + continue + place = Place(kind="locality", slug=slug, name=name, urns=urns, + parent_authority=_parent_authority(group)) + out[place.key] = place + return out + + +def build_place_registry(df) -> dict[str, Place]: + """Every place the site publishes, keyed by ":".""" + if df.empty or "urn" not in df.columns: + return {} + + publishable = _publishable_urns(df) + registry: dict[str, Place] = {} + registry.update(_group(df, "local_authority", "authority", publishable)) + + towns = _group(df, "town", "town", publishable) + registry.update(towns) + + town_slugs = {p.slug for p in towns.values()} + registry.update(_locality_places(df, publishable, town_slugs)) + registry.update(_outcode_places(df, publishable)) + return registry diff --git a/backend/tests/test_places.py b/backend/tests/test_places.py new file mode 100644 index 0000000..bcd7019 --- /dev/null +++ b/backend/tests/test_places.py @@ -0,0 +1,176 @@ +"""Tests for the place registry (spec 2026-08-21). + +The registry is built from the in-memory school DataFrame, so these build a +small frame directly rather than touching a database. +""" + +import numpy as np +import pandas as pd +import pytest + +from backend.places import MIN_SCHOOLS, build_place_registry + + +def _df(rows: list[dict]) -> pd.DataFrame: + base = { + "year": 202425, "ofsted_grade": 2.0, "ofsted_date": None, + "rwm_expected_pct": 60.0, "attainment_8_score": np.nan, + "phase": "Primary", "postcode": "AA1 1AA", + } + return pd.DataFrame([{**base, **r} for r in rows]) + + +def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]: + """`start` offsets the URNs so two calls can describe different schools — + the Bedford case needs two authorities' worth of distinct URNs in one + town.""" + return [ + {"urn": start + i, "school_name": f"{town} School {i}", + "town": town, "local_authority": la, **kw} + for i in range(n) + ] + + +def test_town_clearing_the_threshold_is_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex"))) + assert "town:brentwood" in reg + assert reg["town:brentwood"].name == "Brentwood" + assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS + + +def test_town_below_the_threshold_is_not_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton"))) + assert "town:crosby" not in reg + + +def test_a_town_below_threshold_still_names_its_authority(): + # The route layer needs somewhere to 301 to. + reg = build_place_registry(_df( + _town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton"))) + assert "authority:sefton" in reg + + +def test_town_and_authority_of_the_same_name_are_separate_places(): + # 67 real collisions. Neither set contains the other: Bedford the town has + # 104 schools, Bedford the authority 86, because postal towns cross + # authority boundaries. + rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford") + + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000)) + reg = build_place_registry(_df(rows)) + town, authority = reg["town:bedford"], reg["authority:bedford"] + assert set(town.urns) != set(authority.urns) + assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools + assert len(authority.urns) == MIN_SCHOOLS # only this authority's + + +def test_schools_without_publishable_data_do_not_count_toward_the_threshold(): + rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere") + for r in rows: + r["rwm_expected_pct"] = np.nan + r["ofsted_grade"] = np.nan + reg = build_place_registry(_df(rows)) + assert "town:ghosttown" not in reg + + +def test_blank_town_is_ignored(): + rows = _town(MIN_SCHOOLS, "", "Essex") + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("town:") for k in reg) + + +def test_a_school_is_counted_once_even_with_several_years_of_rows(): + rows = [] + for year in (202324, 202425): + rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")] + reg = build_place_registry(_df(rows)) + assert len(reg["town:beccles"].urns) == MIN_SCHOOLS + + +def test_locality_groups_schools_by_outcode(monkeypatch): + # The GIAS town field collapses 1,819 London schools into "London", so a + # locality is defined by its postcode districts instead. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"battersea": ("Battersea", ("SW11",))}) + rows = _town(MIN_SCHOOLS, "London", "Wandsworth") + for r in rows: + r["postcode"] = "SW11 2AA" + reg = build_place_registry(_df(rows)) + assert reg["locality:battersea"].name == "Battersea" + assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS + + +def test_locality_below_the_threshold_is_not_published(monkeypatch): + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"nowhere": ("Nowhere", ("ZZ99",))}) + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth"))) + assert "locality:nowhere" not in reg + + +def test_a_locality_may_not_shadow_a_viable_town(monkeypatch): + # Silently shadowing a town would lose a page carrying real demand. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"brentwood": ("Brentwood", ("CM13",))}) + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + with pytest.raises(ValueError, match="brentwood"): + build_place_registry(_df(rows)) + + +def test_outcode_places_are_built_from_postcodes(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + reg = build_place_registry(_df(rows)) + assert reg["outcode:cm13"].name == "CM13" + assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS + + +def test_malformed_postcodes_do_not_create_places(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "not a postcode" + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("outcode:") for k in reg) + + +def test_every_curated_locality_is_structurally_valid(): + # Guards the hand-maintained file: real slug, real name, real outcodes. + import re + from backend.localities import LOCALITY_OUTCODES + + assert LOCALITY_OUTCODES, "the curated locality list must not be empty" + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + assert re.fullmatch(r"[a-z0-9-]+", slug), slug + assert name.strip() == name and name, slug + assert outcodes, f"{slug} has no outcodes" + for oc in outcodes: + assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc) + + +def test_the_pipeline_seed_mirrors_the_canonical_module(): + """Two copies with no drift guard is worse than one copy. + + backend/localities.py is canonical because the backend image does not + contain pipeline/. The seed exists so the warehouse can join on the same + definitions, and this is what stops the two diverging — the same + arrangement assert_gias_code_names_match_seed.sql gives gias_codes. + """ + import csv + from pathlib import Path + + from backend.localities import LOCALITY_OUTCODES + + seed_path = (Path(__file__).resolve().parents[2] + / "pipeline/transform/seeds/locality_outcodes.csv") + assert seed_path.exists(), f"missing seed mirror at {seed_path}" + + seed = { + row["locality_slug"]: (row["locality_name"], + tuple(row["outcodes"].split("|"))) + for row in csv.DictReader(seed_path.open()) + } + assert seed == LOCALITY_OUTCODES diff --git a/backend/tests/test_places_api.py b/backend/tests/test_places_api.py new file mode 100644 index 0000000..4eb9822 --- /dev/null +++ b/backend/tests/test_places_api.py @@ -0,0 +1,71 @@ +"""Tests for the places API (spec 2026-08-21).""" + +import numpy as np +import pandas as pd +import pytest +from fastapi.testclient import TestClient + + +def _schools_df() -> pd.DataFrame: + base = { + "local_authority": "Essex", "school_type": "Academy", + "phase": "Primary", "year": 202425, "ofsted_grade": 2.0, + "ofsted_date": None, "attainment_8_score": np.nan, + "town": "Brentwood", "postcode": "CM13 1AA", "status": "Open", + "address": "1 Test Street", "latitude": 51.6, "longitude": 0.3, + } + return pd.DataFrame([ + {**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}", + "rwm_expected_pct": 50.0 + i} + for i in range(6) + ]) + + +@pytest.fixture() +def client(monkeypatch): + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _schools_df) + monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df) + monkeypatch.setattr(app_module, "_place_registry", None) + return TestClient(app_module.app, raise_server_exceptions=False) + + +def test_registry_lists_each_published_place(client): + body = client.get("/api/places").json() + slugs = {(p["kind"], p["slug"]) for p in body["places"]} + assert ("town", "brentwood") in slugs + assert ("authority", "essex") in slugs + assert ("outcode", "cm13") in slugs + + +def test_registry_carries_a_count_per_place(client): + body = client.get("/api/places").json() + town = next(p for p in body["places"] if p["slug"] == "brentwood") + assert town["count"] == 6 + + +def test_place_detail_returns_its_schools_ranked(client): + body = client.get("/api/places/town/brentwood").json() + assert body["place"]["name"] == "Brentwood" + scores = [s["rwm_expected_pct"] for s in body["schools"]] + assert scores == sorted(scores, reverse=True) + + +def test_place_detail_carries_the_local_average(client): + body = client.get("/api/places/town/brentwood").json() + # 50..55 inclusive + assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5) + + +def test_phase_filter_narrows_the_school_list(client): + body = client.get("/api/places/town/brentwood?phase=secondary").json() + assert body["schools"] == [] + + +def test_unknown_place_404s(client): + assert client.get("/api/places/town/atlantis").status_code == 404 + + +def test_unknown_kind_404s(client): + assert client.get("/api/places/planet/mars").status_code == 404 diff --git a/backend/tests/test_sitemap.py b/backend/tests/test_sitemap.py index c3d908a..21d44ff 100644 --- a/backend/tests/test_sitemap.py +++ b/backend/tests/test_sitemap.py @@ -59,9 +59,14 @@ def static_child(monkeypatch) -> str: def test_every_loc_uses_the_www_host(sitemaps): # The apex 301s to www. A that redirects burns a crawl per URL. # Checked across every file, index included, not just one. + # + # A child can legitimately be empty — this fixture holds two schools and no + # town clearing the threshold — so the presence check applies only to files + # that carry URLs. The absence check applies to all of them. for name, xml in sitemaps.items(): - assert "https://www.schoolcompare.co.uk" in xml, name assert "https://schoolcompare.co.uk" not in xml, name + if "" in xml: + assert "https://www.schoolcompare.co.uk" in xml, name def test_school_with_results_is_listed(schools_child): @@ -217,3 +222,49 @@ def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch): monkeypatch.setattr(app_module, "load_school_data", lambda: df) assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"] + + +def _places_df() -> pd.DataFrame: + base = { + "local_authority": "Essex", "school_type": "Academy", + "phase": "Primary", "year": 202425, "ofsted_grade": 2.0, + "ofsted_date": None, "attainment_8_score": np.nan, + "town": "Brentwood", "postcode": "CM13 1AA", + } + return pd.DataFrame([ + {**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}", + "rwm_expected_pct": 60.0} + for i in range(6) + ]) + + +@pytest.fixture() +def place_sitemaps(monkeypatch) -> dict: + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _places_df) + monkeypatch.setattr(app_module, "_place_registry", None) + return app_module.build_sitemaps() + + +def test_place_children_are_listed_in_the_index(place_sitemaps): + index = place_sitemaps["sitemap.xml"] + assert "/sitemaps/places-1.xml" in index + assert "/sitemaps/outcodes-1.xml" in index + + +def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps): + xml = place_sitemaps["places-1.xml"] + assert "https://www.schoolcompare.co.uk/schools/brentwood" in xml + assert "https://www.schoolcompare.co.uk/schools/authority/essex" in xml + + +def test_outcode_urls_live_in_their_own_child(place_sitemaps): + assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"] + assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"] + + +def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps): + for name in ("places-1.xml", "outcodes-1.xml"): + assert "" not in place_sitemaps[name] + assert "" not in place_sitemaps[name] diff --git a/docs/superpowers/plans/2026-08-21-w2-location-layer.md b/docs/superpowers/plans/2026-08-21-w2-location-layer.md new file mode 100644 index 0000000..7dbbd98 --- /dev/null +++ b/docs/superpowers/plans/2026-08-21-w2-location-layer.md @@ -0,0 +1,1837 @@ +# W2: Location Layer Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Publish ~4,000 pages about places — towns, London localities, local +authorities and postcode districts — so the site competes for the location +intent it currently answers from position 49.5. + +**Architecture:** One Python module owns the question "what places exist and +what is in each" (`backend/places.py`), built once at startup from the same +DataFrame the sitemap uses. Two API endpoints expose it. Next App Router +renders four route families from those endpoints under ISR, and the sitemap +index gains two new children so indexation is measurable per family. + +**Tech Stack:** FastAPI + pandas (registry and API), pytest with +`monkeypatch`-injected DataFrames, Next.js 15 App Router with ISR, Jest + +jsdom, Playwright for the staging gate. + +**Spec:** `docs/superpowers/specs/2026-08-21-w2-location-layer-design.md` + +## Global Constraints + +- **Threshold: five schools with publishable data.** Below it a place is not + published and its URL 301s to the parent authority. Reuse + `_has_publishable_data` from `backend/app.py` — do not write a second + definition of "has data". +- **Per-phase thresholds apply independently.** A town with 30 primaries and + 2 secondaries publishes a primary page and no secondary page. +- **No page without a local average.** If a place has too few schools with + results to compute one, it falls back to the authority. +- **Two namespaces, never one:** `/schools/[place]` for towns and localities, + `/schools/authority/[la]` for authorities. 67 town names collide with + authority names and neither set contains the other. +- **A locality slug may not collide with a viable town slug.** The registry + raises rather than silently shadowing a town. +- **Canonical host is `https://www.schoolcompare.co.uk`** — use + `absoluteUrl()` from `@/lib/site`, never a literal. +- **Never push to `main`.** Feature branch and a PR, per `CLAUDE.md`. +- **Backend test command:** + ```bash + uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests -q + ``` +- **Frontend:** `cd nextjs-app && npm test` · `npx tsc --noEmit` · `npx next build` +- **Do not start a local server to test** (`CLAUDE.md`). + +## Deviation from the spec, and why + +The spec puts the locality list in a dbt seed with a dbt test asserting every +outcode matches a school. **The backend Docker image does not contain +`pipeline/`** — it copies only `backend/` and `scripts/` — so the runtime +cannot read a seed file. + +This plan follows the precedent the repo already set for exactly this problem: +`backend/gias_codes.py` is canonical, `pipeline/transform/seeds/` holds a +copy, and `assert_gias_code_names_match_seed.sql` guards drift. Localities get +`backend/localities.py` as canonical. + +The dbt test is dropped rather than duplicated. Its stated purpose was to stop +a typo publishing an empty page, but the five-school threshold already makes +that impossible — a typo'd outcode matches nothing and simply does not +publish. What the threshold does not catch is a locality you *meant* to +publish silently not appearing, so Task 2 logs those at startup instead. + +## File structure + +| File | Responsibility | +|------|----------------| +| `backend/places.py` (new) | The registry. What places exist, what is in each, threshold and collision rules. No HTTP, no SQL. | +| `backend/localities.py` (new) | Curated locality → outcodes data. Data only, no logic. | +| `backend/app.py` (modify) | Two endpoints; two new sitemap children. | +| `nextjs-app/lib/places.ts` (new) | Typed client for the two endpoints. | +| `nextjs-app/components/places/PlaceView.tsx` (new) | Renders one place page. Shared by all four families. | +| `nextjs-app/app/schools/[place]/...` (new) | Place and phase routes. | +| `nextjs-app/app/schools/authority/[la]/...` (new) | Authority routes. | +| `nextjs-app/app/schools/near/[outcode]/page.tsx` (new) | Outcode routes. | + +`PlaceView` is deliberately one component for all four families: they differ +in what fills the registry, not in what the page shows. A second component +would be two places to forget the same change. + +--- + +### Task 1: The place registry — towns and authorities + +**Files:** +- Create: `backend/places.py` +- Test: `backend/tests/test_places.py` + +**Interfaces:** +- Consumes: `_has_publishable_data` and `_slugify` from `backend.app`. +- Produces: + ```python + @dataclass(frozen=True) + class Place: + kind: str # "town" | "locality" | "authority" | "outcode" + slug: str + name: str + urns: tuple[int, ...] + parent_authority: str | None + + MIN_SCHOOLS: int = 5 + def build_place_registry(df) -> dict[str, Place] + ``` + Keys are `":"` — `"town:brentwood"`, `"authority:essex"` — so + the two namespaces cannot collide in the dict either. + +- [ ] **Step 1: Write the failing tests** + +Create `backend/tests/test_places.py`: + +```python +"""Tests for the place registry (spec 2026-08-21). + +The registry is built from the in-memory school DataFrame, so these build a +small frame directly rather than touching a database. +""" + +import numpy as np +import pandas as pd +import pytest + +from backend.places import MIN_SCHOOLS, build_place_registry + + +def _df(rows: list[dict]) -> pd.DataFrame: + base = { + "year": 202425, "ofsted_grade": 2.0, "ofsted_date": None, + "rwm_expected_pct": 60.0, "attainment_8_score": np.nan, + "phase": "Primary", "postcode": "AA1 1AA", + } + return pd.DataFrame([{**base, **r} for r in rows]) + + +def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]: + """`start` offsets the URNs so two calls can describe different schools — + the Bedford case needs two authorities' worth of distinct URNs in one + town.""" + return [ + {"urn": start + i, "school_name": f"{town} School {i}", + "town": town, "local_authority": la, **kw} + for i in range(n) + ] + + +def test_town_clearing_the_threshold_is_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex"))) + assert "town:brentwood" in reg + assert reg["town:brentwood"].name == "Brentwood" + assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS + + +def test_town_below_the_threshold_is_not_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton"))) + assert "town:crosby" not in reg + + +def test_a_town_below_threshold_still_names_its_authority(): + # The route layer needs somewhere to 301 to. + reg = build_place_registry(_df( + _town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton"))) + assert "authority:sefton" in reg + + +def test_town_and_authority_of_the_same_name_are_separate_places(): + # 67 real collisions. Neither set contains the other: Bedford the town has + # 104 schools, Bedford the authority 86, because postal towns cross + # authority boundaries. + rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford") + + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000)) + reg = build_place_registry(_df(rows)) + town, authority = reg["town:bedford"], reg["authority:bedford"] + assert set(town.urns) != set(authority.urns) + assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools + assert len(authority.urns) == MIN_SCHOOLS # only this authority's + + +def test_schools_without_publishable_data_do_not_count_toward_the_threshold(): + rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere") + for r in rows: + r["rwm_expected_pct"] = np.nan + r["ofsted_grade"] = np.nan + reg = build_place_registry(_df(rows)) + assert "town:ghosttown" not in reg + + +def test_blank_town_is_ignored(): + rows = _town(MIN_SCHOOLS, "", "Essex") + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("town:") for k in reg) + + +def test_a_school_is_counted_once_even_with_several_years_of_rows(): + rows = [] + for year in (202324, 202425): + rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")] + reg = build_place_registry(_df(rows)) + assert len(reg["town:beccles"].urns) == MIN_SCHOOLS +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places.py -q +``` +Expected: FAIL — `No module named 'backend.places'`. + +- [ ] **Step 3: Write the registry** + +Create `backend/places.py`: + +```python +"""The place registry: what places the site publishes, and what is in each. + +One module owns this question. The pages, the sitemap and the internal-link +modules all read from here, so the threshold and the collision rules exist in +exactly one place and are testable without a browser or a database. + +Two namespaces, never one. 67 viable town names collide with a local +authority name, and the authority is the larger set in only 43 of them — +postal towns cross authority boundaries, so neither can absorb the other. +Keys are ":" so the collision cannot reappear in the dict. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +# Five schools with publishable data. Below this a place has nothing to say +# that a list of schools does not, and publishing it is index bloat. +MIN_SCHOOLS = 5 + + +@dataclass(frozen=True) +class Place: + kind: str # "town" | "locality" | "authority" | "outcode" + slug: str + name: str + urns: tuple[int, ...] + parent_authority: str | None # authority NAME, for the 301 target + + @property + def key(self) -> str: + return f"{self.kind}:{self.slug}" + + +def _publishable_urns(df): + """URNs with something a page could state, deduplicated across years.""" + from backend.app import _PUBLISHABLE_FIELDS, _slugify # noqa: F401 + + cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns] + if not cols: + return set() + return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int)) + + +def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]: + """One Place per distinct value of `column` that clears the threshold.""" + from backend.app import _slugify + + if column not in df.columns: + return {} + + out: dict[str, Place] = {} + for name, group in df.groupby(column, dropna=True): + name = str(name).strip() + if not name: + continue + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + slug = _slugify(name) + if not slug: + continue + # A town spanning several authorities has no single parent; the most + # common one is the useful 301 target. + parent = None + if kind == "town" and "local_authority" in group.columns: + top = group["local_authority"].dropna() + parent = str(top.mode().iloc[0]) if not top.empty else None + place = Place(kind=kind, slug=slug, name=name, urns=urns, + parent_authority=parent) + out[place.key] = place + return out + + +def build_place_registry(df) -> dict[str, Place]: + """Every place the site publishes, keyed by ":".""" + if df.empty or "urn" not in df.columns: + return {} + + publishable = _publishable_urns(df) + registry: dict[str, Place] = {} + registry.update(_group(df, "local_authority", "authority", publishable)) + registry.update(_group(df, "town", "town", publishable)) + return registry +``` + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places.py -q +``` +Expected: PASS (7 tests). + +- [ ] **Step 5: Commit** + +```bash +git add backend/places.py backend/tests/test_places.py +git commit -m "feat(places): registry of towns and authorities + +Two namespaces because 67 town names collide with an authority name and +neither set contains the other — postal towns cross authority boundaries." +``` + +--- + +### Task 2: The registry — London localities and outcodes + +**Files:** +- Create: `backend/localities.py` +- Create: `pipeline/transform/seeds/locality_outcodes.csv` +- Modify: `backend/places.py` (add locality and outcode groups) +- Test: `backend/tests/test_places.py` (extend) + +**Interfaces:** +- Consumes: `Place`, `MIN_SCHOOLS`, `build_place_registry` (Task 1). +- Produces: + ```python + # backend/localities.py + LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] + # slug -> (display name, outcodes) + ``` + `build_place_registry` gains `"locality:*"` and `"outcode:*"` keys. + +- [ ] **Step 1: Write the failing tests** + +Append to `backend/tests/test_places.py`: + +```python +def test_locality_groups_schools_by_outcode(monkeypatch): + # The GIAS town field collapses 1,819 London schools into "London", so a + # locality is defined by its postcode districts instead. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"battersea": ("Battersea", ("SW11",))}) + rows = _town(MIN_SCHOOLS, "London", "Wandsworth") + for r in rows: + r["postcode"] = "SW11 2AA" + reg = build_place_registry(_df(rows)) + assert reg["locality:battersea"].name == "Battersea" + assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS + + +def test_locality_below_the_threshold_is_not_published(monkeypatch): + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"nowhere": ("Nowhere", ("ZZ99",))}) + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth"))) + assert "locality:nowhere" not in reg + + +def test_a_locality_may_not_shadow_a_viable_town(monkeypatch): + # Silently shadowing a town would lose a page carrying real demand. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"brentwood": ("Brentwood", ("CM13",))}) + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + with pytest.raises(ValueError, match="brentwood"): + build_place_registry(_df(rows)) + + +def test_outcode_places_are_built_from_postcodes(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + reg = build_place_registry(_df(rows)) + assert reg["outcode:cm13"].name == "CM13" + assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS + + +def test_malformed_postcodes_do_not_create_places(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "not a postcode" + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("outcode:") for k in reg) + + +def test_every_curated_locality_is_structurally_valid(): + # Guards the hand-maintained file: real slug, real name, real outcodes. + import re + from backend.localities import LOCALITY_OUTCODES + + assert LOCALITY_OUTCODES, "the curated locality list must not be empty" + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + assert re.fullmatch(r"[a-z0-9-]+", slug), slug + assert name.strip() == name and name, slug + assert outcodes, f"{slug} has no outcodes" + for oc in outcodes: + assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc) +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places.py -q +``` +Expected: FAIL — `No module named 'backend.localities'`. + +- [ ] **Step 3: Write the curated locality data** + +Create `backend/localities.py`: + +```python +"""Curated London localities, defined by the postcode districts they cover. + +The GIAS `town` field puts 1,819 London schools under the single value +"London", so it cannot answer "schools in Battersea" — a query that appears in +the Search Console baseline. No single field can: parliamentary constituency +gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary +Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal +and colloquial rather than administrative. + +So this is curated. Where a locality ends is a judgement, not a fact, and a +reviewable file is the honest place for a judgement. No new ingestion is +needed — the corpus already carries postcodes. + +This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv` +mirrors it for anyone querying the warehouse directly; the backend image does +not contain `pipeline/`, which is why the module rather than the seed is +canonical. Same arrangement as `backend/gias_codes.py`. + +A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not +published, so a typo produces no page rather than an empty one. Places that +fail that check are logged at startup, because a locality you meant to publish +quietly not appearing is the failure worth hearing about. +""" + +# slug -> (display name, outcodes) +LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = { + "battersea": ("Battersea", ("SW11",)), + "canary-wharf": ("Canary Wharf", ("E14",)), + "clapham": ("Clapham", ("SW4",)), + "shoreditch": ("Shoreditch", ("EC2A", "E1")), + "peckham": ("Peckham", ("SE15",)), + "brixton": ("Brixton", ("SW2", "SW9")), + "hackney": ("Hackney", ("E5", "E8", "E9")), + "islington": ("Islington", ("N1", "N5", "N7")), + "camden-town": ("Camden Town", ("NW1",)), + "greenwich": ("Greenwich", ("SE10",)), + "wimbledon": ("Wimbledon", ("SW19",)), + "putney": ("Putney", ("SW15",)), + "fulham": ("Fulham", ("SW6",)), + "chiswick": ("Chiswick", ("W4",)), + "ealing": ("Ealing", ("W5", "W13")), + "richmond": ("Richmond", ("TW9", "TW10")), + "stratford": ("Stratford", ("E15",)), + "walthamstow": ("Walthamstow", ("E17",)), + "tooting": ("Tooting", ("SW17",)), + "dulwich": ("Dulwich", ("SE21", "SE22")), +} +``` + +Create `pipeline/transform/seeds/locality_outcodes.csv` mirroring it, so the +warehouse can join on the same definitions: + +```csv +locality_slug,locality_name,outcodes,region +battersea,Battersea,SW11,London +canary-wharf,Canary Wharf,E14,London +clapham,Clapham,SW4,London +shoreditch,Shoreditch,EC2A|E1,London +peckham,Peckham,SE15,London +brixton,Brixton,SW2|SW9,London +hackney,Hackney,E5|E8|E9,London +islington,Islington,N1|N5|N7,London +camden-town,Camden Town,NW1,London +greenwich,Greenwich,SE10,London +wimbledon,Wimbledon,SW19,London +putney,Putney,SW15,London +fulham,Fulham,SW6,London +chiswick,Chiswick,W4,London +ealing,Ealing,W5|W13,London +richmond,Richmond,TW9|TW10,London +stratford,Stratford,E15,London +walthamstow,Walthamstow,E17,London +tooting,Tooting,SW17,London +dulwich,Dulwich,SE21|SE22,London +``` + +- [ ] **Step 4: Extend the registry** + +In `backend/places.py`, add above `build_place_registry`: + +```python +import logging +import re + +logger = logging.getLogger(__name__) + +# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter. +_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s") + + +def _outcode(postcode) -> str | None: + if not isinstance(postcode, str): + return None + m = _OUTCODE_RE.match(postcode.upper().strip()) + return m.group(1) if m else None + + +def _outcode_places(df, publishable: set[int]) -> dict[str, Place]: + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + working = working[working["_oc"].notna()] + out: dict[str, Place] = {} + for oc, group in working.groupby("_oc"): + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + parent = None + if "local_authority" in group.columns: + top = group["local_authority"].dropna() + parent = str(top.mode().iloc[0]) if not top.empty else None + place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc), + urns=urns, parent_authority=parent) + out[place.key] = place + return out + + +def _locality_places(df, publishable: set[int], + town_slugs: set[str]) -> dict[str, Place]: + from backend.localities import LOCALITY_OUTCODES + + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + out: dict[str, Place] = {} + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + if slug in town_slugs: + raise ValueError( + f"locality {slug!r} collides with a published town of the same " + "slug; publishing both would shadow the town silently" + ) + group = working[working["_oc"].isin(outcodes)] + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + logger.warning( + "locality %s (%s) has %d publishable schools, below the " + "threshold of %d — not published", + slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS) + continue + parent = None + if "local_authority" in group.columns: + top = group["local_authority"].dropna() + parent = str(top.mode().iloc[0]) if not top.empty else None + place = Place(kind="locality", slug=slug, name=name, urns=urns, + parent_authority=parent) + out[place.key] = place + return out +``` + +Replace the body of `build_place_registry` with: + +```python +def build_place_registry(df) -> dict[str, Place]: + """Every place the site publishes, keyed by ":".""" + if df.empty or "urn" not in df.columns: + return {} + + publishable = _publishable_urns(df) + registry: dict[str, Place] = {} + registry.update(_group(df, "local_authority", "authority", publishable)) + towns = _group(df, "town", "town", publishable) + registry.update(towns) + town_slugs = {p.slug for p in towns.values()} + registry.update(_locality_places(df, publishable, town_slugs)) + registry.update(_outcode_places(df, publishable)) + return registry +``` + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places.py -q +``` +Expected: PASS (13 tests). + +- [ ] **Step 6: Commit** + +```bash +git add backend/localities.py backend/places.py backend/tests/test_places.py \ + pipeline/transform/seeds/locality_outcodes.csv +git commit -m "feat(places): London localities and postcode districts + +The GIAS town field puts 1,819 London schools under one value, so a locality +is defined by the postcode districts it covers. Curated, because where a +locality ends is a judgement rather than a fact." +``` + +--- + +### Task 3: The places API + +**Files:** +- Modify: `backend/app.py` (registry cache, two endpoints) +- Test: `backend/tests/test_places_api.py` + +**Interfaces:** +- Consumes: `build_place_registry`, `Place` (Tasks 1–2). +- Produces: + ``` + GET /api/places -> {"places": [{slug, kind, name, count}]} + GET /api/places/{kind}/{slug} -> {"place": {...}, "schools": [...], + "averages": {...}} + ``` + `get_place_registry()` returns the cached `dict[str, Place]` and is what + Task 4's sitemap enumerates. + +- [ ] **Step 1: Write the failing tests** + +Create `backend/tests/test_places_api.py`: + +```python +"""Tests for the places API (spec 2026-08-21).""" + +import numpy as np +import pandas as pd +import pytest +from fastapi.testclient import TestClient + + +def _schools_df() -> pd.DataFrame: + base = { + "local_authority": "Essex", "school_type": "Academy", + "phase": "Primary", "year": 202425, "ofsted_grade": 2.0, + "ofsted_date": None, "attainment_8_score": np.nan, + "town": "Brentwood", "postcode": "CM13 1AA", "status": "Open", + "address": "1 Test Street", "latitude": 51.6, "longitude": 0.3, + } + return pd.DataFrame([ + {**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}", + "rwm_expected_pct": 50.0 + i} + for i in range(6) + ]) + + +@pytest.fixture() +def client(monkeypatch): + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _schools_df) + monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df) + monkeypatch.setattr(app_module, "_place_registry", None) + return TestClient(app_module.app, raise_server_exceptions=False) + + +def test_registry_lists_each_published_place(client): + body = client.get("/api/places").json() + slugs = {(p["kind"], p["slug"]) for p in body["places"]} + assert ("town", "brentwood") in slugs + assert ("authority", "essex") in slugs + assert ("outcode", "cm13") in slugs + + +def test_registry_carries_a_count_per_place(client): + body = client.get("/api/places").json() + town = next(p for p in body["places"] if p["slug"] == "brentwood") + assert town["count"] == 6 + + +def test_place_detail_returns_its_schools_ranked(client): + body = client.get("/api/places/town/brentwood").json() + assert body["place"]["name"] == "Brentwood" + scores = [s["rwm_expected_pct"] for s in body["schools"]] + assert scores == sorted(scores, reverse=True) + + +def test_place_detail_carries_the_local_average(client): + body = client.get("/api/places/town/brentwood").json() + # 50..55 inclusive + assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5) + + +def test_phase_filter_narrows_the_school_list(client): + body = client.get("/api/places/town/brentwood?phase=secondary").json() + assert body["schools"] == [] + + +def test_unknown_place_404s(client): + assert client.get("/api/places/town/atlantis").status_code == 404 + + +def test_unknown_kind_404s(client): + assert client.get("/api/places/planet/mars").status_code == 404 +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places_api.py -q +``` +Expected: FAIL — 404 on every route. + +- [ ] **Step 3: Add the cache and endpoints** + +In `backend/app.py`, beside `_sitemaps`: + +```python +# Built once from the same DataFrame the sitemap uses, and rebuilt by +# /api/admin/regenerate-sitemap after a pipeline run. +_place_registry: dict | None = None +``` + +Add near the other helpers: + +```python +from .places import Place, build_place_registry # noqa: E402 + +VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode") + + +def get_place_registry() -> dict: + global _place_registry + if _place_registry is None: + _place_registry = build_place_registry(load_school_data()) + return _place_registry +``` + +Add the endpoints beside the other `/api` routes: + +```python +@app.get("/api/places") +@limiter.limit(f"{settings.rate_limit_per_minute}/minute") +async def list_places(request: Request): + """Every published place. The sitemap and the link modules read this.""" + registry = get_place_registry() + return {"places": [ + {"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)} + for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)) + ]} + + +@app.get("/api/places/{kind}/{slug}") +@limiter.limit(f"{settings.rate_limit_per_minute}/minute") +async def get_place(request: Request, kind: str, slug: str, + phase: Optional[str] = None): + """One place: its schools ranked, and its local averages.""" + if kind not in VALID_PLACE_KINDS: + raise HTTPException(status_code=404, detail="No such place") + + place = get_place_registry().get(f"{kind}:{slug}") + if place is None: + raise HTTPException(status_code=404, detail="No such place") + + df = load_latest_school_data() + rows = df[df["urn"].isin(place.urns)] + + if phase: + wanted = PHASE_GROUPS.get(phase.lower()) + if wanted and "phase" in rows.columns: + rows = rows[rows["phase"].str.lower().isin(wanted)] + + metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct" + if metric in rows.columns: + rows = rows.sort_values(metric, ascending=False, na_position="last") + + averages = { + m: (None if m not in rows.columns or rows[m].dropna().empty + else float(rows[m].dropna().mean())) + for m in ("rwm_expected_pct", "attainment_8_score") + } + + return { + "place": {"kind": place.kind, "slug": place.slug, "name": place.name, + "count": len(place.urns), + "parent_authority": place.parent_authority}, + "schools": df_to_records(rows, SCHOOL_COLUMNS), + "averages": averages, + } +``` + +If `df_to_records` is not already imported in `app.py`, use the same +serialisation helper the `/api/schools` route uses — match that route rather +than inventing a second shape, so the frontend's `School` type covers both. + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_places_api.py -q +``` +Expected: PASS (7 tests). + +- [ ] **Step 5: Rebuild the registry when the sitemap regenerates** + +In `/api/admin/regenerate-sitemap`, add `_place_registry` to the globals and +reset it, so a pipeline run refreshes places and sitemap together: + +```python + global _sitemaps, _place_registry + _place_registry = None + _sitemaps = build_sitemaps() +``` + +- [ ] **Step 6: Run the whole backend suite** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests -q +``` +Expected: PASS. Baseline before this plan was 70. + +- [ ] **Step 7: Commit** + +```bash +git add backend/app.py backend/tests/test_places_api.py +git commit -m "feat(places): /api/places registry and place detail endpoints" +``` + +--- + +### Task 4: Place sitemaps + +**Files:** +- Modify: `backend/app.py` (`build_sitemaps`) +- Test: `backend/tests/test_sitemap.py` (extend) + +**Interfaces:** +- Consumes: `get_place_registry()` (Task 3), `_url_element`, `_urlset`, + `SITEMAP_CHUNK_SIZE` (existing). +- Produces: `places-{n}.xml` and `outcodes-{n}.xml` children in the index. + +- [ ] **Step 1: Write the failing tests** + +Append to `backend/tests/test_sitemap.py`: + +```python +def _places_df() -> pd.DataFrame: + base = { + "local_authority": "Essex", "school_type": "Academy", + "phase": "Primary", "year": 202425, "ofsted_grade": 2.0, + "ofsted_date": None, "attainment_8_score": np.nan, + "town": "Brentwood", "postcode": "CM13 1AA", + } + return pd.DataFrame([ + {**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}", + "rwm_expected_pct": 60.0} + for i in range(6) + ]) + + +@pytest.fixture() +def place_sitemaps(monkeypatch) -> dict: + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _places_df) + monkeypatch.setattr(app_module, "_place_registry", None) + return app_module.build_sitemaps() + + +def test_place_children_are_listed_in_the_index(place_sitemaps): + index = place_sitemaps["sitemap.xml"] + assert "/sitemaps/places-1.xml" in index + assert "/sitemaps/outcodes-1.xml" in index + + +def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps): + xml = place_sitemaps["places-1.xml"] + assert "https://www.schoolcompare.co.uk/schools/brentwood" in xml + assert "https://www.schoolcompare.co.uk/schools/authority/essex" in xml + + +def test_outcode_urls_live_in_their_own_child(place_sitemaps): + assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"] + assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"] + + +def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps): + for name in ("places-1.xml", "outcodes-1.xml"): + assert "" not in place_sitemaps[name] + assert "" not in place_sitemaps[name] +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_sitemap.py -q +``` +Expected: FAIL — `KeyError: 'places-1.xml'`. + +- [ ] **Step 3: Emit the place children** + +In `backend/app.py`, add above `build_sitemaps`: + +```python +def _place_url(place) -> str: + """The canonical path for a place. Two namespaces, per the spec.""" + if place.kind == "authority": + return f"/schools/authority/{place.slug}" + if place.kind == "outcode": + return f"/schools/near/{place.slug}" + return f"/schools/{place.slug}" # town and locality share one + + +def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]: + return [ + _url_element(BASE_URL + _place_url(p)) + for p in sorted(get_place_registry().values(), + key=lambda p: (p.kind, p.slug)) + if p.kind in kinds + ] +``` + +Inside `build_sitemaps`, after the school children are added: + +```python + # Separate children per family: Search Console reports coverage per + # submitted sitemap, which is how the location layer's indexation is + # measured apart from the school pages'. + for label, kinds in (("places", ("town", "locality", "authority")), + ("outcodes", ("outcode",))): + rows = _place_sitemap_rows(kinds) + chunks = [rows[i:i + SITEMAP_CHUNK_SIZE] + for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]] + for n, chunk in enumerate(chunks, start=1): + children[f"{label}-{n}.xml"] = _urlset(chunk) +``` + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests/test_sitemap.py -q +``` +Expected: PASS. + +- [ ] **Step 5: Widen the Next sitemap proxy's allowlist** + +`nextjs-app/app/sitemaps/[...parts]/route.ts` validates the child name against +a regex that only knows the school families. Replace it: + +```typescript +const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/; +``` + +- [ ] **Step 6: Run the whole backend suite and commit** + +```bash +uv run --quiet --with-requirements requirements.txt --with pytest \ + --with "httpx==0.27.0" python -m pytest backend/tests -q +git add backend/app.py backend/tests/test_sitemap.py \ + "nextjs-app/app/sitemaps/[...parts]/route.ts" +git commit -m "feat(places): submit place and outcode sitemaps + +Separate children per family so Search Console reports the location layer's +indexation apart from the school pages'." +``` + +--- + +### Task 5: The place page + +**Files:** +- Create: `nextjs-app/lib/places.ts` +- Create: `nextjs-app/components/places/PlaceView.tsx` +- Create: `nextjs-app/components/places/PlaceView.module.css` +- Test: `nextjs-app/__tests__/components/PlaceView.test.tsx` + +**Interfaces:** +- Consumes: `/api/places/{kind}/{slug}` (Task 3), `absoluteUrl` from `@/lib/site`. +- Produces: + ```typescript + // lib/places.ts + export interface PlaceSummary { kind: string; slug: string; name: string; count: number } + export interface PlaceDetail { + place: PlaceSummary & { parent_authority: string | null }; + schools: School[]; + averages: { rwm_expected_pct: number | null; attainment_8_score: number | null }; + } + export function placeUrl(kind: string, slug: string, phase?: string): string + export async function fetchPlaces(): Promise + export async function fetchPlace(kind: string, slug: string, phase?: string): Promise + ``` + +- [ ] **Step 1: Write the failing tests** + +Create `nextjs-app/__tests__/components/PlaceView.test.tsx`: + +```tsx +import { render, screen } from '@testing-library/react'; +import { PlaceView } from '@/components/places/PlaceView'; +import type { PlaceDetail } from '@/lib/places'; + +const detail: PlaceDetail = { + place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 29, + parent_authority: 'Essex' }, + schools: [ + { urn: 1, school_name: 'Alpha Primary', rwm_expected_pct: 82, + ofsted_grade: 1 } as never, + { urn: 2, school_name: 'Beta Primary', rwm_expected_pct: 44, + ofsted_grade: 3 } as never, + ], + averages: { rwm_expected_pct: 63, attainment_8_score: null }, +}; + +describe('PlaceView', () => { + it('leads with an H1 that matches how the place is searched', () => { + render(); + expect(screen.getByRole('heading', { level: 1 })) + .toHaveTextContent(/primary schools in brentwood/i); + }); + + it('states the count so the page says something before the table', () => { + render(); + expect(screen.getByText(/29 schools/i)).toBeInTheDocument(); + }); + + it('compares the local average against England, which a list cannot', () => { + render(); + expect(screen.getByTestId('local-vs-england')).toHaveTextContent('63'); + expect(screen.getByTestId('local-vs-england')).toHaveTextContent('61'); + }); + + it('links every school in scope, which is what de-orphans them', () => { + render(); + expect(screen.getAllByRole('link', { name: /Primary$/ })).toHaveLength(2); + }); + + it('links to the parent authority so the place sits in a hierarchy', () => { + render(); + expect(screen.getByRole('link', { name: /Essex/i })) + .toHaveAttribute('href', '/schools/authority/essex'); + }); + + it('shows the Ofsted distribution, not just a count of Outstanding', () => { + render(); + expect(screen.getByTestId('ofsted-distribution')).toBeInTheDocument(); + }); + + it('links to neighbouring places so the page is not a dead end', () => { + render(); + expect(screen.getByRole('link', { name: /Romford/ })) + .toHaveAttribute('href', '/schools/romford'); + }); + + it('says nothing about an average it does not have', () => { + render(); + expect(screen.queryByTestId('local-vs-england')).not.toBeInTheDocument(); + }); +}); +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `cd nextjs-app && npm test -- __tests__/components/PlaceView.test.tsx` +Expected: FAIL — `Cannot find module '@/components/places/PlaceView'`. + +- [ ] **Step 3: Write the client** + +Create `nextjs-app/lib/places.ts`: + +```typescript +/** + * Client for the places API. + * + * Two namespaces, matching the backend: towns and localities share + * /schools/[place]; authorities take /schools/authority/[la]. 67 town names + * collide with an authority name and neither set contains the other, so one + * namespace would publish near-duplicate pages. + */ +import type { School } from '@/lib/types'; + +export interface PlaceSummary { + kind: string; + slug: string; + name: string; + count: number; +} + +export interface PlaceDetail { + place: PlaceSummary & { parent_authority: string | null }; + schools: School[]; + averages: { + rwm_expected_pct: number | null; + attainment_8_score: number | null; + }; +} + +export function placeUrl(kind: string, slug: string, phase?: string): string { + const base = + kind === 'authority' ? `/schools/authority/${slug}` + : kind === 'outcode' ? `/schools/near/${slug}` + : `/schools/${slug}`; + return phase ? `${base}/${phase}` : base; +} + +const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL + || 'http://localhost:8000/api'; + +export async function fetchPlaces(): Promise { + const res = await fetch(`${API}/places`, { next: { revalidate: 604800 } }); + if (!res.ok) return []; + return (await res.json()).places ?? []; +} + +export async function fetchPlace( + kind: string, slug: string, phase?: string, +): Promise { + const q = phase ? `?phase=${encodeURIComponent(phase)}` : ''; + const res = await fetch(`${API}/places/${kind}/${slug}${q}`, + { next: { revalidate: 604800 } }); + if (!res.ok) return null; + return res.json(); +} +``` + +- [ ] **Step 4: Write the component** + +Create `nextjs-app/components/places/PlaceView.tsx`: + +```tsx +/** + * One place page, shared by all four families. + * + * They differ in what fills the registry, not in what the page shows, so a + * second component would be a second place to forget the same change. + * + * The local-versus-England comparison is the reason this page is not a list: + * it is the one number a parent cannot get by reading the schools one by one, + * and it is what keeps the page from reading as a name dropped into a + * template. + */ +import Link from 'next/link'; +import type { PlaceDetail, PlaceSummary } from '@/lib/places'; +import { placeUrl } from '@/lib/places'; +import { schoolUrl } from '@/lib/utils'; +import styles from './PlaceView.module.css'; + +interface Props { + detail: PlaceDetail; + phase?: 'primary' | 'secondary'; + englandAverage: number | null; + /** Nearby places, so the page links onward instead of dead-ending. */ + neighbours: PlaceSummary[]; +} + +// Ofsted grades in the order they are reported. +const OFSTED_LABELS: Array<[number, string]> = [ + [1, 'Outstanding'], [2, 'Good'], + [3, 'Requires improvement'], [4, 'Inadequate'], +]; +const OUTSTANDING = 1; + +export function PlaceView({ detail, phase, englandAverage, neighbours }: Props) { + const { place, schools, averages } = detail; + const metric = phase === 'secondary' ? 'attainment_8_score' : 'rwm_expected_pct'; + const local = averages[metric]; + const outstanding = schools.filter((s) => s.ofsted_grade === OUTSTANDING).length; + const phaseWord = phase === 'secondary' ? 'Secondary schools' + : phase === 'primary' ? 'Primary schools' : 'Schools'; + + return ( +
+

{phaseWord} in {place.name}

+ +

+ {place.count} schools + {outstanding > 0 && <>, {outstanding} rated Outstanding} + {place.parent_authority && ( + <> · + {place.parent_authority} + + )} +

+ + {local != null && englandAverage != null && ( +

+ {place.name} averages {Math.round(local)} against{' '} + {Math.round(englandAverage)} across England. +

+ )} + +
    + {OFSTED_LABELS.map(([grade, label]) => { + const n = schools.filter((s) => s.ofsted_grade === grade).length; + return n === 0 ? null : ( +
  • {label}: {n}
  • + ); + })} +
+ +
+ + + + + + {schools.map((s) => ( + + + + + ))} + +
School{phase === 'secondary' ? 'Attainment 8' : 'RWM expected'}
{s.school_name} + {s[metric] == null ? '—' : Math.round(Number(s[metric]))} +
+
+ + {neighbours.length > 0 && ( + + )} +
+ ); +} +``` + +Create `nextjs-app/components/places/PlaceView.module.css` following the +conventions in `components/RankingsView.module.css` — read that file and match +its token usage rather than introducing new colours. The table must sit in an +`overflow-x: auto` wrapper so the page body never scrolls sideways. + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `cd nextjs-app && npm test -- __tests__/components/PlaceView.test.tsx` +Expected: PASS (6 tests). + +- [ ] **Step 6: Commit** + +```bash +git add nextjs-app/lib/places.ts nextjs-app/components/places/ \ + nextjs-app/__tests__/components/PlaceView.test.tsx +git commit -m "feat(places): place page client and view component" +``` + +--- + +### Task 6: The routes + +**Files:** +- Create: `nextjs-app/app/schools/[place]/page.tsx` +- Create: `nextjs-app/app/schools/[place]/[phase]/page.tsx` +- Create: `nextjs-app/app/schools/authority/[la]/page.tsx` +- Create: `nextjs-app/app/schools/near/[outcode]/page.tsx` +- Test: `nextjs-app/__tests__/app/placeMetadata.test.ts` + +**Interfaces:** +- Consumes: `fetchPlace`, `placeUrl`, `PlaceView` (Task 5), `absoluteUrl` + from `@/lib/site`, `fetchNationalAverages` from `@/lib/api`. +- Produces: the four route families. Nothing later depends on them. + +**Route resolution order matters.** `app/schools/authority/[la]` and +`app/schools/near/[outcode]` are static segments and Next matches them before +the dynamic `app/schools/[place]`, so an authority is never captured by the +place route. Do not rename `authority` or `near` to a bracketed segment. + +- [ ] **Step 1: Write the failing metadata tests** + +Create `nextjs-app/__tests__/app/placeMetadata.test.ts`: + +```typescript +import { generateMetadata as placeMeta } from '@/app/schools/[place]/page'; + +jest.mock('@/lib/places', () => ({ + ...jest.requireActual('@/lib/places'), + fetchPlace: jest.fn(async (kind: string, slug: string) => + slug === 'atlantis' ? null : ({ + place: { kind, slug, name: 'Brentwood', count: 29, + parent_authority: 'Essex' }, + schools: [], averages: { rwm_expected_pct: 63, attainment_8_score: null }, + })), +})); + +describe('place page metadata', () => { + it('titles the page the way the place is searched', async () => { + const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) }); + expect(m.title).toMatch(/schools in brentwood/i); + }); + + it('canonicalises to its own path on the www host', async () => { + const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) }); + expect(m.alternates?.canonical) + .toBe('https://www.schoolcompare.co.uk/schools/brentwood'); + }); + + it('an unknown place gets a not-found title rather than inventing one', async () => { + const m = await placeMeta({ params: Promise.resolve({ place: 'atlantis' }) }); + expect(m.title).toMatch(/not found/i); + }); +}); +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `cd nextjs-app && npm test -- __tests__/app/placeMetadata.test.ts` +Expected: FAIL — module not found. + +- [ ] **Step 3: Write the place route** + +Create `nextjs-app/app/schools/[place]/page.tsx`: + +```tsx +/** + * Town and locality pages. + * + * A place below the five-school threshold is not in the registry, so + * fetchPlace returns null and the request 301s to the parent authority + * rather than rendering a page with nothing to say. + */ +import { notFound, redirect } from 'next/navigation'; +import type { Metadata } from 'next'; +import { fetchPlace, fetchPlaces } from '@/lib/places'; +import { fetchNationalAverages } from '@/lib/api'; +import { PlaceView } from '@/components/places/PlaceView'; +import { absoluteUrl } from '@/lib/site'; + +interface Props { params: Promise<{ place: string }> } + +// ISR: place aggregates change only when the pipeline runs. +export const revalidate = 604800; +export const dynamicParams = true; + +export async function generateStaticParams(): Promise> { + // Off by default: ~2,000 place routes cannot be built in CI on every deploy. + // Matches the PRERENDER_SCHOOLS gate on the school route. + if (process.env.PRERENDER_PLACES !== '1') return []; + try { + return (await fetchPlaces()) + .filter((p) => p.kind === 'town' || p.kind === 'locality') + .map((p) => ({ place: p.slug })); + } catch (error) { + console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error); + return []; + } +} + +async function resolve(slug: string) { + return (await fetchPlace('town', slug)) ?? (await fetchPlace('locality', slug)); +} + +/** Other towns in the same authority — the cheapest honest definition of + * "nearby", and enough to stop each place page being a dead end. */ +async function neighboursOf(detail: { place: { slug: string; parent_authority: string | null } }) { + if (!detail.place.parent_authority) return []; + const all = await fetchPlaces(); + return all + .filter((p) => p.kind === 'town' && p.slug !== detail.place.slug) + .slice(0, 12); +} + +export async function generateMetadata({ params }: Props): Promise { + const { place: slug } = await params; + const detail = await resolve(slug); + if (!detail) return { title: 'Place Not Found' }; + + const { name, count } = detail.place; + return { + title: `Schools in ${name} — Compare ${count} Schools | schoolcompare`, + description: + `Every school in ${name} ranked by SATs and GCSE results, with Ofsted grades, ` + + `the local average against England, and how close you had to live to get a place.`, + alternates: { canonical: absoluteUrl(`/schools/${slug}`) }, + }; +} + +export default async function PlacePage({ params }: Props) { + const { place: slug } = await params; + const detail = await resolve(slug); + if (!detail) notFound(); + + // Global constraint: no page without a local average. A place with too few + // schools carrying results has nothing to say that a list does not, so it + // defers to its authority rather than publishing a thin page. + if (detail.averages.rwm_expected_pct == null + && detail.averages.attainment_8_score == null) { + if (detail.place.parent_authority) { + redirect(`/schools/authority/${detail.place.parent_authority + .toLowerCase().replace(/\s+/g, '-')}`); + } + notFound(); + } + + const national = await fetchNationalAverages().catch(() => null); + // NationalAverages is nested by phase — { primary: {...}, secondary: {...} } + // — not flat. Reading it flat silently yields undefined and the page renders + // with no comparison, which is the one thing that makes it not a list. + return ( + + ); +} +``` + +- [ ] **Step 4: Write the phase, authority and outcode routes** + +Create `nextjs-app/app/schools/[place]/[phase]/page.tsx`. It is the same +shape, with two differences: it validates the phase segment, and it passes +that phase to both `fetchPlace` and `PlaceView`. + +```tsx +import { notFound } from 'next/navigation'; +import type { Metadata } from 'next'; +import { fetchPlace } from '@/lib/places'; +import { fetchNationalAverages } from '@/lib/api'; +import { PlaceView } from '@/components/places/PlaceView'; +import { absoluteUrl } from '@/lib/site'; + +interface Props { params: Promise<{ place: string; phase: string }> } + +export const revalidate = 604800; +export const dynamicParams = true; + +const PHASES = ['primary', 'secondary'] as const; +type Phase = (typeof PHASES)[number]; + +const isPhase = (v: string): v is Phase => (PHASES as readonly string[]).includes(v); + +async function resolve(slug: string, phase: Phase) { + return (await fetchPlace('town', slug, phase)) + ?? (await fetchPlace('locality', slug, phase)); +} + +export async function generateMetadata({ params }: Props): Promise { + const { place: slug, phase } = await params; + if (!isPhase(phase)) return { title: 'Place Not Found' }; + const detail = await resolve(slug, phase); + // A place with no schools of this phase has no page — the per-phase + // threshold, not an error. + if (!detail || detail.schools.length === 0) return { title: 'Place Not Found' }; + + const word = phase === 'secondary' ? 'Secondary' : 'Primary'; + const { name } = detail.place; + return { + title: `${word} Schools in ${name} — Ranked | schoolcompare`, + description: + `Every ${phase} school in ${name} ranked by results, with Ofsted grades and ` + + `the local average against England.`, + alternates: { canonical: absoluteUrl(`/schools/${slug}/${phase}`) }, + }; +} + +export default async function PlacePhasePage({ params }: Props) { + const { place: slug, phase } = await params; + if (!isPhase(phase)) notFound(); + const detail = await resolve(slug, phase); + if (!detail || detail.schools.length === 0) notFound(); + + const national = await fetchNationalAverages().catch(() => null); + const englandAverage = phase === 'secondary' + ? national?.secondary?.attainment_8_score ?? null + : national?.primary?.rwm_expected_pct ?? null; + + return ; +} +``` + +Create `nextjs-app/app/schools/authority/[la]/page.tsx`: + +```tsx +/** + * Local authority pages. + * + * A separate namespace from /schools/[place] because 67 town names collide + * with an authority name and neither set contains the other — Bedford the + * town holds 104 schools, Bedford the authority 86, because postal towns + * cross authority boundaries. The title says "Local Authority" so a reader + * landing on both knows which set each covers. + */ +import { notFound } from 'next/navigation'; +import type { Metadata } from 'next'; +import { fetchPlace, fetchPlaces } from '@/lib/places'; +import { fetchNationalAverages } from '@/lib/api'; +import { PlaceView } from '@/components/places/PlaceView'; +import { absoluteUrl } from '@/lib/site'; + +interface Props { params: Promise<{ la: string }> } + +export const revalidate = 604800; +export const dynamicParams = true; + +export async function generateStaticParams(): Promise> { + // Gated like every other prerender in this app. There are only ~154 + // authorities, but "few enough to always build" still means the API must be + // reachable at build time, and in CI it is not — the build fails with + // ECONNREFUSED rather than degrading. The catch is the same fallback the + // school route uses. + if (process.env.PRERENDER_PLACES !== '1') return []; + try { + return (await fetchPlaces()) + .filter((p) => p.kind === 'authority') + .map((p) => ({ la: p.slug })); + } catch (error) { + console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error); + return []; + } +} + +export async function generateMetadata({ params }: Props): Promise { + const { la } = await params; + const detail = await fetchPlace('authority', la); + if (!detail) return { title: 'Place Not Found' }; + + const { name, count } = detail.place; + return { + title: `Schools in ${name} — Local Authority | schoolcompare`, + description: + `All ${count} schools in the ${name} local authority, ranked by SATs and GCSE ` + + `results, with Ofsted grades and the authority average against England.`, + alternates: { canonical: absoluteUrl(`/schools/authority/${la}`) }, + }; +} + +export default async function AuthorityPage({ params }: Props) { + const { la } = await params; + const detail = await fetchPlace('authority', la); + if (!detail) notFound(); + + const national = await fetchNationalAverages().catch(() => null); + return ( + + ); +} +``` + +Create `nextjs-app/app/schools/near/[outcode]/page.tsx`: + +```tsx +/** + * Postcode district pages. + * + * No phase variants: nobody searches "primary schools in SW11", so the + * variants would be pages without demand. These exist to catch + * "schools near " and to give London districts a geographic page + * where the GIAS town field cannot. + */ +import { notFound } from 'next/navigation'; +import type { Metadata } from 'next'; +import { fetchPlace, fetchPlaces } from '@/lib/places'; +import { fetchNationalAverages } from '@/lib/api'; +import { PlaceView } from '@/components/places/PlaceView'; +import { absoluteUrl } from '@/lib/site'; + +interface Props { params: Promise<{ outcode: string }> } + +export const revalidate = 604800; +export const dynamicParams = true; + +export async function generateStaticParams(): Promise> { + // 1,760 of these; same CI budget argument as the town routes. + if (process.env.PRERENDER_PLACES !== '1') return []; + try { + return (await fetchPlaces()) + .filter((p) => p.kind === 'outcode') + .map((p) => ({ outcode: p.slug })); + } catch (error) { + console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error); + return []; + } +} + +export async function generateMetadata({ params }: Props): Promise { + const { outcode } = await params; + const detail = await fetchPlace('outcode', outcode); + if (!detail) return { title: 'Place Not Found' }; + + const { name, count } = detail.place; + return { + title: `Schools near ${name} | schoolcompare`, + description: + `${count} schools in the ${name} postcode district, ranked by results, with ` + + `Ofsted grades and how close you had to live to get a place.`, + alternates: { canonical: absoluteUrl(`/schools/near/${outcode}`) }, + }; +} + +export default async function OutcodePage({ params }: Props) { + const { outcode } = await params; + const detail = await fetchPlace('outcode', outcode); + if (!detail) notFound(); + + const national = await fetchNationalAverages().catch(() => null); + return ( + + ); +} +``` + +- [ ] **Step 5: Run the tests, typecheck and build** + +Run: +```bash +cd nextjs-app && npm test && npx tsc --noEmit && npx next build +``` +Expected: Jest green; typecheck clean; the build lists `/schools/[place]`, +`/schools/[place]/[phase]`, `/schools/authority/[la]` and +`/schools/near/[outcode]` as dynamic routes. + +- [ ] **Step 6: Commit** + +```bash +git add nextjs-app/app/schools nextjs-app/__tests__/app/placeMetadata.test.ts +git commit -m "feat(places): town, locality, authority and outcode routes" +``` + +--- + +### Task 7: Structured data and the e2e gate + +**Files:** +- Modify: `nextjs-app/components/places/PlaceView.tsx` (JSON-LD) +- Modify: `e2e/tests/journeys.spec.ts` + +**Interfaces:** +- Consumes: everything above. +- Produces: nothing. + +- [ ] **Step 1: Add ItemList and BreadcrumbList to the place page** + +In `PlaceView.tsx`, build the structured data above the returned markup and +render it as the first child: + +```tsx + const jsonLd = { + '@context': 'https://schema.org', + '@graph': [ + { + '@type': 'ItemList', + name: `${phaseWord} in ${place.name}`, + numberOfItems: schools.length, + itemListElement: schools.slice(0, 20).map((s, i) => ({ + '@type': 'ListItem', + position: i + 1, + url: `https://www.schoolcompare.co.uk${schoolUrl(s.urn, s.school_name)}`, + name: s.school_name, + })), + }, + { + '@type': 'BreadcrumbList', + itemListElement: [ + { '@type': 'ListItem', position: 1, name: 'Schools', + item: 'https://www.schoolcompare.co.uk/' }, + { '@type': 'ListItem', position: 2, name: place.name }, + ], + }, + ], + }; +``` + +```tsx +