Sitemap regeneration failed on staging. 'richmond' in the curated locality list collides with the GIAS town Richmond in North Yorkshire (37 schools), the registry raised, and the admin endpoint 500d — taking down sitemap generation for all 25,000 school pages over one bad row of curated data. The guard now skips the colliding locality and logs an error. Skipping still achieves what the guard was for — a locality never silently shadows a town — without letting curated data break the site. That matters beyond this bug: GIAS town names change with no code change here, so a raise could fire spontaneously in production later. Also removes four localities that were London boroughs rather than districts. Hackney, Islington, Greenwich and Ealing are local authorities with 104, 72, 108 and 115 schools and already have authority pages; a locality defined by two or three outcodes would have been a partial near-duplicate of one — the thin-content failure the two-namespace design exists to avoid. A test now guards the whole borough list. Validated against the live corpus: 15 localities, no town collisions, no authority duplicates, all 15 clear the threshold. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
232 lines
8.9 KiB
Python
232 lines
8.9 KiB
Python
"""Tests for the place registry (spec 2026-08-21).
|
|
|
|
The registry is built from the in-memory school DataFrame, so these build a
|
|
small frame directly rather than touching a database.
|
|
"""
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
import pytest
|
|
|
|
from backend.places import MIN_SCHOOLS, build_place_registry
|
|
|
|
|
|
def _df(rows: list[dict]) -> pd.DataFrame:
|
|
base = {
|
|
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
|
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
|
|
"phase": "Primary", "postcode": "AA1 1AA",
|
|
}
|
|
return pd.DataFrame([{**base, **r} for r in rows])
|
|
|
|
|
|
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
|
"""`start` offsets the URNs so two calls can describe different schools —
|
|
the Bedford case needs two authorities' worth of distinct URNs in one
|
|
town."""
|
|
return [
|
|
{"urn": start + i, "school_name": f"{town} School {i}",
|
|
"town": town, "local_authority": la, **kw}
|
|
for i in range(n)
|
|
]
|
|
|
|
|
|
def test_town_clearing_the_threshold_is_published():
|
|
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
|
assert "town:brentwood" in reg
|
|
assert reg["town:brentwood"].name == "Brentwood"
|
|
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
|
|
|
|
|
|
def test_town_below_the_threshold_is_not_published():
|
|
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
|
|
assert "town:crosby" not in reg
|
|
|
|
|
|
def test_a_town_below_threshold_still_names_its_authority():
|
|
# The route layer needs somewhere to 301 to.
|
|
reg = build_place_registry(_df(
|
|
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
|
|
assert "authority:sefton" in reg
|
|
|
|
|
|
def test_town_and_authority_of_the_same_name_are_separate_places():
|
|
# 67 real collisions. Neither set contains the other: Bedford the town has
|
|
# 104 schools, Bedford the authority 86, because postal towns cross
|
|
# authority boundaries.
|
|
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
|
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
|
reg = build_place_registry(_df(rows))
|
|
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
|
assert set(town.urns) != set(authority.urns)
|
|
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
|
|
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
|
|
|
|
|
|
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
|
|
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
|
|
for r in rows:
|
|
r["rwm_expected_pct"] = np.nan
|
|
r["ofsted_grade"] = np.nan
|
|
reg = build_place_registry(_df(rows))
|
|
assert "town:ghosttown" not in reg
|
|
|
|
|
|
def test_blank_town_is_ignored():
|
|
rows = _town(MIN_SCHOOLS, "", "Essex")
|
|
reg = build_place_registry(_df(rows))
|
|
assert not any(k.startswith("town:") for k in reg)
|
|
|
|
|
|
def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
|
rows = []
|
|
for year in (202324, 202425):
|
|
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
|
reg = build_place_registry(_df(rows))
|
|
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
|
|
|
|
|
def test_locality_groups_schools_by_outcode(monkeypatch):
|
|
# The GIAS town field collapses 1,819 London schools into "London", so a
|
|
# locality is defined by its postcode districts instead.
|
|
from backend import localities
|
|
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
|
{"battersea": ("Battersea", ("SW11",))})
|
|
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
|
for r in rows:
|
|
r["postcode"] = "SW11 2AA"
|
|
reg = build_place_registry(_df(rows))
|
|
assert reg["locality:battersea"].name == "Battersea"
|
|
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
|
|
|
|
|
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
|
from backend import localities
|
|
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
|
{"nowhere": ("Nowhere", ("ZZ99",))})
|
|
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
|
assert "locality:nowhere" not in reg
|
|
|
|
|
|
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch, caplog):
|
|
"""A colliding locality is skipped loudly, and the town survives.
|
|
|
|
This used to raise, which took down sitemap generation for all 25,000
|
|
school pages the first time a curated slug met a real GIAS town. Curated
|
|
data must not be able to break the site — and GIAS town names change with
|
|
no code change at all, so the raise could fire spontaneously.
|
|
"""
|
|
import logging
|
|
|
|
from backend import localities
|
|
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
|
{"brentwood": ("Brentwood", ("CM13",))})
|
|
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
|
for r in rows:
|
|
r["postcode"] = "CM13 1AA"
|
|
|
|
with caplog.at_level(logging.ERROR):
|
|
reg = build_place_registry(_df(rows))
|
|
|
|
assert "locality:brentwood" not in reg # skipped
|
|
assert "town:brentwood" in reg # the town is untouched
|
|
assert "brentwood" in caplog.text # and it was loud about it
|
|
|
|
|
|
def test_a_locality_collision_does_not_break_the_rest_of_the_registry(monkeypatch):
|
|
# The whole point of skipping rather than raising.
|
|
from backend import localities
|
|
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
|
{"brentwood": ("Brentwood", ("CM13",))})
|
|
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
|
for r in rows:
|
|
r["postcode"] = "CM13 1AA"
|
|
reg = build_place_registry(_df(rows))
|
|
assert "authority:essex" in reg
|
|
assert "outcode:cm13" in reg
|
|
|
|
|
|
def test_outcode_places_are_built_from_postcodes():
|
|
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
|
for r in rows:
|
|
r["postcode"] = "CM13 1AA"
|
|
reg = build_place_registry(_df(rows))
|
|
assert reg["outcode:cm13"].name == "CM13"
|
|
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
|
|
|
|
|
def test_malformed_postcodes_do_not_create_places():
|
|
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
|
for r in rows:
|
|
r["postcode"] = "not a postcode"
|
|
reg = build_place_registry(_df(rows))
|
|
assert not any(k.startswith("outcode:") for k in reg)
|
|
|
|
|
|
def test_every_curated_locality_is_structurally_valid():
|
|
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
|
import re
|
|
from backend.localities import LOCALITY_OUTCODES
|
|
|
|
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
|
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
|
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
|
assert name.strip() == name and name, slug
|
|
assert outcodes, f"{slug} has no outcodes"
|
|
for oc in outcodes:
|
|
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
|
|
|
|
|
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
|
"""Two copies with no drift guard is worse than one copy.
|
|
|
|
backend/localities.py is canonical because the backend image does not
|
|
contain pipeline/. The seed exists so the warehouse can join on the same
|
|
definitions, and this is what stops the two diverging — the same
|
|
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
|
"""
|
|
import csv
|
|
from pathlib import Path
|
|
|
|
from backend.localities import LOCALITY_OUTCODES
|
|
|
|
seed_path = (Path(__file__).resolve().parents[2]
|
|
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
|
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
|
|
|
seed = {
|
|
row["locality_slug"]: (row["locality_name"],
|
|
tuple(row["outcodes"].split("|")))
|
|
for row in csv.DictReader(seed_path.open())
|
|
}
|
|
assert seed == LOCALITY_OUTCODES
|
|
|
|
|
|
def test_no_curated_locality_names_a_london_borough():
|
|
"""Boroughs are authorities and already have a page.
|
|
|
|
A locality defined by two or three outcodes inside a borough would be a
|
|
partial, near-duplicate subset of that authority page — the exact
|
|
thin-content failure the two-namespace design exists to avoid. Hackney,
|
|
Islington, Greenwich and Ealing were all in the first draft.
|
|
|
|
Hardcoded rather than read from the corpus because this must fail in CI,
|
|
where there is no database.
|
|
"""
|
|
from backend.localities import LOCALITY_OUTCODES
|
|
|
|
boroughs = {
|
|
"barking-and-dagenham", "barnet", "bexley", "brent", "bromley",
|
|
"camden", "croydon", "ealing", "enfield", "greenwich", "hackney",
|
|
"hammersmith-and-fulham", "haringey", "harrow", "havering",
|
|
"hillingdon", "hounslow", "islington", "kensington-and-chelsea",
|
|
"kingston-upon-thames", "lambeth", "lewisham", "merton", "newham",
|
|
"redbridge", "richmond-upon-thames", "southwark", "sutton",
|
|
"tower-hamlets", "waltham-forest", "wandsworth", "westminster",
|
|
}
|
|
named = boroughs & set(LOCALITY_OUTCODES)
|
|
assert not named, (
|
|
f"these are boroughs, not districts: {sorted(named)} - they already "
|
|
"have an authority page covering every school"
|
|
)
|