Files
school_compare/backend/tests/test_places.py
T
TudorandClaude Opus 5 de853b90b3 feat(places): London localities and postcode districts
The GIAS town field puts 1,819 London schools under the single value
'London', so it cannot answer 'schools in Battersea' — a query that appears in
the baseline. No single field can: parliamentary constituency gives Battersea
but not Canary Wharf, admin_ward gives Canary Wharf but not Battersea, and
neither gives Clapham or Shoreditch. So a locality is curated, defined by the
postcode districts it covers, which needs no new ingestion.

A locality may not shadow a published town: the registry raises rather than
silently costing a page that carries real demand. One below the threshold is
logged rather than raising, because a locality can legitimately be too small.

The pipeline seed mirrors the module, with a test guarding the drift — the
same arrangement gias_codes has, and for the same reason: the backend image
does not contain pipeline/.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-21 18:11:56 +01:00

177 lines
6.6 KiB
Python

"""Tests for the place registry (spec 2026-08-21).
The registry is built from the in-memory school DataFrame, so these build a
small frame directly rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
from backend.places import MIN_SCHOOLS, build_place_registry
def _df(rows: list[dict]) -> pd.DataFrame:
base = {
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
"phase": "Primary", "postcode": "AA1 1AA",
}
return pd.DataFrame([{**base, **r} for r in rows])
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
"""`start` offsets the URNs so two calls can describe different schools —
the Bedford case needs two authorities' worth of distinct URNs in one
town."""
return [
{"urn": start + i, "school_name": f"{town} School {i}",
"town": town, "local_authority": la, **kw}
for i in range(n)
]
def test_town_clearing_the_threshold_is_published():
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
assert "town:brentwood" in reg
assert reg["town:brentwood"].name == "Brentwood"
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
def test_town_below_the_threshold_is_not_published():
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
assert "town:crosby" not in reg
def test_a_town_below_threshold_still_names_its_authority():
# The route layer needs somewhere to 301 to.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
assert "authority:sefton" in reg
def test_town_and_authority_of_the_same_name_are_separate_places():
# 67 real collisions. Neither set contains the other: Bedford the town has
# 104 schools, Bedford the authority 86, because postal towns cross
# authority boundaries.
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
reg = build_place_registry(_df(rows))
town, authority = reg["town:bedford"], reg["authority:bedford"]
assert set(town.urns) != set(authority.urns)
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
for r in rows:
r["rwm_expected_pct"] = np.nan
r["ofsted_grade"] = np.nan
reg = build_place_registry(_df(rows))
assert "town:ghosttown" not in reg
def test_blank_town_is_ignored():
rows = _town(MIN_SCHOOLS, "", "Essex")
reg = build_place_registry(_df(rows))
assert not any(k.startswith("town:") for k in reg)
def test_a_school_is_counted_once_even_with_several_years_of_rows():
rows = []
for year in (202324, 202425):
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
reg = build_place_registry(_df(rows))
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
def test_locality_groups_schools_by_outcode(monkeypatch):
# The GIAS town field collapses 1,819 London schools into "London", so a
# locality is defined by its postcode districts instead.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"battersea": ("Battersea", ("SW11",))})
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
for r in rows:
r["postcode"] = "SW11 2AA"
reg = build_place_registry(_df(rows))
assert reg["locality:battersea"].name == "Battersea"
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
def test_locality_below_the_threshold_is_not_published(monkeypatch):
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"nowhere": ("Nowhere", ("ZZ99",))})
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
assert "locality:nowhere" not in reg
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch):
# Silently shadowing a town would lose a page carrying real demand.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"brentwood": ("Brentwood", ("CM13",))})
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
with pytest.raises(ValueError, match="brentwood"):
build_place_registry(_df(rows))
def test_outcode_places_are_built_from_postcodes():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
reg = build_place_registry(_df(rows))
assert reg["outcode:cm13"].name == "CM13"
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
def test_malformed_postcodes_do_not_create_places():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "not a postcode"
reg = build_place_registry(_df(rows))
assert not any(k.startswith("outcode:") for k in reg)
def test_every_curated_locality_is_structurally_valid():
# Guards the hand-maintained file: real slug, real name, real outcodes.
import re
from backend.localities import LOCALITY_OUTCODES
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
assert name.strip() == name and name, slug
assert outcodes, f"{slug} has no outcodes"
for oc in outcodes:
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
def test_the_pipeline_seed_mirrors_the_canonical_module():
"""Two copies with no drift guard is worse than one copy.
backend/localities.py is canonical because the backend image does not
contain pipeline/. The seed exists so the warehouse can join on the same
definitions, and this is what stops the two diverging — the same
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
"""
import csv
from pathlib import Path
from backend.localities import LOCALITY_OUTCODES
seed_path = (Path(__file__).resolve().parents[2]
/ "pipeline/transform/seeds/locality_outcodes.csv")
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
seed = {
row["locality_slug"]: (row["locality_name"],
tuple(row["outcodes"].split("|")))
for row in csv.DictReader(seed_path.open())
}
assert seed == LOCALITY_OUTCODES