"""Tests for the place registry (spec 2026-08-21). The registry is built from the in-memory school DataFrame, so these build a small frame directly rather than touching a database. """ import numpy as np import pandas as pd import pytest from backend.places import MIN_SCHOOLS, build_place_registry def _df(rows: list[dict]) -> pd.DataFrame: base = { "year": 202425, "ofsted_grade": 2.0, "ofsted_date": None, "rwm_expected_pct": 60.0, "attainment_8_score": np.nan, "phase": "Primary", "postcode": "AA1 1AA", } return pd.DataFrame([{**base, **r} for r in rows]) def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]: """`start` offsets the URNs so two calls can describe different schools — the Bedford case needs two authorities' worth of distinct URNs in one town.""" return [ {"urn": start + i, "school_name": f"{town} School {i}", "town": town, "local_authority": la, **kw} for i in range(n) ] def test_town_clearing_the_threshold_is_published(): reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex"))) assert "town:brentwood" in reg assert reg["town:brentwood"].name == "Brentwood" assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS def test_town_below_the_threshold_is_not_published(): reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton"))) assert "town:crosby" not in reg def test_a_town_below_threshold_still_names_its_authority(): # The route layer needs somewhere to 301 to. reg = build_place_registry(_df( _town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton"))) assert "authority:sefton" in reg def test_town_and_authority_of_the_same_name_are_separate_places(): # 67 real collisions. Neither set contains the other: Bedford the town has # 104 schools, Bedford the authority 86, because postal towns cross # authority boundaries. rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford") + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000)) reg = build_place_registry(_df(rows)) town, authority = reg["town:bedford"], reg["authority:bedford"] assert set(town.urns) != set(authority.urns) assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools assert len(authority.urns) == MIN_SCHOOLS # only this authority's def test_schools_without_publishable_data_do_not_count_toward_the_threshold(): rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere") for r in rows: r["rwm_expected_pct"] = np.nan r["ofsted_grade"] = np.nan reg = build_place_registry(_df(rows)) assert "town:ghosttown" not in reg def test_blank_town_is_ignored(): rows = _town(MIN_SCHOOLS, "", "Essex") reg = build_place_registry(_df(rows)) assert not any(k.startswith("town:") for k in reg) def test_a_school_is_counted_once_even_with_several_years_of_rows(): rows = [] for year in (202324, 202425): rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")] reg = build_place_registry(_df(rows)) assert len(reg["town:beccles"].urns) == MIN_SCHOOLS def test_locality_groups_schools_by_outcode(monkeypatch): # The GIAS town field collapses 1,819 London schools into "London", so a # locality is defined by its postcode districts instead. from backend import localities monkeypatch.setattr(localities, "LOCALITY_OUTCODES", {"battersea": ("Battersea", ("SW11",))}) rows = _town(MIN_SCHOOLS, "London", "Wandsworth") for r in rows: r["postcode"] = "SW11 2AA" reg = build_place_registry(_df(rows)) assert reg["locality:battersea"].name == "Battersea" assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS def test_locality_below_the_threshold_is_not_published(monkeypatch): from backend import localities monkeypatch.setattr(localities, "LOCALITY_OUTCODES", {"nowhere": ("Nowhere", ("ZZ99",))}) reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth"))) assert "locality:nowhere" not in reg def test_a_locality_may_not_shadow_a_viable_town(monkeypatch): # Silently shadowing a town would lose a page carrying real demand. from backend import localities monkeypatch.setattr(localities, "LOCALITY_OUTCODES", {"brentwood": ("Brentwood", ("CM13",))}) rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") for r in rows: r["postcode"] = "CM13 1AA" with pytest.raises(ValueError, match="brentwood"): build_place_registry(_df(rows)) def test_outcode_places_are_built_from_postcodes(): rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") for r in rows: r["postcode"] = "CM13 1AA" reg = build_place_registry(_df(rows)) assert reg["outcode:cm13"].name == "CM13" assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS def test_malformed_postcodes_do_not_create_places(): rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") for r in rows: r["postcode"] = "not a postcode" reg = build_place_registry(_df(rows)) assert not any(k.startswith("outcode:") for k in reg) def test_every_curated_locality_is_structurally_valid(): # Guards the hand-maintained file: real slug, real name, real outcodes. import re from backend.localities import LOCALITY_OUTCODES assert LOCALITY_OUTCODES, "the curated locality list must not be empty" for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): assert re.fullmatch(r"[a-z0-9-]+", slug), slug assert name.strip() == name and name, slug assert outcodes, f"{slug} has no outcodes" for oc in outcodes: assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc) def test_the_pipeline_seed_mirrors_the_canonical_module(): """Two copies with no drift guard is worse than one copy. backend/localities.py is canonical because the backend image does not contain pipeline/. The seed exists so the warehouse can join on the same definitions, and this is what stops the two diverging — the same arrangement assert_gias_code_names_match_seed.sql gives gias_codes. """ import csv from pathlib import Path from backend.localities import LOCALITY_OUTCODES seed_path = (Path(__file__).resolve().parents[2] / "pipeline/transform/seeds/locality_outcodes.csv") assert seed_path.exists(), f"missing seed mirror at {seed_path}" seed = { row["locality_slug"]: (row["locality_name"], tuple(row["outcodes"].split("|"))) for row in csv.DictReader(seed_path.open()) } assert seed == LOCALITY_OUTCODES