feat(places): London localities and postcode districts
The GIAS town field puts 1,819 London schools under the single value 'London', so it cannot answer 'schools in Battersea' — a query that appears in the baseline. No single field can: parliamentary constituency gives Battersea but not Canary Wharf, admin_ward gives Canary Wharf but not Battersea, and neither gives Clapham or Shoreditch. So a locality is curated, defined by the postcode districts it covers, which needs no new ingestion. A locality may not shadow a published town: the registry raises rather than silently costing a page that carries real demand. One below the threshold is logged rather than raising, because a locality can legitimately be too small. The pipeline seed mirrors the module, with a test guarding the drift — the same arrangement gias_codes has, and for the same reason: the backend image does not contain pipeline/. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
1 parent
759d9f5cea
commit
de853b90b3
4 files changed
+234
-1
No files matched your search
@@ -84,3 +84,93 @@ def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_groups_schools_by_outcode(monkeypatch):
|
||||
# The GIAS town field collapses 1,819 London schools into "London", so a
|
||||
# locality is defined by its postcode districts instead.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"battersea": ("Battersea", ("SW11",))})
|
||||
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
||||
for r in rows:
|
||||
r["postcode"] = "SW11 2AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["locality:battersea"].name == "Battersea"
|
||||
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"nowhere": ("Nowhere", ("ZZ99",))})
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
||||
assert "locality:nowhere" not in reg
|
||||
|
||||
|
||||
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch):
|
||||
# Silently shadowing a town would lose a page carrying real demand.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
with pytest.raises(ValueError, match="brentwood"):
|
||||
build_place_registry(_df(rows))
|
||||
|
||||
|
||||
def test_outcode_places_are_built_from_postcodes():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:cm13"].name == "CM13"
|
||||
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_malformed_postcodes_do_not_create_places():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "not a postcode"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("outcode:") for k in reg)
|
||||
|
||||
|
||||
def test_every_curated_locality_is_structurally_valid():
|
||||
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
||||
import re
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
||||
assert name.strip() == name and name, slug
|
||||
assert outcodes, f"{slug} has no outcodes"
|
||||
for oc in outcodes:
|
||||
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
||||
|
||||
|
||||
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
||||
"""Two copies with no drift guard is worse than one copy.
|
||||
|
||||
backend/localities.py is canonical because the backend image does not
|
||||
contain pipeline/. The seed exists so the warehouse can join on the same
|
||||
definitions, and this is what stops the two diverging — the same
|
||||
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
||||
"""
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
seed_path = (Path(__file__).resolve().parents[2]
|
||||
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
||||
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
||||
|
||||
seed = {
|
||||
row["locality_slug"]: (row["locality_name"],
|
||||
tuple(row["outcodes"].split("|")))
|
||||
for row in csv.DictReader(seed_path.open())
|
||||
}
|
||||
assert seed == LOCALITY_OUTCODES
|
||||
Reference in new issue
Block a user