diff --git a/backend/localities.py b/backend/localities.py new file mode 100644 index 0000000..ecd3077 --- /dev/null +++ b/backend/localities.py @@ -0,0 +1,47 @@ +"""Curated London localities, defined by the postcode districts they cover. + +The GIAS `town` field puts 1,819 London schools under the single value +"London", so it cannot answer "schools in Battersea" — a query that appears in +the Search Console baseline. No single field can: parliamentary constituency +gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary +Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal +and colloquial rather than administrative. + +So this is curated. Where a locality ends is a judgement, not a fact, and a +reviewable file is the honest place for a judgement. No new ingestion is +needed — the corpus already carries postcodes. + +This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv` +mirrors it for anyone querying the warehouse directly; the backend image does +not contain `pipeline/`, which is why the module rather than the seed is +canonical. Same arrangement as `backend/gias_codes.py`. + +A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not +published, so a typo produces no page rather than an empty one. Places that +fail that check are logged at startup, because a locality you meant to publish +quietly not appearing is the failure worth hearing about. +""" + +# slug -> (display name, outcodes) +LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = { + "battersea": ("Battersea", ("SW11",)), + "canary-wharf": ("Canary Wharf", ("E14",)), + "clapham": ("Clapham", ("SW4",)), + "shoreditch": ("Shoreditch", ("EC2A", "E1")), + "peckham": ("Peckham", ("SE15",)), + "brixton": ("Brixton", ("SW2", "SW9")), + "hackney": ("Hackney", ("E5", "E8", "E9")), + "islington": ("Islington", ("N1", "N5", "N7")), + "camden-town": ("Camden Town", ("NW1",)), + "greenwich": ("Greenwich", ("SE10",)), + "wimbledon": ("Wimbledon", ("SW19",)), + "putney": ("Putney", ("SW15",)), + "fulham": ("Fulham", ("SW6",)), + "chiswick": ("Chiswick", ("W4",)), + "ealing": ("Ealing", ("W5", "W13")), + "richmond": ("Richmond", ("TW9", "TW10")), + "stratford": ("Stratford", ("E15",)), + "walthamstow": ("Walthamstow", ("E17",)), + "tooting": ("Tooting", ("SW17",)), + "dulwich": ("Dulwich", ("SE21", "SE22")), +} diff --git a/backend/places.py b/backend/places.py index c7c528c..12eb84a 100644 --- a/backend/places.py +++ b/backend/places.py @@ -12,8 +12,12 @@ Keys are ":" so the collision cannot reappear in the dict. from __future__ import annotations +import logging +import re from dataclasses import dataclass +logger = logging.getLogger(__name__) + # Five schools with publishable data. Below this a place has nothing to say # that a list of schools does not, and publishing it is index bloat. MIN_SCHOOLS = 5 @@ -80,6 +84,71 @@ def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place return out +# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter. +_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s") + + +def _outcode(postcode) -> str | None: + if not isinstance(postcode, str): + return None + m = _OUTCODE_RE.match(postcode.upper().strip()) + return m.group(1) if m else None + + +def _outcode_places(df, publishable: set[int]) -> dict[str, Place]: + """One Place per postcode district clearing the threshold. + + These carry no phase variants: nobody searches "primary schools in SW11". + """ + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + working = working[working["_oc"].notna()] + + out: dict[str, Place] = {} + for oc, group in working.groupby("_oc"): + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc), + urns=urns, parent_authority=_parent_authority(group)) + out[place.key] = place + return out + + +def _locality_places(df, publishable: set[int], + town_slugs: set[str]) -> dict[str, Place]: + """One Place per curated locality clearing the threshold.""" + from backend.localities import LOCALITY_OUTCODES + + if "postcode" not in df.columns: + return {} + working = df.assign(_oc=df["postcode"].map(_outcode)) + + out: dict[str, Place] = {} + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + if slug in town_slugs: + raise ValueError( + f"locality {slug!r} collides with a published town of the same " + "slug; publishing both would shadow the town silently" + ) + group = working[working["_oc"].isin(outcodes)] + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + # Not an error — a locality can legitimately be too small. Logged + # because one you meant to publish quietly vanishing is the + # failure worth hearing about. + logger.warning( + "locality %s (%s) has %d publishable schools, below the " + "threshold of %d - not published", + slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS) + continue + place = Place(kind="locality", slug=slug, name=name, urns=urns, + parent_authority=_parent_authority(group)) + out[place.key] = place + return out + + def build_place_registry(df) -> dict[str, Place]: """Every place the site publishes, keyed by ":".""" if df.empty or "urn" not in df.columns: @@ -88,5 +157,11 @@ def build_place_registry(df) -> dict[str, Place]: publishable = _publishable_urns(df) registry: dict[str, Place] = {} registry.update(_group(df, "local_authority", "authority", publishable)) - registry.update(_group(df, "town", "town", publishable)) + + towns = _group(df, "town", "town", publishable) + registry.update(towns) + + town_slugs = {p.slug for p in towns.values()} + registry.update(_locality_places(df, publishable, town_slugs)) + registry.update(_outcode_places(df, publishable)) return registry diff --git a/backend/tests/test_places.py b/backend/tests/test_places.py index c7b888c..bcd7019 100644 --- a/backend/tests/test_places.py +++ b/backend/tests/test_places.py @@ -84,3 +84,93 @@ def test_a_school_is_counted_once_even_with_several_years_of_rows(): rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")] reg = build_place_registry(_df(rows)) assert len(reg["town:beccles"].urns) == MIN_SCHOOLS + + +def test_locality_groups_schools_by_outcode(monkeypatch): + # The GIAS town field collapses 1,819 London schools into "London", so a + # locality is defined by its postcode districts instead. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"battersea": ("Battersea", ("SW11",))}) + rows = _town(MIN_SCHOOLS, "London", "Wandsworth") + for r in rows: + r["postcode"] = "SW11 2AA" + reg = build_place_registry(_df(rows)) + assert reg["locality:battersea"].name == "Battersea" + assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS + + +def test_locality_below_the_threshold_is_not_published(monkeypatch): + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"nowhere": ("Nowhere", ("ZZ99",))}) + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth"))) + assert "locality:nowhere" not in reg + + +def test_a_locality_may_not_shadow_a_viable_town(monkeypatch): + # Silently shadowing a town would lose a page carrying real demand. + from backend import localities + monkeypatch.setattr(localities, "LOCALITY_OUTCODES", + {"brentwood": ("Brentwood", ("CM13",))}) + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + with pytest.raises(ValueError, match="brentwood"): + build_place_registry(_df(rows)) + + +def test_outcode_places_are_built_from_postcodes(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "CM13 1AA" + reg = build_place_registry(_df(rows)) + assert reg["outcode:cm13"].name == "CM13" + assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS + + +def test_malformed_postcodes_do_not_create_places(): + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + for r in rows: + r["postcode"] = "not a postcode" + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("outcode:") for k in reg) + + +def test_every_curated_locality_is_structurally_valid(): + # Guards the hand-maintained file: real slug, real name, real outcodes. + import re + from backend.localities import LOCALITY_OUTCODES + + assert LOCALITY_OUTCODES, "the curated locality list must not be empty" + for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): + assert re.fullmatch(r"[a-z0-9-]+", slug), slug + assert name.strip() == name and name, slug + assert outcodes, f"{slug} has no outcodes" + for oc in outcodes: + assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc) + + +def test_the_pipeline_seed_mirrors_the_canonical_module(): + """Two copies with no drift guard is worse than one copy. + + backend/localities.py is canonical because the backend image does not + contain pipeline/. The seed exists so the warehouse can join on the same + definitions, and this is what stops the two diverging — the same + arrangement assert_gias_code_names_match_seed.sql gives gias_codes. + """ + import csv + from pathlib import Path + + from backend.localities import LOCALITY_OUTCODES + + seed_path = (Path(__file__).resolve().parents[2] + / "pipeline/transform/seeds/locality_outcodes.csv") + assert seed_path.exists(), f"missing seed mirror at {seed_path}" + + seed = { + row["locality_slug"]: (row["locality_name"], + tuple(row["outcodes"].split("|"))) + for row in csv.DictReader(seed_path.open()) + } + assert seed == LOCALITY_OUTCODES diff --git a/pipeline/transform/seeds/locality_outcodes.csv b/pipeline/transform/seeds/locality_outcodes.csv new file mode 100644 index 0000000..968f6ce --- /dev/null +++ b/pipeline/transform/seeds/locality_outcodes.csv @@ -0,0 +1,21 @@ +locality_slug,locality_name,outcodes,region +battersea,Battersea,SW11,London +canary-wharf,Canary Wharf,E14,London +clapham,Clapham,SW4,London +shoreditch,Shoreditch,EC2A|E1,London +peckham,Peckham,SE15,London +brixton,Brixton,SW2|SW9,London +hackney,Hackney,E5|E8|E9,London +islington,Islington,N1|N5|N7,London +camden-town,Camden Town,NW1,London +greenwich,Greenwich,SE10,London +wimbledon,Wimbledon,SW19,London +putney,Putney,SW15,London +fulham,Fulham,SW6,London +chiswick,Chiswick,W4,London +ealing,Ealing,W5|W13,London +richmond,Richmond,TW9|TW10,London +stratford,Stratford,E15,London +walthamstow,Walthamstow,E17,London +tooting,Tooting,SW17,London +dulwich,Dulwich,SE21|SE22,London