feat(places): London localities and postcode districts
The GIAS town field puts 1,819 London schools under the single value 'London', so it cannot answer 'schools in Battersea' — a query that appears in the baseline. No single field can: parliamentary constituency gives Battersea but not Canary Wharf, admin_ward gives Canary Wharf but not Battersea, and neither gives Clapham or Shoreditch. So a locality is curated, defined by the postcode districts it covers, which needs no new ingestion. A locality may not shadow a published town: the registry raises rather than silently costing a page that carries real demand. One below the threshold is logged rather than raising, because a locality can legitimately be too small. The pipeline seed mirrors the module, with a test guarding the drift — the same arrangement gias_codes has, and for the same reason: the backend image does not contain pipeline/. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
1 parent
759d9f5cea
commit
de853b90b3
4 files changed
+234
-1
No files matched your search
@@ -0,0 +1,47 @@
|
||||
"""Curated London localities, defined by the postcode districts they cover.
|
||||
|
||||
The GIAS `town` field puts 1,819 London schools under the single value
|
||||
"London", so it cannot answer "schools in Battersea" — a query that appears in
|
||||
the Search Console baseline. No single field can: parliamentary constituency
|
||||
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
|
||||
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
|
||||
and colloquial rather than administrative.
|
||||
|
||||
So this is curated. Where a locality ends is a judgement, not a fact, and a
|
||||
reviewable file is the honest place for a judgement. No new ingestion is
|
||||
needed — the corpus already carries postcodes.
|
||||
|
||||
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
|
||||
mirrors it for anyone querying the warehouse directly; the backend image does
|
||||
not contain `pipeline/`, which is why the module rather than the seed is
|
||||
canonical. Same arrangement as `backend/gias_codes.py`.
|
||||
|
||||
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
|
||||
published, so a typo produces no page rather than an empty one. Places that
|
||||
fail that check are logged at startup, because a locality you meant to publish
|
||||
quietly not appearing is the failure worth hearing about.
|
||||
"""
|
||||
|
||||
# slug -> (display name, outcodes)
|
||||
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
|
||||
"battersea": ("Battersea", ("SW11",)),
|
||||
"canary-wharf": ("Canary Wharf", ("E14",)),
|
||||
"clapham": ("Clapham", ("SW4",)),
|
||||
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
|
||||
"peckham": ("Peckham", ("SE15",)),
|
||||
"brixton": ("Brixton", ("SW2", "SW9")),
|
||||
"hackney": ("Hackney", ("E5", "E8", "E9")),
|
||||
"islington": ("Islington", ("N1", "N5", "N7")),
|
||||
"camden-town": ("Camden Town", ("NW1",)),
|
||||
"greenwich": ("Greenwich", ("SE10",)),
|
||||
"wimbledon": ("Wimbledon", ("SW19",)),
|
||||
"putney": ("Putney", ("SW15",)),
|
||||
"fulham": ("Fulham", ("SW6",)),
|
||||
"chiswick": ("Chiswick", ("W4",)),
|
||||
"ealing": ("Ealing", ("W5", "W13")),
|
||||
"richmond": ("Richmond", ("TW9", "TW10")),
|
||||
"stratford": ("Stratford", ("E15",)),
|
||||
"walthamstow": ("Walthamstow", ("E17",)),
|
||||
"tooting": ("Tooting", ("SW17",)),
|
||||
"dulwich": ("Dulwich", ("SE21", "SE22")),
|
||||
}
|
||||
+76
-1
@@ -12,8 +12,12 @@ Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Five schools with publishable data. Below this a place has nothing to say
|
||||
# that a list of schools does not, and publishing it is index bloat.
|
||||
MIN_SCHOOLS = 5
|
||||
@@ -80,6 +84,71 @@ def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place
|
||||
return out
|
||||
|
||||
|
||||
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
||||
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
||||
|
||||
|
||||
def _outcode(postcode) -> str | None:
|
||||
if not isinstance(postcode, str):
|
||||
return None
|
||||
m = _OUTCODE_RE.match(postcode.upper().strip())
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per postcode district clearing the threshold.
|
||||
|
||||
These carry no phase variants: nobody searches "primary schools in SW11".
|
||||
"""
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
working = working[working["_oc"].notna()]
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for oc, group in working.groupby("_oc"):
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
||||
urns=urns, parent_authority=_parent_authority(group))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def _locality_places(df, publishable: set[int],
|
||||
town_slugs: set[str]) -> dict[str, Place]:
|
||||
"""One Place per curated locality clearing the threshold."""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
if slug in town_slugs:
|
||||
raise ValueError(
|
||||
f"locality {slug!r} collides with a published town of the same "
|
||||
"slug; publishing both would shadow the town silently"
|
||||
)
|
||||
group = working[working["_oc"].isin(outcodes)]
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
# Not an error — a locality can legitimately be too small. Logged
|
||||
# because one you meant to publish quietly vanishing is the
|
||||
# failure worth hearing about.
|
||||
logger.warning(
|
||||
"locality %s (%s) has %d publishable schools, below the "
|
||||
"threshold of %d - not published",
|
||||
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
||||
continue
|
||||
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def build_place_registry(df) -> dict[str, Place]:
|
||||
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
||||
if df.empty or "urn" not in df.columns:
|
||||
@@ -88,5 +157,11 @@ def build_place_registry(df) -> dict[str, Place]:
|
||||
publishable = _publishable_urns(df)
|
||||
registry: dict[str, Place] = {}
|
||||
registry.update(_group(df, "local_authority", "authority", publishable))
|
||||
registry.update(_group(df, "town", "town", publishable))
|
||||
|
||||
towns = _group(df, "town", "town", publishable)
|
||||
registry.update(towns)
|
||||
|
||||
town_slugs = {p.slug for p in towns.values()}
|
||||
registry.update(_locality_places(df, publishable, town_slugs))
|
||||
registry.update(_outcode_places(df, publishable))
|
||||
return registry
|
||||
@@ -84,3 +84,93 @@ def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_groups_schools_by_outcode(monkeypatch):
|
||||
# The GIAS town field collapses 1,819 London schools into "London", so a
|
||||
# locality is defined by its postcode districts instead.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"battersea": ("Battersea", ("SW11",))})
|
||||
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
||||
for r in rows:
|
||||
r["postcode"] = "SW11 2AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["locality:battersea"].name == "Battersea"
|
||||
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"nowhere": ("Nowhere", ("ZZ99",))})
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
||||
assert "locality:nowhere" not in reg
|
||||
|
||||
|
||||
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch):
|
||||
# Silently shadowing a town would lose a page carrying real demand.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
with pytest.raises(ValueError, match="brentwood"):
|
||||
build_place_registry(_df(rows))
|
||||
|
||||
|
||||
def test_outcode_places_are_built_from_postcodes():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:cm13"].name == "CM13"
|
||||
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_malformed_postcodes_do_not_create_places():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "not a postcode"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("outcode:") for k in reg)
|
||||
|
||||
|
||||
def test_every_curated_locality_is_structurally_valid():
|
||||
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
||||
import re
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
||||
assert name.strip() == name and name, slug
|
||||
assert outcodes, f"{slug} has no outcodes"
|
||||
for oc in outcodes:
|
||||
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
||||
|
||||
|
||||
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
||||
"""Two copies with no drift guard is worse than one copy.
|
||||
|
||||
backend/localities.py is canonical because the backend image does not
|
||||
contain pipeline/. The seed exists so the warehouse can join on the same
|
||||
definitions, and this is what stops the two diverging — the same
|
||||
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
||||
"""
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
seed_path = (Path(__file__).resolve().parents[2]
|
||||
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
||||
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
||||
|
||||
seed = {
|
||||
row["locality_slug"]: (row["locality_name"],
|
||||
tuple(row["outcodes"].split("|")))
|
||||
for row in csv.DictReader(seed_path.open())
|
||||
}
|
||||
assert seed == LOCALITY_OUTCODES
|
||||
@@ -0,0 +1,21 @@
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,SW11,London
|
||||
canary-wharf,Canary Wharf,E14,London
|
||||
clapham,Clapham,SW4,London
|
||||
shoreditch,Shoreditch,EC2A|E1,London
|
||||
peckham,Peckham,SE15,London
|
||||
brixton,Brixton,SW2|SW9,London
|
||||
hackney,Hackney,E5|E8|E9,London
|
||||
islington,Islington,N1|N5|N7,London
|
||||
camden-town,Camden Town,NW1,London
|
||||
greenwich,Greenwich,SE10,London
|
||||
wimbledon,Wimbledon,SW19,London
|
||||
putney,Putney,SW15,London
|
||||
fulham,Fulham,SW6,London
|
||||
chiswick,Chiswick,W4,London
|
||||
ealing,Ealing,W5|W13,London
|
||||
richmond,Richmond,TW9|TW10,London
|
||||
stratford,Stratford,E15,London
|
||||
walthamstow,Walthamstow,E17,London
|
||||
tooting,Tooting,SW17,London
|
||||
dulwich,Dulwich,SE21|SE22,London
|
||||
|
Reference in new issue
Block a user