feat(places): London localities and postcode districts

The GIAS town field puts 1,819 London schools under the single value
'London', so it cannot answer 'schools in Battersea' — a query that appears in
the baseline. No single field can: parliamentary constituency gives Battersea
but not Canary Wharf, admin_ward gives Canary Wharf but not Battersea, and
neither gives Clapham or Shoreditch. So a locality is curated, defined by the
postcode districts it covers, which needs no new ingestion.

A locality may not shadow a published town: the registry raises rather than
silently costing a page that carries real demand. One below the threshold is
logged rather than raising, because a locality can legitimately be too small.

The pipeline seed mirrors the module, with a test guarding the drift — the
same arrangement gias_codes has, and for the same reason: the backend image
does not contain pipeline/.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-21 18:11:56 +01:00
1 parent 759d9f5cea
commit de853b90b3
4 files changed
+234 -1

No files matched your search

+47
View File
@@ -0,0 +1,47 @@
"""Curated London localities, defined by the postcode districts they cover.
The GIAS `town` field puts 1,819 London schools under the single value
"London", so it cannot answer "schools in Battersea" — a query that appears in
the Search Console baseline. No single field can: parliamentary constituency
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
and colloquial rather than administrative.
So this is curated. Where a locality ends is a judgement, not a fact, and a
reviewable file is the honest place for a judgement. No new ingestion is
needed — the corpus already carries postcodes.
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
mirrors it for anyone querying the warehouse directly; the backend image does
not contain `pipeline/`, which is why the module rather than the seed is
canonical. Same arrangement as `backend/gias_codes.py`.
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
published, so a typo produces no page rather than an empty one. Places that
fail that check are logged at startup, because a locality you meant to publish
quietly not appearing is the failure worth hearing about.
"""
# slug -> (display name, outcodes)
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
"battersea": ("Battersea", ("SW11",)),
"canary-wharf": ("Canary Wharf", ("E14",)),
"clapham": ("Clapham", ("SW4",)),
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
"peckham": ("Peckham", ("SE15",)),
"brixton": ("Brixton", ("SW2", "SW9")),
"hackney": ("Hackney", ("E5", "E8", "E9")),
"islington": ("Islington", ("N1", "N5", "N7")),
"camden-town": ("Camden Town", ("NW1",)),
"greenwich": ("Greenwich", ("SE10",)),
"wimbledon": ("Wimbledon", ("SW19",)),
"putney": ("Putney", ("SW15",)),
"fulham": ("Fulham", ("SW6",)),
"chiswick": ("Chiswick", ("W4",)),
"ealing": ("Ealing", ("W5", "W13")),
"richmond": ("Richmond", ("TW9", "TW10")),
"stratford": ("Stratford", ("E15",)),
"walthamstow": ("Walthamstow", ("E17",)),
"tooting": ("Tooting", ("SW17",)),
"dulwich": ("Dulwich", ("SE21", "SE22")),
}
+76 -1
View File
@@ -12,8 +12,12 @@ Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
logger = logging.getLogger(__name__)
# Five schools with publishable data. Below this a place has nothing to say
# that a list of schools does not, and publishing it is index bloat.
MIN_SCHOOLS = 5
@@ -80,6 +84,71 @@ def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place
return out
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
def _outcode(postcode) -> str | None:
if not isinstance(postcode, str):
return None
m = _OUTCODE_RE.match(postcode.upper().strip())
return m.group(1) if m else None
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
"""One Place per postcode district clearing the threshold.
These carry no phase variants: nobody searches "primary schools in SW11".
"""
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
working = working[working["_oc"].notna()]
out: dict[str, Place] = {}
for oc, group in working.groupby("_oc"):
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
urns=urns, parent_authority=_parent_authority(group))
out[place.key] = place
return out
def _locality_places(df, publishable: set[int],
town_slugs: set[str]) -> dict[str, Place]:
"""One Place per curated locality clearing the threshold."""
from backend.localities import LOCALITY_OUTCODES
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
out: dict[str, Place] = {}
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
if slug in town_slugs:
raise ValueError(
f"locality {slug!r} collides with a published town of the same "
"slug; publishing both would shadow the town silently"
)
group = working[working["_oc"].isin(outcodes)]
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
# Not an error — a locality can legitimately be too small. Logged
# because one you meant to publish quietly vanishing is the
# failure worth hearing about.
logger.warning(
"locality %s (%s) has %d publishable schools, below the "
"threshold of %d - not published",
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
continue
place = Place(kind="locality", slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group))
out[place.key] = place
return out
def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns:
@@ -88,5 +157,11 @@ def build_place_registry(df) -> dict[str, Place]:
publishable = _publishable_urns(df)
registry: dict[str, Place] = {}
registry.update(_group(df, "local_authority", "authority", publishable))
registry.update(_group(df, "town", "town", publishable))
towns = _group(df, "town", "town", publishable)
registry.update(towns)
town_slugs = {p.slug for p in towns.values()}
registry.update(_locality_places(df, publishable, town_slugs))
registry.update(_outcode_places(df, publishable))
return registry
+90
View File
@@ -84,3 +84,93 @@ def test_a_school_is_counted_once_even_with_several_years_of_rows():
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
reg = build_place_registry(_df(rows))
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
def test_locality_groups_schools_by_outcode(monkeypatch):
# The GIAS town field collapses 1,819 London schools into "London", so a
# locality is defined by its postcode districts instead.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"battersea": ("Battersea", ("SW11",))})
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
for r in rows:
r["postcode"] = "SW11 2AA"
reg = build_place_registry(_df(rows))
assert reg["locality:battersea"].name == "Battersea"
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
def test_locality_below_the_threshold_is_not_published(monkeypatch):
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"nowhere": ("Nowhere", ("ZZ99",))})
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
assert "locality:nowhere" not in reg
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch):
# Silently shadowing a town would lose a page carrying real demand.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"brentwood": ("Brentwood", ("CM13",))})
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
with pytest.raises(ValueError, match="brentwood"):
build_place_registry(_df(rows))
def test_outcode_places_are_built_from_postcodes():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
reg = build_place_registry(_df(rows))
assert reg["outcode:cm13"].name == "CM13"
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
def test_malformed_postcodes_do_not_create_places():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "not a postcode"
reg = build_place_registry(_df(rows))
assert not any(k.startswith("outcode:") for k in reg)
def test_every_curated_locality_is_structurally_valid():
# Guards the hand-maintained file: real slug, real name, real outcodes.
import re
from backend.localities import LOCALITY_OUTCODES
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
assert name.strip() == name and name, slug
assert outcodes, f"{slug} has no outcodes"
for oc in outcodes:
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
def test_the_pipeline_seed_mirrors_the_canonical_module():
"""Two copies with no drift guard is worse than one copy.
backend/localities.py is canonical because the backend image does not
contain pipeline/. The seed exists so the warehouse can join on the same
definitions, and this is what stops the two diverging — the same
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
"""
import csv
from pathlib import Path
from backend.localities import LOCALITY_OUTCODES
seed_path = (Path(__file__).resolve().parents[2]
/ "pipeline/transform/seeds/locality_outcodes.csv")
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
seed = {
row["locality_slug"]: (row["locality_name"],
tuple(row["outcodes"].split("|")))
for row in csv.DictReader(seed_path.open())
}
assert seed == LOCALITY_OUTCODES
@@ -0,0 +1,21 @@
locality_slug,locality_name,outcodes,region
battersea,Battersea,SW11,London
canary-wharf,Canary Wharf,E14,London
clapham,Clapham,SW4,London
shoreditch,Shoreditch,EC2A|E1,London
peckham,Peckham,SE15,London
brixton,Brixton,SW2|SW9,London
hackney,Hackney,E5|E8|E9,London
islington,Islington,N1|N5|N7,London
camden-town,Camden Town,NW1,London
greenwich,Greenwich,SE10,London
wimbledon,Wimbledon,SW19,London
putney,Putney,SW15,London
fulham,Fulham,SW6,London
chiswick,Chiswick,W4,London
ealing,Ealing,W5|W13,London
richmond,Richmond,TW9|TW10,London
stratford,Stratford,E15,London
walthamstow,Walthamstow,E17,London
tooting,Tooting,SW17,London
dulwich,Dulwich,SE21|SE22,London
1 locality_slug locality_name outcodes region
2 battersea Battersea SW11 London
3 canary-wharf Canary Wharf E14 London
4 clapham Clapham SW4 London
5 shoreditch Shoreditch EC2A|E1 London
6 peckham Peckham SE15 London
7 brixton Brixton SW2|SW9 London
8 hackney Hackney E5|E8|E9 London
9 islington Islington N1|N5|N7 London
10 camden-town Camden Town NW1 London
11 greenwich Greenwich SE10 London
12 wimbledon Wimbledon SW19 London
13 putney Putney SW15 London
14 fulham Fulham SW6 London
15 chiswick Chiswick W4 London
16 ealing Ealing W5|W13 London
17 richmond Richmond TW9|TW10 London
18 stratford Stratford E15 London
19 walthamstow Walthamstow E17 London
20 tooting Tooting SW17 London
21 dulwich Dulwich SE21|SE22 London