feat(places): name every authority a place sits in

SW19 is mostly Merton but partly Wandsworth, and the page said only Merton.
The cause was one field doing two jobs: _parent_authority takes the modal
authority, which is right for a 301 target and wrong as a statement about
where a place is.

This is not a corner case. A quarter of viable outcodes (425 of 1,760) and a
third of viable towns (263 of 783) cross an authority boundary — Bedford the
town spans Bedford and Central Bedfordshire.

Place now carries `authorities`, every authority holding at least a tenth of
the schools and at least two of them, largest first. parent_authority stays
single and unchanged, because a redirect still needs one target.

The share threshold exists because GIAS carries postcode errors: EN6 lists two
Shropshire schools among fourteen in Hertfordshire, and a bare "any authority
present" rule would print those as though they were real. A place too small or
too fragmented to clear the threshold still names its largest, so the page
never goes silent about where it is.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-21 22:40:58 +01:00
1 parent 4cea26b813
commit 1cb5314c53
7 files changed
+214 -6

No files matched your search

+7
View File
@@ -1232,6 +1232,13 @@ async def get_place(request: Request, kind: str, slug: str,
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
"count": len(place.urns),
"parent_authority": place.parent_authority,
# Every authority the place meaningfully sits in. SW19 is
# mostly Merton but partly Wandsworth; naming one asserts
# something false.
"authorities": [
{"name": name, "slug": _slugify(name), "count": n}
for name, n in place.authorities
],
# Only phases that clear the threshold, so the page links
# variants that exist rather than 404s.
"phases": [ph for ph in ("primary", "secondary")
+45
View File
@@ -30,6 +30,12 @@ class Place:
name: str
urns: tuple[int, ...]
parent_authority: str | None # authority NAME, for the 301 target
# Every authority the place meaningfully sits in, largest first. A quarter
# of outcodes and a third of towns straddle a boundary — SW19 is mostly
# Merton but partly Wandsworth — so naming only one asserts something
# false. parent_authority stays single because a redirect needs one
# target; this is what the page shows.
authorities: tuple[tuple[str, int], ...] = ()
# URNs per phase, so the per-phase threshold can be applied without
# re-querying. A place with 30 primaries and 2 secondaries publishes a
# primary variant and no secondary one.
@@ -71,6 +77,42 @@ def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]:
return out
# A place is described by an authority when it holds at least a tenth of the
# schools, and at least two. GIAS carries occasional postcode errors — EN6
# lists two Shropshire schools among fourteen in Hertfordshire — and a bare
# "any authority present" rule would print those as though they were real.
_AUTHORITY_MIN_SHARE = 0.10
_AUTHORITY_MIN_SCHOOLS = 2
_AUTHORITY_MAX_SHOWN = 3
def _authorities(group) -> tuple[tuple[str, int], ...]:
"""Authorities this place meaningfully sits in, largest first."""
from backend.app import EXCLUDED_FILTER_VALUES
if "local_authority" not in group.columns:
return ()
counts = group["local_authority"].dropna().value_counts()
total = int(counts.sum())
if not total:
return ()
kept = [
(str(name), int(n)) for name, n in counts.items()
if str(name) not in EXCLUDED_FILTER_VALUES
and n >= _AUTHORITY_MIN_SCHOOLS
and n / total >= _AUTHORITY_MIN_SHARE
]
# A place too small or too fragmented for the share rule still names its
# largest authority, or the page would say nothing about where it is.
if not kept:
for name, n in counts.items():
if str(name) not in EXCLUDED_FILTER_VALUES:
return ((str(name), int(n)),)
return ()
return tuple(kept[:_AUTHORITY_MAX_SHOWN])
def _parent_authority(group) -> str | None:
"""The most common authority in a group — the useful 301 target.
@@ -104,6 +146,7 @@ def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place
place = Place(
kind=kind, slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group) if kind == "town" else None,
authorities=() if kind == "authority" else _authorities(group),
phase_urns=_phase_urns(group, publishable),
)
out[place.key] = place
@@ -138,6 +181,7 @@ def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
continue
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
urns=urns, parent_authority=_parent_authority(group),
authorities=_authorities(group),
phase_urns=_phase_urns(group, publishable))
out[place.key] = place
return out
@@ -181,6 +225,7 @@ def _locality_places(df, publishable: set[int],
continue
place = Place(kind="locality", slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group),
authorities=_authorities(group),
phase_urns=_phase_urns(group, publishable))
out[place.key] = place
return out
+61
View File
@@ -229,3 +229,64 @@ def test_no_curated_locality_names_a_london_borough():
f"these are boroughs, not districts: {sorted(named)} - they already "
"have an authority page covering every school"
)
def test_a_place_names_every_authority_it_straddles():
"""SW19 is mostly Merton but partly Wandsworth.
A quarter of viable outcodes and a third of viable towns cross an
authority boundary, so naming only the largest asserts something false.
"""
rows = (_town(26, "London", "Merton", start=300000)
+ _town(7, "London", "Wandsworth", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
names = [n for n, _ in reg["outcode:sw19"].authorities]
assert names == ["Merton", "Wandsworth"] # largest first
assert dict(reg["outcode:sw19"].authorities)["Wandsworth"] == 7
def test_the_redirect_target_stays_a_single_authority():
# parent_authority and authorities do different jobs: a 301 needs one
# target, the page needs the truth.
rows = (_town(26, "London", "Merton", start=300000)
+ _town(7, "London", "Wandsworth", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
assert reg["outcode:sw19"].parent_authority == "Merton"
def test_a_stray_authority_below_the_share_threshold_is_not_named():
# GIAS carries postcode errors — EN6 lists two Shropshire schools among
# fourteen in Hertfordshire. Printing those as though real would be worse
# than omitting them.
rows = (_town(30, "Barnet", "Hertfordshire", start=300000)
+ _town(1, "Barnet", "Shropshire", start=400000))
for r in rows:
r["postcode"] = "EN6 1AA"
reg = build_place_registry(_df(rows))
assert [n for n, _ in reg["outcode:en6"].authorities] == ["Hertfordshire"]
def test_a_sentinel_authority_is_never_named():
rows = (_town(20, "London", "Merton", start=300000)
+ _town(6, "London", "Does not apply", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
assert [n for n, _ in reg["outcode:sw19"].authorities] == ["Merton"]
def test_a_place_always_names_at_least_one_authority():
# Even when every authority is below the share threshold, the page has to
# say where the place is.
rows = []
for i, la in enumerate(["A", "B", "C", "D", "E", "F", "G"]):
rows += _town(1, "Fragmented", la, start=300000 + i * 100)
reg = build_place_registry(_df(rows))
place = reg.get("town:fragmented")
assert place is not None
assert len(place.authorities) == 1