feat(places): registry of towns and authorities
Two namespaces because 67 town names collide with an authority name and neither set contains the other — postal towns cross authority boundaries, so Bedford the town holds 104 schools against the authority's 86. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
1 parent
555d3f0a7d
commit
759d9f5cea
3 files changed
+184
-3
No files matched your search
@@ -0,0 +1,92 @@
|
||||
"""The place registry: what places the site publishes, and what is in each.
|
||||
|
||||
One module owns this question. The pages, the sitemap and the internal-link
|
||||
modules all read from here, so the threshold and the collision rules exist in
|
||||
exactly one place and are testable without a browser or a database.
|
||||
|
||||
Two namespaces, never one. 67 viable town names collide with a local
|
||||
authority name, and the authority is the larger set in only 43 of them —
|
||||
postal towns cross authority boundaries, so neither can absorb the other.
|
||||
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
# Five schools with publishable data. Below this a place has nothing to say
|
||||
# that a list of schools does not, and publishing it is index bloat.
|
||||
MIN_SCHOOLS = 5
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Place:
|
||||
kind: str # "town" | "locality" | "authority" | "outcode"
|
||||
slug: str
|
||||
name: str
|
||||
urns: tuple[int, ...]
|
||||
parent_authority: str | None # authority NAME, for the 301 target
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
return f"{self.kind}:{self.slug}"
|
||||
|
||||
|
||||
def _publishable_urns(df) -> set[int]:
|
||||
"""URNs with something a page could state, deduplicated across years."""
|
||||
from backend.app import _PUBLISHABLE_FIELDS
|
||||
|
||||
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
if not cols:
|
||||
return set()
|
||||
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
||||
|
||||
|
||||
def _parent_authority(group) -> str | None:
|
||||
"""The most common authority in a group — the useful 301 target.
|
||||
|
||||
A town spanning several authorities has no single parent, so the mode is
|
||||
the honest answer rather than an arbitrary first row.
|
||||
"""
|
||||
if "local_authority" not in group.columns:
|
||||
return None
|
||||
top = group["local_authority"].dropna()
|
||||
return str(top.mode().iloc[0]) if not top.empty else None
|
||||
|
||||
|
||||
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per distinct value of `column` that clears the threshold."""
|
||||
from backend.app import _slugify
|
||||
|
||||
if column not in df.columns:
|
||||
return {}
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for name, group in df.groupby(column, dropna=True):
|
||||
name = str(name).strip()
|
||||
if not name:
|
||||
continue
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
slug = _slugify(name)
|
||||
if not slug:
|
||||
continue
|
||||
place = Place(
|
||||
kind=kind, slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(group) if kind == "town" else None,
|
||||
)
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def build_place_registry(df) -> dict[str, Place]:
|
||||
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
||||
if df.empty or "urn" not in df.columns:
|
||||
return {}
|
||||
|
||||
publishable = _publishable_urns(df)
|
||||
registry: dict[str, Place] = {}
|
||||
registry.update(_group(df, "local_authority", "authority", publishable))
|
||||
registry.update(_group(df, "town", "town", publishable))
|
||||
return registry
|
||||
@@ -0,0 +1,86 @@
|
||||
"""Tests for the place registry (spec 2026-08-21).
|
||||
|
||||
The registry is built from the in-memory school DataFrame, so these build a
|
||||
small frame directly rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from backend.places import MIN_SCHOOLS, build_place_registry
|
||||
|
||||
|
||||
def _df(rows: list[dict]) -> pd.DataFrame:
|
||||
base = {
|
||||
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
|
||||
"phase": "Primary", "postcode": "AA1 1AA",
|
||||
}
|
||||
return pd.DataFrame([{**base, **r} for r in rows])
|
||||
|
||||
|
||||
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
||||
"""`start` offsets the URNs so two calls can describe different schools —
|
||||
the Bedford case needs two authorities' worth of distinct URNs in one
|
||||
town."""
|
||||
return [
|
||||
{"urn": start + i, "school_name": f"{town} School {i}",
|
||||
"town": town, "local_authority": la, **kw}
|
||||
for i in range(n)
|
||||
]
|
||||
|
||||
|
||||
def test_town_clearing_the_threshold_is_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
assert "town:brentwood" in reg
|
||||
assert reg["town:brentwood"].name == "Brentwood"
|
||||
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_town_below_the_threshold_is_not_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
|
||||
assert "town:crosby" not in reg
|
||||
|
||||
|
||||
def test_a_town_below_threshold_still_names_its_authority():
|
||||
# The route layer needs somewhere to 301 to.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
|
||||
assert "authority:sefton" in reg
|
||||
|
||||
|
||||
def test_town_and_authority_of_the_same_name_are_separate_places():
|
||||
# 67 real collisions. Neither set contains the other: Bedford the town has
|
||||
# 104 schools, Bedford the authority 86, because postal towns cross
|
||||
# authority boundaries.
|
||||
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
||||
assert set(town.urns) != set(authority.urns)
|
||||
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
|
||||
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
|
||||
|
||||
|
||||
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
|
||||
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
|
||||
for r in rows:
|
||||
r["rwm_expected_pct"] = np.nan
|
||||
r["ofsted_grade"] = np.nan
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "town:ghosttown" not in reg
|
||||
|
||||
|
||||
def test_blank_town_is_ignored():
|
||||
rows = _town(MIN_SCHOOLS, "", "Essex")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("town:") for k in reg)
|
||||
|
||||
|
||||
def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows = []
|
||||
for year in (202324, 202425):
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
@@ -132,9 +132,12 @@ def _df(rows: list[dict]) -> pd.DataFrame:
|
||||
return pd.DataFrame([{**base, **r} for r in rows])
|
||||
|
||||
|
||||
def _town(n: int, town: str, la: str, **kw) -> list[dict]:
|
||||
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
||||
"""`start` offsets the URNs so two calls can describe different schools —
|
||||
the Bedford case needs two authorities' worth of distinct URNs in one
|
||||
town."""
|
||||
return [
|
||||
{"urn": 100000 + i, "school_name": f"{town} School {i}",
|
||||
{"urn": start + i, "school_name": f"{town} School {i}",
|
||||
"town": town, "local_authority": la, **kw}
|
||||
for i in range(n)
|
||||
]
|
||||
@@ -164,7 +167,7 @@ def test_town_and_authority_of_the_same_name_are_separate_places():
|
||||
# 104 schools, Bedford the authority 86, because postal towns cross
|
||||
# authority boundaries.
|
||||
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire"))
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
||||
assert set(town.urns) != set(authority.urns)
|
||||
|
||||
Reference in new issue
Block a user