From 759d9f5ceaad653cff3a07281b745dabd43b382c Mon Sep 17 00:00:00 2001 From: Tudor Date: Fri, 21 Aug 2026 18:10:37 +0100 Subject: [PATCH] feat(places): registry of towns and authorities MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two namespaces because 67 town names collide with an authority name and neither set contains the other — postal towns cross authority boundaries, so Bedford the town holds 104 schools against the authority's 86. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- backend/places.py | 92 +++++++++++++++++++ backend/tests/test_places.py | 86 +++++++++++++++++ .../plans/2026-08-21-w2-location-layer.md | 9 +- 3 files changed, 184 insertions(+), 3 deletions(-) create mode 100644 backend/places.py create mode 100644 backend/tests/test_places.py diff --git a/backend/places.py b/backend/places.py new file mode 100644 index 0000000..c7c528c --- /dev/null +++ b/backend/places.py @@ -0,0 +1,92 @@ +"""The place registry: what places the site publishes, and what is in each. + +One module owns this question. The pages, the sitemap and the internal-link +modules all read from here, so the threshold and the collision rules exist in +exactly one place and are testable without a browser or a database. + +Two namespaces, never one. 67 viable town names collide with a local +authority name, and the authority is the larger set in only 43 of them — +postal towns cross authority boundaries, so neither can absorb the other. +Keys are ":" so the collision cannot reappear in the dict. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +# Five schools with publishable data. Below this a place has nothing to say +# that a list of schools does not, and publishing it is index bloat. +MIN_SCHOOLS = 5 + + +@dataclass(frozen=True) +class Place: + kind: str # "town" | "locality" | "authority" | "outcode" + slug: str + name: str + urns: tuple[int, ...] + parent_authority: str | None # authority NAME, for the 301 target + + @property + def key(self) -> str: + return f"{self.kind}:{self.slug}" + + +def _publishable_urns(df) -> set[int]: + """URNs with something a page could state, deduplicated across years.""" + from backend.app import _PUBLISHABLE_FIELDS + + cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns] + if not cols: + return set() + return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int)) + + +def _parent_authority(group) -> str | None: + """The most common authority in a group — the useful 301 target. + + A town spanning several authorities has no single parent, so the mode is + the honest answer rather than an arbitrary first row. + """ + if "local_authority" not in group.columns: + return None + top = group["local_authority"].dropna() + return str(top.mode().iloc[0]) if not top.empty else None + + +def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]: + """One Place per distinct value of `column` that clears the threshold.""" + from backend.app import _slugify + + if column not in df.columns: + return {} + + out: dict[str, Place] = {} + for name, group in df.groupby(column, dropna=True): + name = str(name).strip() + if not name: + continue + urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) + if len(urns) < MIN_SCHOOLS: + continue + slug = _slugify(name) + if not slug: + continue + place = Place( + kind=kind, slug=slug, name=name, urns=urns, + parent_authority=_parent_authority(group) if kind == "town" else None, + ) + out[place.key] = place + return out + + +def build_place_registry(df) -> dict[str, Place]: + """Every place the site publishes, keyed by ":".""" + if df.empty or "urn" not in df.columns: + return {} + + publishable = _publishable_urns(df) + registry: dict[str, Place] = {} + registry.update(_group(df, "local_authority", "authority", publishable)) + registry.update(_group(df, "town", "town", publishable)) + return registry diff --git a/backend/tests/test_places.py b/backend/tests/test_places.py new file mode 100644 index 0000000..c7b888c --- /dev/null +++ b/backend/tests/test_places.py @@ -0,0 +1,86 @@ +"""Tests for the place registry (spec 2026-08-21). + +The registry is built from the in-memory school DataFrame, so these build a +small frame directly rather than touching a database. +""" + +import numpy as np +import pandas as pd +import pytest + +from backend.places import MIN_SCHOOLS, build_place_registry + + +def _df(rows: list[dict]) -> pd.DataFrame: + base = { + "year": 202425, "ofsted_grade": 2.0, "ofsted_date": None, + "rwm_expected_pct": 60.0, "attainment_8_score": np.nan, + "phase": "Primary", "postcode": "AA1 1AA", + } + return pd.DataFrame([{**base, **r} for r in rows]) + + +def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]: + """`start` offsets the URNs so two calls can describe different schools — + the Bedford case needs two authorities' worth of distinct URNs in one + town.""" + return [ + {"urn": start + i, "school_name": f"{town} School {i}", + "town": town, "local_authority": la, **kw} + for i in range(n) + ] + + +def test_town_clearing_the_threshold_is_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex"))) + assert "town:brentwood" in reg + assert reg["town:brentwood"].name == "Brentwood" + assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS + + +def test_town_below_the_threshold_is_not_published(): + reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton"))) + assert "town:crosby" not in reg + + +def test_a_town_below_threshold_still_names_its_authority(): + # The route layer needs somewhere to 301 to. + reg = build_place_registry(_df( + _town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton"))) + assert "authority:sefton" in reg + + +def test_town_and_authority_of_the_same_name_are_separate_places(): + # 67 real collisions. Neither set contains the other: Bedford the town has + # 104 schools, Bedford the authority 86, because postal towns cross + # authority boundaries. + rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford") + + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000)) + reg = build_place_registry(_df(rows)) + town, authority = reg["town:bedford"], reg["authority:bedford"] + assert set(town.urns) != set(authority.urns) + assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools + assert len(authority.urns) == MIN_SCHOOLS # only this authority's + + +def test_schools_without_publishable_data_do_not_count_toward_the_threshold(): + rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere") + for r in rows: + r["rwm_expected_pct"] = np.nan + r["ofsted_grade"] = np.nan + reg = build_place_registry(_df(rows)) + assert "town:ghosttown" not in reg + + +def test_blank_town_is_ignored(): + rows = _town(MIN_SCHOOLS, "", "Essex") + reg = build_place_registry(_df(rows)) + assert not any(k.startswith("town:") for k in reg) + + +def test_a_school_is_counted_once_even_with_several_years_of_rows(): + rows = [] + for year in (202324, 202425): + rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")] + reg = build_place_registry(_df(rows)) + assert len(reg["town:beccles"].urns) == MIN_SCHOOLS diff --git a/docs/superpowers/plans/2026-08-21-w2-location-layer.md b/docs/superpowers/plans/2026-08-21-w2-location-layer.md index d9a5923..fd101ac 100644 --- a/docs/superpowers/plans/2026-08-21-w2-location-layer.md +++ b/docs/superpowers/plans/2026-08-21-w2-location-layer.md @@ -132,9 +132,12 @@ def _df(rows: list[dict]) -> pd.DataFrame: return pd.DataFrame([{**base, **r} for r in rows]) -def _town(n: int, town: str, la: str, **kw) -> list[dict]: +def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]: + """`start` offsets the URNs so two calls can describe different schools — + the Bedford case needs two authorities' worth of distinct URNs in one + town.""" return [ - {"urn": 100000 + i, "school_name": f"{town} School {i}", + {"urn": start + i, "school_name": f"{town} School {i}", "town": town, "local_authority": la, **kw} for i in range(n) ] @@ -164,7 +167,7 @@ def test_town_and_authority_of_the_same_name_are_separate_places(): # 104 schools, Bedford the authority 86, because postal towns cross # authority boundaries. rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford") - + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire")) + + _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000)) reg = build_place_registry(_df(rows)) town, authority = reg["town:bedford"], reg["authority:bedford"] assert set(town.urns) != set(authority.urns)