The GIAS town field puts 1,819 London schools under the single value 'London', so it cannot answer 'schools in Battersea' — a query that appears in the baseline. No single field can: parliamentary constituency gives Battersea but not Canary Wharf, admin_ward gives Canary Wharf but not Battersea, and neither gives Clapham or Shoreditch. So a locality is curated, defined by the postcode districts it covers, which needs no new ingestion. A locality may not shadow a published town: the registry raises rather than silently costing a page that carries real demand. One below the threshold is logged rather than raising, because a locality can legitimately be too small. The pipeline seed mirrors the module, with a test guarding the drift — the same arrangement gias_codes has, and for the same reason: the backend image does not contain pipeline/. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
168 lines
5.9 KiB
Python
168 lines
5.9 KiB
Python
"""The place registry: what places the site publishes, and what is in each.
|
|
|
|
One module owns this question. The pages, the sitemap and the internal-link
|
|
modules all read from here, so the threshold and the collision rules exist in
|
|
exactly one place and are testable without a browser or a database.
|
|
|
|
Two namespaces, never one. 67 viable town names collide with a local
|
|
authority name, and the authority is the larger set in only 43 of them —
|
|
postal towns cross authority boundaries, so neither can absorb the other.
|
|
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Five schools with publishable data. Below this a place has nothing to say
|
|
# that a list of schools does not, and publishing it is index bloat.
|
|
MIN_SCHOOLS = 5
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Place:
|
|
kind: str # "town" | "locality" | "authority" | "outcode"
|
|
slug: str
|
|
name: str
|
|
urns: tuple[int, ...]
|
|
parent_authority: str | None # authority NAME, for the 301 target
|
|
|
|
@property
|
|
def key(self) -> str:
|
|
return f"{self.kind}:{self.slug}"
|
|
|
|
|
|
def _publishable_urns(df) -> set[int]:
|
|
"""URNs with something a page could state, deduplicated across years."""
|
|
from backend.app import _PUBLISHABLE_FIELDS
|
|
|
|
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
|
if not cols:
|
|
return set()
|
|
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
|
|
|
|
|
def _parent_authority(group) -> str | None:
|
|
"""The most common authority in a group — the useful 301 target.
|
|
|
|
A town spanning several authorities has no single parent, so the mode is
|
|
the honest answer rather than an arbitrary first row.
|
|
"""
|
|
if "local_authority" not in group.columns:
|
|
return None
|
|
top = group["local_authority"].dropna()
|
|
return str(top.mode().iloc[0]) if not top.empty else None
|
|
|
|
|
|
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
|
"""One Place per distinct value of `column` that clears the threshold."""
|
|
from backend.app import _slugify
|
|
|
|
if column not in df.columns:
|
|
return {}
|
|
|
|
out: dict[str, Place] = {}
|
|
for name, group in df.groupby(column, dropna=True):
|
|
name = str(name).strip()
|
|
if not name:
|
|
continue
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
continue
|
|
slug = _slugify(name)
|
|
if not slug:
|
|
continue
|
|
place = Place(
|
|
kind=kind, slug=slug, name=name, urns=urns,
|
|
parent_authority=_parent_authority(group) if kind == "town" else None,
|
|
)
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
|
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
|
|
|
|
|
def _outcode(postcode) -> str | None:
|
|
if not isinstance(postcode, str):
|
|
return None
|
|
m = _OUTCODE_RE.match(postcode.upper().strip())
|
|
return m.group(1) if m else None
|
|
|
|
|
|
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
|
"""One Place per postcode district clearing the threshold.
|
|
|
|
These carry no phase variants: nobody searches "primary schools in SW11".
|
|
"""
|
|
if "postcode" not in df.columns:
|
|
return {}
|
|
working = df.assign(_oc=df["postcode"].map(_outcode))
|
|
working = working[working["_oc"].notna()]
|
|
|
|
out: dict[str, Place] = {}
|
|
for oc, group in working.groupby("_oc"):
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
continue
|
|
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
|
urns=urns, parent_authority=_parent_authority(group))
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
def _locality_places(df, publishable: set[int],
|
|
town_slugs: set[str]) -> dict[str, Place]:
|
|
"""One Place per curated locality clearing the threshold."""
|
|
from backend.localities import LOCALITY_OUTCODES
|
|
|
|
if "postcode" not in df.columns:
|
|
return {}
|
|
working = df.assign(_oc=df["postcode"].map(_outcode))
|
|
|
|
out: dict[str, Place] = {}
|
|
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
|
if slug in town_slugs:
|
|
raise ValueError(
|
|
f"locality {slug!r} collides with a published town of the same "
|
|
"slug; publishing both would shadow the town silently"
|
|
)
|
|
group = working[working["_oc"].isin(outcodes)]
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
# Not an error — a locality can legitimately be too small. Logged
|
|
# because one you meant to publish quietly vanishing is the
|
|
# failure worth hearing about.
|
|
logger.warning(
|
|
"locality %s (%s) has %d publishable schools, below the "
|
|
"threshold of %d - not published",
|
|
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
|
continue
|
|
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
|
parent_authority=_parent_authority(group))
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
def build_place_registry(df) -> dict[str, Place]:
|
|
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
|
if df.empty or "urn" not in df.columns:
|
|
return {}
|
|
|
|
publishable = _publishable_urns(df)
|
|
registry: dict[str, Place] = {}
|
|
registry.update(_group(df, "local_authority", "authority", publishable))
|
|
|
|
towns = _group(df, "town", "town", publishable)
|
|
registry.update(towns)
|
|
|
|
town_slugs = {p.slug for p in towns.values()}
|
|
registry.update(_locality_places(df, publishable, town_slugs))
|
|
registry.update(_outcode_places(df, publishable))
|
|
return registry
|