Sitemap regeneration failed on staging. 'richmond' in the curated locality list collides with the GIAS town Richmond in North Yorkshire (37 schools), the registry raised, and the admin endpoint 500d — taking down sitemap generation for all 25,000 school pages over one bad row of curated data. The guard now skips the colliding locality and logs an error. Skipping still achieves what the guard was for — a locality never silently shadows a town — without letting curated data break the site. That matters beyond this bug: GIAS town names change with no code change here, so a raise could fire spontaneously in production later. Also removes four localities that were London boroughs rather than districts. Hackney, Islington, Greenwich and Ealing are local authorities with 104, 72, 108 and 115 schools and already have authority pages; a locality defined by two or three outcodes would have been a partial near-duplicate of one — the thin-content failure the two-namespace design exists to avoid. A test now guards the whole borough list. Validated against the live corpus: 15 localities, no town collisions, no authority duplicates, all 15 clear the threshold. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
177 lines
6.4 KiB
Python
177 lines
6.4 KiB
Python
"""The place registry: what places the site publishes, and what is in each.
|
|
|
|
One module owns this question. The pages, the sitemap and the internal-link
|
|
modules all read from here, so the threshold and the collision rules exist in
|
|
exactly one place and are testable without a browser or a database.
|
|
|
|
Two namespaces, never one. 67 viable town names collide with a local
|
|
authority name, and the authority is the larger set in only 43 of them —
|
|
postal towns cross authority boundaries, so neither can absorb the other.
|
|
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Five schools with publishable data. Below this a place has nothing to say
|
|
# that a list of schools does not, and publishing it is index bloat.
|
|
MIN_SCHOOLS = 5
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Place:
|
|
kind: str # "town" | "locality" | "authority" | "outcode"
|
|
slug: str
|
|
name: str
|
|
urns: tuple[int, ...]
|
|
parent_authority: str | None # authority NAME, for the 301 target
|
|
|
|
@property
|
|
def key(self) -> str:
|
|
return f"{self.kind}:{self.slug}"
|
|
|
|
|
|
def _publishable_urns(df) -> set[int]:
|
|
"""URNs with something a page could state, deduplicated across years."""
|
|
from backend.app import _PUBLISHABLE_FIELDS
|
|
|
|
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
|
if not cols:
|
|
return set()
|
|
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
|
|
|
|
|
def _parent_authority(group) -> str | None:
|
|
"""The most common authority in a group — the useful 301 target.
|
|
|
|
A town spanning several authorities has no single parent, so the mode is
|
|
the honest answer rather than an arbitrary first row.
|
|
"""
|
|
if "local_authority" not in group.columns:
|
|
return None
|
|
top = group["local_authority"].dropna()
|
|
return str(top.mode().iloc[0]) if not top.empty else None
|
|
|
|
|
|
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
|
"""One Place per distinct value of `column` that clears the threshold."""
|
|
from backend.app import _slugify
|
|
|
|
if column not in df.columns:
|
|
return {}
|
|
|
|
out: dict[str, Place] = {}
|
|
for name, group in df.groupby(column, dropna=True):
|
|
name = str(name).strip()
|
|
if not name:
|
|
continue
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
continue
|
|
slug = _slugify(name)
|
|
if not slug:
|
|
continue
|
|
place = Place(
|
|
kind=kind, slug=slug, name=name, urns=urns,
|
|
parent_authority=_parent_authority(group) if kind == "town" else None,
|
|
)
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
|
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
|
|
|
|
|
def _outcode(postcode) -> str | None:
|
|
if not isinstance(postcode, str):
|
|
return None
|
|
m = _OUTCODE_RE.match(postcode.upper().strip())
|
|
return m.group(1) if m else None
|
|
|
|
|
|
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
|
"""One Place per postcode district clearing the threshold.
|
|
|
|
These carry no phase variants: nobody searches "primary schools in SW11".
|
|
"""
|
|
if "postcode" not in df.columns:
|
|
return {}
|
|
working = df.assign(_oc=df["postcode"].map(_outcode))
|
|
working = working[working["_oc"].notna()]
|
|
|
|
out: dict[str, Place] = {}
|
|
for oc, group in working.groupby("_oc"):
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
continue
|
|
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
|
urns=urns, parent_authority=_parent_authority(group))
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
def _locality_places(df, publishable: set[int],
|
|
town_slugs: set[str]) -> dict[str, Place]:
|
|
"""One Place per curated locality clearing the threshold."""
|
|
from backend.localities import LOCALITY_OUTCODES
|
|
|
|
if "postcode" not in df.columns:
|
|
return {}
|
|
working = df.assign(_oc=df["postcode"].map(_outcode))
|
|
|
|
out: dict[str, Place] = {}
|
|
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
|
if slug in town_slugs:
|
|
# Skip, do not raise. The guard exists so a locality never
|
|
# silently shadows a town — skipping achieves that, and the error
|
|
# log makes it loud.
|
|
#
|
|
# Raising here took down sitemap generation for all 25,000 school
|
|
# pages when "richmond" met the GIAS town Richmond in North
|
|
# Yorkshire. Worse, GIAS town names change without any code change,
|
|
# so a raise means curated data can break the site spontaneously.
|
|
# A curation mistake must cost one page, not the sitemap.
|
|
logger.error(
|
|
"locality %r collides with the published town of the same "
|
|
"slug and has been skipped; rename it or remove it", slug)
|
|
continue
|
|
group = working[working["_oc"].isin(outcodes)]
|
|
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
|
if len(urns) < MIN_SCHOOLS:
|
|
# Not an error — a locality can legitimately be too small. Logged
|
|
# because one you meant to publish quietly vanishing is the
|
|
# failure worth hearing about.
|
|
logger.warning(
|
|
"locality %s (%s) has %d publishable schools, below the "
|
|
"threshold of %d - not published",
|
|
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
|
continue
|
|
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
|
parent_authority=_parent_authority(group))
|
|
out[place.key] = place
|
|
return out
|
|
|
|
|
|
def build_place_registry(df) -> dict[str, Place]:
|
|
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
|
if df.empty or "urn" not in df.columns:
|
|
return {}
|
|
|
|
publishable = _publishable_urns(df)
|
|
registry: dict[str, Place] = {}
|
|
registry.update(_group(df, "local_authority", "authority", publishable))
|
|
|
|
towns = _group(df, "town", "town", publishable)
|
|
registry.update(towns)
|
|
|
|
town_slugs = {p.slug for p in towns.values()}
|
|
registry.update(_locality_places(df, publishable, town_slugs))
|
|
registry.update(_outcode_places(df, publishable))
|
|
return registry
|