Files
school_compare/backend/places.py
TudorandClaude Opus 5 6f749ed21f
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m3s
PR Checks / Backend Smoke (pull_request) Successful in 8s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 45s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Failing after 2m39s
fix(places): submit and link the phase variants
/schools/[place]/[phase] shipped as routes but reached nothing. The sitemap
emitted one URL per registry entry and the registry had no phase dimension, so
~950 pages were absent from every sitemap — and PlaceView did not link them
either, leaving them reachable by nothing at all.

That is the query shape the baseline actually showed: 'primary schools in
beccles', 'secondary schools in brentwood'. Publishing the routes without a
path in meant building for the demand and then hiding from it.

Place now carries phase_urns so the per-phase threshold can be applied without
re-querying, the sitemap emits a variant wherever a phase clears the threshold
on its own, and the API exposes the qualifying phases so the place page links
only variants that exist. Outcodes are excluded: nobody searches 'primary
schools in SW11' and those routes do not exist.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-21 20:43:07 +01:00

205 lines
7.6 KiB
Python

"""The place registry: what places the site publishes, and what is in each.
One module owns this question. The pages, the sitemap and the internal-link
modules all read from here, so the threshold and the collision rules exist in
exactly one place and are testable without a browser or a database.
Two namespaces, never one. 67 viable town names collide with a local
authority name, and the authority is the larger set in only 43 of them —
postal towns cross authority boundaries, so neither can absorb the other.
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
"""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass, field
logger = logging.getLogger(__name__)
# Five schools with publishable data. Below this a place has nothing to say
# that a list of schools does not, and publishing it is index bloat.
MIN_SCHOOLS = 5
@dataclass(frozen=True)
class Place:
kind: str # "town" | "locality" | "authority" | "outcode"
slug: str
name: str
urns: tuple[int, ...]
parent_authority: str | None # authority NAME, for the 301 target
# URNs per phase, so the per-phase threshold can be applied without
# re-querying. A place with 30 primaries and 2 secondaries publishes a
# primary variant and no secondary one.
phase_urns: dict[str, tuple[int, ...]] = field(default_factory=dict)
def publishes_phase(self, phase: str) -> bool:
return len(self.phase_urns.get(phase, ())) >= MIN_SCHOOLS
@property
def key(self) -> str:
return f"{self.kind}:{self.slug}"
def _publishable_urns(df) -> set[int]:
"""URNs with something a page could state, deduplicated across years."""
from backend.app import _PUBLISHABLE_FIELDS
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
if not cols:
return set()
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]:
"""URNs per phase. All-through schools count toward both, matching the
PHASE_GROUPS mapping the search filters already use."""
from backend.app import PHASE_GROUPS
if "phase" not in group.columns:
return {}
lowered = group["phase"].fillna("").str.lower()
out: dict[str, tuple[int, ...]] = {}
for phase in ("primary", "secondary"):
wanted = PHASE_GROUPS.get(phase, set())
subset = group[lowered.isin(wanted)]
urns = tuple(sorted({int(u) for u in subset["urn"]} & publishable))
if urns:
out[phase] = urns
return out
def _parent_authority(group) -> str | None:
"""The most common authority in a group — the useful 301 target.
A town spanning several authorities has no single parent, so the mode is
the honest answer rather than an arbitrary first row.
"""
if "local_authority" not in group.columns:
return None
top = group["local_authority"].dropna()
return str(top.mode().iloc[0]) if not top.empty else None
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
"""One Place per distinct value of `column` that clears the threshold."""
from backend.app import _slugify
if column not in df.columns:
return {}
out: dict[str, Place] = {}
for name, group in df.groupby(column, dropna=True):
name = str(name).strip()
if not name:
continue
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
slug = _slugify(name)
if not slug:
continue
place = Place(
kind=kind, slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group) if kind == "town" else None,
phase_urns=_phase_urns(group, publishable),
)
out[place.key] = place
return out
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
def _outcode(postcode) -> str | None:
if not isinstance(postcode, str):
return None
m = _OUTCODE_RE.match(postcode.upper().strip())
return m.group(1) if m else None
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
"""One Place per postcode district clearing the threshold.
These carry no phase variants: nobody searches "primary schools in SW11".
"""
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
working = working[working["_oc"].notna()]
out: dict[str, Place] = {}
for oc, group in working.groupby("_oc"):
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
urns=urns, parent_authority=_parent_authority(group),
phase_urns=_phase_urns(group, publishable))
out[place.key] = place
return out
def _locality_places(df, publishable: set[int],
town_slugs: set[str]) -> dict[str, Place]:
"""One Place per curated locality clearing the threshold."""
from backend.localities import LOCALITY_OUTCODES
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
out: dict[str, Place] = {}
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
if slug in town_slugs:
# Skip, do not raise. The guard exists so a locality never
# silently shadows a town — skipping achieves that, and the error
# log makes it loud.
#
# Raising here took down sitemap generation for all 25,000 school
# pages when "richmond" met the GIAS town Richmond in North
# Yorkshire. Worse, GIAS town names change without any code change,
# so a raise means curated data can break the site spontaneously.
# A curation mistake must cost one page, not the sitemap.
logger.error(
"locality %r collides with the published town of the same "
"slug and has been skipped; rename it or remove it", slug)
continue
group = working[working["_oc"].isin(outcodes)]
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
# Not an error — a locality can legitimately be too small. Logged
# because one you meant to publish quietly vanishing is the
# failure worth hearing about.
logger.warning(
"locality %s (%s) has %d publishable schools, below the "
"threshold of %d - not published",
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
continue
place = Place(kind="locality", slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group),
phase_urns=_phase_urns(group, publishable))
out[place.key] = place
return out
def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns:
return {}
publishable = _publishable_urns(df)
registry: dict[str, Place] = {}
registry.update(_group(df, "local_authority", "authority", publishable))
towns = _group(df, "town", "town", publishable)
registry.update(towns)
town_slugs = {p.slug for p in towns.values()}
registry.update(_locality_places(df, publishable, town_slugs))
registry.update(_outcode_places(df, publishable))
return registry