"""The place registry: what places the site publishes, and what is in each. One module owns this question. The pages, the sitemap and the internal-link modules all read from here, so the threshold and the collision rules exist in exactly one place and are testable without a browser or a database. Two namespaces, never one. 67 viable town names collide with a local authority name, and the authority is the larger set in only 43 of them — postal towns cross authority boundaries, so neither can absorb the other. Keys are ":" so the collision cannot reappear in the dict. """ from __future__ import annotations import logging import re from dataclasses import dataclass logger = logging.getLogger(__name__) # Five schools with publishable data. Below this a place has nothing to say # that a list of schools does not, and publishing it is index bloat. MIN_SCHOOLS = 5 @dataclass(frozen=True) class Place: kind: str # "town" | "locality" | "authority" | "outcode" slug: str name: str urns: tuple[int, ...] parent_authority: str | None # authority NAME, for the 301 target @property def key(self) -> str: return f"{self.kind}:{self.slug}" def _publishable_urns(df) -> set[int]: """URNs with something a page could state, deduplicated across years.""" from backend.app import _PUBLISHABLE_FIELDS cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns] if not cols: return set() return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int)) def _parent_authority(group) -> str | None: """The most common authority in a group — the useful 301 target. A town spanning several authorities has no single parent, so the mode is the honest answer rather than an arbitrary first row. """ if "local_authority" not in group.columns: return None top = group["local_authority"].dropna() return str(top.mode().iloc[0]) if not top.empty else None def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]: """One Place per distinct value of `column` that clears the threshold.""" from backend.app import _slugify if column not in df.columns: return {} out: dict[str, Place] = {} for name, group in df.groupby(column, dropna=True): name = str(name).strip() if not name: continue urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) if len(urns) < MIN_SCHOOLS: continue slug = _slugify(name) if not slug: continue place = Place( kind=kind, slug=slug, name=name, urns=urns, parent_authority=_parent_authority(group) if kind == "town" else None, ) out[place.key] = place return out # "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter. _OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s") def _outcode(postcode) -> str | None: if not isinstance(postcode, str): return None m = _OUTCODE_RE.match(postcode.upper().strip()) return m.group(1) if m else None def _outcode_places(df, publishable: set[int]) -> dict[str, Place]: """One Place per postcode district clearing the threshold. These carry no phase variants: nobody searches "primary schools in SW11". """ if "postcode" not in df.columns: return {} working = df.assign(_oc=df["postcode"].map(_outcode)) working = working[working["_oc"].notna()] out: dict[str, Place] = {} for oc, group in working.groupby("_oc"): urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) if len(urns) < MIN_SCHOOLS: continue place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc), urns=urns, parent_authority=_parent_authority(group)) out[place.key] = place return out def _locality_places(df, publishable: set[int], town_slugs: set[str]) -> dict[str, Place]: """One Place per curated locality clearing the threshold.""" from backend.localities import LOCALITY_OUTCODES if "postcode" not in df.columns: return {} working = df.assign(_oc=df["postcode"].map(_outcode)) out: dict[str, Place] = {} for slug, (name, outcodes) in LOCALITY_OUTCODES.items(): if slug in town_slugs: # Skip, do not raise. The guard exists so a locality never # silently shadows a town — skipping achieves that, and the error # log makes it loud. # # Raising here took down sitemap generation for all 25,000 school # pages when "richmond" met the GIAS town Richmond in North # Yorkshire. Worse, GIAS town names change without any code change, # so a raise means curated data can break the site spontaneously. # A curation mistake must cost one page, not the sitemap. logger.error( "locality %r collides with the published town of the same " "slug and has been skipped; rename it or remove it", slug) continue group = working[working["_oc"].isin(outcodes)] urns = tuple(sorted({int(u) for u in group["urn"]} & publishable)) if len(urns) < MIN_SCHOOLS: # Not an error — a locality can legitimately be too small. Logged # because one you meant to publish quietly vanishing is the # failure worth hearing about. logger.warning( "locality %s (%s) has %d publishable schools, below the " "threshold of %d - not published", slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS) continue place = Place(kind="locality", slug=slug, name=name, urns=urns, parent_authority=_parent_authority(group)) out[place.key] = place return out def build_place_registry(df) -> dict[str, Place]: """Every place the site publishes, keyed by ":".""" if df.empty or "urn" not in df.columns: return {} publishable = _publishable_urns(df) registry: dict[str, Place] = {} registry.update(_group(df, "local_authority", "authority", publishable)) towns = _group(df, "town", "town", publishable) registry.update(towns) town_slugs = {p.slug for p in towns.values()} registry.update(_locality_places(df, publishable, town_slugs)) registry.update(_outcode_places(df, publishable)) return registry