Files
school_compare/backend/places.py
T

92 lines
3.1 KiB
Python
Raw Normal View History

"""The place registry: what places the site publishes, and what is in each.
One module owns this question. The pages, the sitemap and the internal-link
modules all read from here, so the threshold and the collision rules exist in
exactly one place and are testable without a browser or a database.
Two namespaces, never one. 67 viable town names collide with a local
authority name, and the authority is the larger set in only 43 of them —
postal towns cross authority boundaries, so neither can absorb the other.
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
"""
from __future__ import annotations
from dataclasses import dataclass
# Five schools with publishable data. Below this a place has nothing to say
# that a list of schools does not, and publishing it is index bloat.
MIN_SCHOOLS = 5
@dataclass(frozen=True)
class Place:
kind: str # "town" | "locality" | "authority" | "outcode"
slug: str
name: str
urns: tuple[int, ...]
parent_authority: str | None # authority NAME, for the 301 target
@property
def key(self) -> str:
return f"{self.kind}:{self.slug}"
def _publishable_urns(df) -> set[int]:
"""URNs with something a page could state, deduplicated across years."""
from backend.app import _PUBLISHABLE_FIELDS
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
if not cols:
return set()
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
def _parent_authority(group) -> str | None:
"""The most common authority in a group — the useful 301 target.
A town spanning several authorities has no single parent, so the mode is
the honest answer rather than an arbitrary first row.
"""
if "local_authority" not in group.columns:
return None
top = group["local_authority"].dropna()
return str(top.mode().iloc[0]) if not top.empty else None
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
"""One Place per distinct value of `column` that clears the threshold."""
from backend.app import _slugify
if column not in df.columns:
return {}
out: dict[str, Place] = {}
for name, group in df.groupby(column, dropna=True):
name = str(name).strip()
if not name:
continue
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
slug = _slugify(name)
if not slug:
continue
place = Place(
kind=kind, slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(group) if kind == "town" else None,
)
out[place.key] = place
return out
def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns:
return {}
publishable = _publishable_urns(df)
registry: dict[str, Place] = {}
registry.update(_group(df, "local_authority", "authority", publishable))
registry.update(_group(df, "town", "town", publishable))
return registry