feat(places): /api/places registry and place detail endpoints

The registry is cached for the process and reset by the same admin endpoint
that rebuilds the sitemaps, so places and sitemap always describe the same
corpus rather than drifting apart.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-21 18:13:05 +01:00
1 parent de853b90b3
commit c5af476213
2 files changed
+147 -1

No files matched your search

+76 -1
View File
@@ -35,6 +35,7 @@ from .data_loader import (
search_schools_typesense,
)
from .data_loader import get_data_info as get_db_info
from .places import build_place_registry
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
from .utils import clean_for_json, convert_to_native
@@ -58,6 +59,12 @@ MAX_SLUG_LENGTH = 60
# regenerate endpoint after a pipeline run.
_sitemaps: dict[str, str] | None = None
# Built from the same DataFrame the sitemap uses, so places and sitemap can
# never describe different corpora. Reset by the same admin endpoint.
_place_registry: dict | None = None
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
def _slugify(text: str) -> str:
text = text.lower()
@@ -170,6 +177,14 @@ SITEMAP_CHUNK_SIZE = 10_000
SITEMAP_CHILD_PREFIX = "/sitemaps"
def get_place_registry() -> dict:
"""The place registry, built once and cached for the process."""
global _place_registry
if _place_registry is None:
_place_registry = build_place_registry(load_school_data())
return _place_registry
def _urlset(rows: list[str]) -> str:
return "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
@@ -1117,6 +1132,62 @@ async def get_rankings(
}
@app.get("/api/places")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def list_places(request: Request):
"""Every published place. The sitemap and the link modules read this."""
registry = get_place_registry()
return {"places": [
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
]}
@app.get("/api/places/{kind}/{slug}")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_place(request: Request, kind: str, slug: str,
phase: Optional[str] = None):
"""One place: its schools ranked, and its local averages."""
if kind not in VALID_PLACE_KINDS:
raise HTTPException(status_code=404, detail="No such place")
place = get_place_registry().get(f"{kind}:{slug}")
if place is None:
raise HTTPException(status_code=404, detail="No such place")
df = load_latest_school_data()
rows = df[df["urn"].isin(place.urns)]
if phase:
wanted = PHASE_GROUPS.get(phase.lower())
if wanted and "phase" in rows.columns:
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
# The metric the page ranks on, which is also the one it averages.
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
if metric in rows.columns:
rows = rows.sort_values(metric, ascending=False, na_position="last")
averages = {
m: (None if m not in rows.columns or rows[m].dropna().empty
else float(rows[m].dropna().mean()))
for m in ("rwm_expected_pct", "attainment_8_score")
}
cols = [c for c in SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
"rwm_expected_pct", "attainment_8_score",
"total_pupils"]
if c in rows.columns]
return {
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
"count": len(place.urns),
"parent_authority": place.parent_authority},
"schools": clean_for_json(rows[cols]),
"averages": averages,
}
@app.get("/api/data-info")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_data_info(request: Request):
@@ -1228,7 +1299,11 @@ async def regenerate_sitemap(
_: bool = Depends(verify_admin_api_key),
):
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
global _sitemaps
global _sitemaps, _place_registry
# Places and sitemap are rebuilt together — they read the same marts, and
# letting them drift apart would submit URLs for places that no longer
# exist.
_place_registry = None
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}