feat(places): /api/places registry and place detail endpoints
The registry is cached for the process and reset by the same admin endpoint that rebuilds the sitemaps, so places and sitemap always describe the same corpus rather than drifting apart. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
1 parent
de853b90b3
commit
c5af476213
2 files changed
+147
-1
No files matched your search
+76
-1
@@ -35,6 +35,7 @@ from .data_loader import (
|
||||
search_schools_typesense,
|
||||
)
|
||||
from .data_loader import get_data_info as get_db_info
|
||||
from .places import build_place_registry
|
||||
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||
from .utils import clean_for_json, convert_to_native
|
||||
|
||||
@@ -58,6 +59,12 @@ MAX_SLUG_LENGTH = 60
|
||||
# regenerate endpoint after a pipeline run.
|
||||
_sitemaps: dict[str, str] | None = None
|
||||
|
||||
# Built from the same DataFrame the sitemap uses, so places and sitemap can
|
||||
# never describe different corpora. Reset by the same admin endpoint.
|
||||
_place_registry: dict | None = None
|
||||
|
||||
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
text = text.lower()
|
||||
@@ -170,6 +177,14 @@ SITEMAP_CHUNK_SIZE = 10_000
|
||||
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
||||
|
||||
|
||||
def get_place_registry() -> dict:
|
||||
"""The place registry, built once and cached for the process."""
|
||||
global _place_registry
|
||||
if _place_registry is None:
|
||||
_place_registry = build_place_registry(load_school_data())
|
||||
return _place_registry
|
||||
|
||||
|
||||
def _urlset(rows: list[str]) -> str:
|
||||
return "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
@@ -1117,6 +1132,62 @@ async def get_rankings(
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/places")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def list_places(request: Request):
|
||||
"""Every published place. The sitemap and the link modules read this."""
|
||||
registry = get_place_registry()
|
||||
return {"places": [
|
||||
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
|
||||
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
|
||||
]}
|
||||
|
||||
|
||||
@app.get("/api/places/{kind}/{slug}")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_place(request: Request, kind: str, slug: str,
|
||||
phase: Optional[str] = None):
|
||||
"""One place: its schools ranked, and its local averages."""
|
||||
if kind not in VALID_PLACE_KINDS:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
place = get_place_registry().get(f"{kind}:{slug}")
|
||||
if place is None:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
df = load_latest_school_data()
|
||||
rows = df[df["urn"].isin(place.urns)]
|
||||
|
||||
if phase:
|
||||
wanted = PHASE_GROUPS.get(phase.lower())
|
||||
if wanted and "phase" in rows.columns:
|
||||
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
|
||||
|
||||
# The metric the page ranks on, which is also the one it averages.
|
||||
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
|
||||
if metric in rows.columns:
|
||||
rows = rows.sort_values(metric, ascending=False, na_position="last")
|
||||
|
||||
averages = {
|
||||
m: (None if m not in rows.columns or rows[m].dropna().empty
|
||||
else float(rows[m].dropna().mean()))
|
||||
for m in ("rwm_expected_pct", "attainment_8_score")
|
||||
}
|
||||
|
||||
cols = [c for c in SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
|
||||
"rwm_expected_pct", "attainment_8_score",
|
||||
"total_pupils"]
|
||||
if c in rows.columns]
|
||||
|
||||
return {
|
||||
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
|
||||
"count": len(place.urns),
|
||||
"parent_authority": place.parent_authority},
|
||||
"schools": clean_for_json(rows[cols]),
|
||||
"averages": averages,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/data-info")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_data_info(request: Request):
|
||||
@@ -1228,7 +1299,11 @@ async def regenerate_sitemap(
|
||||
_: bool = Depends(verify_admin_api_key),
|
||||
):
|
||||
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
|
||||
global _sitemaps
|
||||
global _sitemaps, _place_registry
|
||||
# Places and sitemap are rebuilt together — they read the same marts, and
|
||||
# letting them drift apart would submit URLs for places that no longer
|
||||
# exist.
|
||||
_place_registry = None
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
|
||||
|
||||
Reference in new issue
Block a user