fix(api): publish a validated dataset, and stop truncating search

Reload cleared the caches first and rebuilt afterwards, so any failure
left the API serving nothing, and requests arriving mid-reload saw a
half-swapped state. It now builds and validates the replacement frames,
place registry, reverse index and sitemaps off the request loop, then
publishes them in one synchronous step under a lock. A failed reload
returns 503 and keeps the previous data. Sitemap regeneration takes the
same path rather than clearing the live registry up front.

Typesense search returned at most one page of hits and used an empty
list for both "no matches" and "search is down", so a genuine empty
result silently fell back to substring matching. It now pages through
every candidate and returns None only on failure; the fallback matches
literally, since a query containing regex metacharacters used to throw.

Empty datasets answer 503 rather than 200-with-nothing or a misleading
404, so callers can tell an outage from an absent school. Adds
/api/release, which reports the build identity baked into the image.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5 committed 2026-09-15 10:16:58 +01:00
1 parent 38bc17cab3
commit 9b75f54206
4 files changed
+253 -47

No files matched your search

+78 -33
View File
@@ -26,7 +26,8 @@ from starlette.middleware.base import BaseHTTPMiddleware
import asyncio
from .config import settings
from .data_loader import (
clear_cache,
build_latest_school_data,
load_school_data_as_dataframe,
compute_benchmarks,
load_school_data,
load_latest_school_data,
@@ -272,7 +273,7 @@ def _places_payload(urn: int) -> list[dict]:
return payload
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
"""A <url> per place, plus a phase variant wherever that phase clears the
threshold on its own.
@@ -282,7 +283,9 @@ def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
linked from the place page either.
"""
rows: list[str] = []
for p in sorted(get_place_registry().values(), key=lambda p: (p.kind, p.slug)):
if registry is None:
registry = get_place_registry()
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)):
if p.kind not in kinds:
continue
rows.append(_url_element(BASE_URL + _place_url(p)))
@@ -296,9 +299,10 @@ def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
return rows
def build_sitemaps() -> dict[str, str]:
def build_sitemaps(df=None, registry=None) -> dict[str, str]:
"""Build the sitemap index and every child, keyed by name."""
df = load_school_data()
if df is None:
df = load_school_data()
children: dict[str, str] = {
"static.xml": _urlset(
@@ -318,7 +322,7 @@ def build_sitemaps() -> dict[str, str]:
# measured apart from the school pages'.
for label, kinds in (("places", ("town", "locality", "authority")),
("outcodes", ("outcode",))):
rows = _place_sitemap_rows(kinds)
rows = _place_sitemap_rows(kinds, registry)
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1):
@@ -700,6 +704,15 @@ async def get_config():
}
@app.get("/api/release")
async def release_identity():
import json
from pathlib import Path
path = Path(__file__).with_name("build-info.json")
identity = json.loads(path.read_text()) if path.exists() else {"sha": "development", "build_id": "development"}
return JSONResponse(identity, headers={"Cache-Control": "no-store"})
@app.get("/api/schools")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_schools(
@@ -736,7 +749,7 @@ async def get_schools(
df_latest = load_latest_school_data()
if df_latest.empty:
return {"schools": [], "total": 0, "page": page, "page_size": 0}
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
# Use configured default if not specified
if page_size is None:
@@ -835,8 +848,8 @@ async def get_schools(
# Apply filters
if search:
ts_urns = search_schools_typesense(search)
if ts_urns:
ts_urns = await asyncio.to_thread(search_schools_typesense, search)
if ts_urns is not None:
urn_order = {urn: i for i, urn in enumerate(ts_urns)}
schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy()
schools_df["_ts_rank"] = schools_df["urn"].map(urn_order)
@@ -844,9 +857,9 @@ async def get_schools(
else:
# Fallback: Typesense unavailable, use substring match
search_lower = search.lower()
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False)
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False, regex=False)
if "address" in schools_df.columns:
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False)
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False, regex=False)
schools_df = schools_df[mask]
if local_authority:
@@ -905,7 +918,7 @@ async def get_school_details(request: Request, urn: int):
df = load_school_data()
if df.empty:
raise HTTPException(status_code=404, detail="No data available")
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
school_data = df[df["urn"] == urn]
@@ -1542,20 +1555,51 @@ async def get_data_info(request: Request):
}
_publication_lock = asyncio.Lock()
def _prepare_publication(df):
if df.empty:
raise ValueError("Refusing to publish an empty school dataset")
if not {"urn", "year", "school_name"}.issubset(df.columns):
raise ValueError("School dataset is missing required columns")
if df["urn"].isna().any() or df.duplicated(["urn", "year"]).any():
raise ValueError("School dataset has missing URNs or duplicate school years")
latest = build_latest_school_data(df)
registry = build_place_registry(df)
index = build_place_index(registry)
sitemaps = build_sitemaps(df, registry)
return df, latest, registry, index, sitemaps
def _publish(prepared):
# Called on the event loop with no await: routes cannot observe half a swap.
# The application currently runs one worker; replicas require coordination.
from . import data_loader
global _place_registry, _place_index, _place_index_source, _sitemaps
df, latest, registry, index, sitemaps = prepared
data_loader._df_cache = df
data_loader._df_latest_cache = latest
_place_registry = registry
_place_index = index
_place_index_source = registry
_sitemaps = sitemaps
@app.post("/api/admin/reload")
@limiter.limit("5/minute")
async def reload_data(
request: Request,
_: bool = Depends(verify_admin_api_key)
):
"""
Admin endpoint to force data reload (useful after data updates).
Requires X-API-Key header with valid admin API key.
"""
clear_cache()
await asyncio.to_thread(load_school_data)
await asyncio.to_thread(load_latest_school_data)
return {"status": "reloaded"}
async def reload_data(request: Request, _: bool = Depends(verify_admin_api_key)):
"""Validate a complete replacement before publishing it; retain data on failure."""
async with _publication_lock:
try:
df = await asyncio.to_thread(load_school_data_as_dataframe)
prepared = await asyncio.to_thread(_prepare_publication, df)
except Exception as exc:
import logging
logging.getLogger(__name__).exception("Dataset reload failed")
raise HTTPException(status_code=503, detail="Dataset reload failed; previous data retained") from exc
_publish(prepared)
return {"status": "reloaded", "schools": len(prepared[1])}
@@ -1607,15 +1651,16 @@ async def regenerate_sitemap(
request: Request,
_: bool = Depends(verify_admin_api_key),
):
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
global _sitemaps, _place_registry
# Places and sitemap are rebuilt together — they read the same marts, and
# letting them drift apart would submit URLs for places that no longer
# exist.
_place_registry = None
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
"""Rebuild derived publication data without clearing the live registry."""
async with _publication_lock:
try:
prepared = await asyncio.to_thread(_prepare_publication, load_school_data())
except Exception as exc:
raise HTTPException(status_code=503, detail="Sitemap rebuild failed; previous data retained") from exc
_publish(prepared)
n = sum(x.count("<url>") for x in prepared[4].values())
return {"status": "ok", "urls": n, "sitemaps": len(prepared[4])}
# Mount static files directly (must be after all routes to avoid catching API calls)