""" SchoolCompare.co.uk API Serves primary and secondary school performance data for comparing schools. Uses real data from UK Government Compare School Performance downloads. """ import hashlib import re import time from contextlib import asynccontextmanager from datetime import datetime, timezone from typing import Optional import numpy as np import pandas as pd from fastapi import FastAPI, HTTPException, Query, Request, Depends, Header from fastapi.middleware.cors import CORSMiddleware from fastapi.middleware.gzip import GZipMiddleware from fastapi.responses import FileResponse, JSONResponse, Response from fastapi.staticfiles import StaticFiles from slowapi import Limiter, _rate_limit_exceeded_handler from slowapi.util import get_remote_address from slowapi.errors import RateLimitExceeded from starlette.middleware.base import BaseHTTPMiddleware import asyncio from .config import settings from .data_loader import ( build_latest_school_data, load_school_data_as_dataframe, compute_benchmarks, load_school_data, load_latest_school_data, geocode_single_postcode, get_supplementary_data, get_supplementary_data_batch, search_schools_typesense, suggest_schools_typesense, ) from .data_loader import get_data_info as get_db_info from . import flags from .places import build_place_index, build_place_registry, places_for_urn from .schemas import METRIC_DEFINITIONS, PHASE_GROUPS, RANKING_COLUMNS, SCHOOL_COLUMNS from .similar_schools import is_secondary_phase, select_similar from .utils import clean_for_json, convert_to_native # Values to exclude from filter dropdowns (empty strings, non-applicable labels) EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"} # Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a # sitemap that redirects wastes a crawl on every URL it lists. BASE_URL = "https://www.schoolcompare.co.uk" MAX_SLUG_LENGTH = 60 # In-memory sitemap cache: name -> XML. Populated on startup and by the admin # regenerate endpoint after a pipeline run. _sitemaps: dict[str, str] | None = None # Built from the same DataFrame the sitemap uses, so places and sitemap can # never describe different corpora. Reset by the same admin endpoint. _place_registry: dict | None = None # Cached beside the registry, and invalidated by identity against it — see # get_place_index. Never cleared independently. _place_index: dict | None = None _place_index_source: dict | None = None VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode") def _slugify(text: str) -> str: text = text.lower() text = re.sub(r"[^\w\s-]", "", text) text = re.sub(r"\s+", "-", text) text = re.sub(r"-+", "-", text) return text.strip("-") def _school_url(urn: int, school_name: str) -> str: slug = _slugify(school_name) if len(slug) > MAX_SLUG_LENGTH: slug = slug[:MAX_SLUG_LENGTH].rstrip("-") return f"/school/{urn}-{slug}" # Routes worth submitting that are not a school page. /admissions was missing # from the sitemap entirely despite being a static, indexable guide. STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions") # A page has something a search result could state if any of these is present # in any year. Shared by _has_publishable_data and the per-school check in # _school_sitemap_rows so the two can never drift. _PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade") def _has_publishable_data(row) -> bool: """True when a school page has something a search result could state. A school with no results in any year and no Ofsted grade renders an empty page. Submitting it spends crawl budget and drags the corpus-wide quality signal down, so it stays out of the sitemap. The page itself still resolves for anyone who has the URL. """ for field in _PUBLISHABLE_FIELDS: value = row.get(field) if value is not None and not pd.isna(value): return True return False def _url_element(loc: str, lastmod: str | None = None) -> str: """One entry. No priority or changefreq — Google ignores both.""" body = f"{loc}" if lastmod: body += f"{lastmod}" return f" {body}" def _school_sitemap_rows(df) -> list[str]: """A element per school that has something to show. lastmod comes from the school's Ofsted date where there is one and is omitted otherwise. An always-now lastmod is a claim Google learns to distrust; an absent one honestly means "unknown". """ if df.empty or "urn" not in df.columns or "school_name" not in df.columns: return [] rows: list[str] = [] seen: set[int] = set() # Publishable is a property of the SCHOOL, not of its latest row. # # The first cut tested the latest year's row alone, which quietly dropped # every school that has results in its history but a null row for the most # recent year — a school that stopped reporting, or whose figures were # suppressed for small-cohort disclosure. The Mallard Academy (150367) is # the case that caught it: real KS2 results for 2015-16 through 2018-19, # then null rows for 2022-23 onward. Its page shows all four years; the # sitemap omitted it. Roughly 220 schools were affected. publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns] publishable: set[int] = ( set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int)) if publishable_cols else set() ) # Latest row per URN first, so a school's most recent Ofsted date wins. ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df for _, row in ordered.iterrows(): urn = int(row["urn"]) if urn in seen: continue seen.add(urn) if urn not in publishable: continue lastmod = None ofsted_date = row.get("ofsted_date") if ofsted_date is not None and not pd.isna(ofsted_date): lastmod = pd.Timestamp(ofsted_date).date().isoformat() rows.append(_url_element( BASE_URL + _school_url(urn, str(row["school_name"])), lastmod)) return rows # Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to # scan by eye in Search Console, which is the point of splitting at all: # coverage is reported per submitted sitemap, so one file per page family is # what makes an indexation problem attributable to a family. SITEMAP_CHUNK_SIZE = 10_000 # Children are served under /sitemaps/ because Next.js only treats a whole # bracketed path segment as dynamic — a route folder named "sitemap-[...parts]" # is read as a literal static segment and never matches. SITEMAP_CHILD_PREFIX = "/sitemaps" def get_place_registry() -> dict: """The place registry, built once and cached for the process.""" global _place_registry if _place_registry is None: _place_registry = build_place_registry(load_school_data()) return _place_registry def get_place_index() -> dict: """URN → its published places, cached against the registry it came from. Invalidation is an identity check rather than a second flag to remember to clear. Anything that drops `_place_registry` — the tests all do — gets a fresh registry object here, which no longer matches the one the index was built from, so the index rebuilds with it. A separate `_place_index = None` would be one more thing to forget, and a stale reverse index is exactly the bug that would put links to another dataset's places on a school page. """ global _place_index, _place_index_source registry = get_place_registry() if _place_index is None or _place_index_source is not registry: _place_index = build_place_index(registry) _place_index_source = registry return _place_index def _urlset(rows: list[str]) -> str: return "\n".join([ '', '', *rows, "", ]) def _place_url(place) -> str: """The canonical path for a place. Two namespaces, per the spec. Towns and localities share /schools/[place]; authorities take their own prefix because 67 town names collide with an authority name and neither set contains the other. """ if place.kind == "authority": return f"/schools/authority/{place.slug}" if place.kind == "outcode": return f"/schools/near/{place.slug}" return f"/schools/{place.slug}" def _places_payload(urn: int) -> list[dict]: """The published places containing this school, as the school page needs them: a name to write in the link, a count so the anchor can say what it leads to, and the canonical path. `phases` carries the phase variants this school actually appears on, which is usually one and is two for an all-through school — it is listed on both pages, so there is no tie to break. Membership is read straight from the registry's own `phase_urns` rather than re-derived from the school's phase string. The registry is the one place that decides which phases a place publishes and who is on them; computing it a second time here is how a page comes to link a school to a phase page that does not list it, or to a route that does not exist. That is also why outcodes need no special case: they carry empty `phase_urns`, so they report no phase links on their own. """ payload = [] for place in places_for_urn(get_place_index(), int(urn)): phases = [ { "phase": phase, "count": len(phase_urns), "url": f"{_place_url(place)}/{phase}", } for phase, phase_urns in sorted(place.phase_urns.items()) if int(urn) in phase_urns ] payload.append({ "kind": place.kind, "slug": place.slug, "name": place.name, "count": len(place.urns), "url": _place_url(place), "phases": phases, }) return payload def _similar_schools_payload(urn: int, phase: str | None) -> list[dict]: """Nearby schools this page may offer as alternatives. Wrapped: a failure in selection must never 500 a page that is otherwise complete, which is the posture get_supplementary_data already takes. The section simply does not render. """ try: # Decided in similar_schools, beside the PHASE_GROUPS bucket it selects # from, so the two cannot drift. A substring test for "secondary" here # would miss "16 plus" and hand a sixth-form college the primary bucket. return select_similar( load_latest_school_data(), int(urn), is_secondary_phase(phase) ) except Exception: import logging logging.getLogger(__name__).exception( "Similar schools selection failed for urn=%s", urn ) return [] def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]: """A per place, plus a phase variant wherever that phase clears the threshold on its own. Phase is part of the query — "primary schools in beccles" — so each variant is its own indexable page. Submitting only the bare place URL left ~950 of them reachable by nothing: absent from every sitemap, and not linked from the place page either. """ rows: list[str] = [] if registry is None: registry = get_place_registry() for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)): if p.kind not in kinds: continue rows.append(_url_element(BASE_URL + _place_url(p))) # Which phases a place publishes is the registry's decision alone — # outcodes report none, because the spec gives them no phase route. # Repeating that rule here was how the page and the sitemap came to # disagree about which URLs exist. for phase in ("primary", "secondary"): if p.publishes_phase(phase): rows.append(_url_element(f"{BASE_URL}{_place_url(p)}/{phase}")) return rows def build_sitemaps(df=None, registry=None) -> dict[str, str]: """Build the sitemap index and every child, keyed by name.""" if df is None: df = load_school_data() children: dict[str, str] = { "static.xml": _urlset( [_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]), } school_rows = _school_sitemap_rows(df) # Always emit at least one school child, so the index shape is stable even # on an empty database. chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE] for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]] for n, chunk in enumerate(chunks, start=1): children[f"schools-{n}.xml"] = _urlset(chunk) # Separate children per family: Search Console reports coverage per # submitted sitemap, which is how the location layer's indexation is # measured apart from the school pages'. for label, kinds in (("places", ("town", "locality", "authority")), ("outcodes", ("outcode",))): rows = _place_sitemap_rows(kinds, registry) chunks = [rows[i:i + SITEMAP_CHUNK_SIZE] for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]] for n, chunk in enumerate(chunks, start=1): children[f"{label}-{n}.xml"] = _urlset(chunk) # On a sitemap index, lastmod means "when this sitemap file last changed", # so generation time is the correct value here — unlike on a , where # it would be a claim about content we cannot support. generated = datetime.now(timezone.utc).date().isoformat() index_rows = [ f" {BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}" f"{generated}" for name in children ] index = "\n".join([ '', '', *index_rows, "", ]) return {**children, "sitemap.xml": index} def build_sitemap() -> str: """The sitemap index. Kept for `lifespan` and the admin endpoint.""" return build_sitemaps()["sitemap.xml"] def clean_filter_values(series: pd.Series) -> list[str]: """Return sorted unique values from a Series, excluding NaN and junk labels.""" return sorted( v for v in series.dropna().unique().tolist() if v not in EXCLUDED_FILTER_VALUES ) # ============================================================================= # SECURITY MIDDLEWARE & HELPERS # ============================================================================= def client_key(request: Request) -> str: """The rate-limit bucket: the real caller, not the proxy in front of them. `get_remote_address` reads request.client.host. In staging and production the backend has no published ports and sits on the internal network, so its only caller is the Next proxy — meaning every browser user on the site shared one bucket. Measured before this fix: 70 concurrent requests to /api/schools returned 60 OK and 10 refused. CF-Connecting-IP first, because Cloudflare (in front of both environments) sets it on every origin request and *overwrites* any client-supplied value, which a parsed X-Forwarded-For chain does not guarantee. The XFF fallback is forgeable, but only by a caller already inside the Docker network, which is the one place nothing untrusted can reach. """ cf = request.headers.get("cf-connecting-ip") if cf: return cf.strip() xff = request.headers.get("x-forwarded-for") if xff: return xff.split(",")[0].strip() return get_remote_address(request) # Per-client limiter. Paired with the global ceiling below — the two do # different jobs and neither substitutes for the other. limiter = Limiter(key_func=client_key) # --- The ceiling no header can raise ---------------------------------------- # # client_key trusts CF-Connecting-IP, and nothing in this process can tell an # edge-set header from an attacker-set one. That distinction can only be made # at Cloudflare, with Authenticated Origin Pulls or an origin firewall. A # caller reaching the origin directly could otherwise mint a fresh rate-limit # bucket per request and evade per-client limits entirely — which would make # correct keying a net regression against abuse, since the single shared bucket # it replaced at least capped everyone at 60/minute together. # # So per-client limits give fairness, and this gives the origin a hard total. # It does not make the header trustworthy; it bounds what trusting it can cost. # The header problem itself is closed at Cloudflare, not here. # # [window_start_monotonic, count], or None before the first request. A fixed # window is crude, which is right for a backstop: it has to be obviously # correct rather than fair. _global_window: Optional[list] = None # The container healthcheck runs `curl http://localhost:80/api/data-info` from # inside the container. Starving it would fail the check, restart the # container, and turn a load spike into an outage loop — the ceiling exists to # protect the origin, not to kill it. _LOCAL_HOSTS = frozenset({"127.0.0.1", "::1", "localhost"}) def exempt_from_ceiling(request: Request) -> bool: """Whether the ceiling should ignore this request. Its own function so the rule is testable without standing up a server — and so the healthcheck exemption is somewhere a reader can find it. """ if not request.url.path.startswith("/api/"): return True # The peer address, never the Host header: Host is set by the caller and # would hand every attacker an exemption. return (request.client.host if request.client else "") in _LOCAL_HOSTS class GlobalRateLimitMiddleware(BaseHTTPMiddleware): """A cap on total /api/ traffic, independent of any client identity.""" async def dispatch(self, request: Request, call_next): global _global_window if exempt_from_ceiling(request): return await call_next(request) now = time.monotonic() # One event loop, and no await between the read and the write, so this # sequence is atomic without a lock. if _global_window is None or now - _global_window[0] >= 60: _global_window = [now, 0] _global_window[1] += 1 if _global_window[1] > settings.global_rate_limit_per_minute: return JSONResponse( # Distinguishable from slowapi's per-client 429: an operator # reading logs has to be able to tell "one noisy client" from # "the origin is saturated". {"detail": "The service is at capacity. Please retry shortly."}, status_code=429, headers={"Retry-After": str(max(1, int(60 - (now - _global_window[0]))))}, ) return await call_next(request) class SecurityHeadersMiddleware(BaseHTTPMiddleware): """Add security headers to all responses.""" async def dispatch(self, request: Request, call_next): response = await call_next(request) # Prevent clickjacking response.headers["X-Frame-Options"] = "DENY" # Prevent MIME type sniffing response.headers["X-Content-Type-Options"] = "nosniff" # XSS Protection (legacy browsers) response.headers["X-XSS-Protection"] = "1; mode=block" # Referrer policy response.headers["Referrer-Policy"] = "strict-origin-when-cross-origin" # Permissions policy (restrict browser features) response.headers["Permissions-Policy"] = ( "geolocation=(), microphone=(), camera=(), payment=()" ) # Content Security Policy response.headers["Content-Security-Policy"] = ( "default-src 'self'; " "script-src 'self' 'unsafe-inline' https://cdn.jsdelivr.net https://unpkg.com https://analytics.schoolcompare.co.uk; " "style-src 'self' 'unsafe-inline' https://fonts.googleapis.com https://cdn.jsdelivr.net https://unpkg.com; " "font-src 'self' https://fonts.gstatic.com; " "img-src 'self' data: https://*.tile.openstreetmap.org https://unpkg.com; " "connect-src 'self' https://cdn.jsdelivr.net https://*.tile.openstreetmap.org https://unpkg.com https://analytics.schoolcompare.co.uk; " "frame-ancestors 'none'; " "base-uri 'self'; " "form-action 'self' https://formsubmit.co;" ) # HSTS (only enable if using HTTPS in production) response.headers["Strict-Transport-Security"] = ( "max-age=31536000; includeSubDomains" ) return response # Per-path Cache-Control rules. Keys are matched as path prefixes (longest wins). # Values: (max_age, s_maxage, stale_while_revalidate) CACHE_RULES: list[tuple[str, tuple[int, int, int]]] = [ ("/api/filters", (300, 86400, 604800)), ("/api/metrics", (300, 86400, 604800)), ("/api/national-averages", (300, 86400, 604800)), ("/api/la-averages", (300, 86400, 604800)), ("/api/data-info", (300, 86400, 604800)), ("/api/schools/", (300, 3600, 86400)), # /api/schools/{urn} ("/api/rankings", (60, 600, 3600)), ("/api/compare", (60, 600, 3600)), ("/api/suggest", (60, 3600, 86400)), # autosuggest ("/api/schools", (30, 300, 1800)), # search list ] def _cache_control_for_path(path: str) -> Optional[str]: # Longest-prefix match best: Optional[tuple[int, tuple[int, int, int]]] = None for prefix, vals in CACHE_RULES: if path.startswith(prefix) and (best is None or len(prefix) > best[0]): best = (len(prefix), vals) if best is None: return None max_age, s_maxage, swr = best[1] return f"public, max-age={max_age}, s-maxage={s_maxage}, stale-while-revalidate={swr}" class CacheAndETagMiddleware(BaseHTTPMiddleware): """Set Cache-Control on cacheable API responses and serve 304s via ETag.""" async def dispatch(self, request: Request, call_next): response = await call_next(request) # Only cache GETs that succeeded. if request.method != "GET" or response.status_code != 200: return response cache_header = _cache_control_for_path(request.url.path) if cache_header is None: return response # Drain body so we can hash it for ETag. body_chunks = [] async for chunk in response.body_iterator: body_chunks.append(chunk) body = b"".join(body_chunks) etag = '"' + hashlib.md5(body).hexdigest() + '"' headers = dict(response.headers) headers["Cache-Control"] = cache_header headers["ETag"] = etag headers["Vary"] = ", ".join(filter(None, [headers.get("Vary"), "Accept-Encoding"])) inm = request.headers.get("if-none-match") if inm and inm == etag: # Strip content headers on 304. for h in ("Content-Length", "content-length", "Content-Type", "content-type"): headers.pop(h, None) return Response(status_code=304, headers=headers) return Response(content=body, status_code=200, headers=headers, media_type=response.media_type) class RequestSizeLimitMiddleware(BaseHTTPMiddleware): """Limit request body size to prevent DoS attacks.""" async def dispatch(self, request: Request, call_next): content_length = request.headers.get("content-length") if content_length: if int(content_length) > settings.max_request_size: return Response( content="Request too large", status_code=413, ) return await call_next(request) def verify_admin_api_key(x_api_key: str = Header(None)) -> bool: """Verify admin API key for protected endpoints.""" if not x_api_key or x_api_key != settings.admin_api_key: raise HTTPException( status_code=401, detail="Invalid or missing API key", headers={"WWW-Authenticate": "ApiKey"}, ) return True # Input validation helpers def sanitize_search_input(value: Optional[str], max_length: int = 100) -> Optional[str]: """Sanitize search input to prevent injection attacks.""" if value is None: return None # Strip whitespace and limit length value = value.strip()[:max_length] # Remove potentially dangerous characters (allow alphanumeric, spaces, common punctuation) value = re.sub(r"[^\w\s\-\',\.]", "", value) return value if value else None def validate_postcode(postcode: Optional[str]) -> Optional[str]: """Validate and normalize UK postcode format.""" if not postcode: return None postcode = postcode.strip().upper() # UK postcode pattern pattern = r"^[A-Z]{1,2}[0-9][A-Z0-9]?\s*[0-9][A-Z]{2}$" if not re.match(pattern, postcode): return None return postcode @asynccontextmanager async def lifespan(app: FastAPI): """Application lifespan - startup and shutdown events.""" global _sitemaps flags.init() print("Loading school data from marts...") df = load_school_data() if df.empty: print("Warning: No data in marts. Run the annual EES pipeline to populate KS2 data.") else: print(f"Data loaded successfully: {len(df)} records.") # Pre-compute the latest-year snapshot so the first search request is fast await asyncio.to_thread(load_latest_school_data) try: _sitemaps = build_sitemaps() n = sum(x.count("") for x in _sitemaps.values()) print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.") except Exception as e: print(f"Warning: sitemap build failed on startup: {e}") yield print("Shutting down...") app = FastAPI( title="SchoolCompare API", description="API for comparing primary and secondary school performance data - schoolcompare.co.uk", version="2.0.0", lifespan=lifespan, # Disable docs in production for security docs_url="/docs" if settings.debug else None, redoc_url="/redoc" if settings.debug else None, openapi_url="/openapi.json" if settings.debug else None, ) # Add rate limiter app.state.limiter = limiter app.add_exception_handler(RateLimitExceeded, _rate_limit_exceeded_handler) # Middleware (Starlette runs the last-added middleware first on the way out, # so list outermost-last: GZip wraps everything and compresses the final body). app.add_middleware(CacheAndETagMiddleware) app.add_middleware(SecurityHeadersMiddleware) app.add_middleware(RequestSizeLimitMiddleware) app.add_middleware(GZipMiddleware, minimum_size=512) # Added last, so it is outermost and refuses before anything downstream does # work. A ceiling that only applies after the expensive part has run is not a # ceiling. app.add_middleware(GlobalRateLimitMiddleware) # CORS middleware - restricted for production app.add_middleware( CORSMiddleware, allow_origins=settings.allowed_origins, allow_credentials=False, # Don't allow credentials unless needed allow_methods=["GET", "POST"], # Only allow needed methods allow_headers=["Content-Type", "X-API-Key"], # Only allow needed headers ) @app.get("/") async def root(): """Serve the frontend.""" return FileResponse(settings.frontend_dir / "index.html") @app.get("/compare") async def serve_compare(): """Serve the frontend for /compare route (SPA routing).""" return FileResponse(settings.frontend_dir / "index.html") @app.get("/rankings") async def serve_rankings(): """Serve the frontend for /rankings route (SPA routing).""" return FileResponse(settings.frontend_dir / "index.html") @app.get("/api/config") async def get_config(): """Return public configuration for the frontend.""" return { "ga_measurement_id": settings.ga_measurement_id } @app.get("/api/release") async def release_identity(): import json from pathlib import Path path = Path(__file__).with_name("build-info.json") identity = json.loads(path.read_text()) if path.exists() else {"sha": "development", "build_id": "development"} return JSONResponse(identity, headers={"Cache-Control": "no-store"}) @app.get("/api/schools") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_schools( request: Request, search: Optional[str] = Query(None, description="Search by school name", max_length=100), local_authority: Optional[str] = Query( None, description="Filter by local authority", max_length=100 ), school_type: Optional[str] = Query(None, description="Filter by school type", max_length=100), phase: Optional[str] = Query(None, description="Filter by phase: primary, secondary, all-through", max_length=50), postcode: Optional[str] = Query(None, description="Search near postcode", max_length=10), radius: float = Query(5.0, ge=0.1, le=5, description="Search radius in miles"), page: int = Query(1, ge=1, le=1000, description="Page number"), page_size: int = Query(25, ge=1, le=500, description="Results per page"), gender: Optional[str] = Query(None, description="Filter by gender (Mixed/Boys/Girls)", max_length=50), admissions_policy: Optional[str] = Query(None, description="Filter by admissions policy", max_length=100), has_sixth_form: Optional[str] = Query(None, description="Filter by sixth form presence: yes/no", max_length=3), ): """ Get list of schools with pagination. Returns paginated results with total count for efficient loading. Supports location-based search using postcode and phase filtering. """ # Sanitize inputs search = sanitize_search_input(search) local_authority = sanitize_search_input(local_authority) school_type = sanitize_search_input(school_type) phase = sanitize_search_input(phase) postcode = validate_postcode(postcode) # Load the pre-computed latest-year snapshot (cached after first request / startup). # This avoids rebuilding the expensive groupby + prev-year merge on every search. df_latest = load_latest_school_data() if df_latest.empty: raise HTTPException(status_code=503, detail="School data temporarily unavailable") # Use configured default if not specified if page_size is None: page_size = settings.default_page_size # Phase filter — uses PHASE_GROUPS so all-through/middle schools appear # in the correct phase(s) rather than being invisible to both filters. if phase: phase_lower = phase.lower().replace("_", "-") allowed = PHASE_GROUPS.get(phase_lower) if allowed: df_latest = df_latest[df_latest["phase"].str.lower().isin(allowed)] # Secondary-specific filters (after phase filter) if gender: df_latest = df_latest[df_latest["gender"].str.lower() == gender.lower()] if admissions_policy: df_latest = df_latest[df_latest["admissions_policy"].str.lower() == admissions_policy.lower()] # GIAS OfficialSixthForm flag (dim_school.has_sixth_form). NULL (flag not # yet populated by the pipeline) is treated as "no sixth form". if has_sixth_form in ("yes", "no"): if "has_sixth_form" in df_latest.columns: flag = df_latest["has_sixth_form"].eq(True) else: # Defensive fallback only — data_loader now always synthesizes # has_sixth_form as NULL when the DB predates the pipeline re-run, # so this branch shouldn't normally trigger. Falls back to age # range if the column is somehow absent anyway. flag = df_latest["age_range"].str.contains("18", na=False) df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag] # Include key result metrics for display on cards location_cols = ["latitude", "longitude"] result_cols = [ "phase", "year", "rwm_expected_pct", "rwm_high_pct", "prev_rwm_expected_pct", "prev_attainment_8_score", "reading_expected_pct", "writing_expected_pct", "maths_expected_pct", "total_pupils", "attainment_8_score", "english_maths_standard_pass_pct", ] available_cols = [ c for c in SCHOOL_COLUMNS + location_cols + result_cols if c in df_latest.columns ] # fact_performance guarantees one row per (urn, year); df_latest has one row per urn. schools_df = df_latest[available_cols] # Location-based search (uses pre-geocoded data from database) search_coords = None if postcode: # Offload the synchronous HTTP call to a thread so the event loop stays free coords = await asyncio.to_thread(geocode_single_postcode, postcode) if coords: search_coords = coords schools_df = schools_df.copy() # Filter by distance using pre-geocoded lat/long from database # Use vectorized haversine calculation for better performance lat1, lon1 = search_coords # Handle potential duplicate columns by taking first occurrence lat_col = schools_df.loc[:, "latitude"] lon_col = schools_df.loc[:, "longitude"] if isinstance(lat_col, pd.DataFrame): lat_col = lat_col.iloc[:, 0] if isinstance(lon_col, pd.DataFrame): lon_col = lon_col.iloc[:, 0] lat2 = lat_col.values lon2 = lon_col.values # Vectorized haversine formula R = 3959 # Earth's radius in miles lat1_rad = np.radians(lat1) lat2_rad = np.radians(lat2) dlat = np.radians(lat2 - lat1) dlon = np.radians(lon2 - lon1) a = np.sin(dlat / 2) ** 2 + np.cos(lat1_rad) * np.cos(lat2_rad) * np.sin(dlon / 2) ** 2 c = 2 * np.arctan2(np.sqrt(a), np.sqrt(1 - a)) distances = R * c # Handle missing coordinates has_coords = ~(pd.isna(lat_col) | pd.isna(lon_col)) distances = np.where(has_coords.values, distances, float("inf")) schools_df["distance"] = distances schools_df = schools_df[schools_df["distance"] <= radius] schools_df = schools_df.sort_values("distance") # Apply filters if search: ts_urns = await asyncio.to_thread(search_schools_typesense, search) if ts_urns is not None: urn_order = {urn: i for i, urn in enumerate(ts_urns)} schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy() schools_df["_ts_rank"] = schools_df["urn"].map(urn_order) schools_df = schools_df.sort_values("_ts_rank").drop(columns=["_ts_rank"]) else: # Fallback: Typesense unavailable, use substring match search_lower = search.lower() mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False, regex=False) if "address" in schools_df.columns: mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False, regex=False) schools_df = schools_df[mask] if local_authority: schools_df = schools_df[ schools_df["local_authority"].str.lower() == local_authority.lower() ] if school_type: schools_df = schools_df[ schools_df["school_type"].str.lower() == school_type.lower() ] # Compute result-scoped filter values (before pagination). # Gender and admissions are secondary-only filters — scope them to schools # with KS4 data so they don't appear for purely primary result sets. _sec_mask = schools_df["attainment_8_score"].notna() if "attainment_8_score" in schools_df.columns else pd.Series(False, index=schools_df.index) result_filters = { "local_authorities": clean_filter_values(schools_df["local_authority"]) if "local_authority" in schools_df.columns else [], "school_types": clean_filter_values(schools_df["school_type"]) if "school_type" in schools_df.columns else [], "phases": clean_filter_values(schools_df["phase"]) if "phase" in schools_df.columns else [], "genders": clean_filter_values(schools_df.loc[_sec_mask, "gender"]) if "gender" in schools_df.columns and _sec_mask.any() else [], "admissions_policies": clean_filter_values(schools_df.loc[_sec_mask, "admissions_policy"]) if "admissions_policy" in schools_df.columns and _sec_mask.any() else [], } # Pagination total = len(schools_df) start_idx = (page - 1) * page_size end_idx = start_idx + page_size schools_df = schools_df.iloc[start_idx:end_idx] return { "schools": clean_for_json(schools_df), "total": total, "page": page, "page_size": page_size, "total_pages": (total + page_size - 1) // page_size if page_size > 0 else 0, "result_filters": result_filters, "location_info": { "postcode": postcode, "radius": radius * 1.60934, # Convert miles to km for frontend display "coordinates": [search_coords[0], search_coords[1]] } if search_coords else None, } @app.get("/api/schools/{urn}") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_school_details(request: Request, urn: int): """Get detailed performance data for a specific school across all years.""" # Validate URN range (UK school URNs are 6 digits) if not (100000 <= urn <= 999999): raise HTTPException(status_code=400, detail="Invalid URN format") df = load_school_data() if df.empty: raise HTTPException(status_code=503, detail="School data temporarily unavailable") school_data = df[df["urn"] == urn] if school_data.empty: raise HTTPException(status_code=404, detail="School not found") # Sort by year school_data = school_data.sort_values("year") # Get latest info for the school latest = school_data.iloc[-1] # Fetch supplementary data (Ofsted, admissions, etc.) from .database import SessionLocal supplementary = {} try: db = SessionLocal() supplementary = get_supplementary_data(db, urn) db.close() except Exception: pass # Schools with no performance rows (post-16 institutions, PRUs, new # schools) carry NaN in every LEFT-JOINed numeric column; NaN reaching # JSONResponse raises ValueError, so school_info needs the same # conversion yearly_data gets from clean_for_json. school_info = { k: convert_to_native(v) for k, v in { "urn": urn, "school_name": latest.get("school_name", ""), "local_authority": latest.get("local_authority", ""), "school_type": latest.get("school_type", ""), "address": latest.get("address", ""), "religious_denomination": latest.get("religious_denomination", ""), "age_range": latest.get("age_range", ""), "has_sixth_form": latest.get("has_sixth_form"), "nursery_provision": latest.get("nursery_provision"), "status": latest.get("status"), "latitude": latest.get("latitude"), "longitude": latest.get("longitude"), "phase": latest.get("phase"), # GIAS fields "website": latest.get("website"), "telephone": latest.get("telephone"), "headteacher_name": latest.get("headteacher_name"), "capacity": latest.get("capacity"), "total_pupils": latest.get("gias_total_pupils"), "trust_name": latest.get("trust_name"), "gender": latest.get("gender"), "county": latest.get("county"), "parliamentary_constituency": latest.get("parliamentary_constituency"), }.items() } return { "school_info": school_info, # Where this school sits in the location layer, for the page's link # module and breadcrumb. Derived from the same registry the place # pages and the sitemap use, so a link is never offered for a page # that does not exist. Empty is a valid answer: a school whose town # and authority both fall below the publish threshold has nowhere to # point, and the page renders without the module. "places": _places_payload(urn), # Nearby schools of the same phase and a comparable intake. Always # present on a build with this code; the frontend treats absent and # empty identically, which is what lets the two images deploy # independently. "similar_schools": _similar_schools_payload(urn, latest.get("phase")), "yearly_data": clean_for_json(school_data), # Supplementary data (null if not yet populated by Kestra) "ofsted": supplementary.get("ofsted"), "census": supplementary.get("census"), "admissions": supplementary.get("admissions"), "admissions_history": supplementary.get("admissions_history") or [], # Behind a flag, and withheld at the source rather than rendered-but- # hidden: this endpoint is public and unauthenticated, so a field left # in the payload is a published field. The key is absent, not null — # null would state that this school has no cut-off, which is a # different claim from "we are not publishing cut-offs". **({"admission_distance": supplementary.get("admission_distance")} if flags.is_enabled("admission_distance") else {}), "sen_detail": supplementary.get("sen_detail"), "phonics": supplementary.get("phonics"), "deprivation": supplementary.get("deprivation"), "finance": supplementary.get("finance"), "destinations": supplementary.get("destinations"), } @app.get("/api/compare") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def compare_schools( request: Request, urns: str = Query(..., description="Comma-separated URNs", max_length=100) ): """Compare multiple schools side by side.""" df = load_school_data() if df.empty: raise HTTPException(status_code=404, detail="No data available") try: urn_list = [int(u.strip()) for u in urns.split(",")] # Limit number of schools to compare if len(urn_list) > 10: raise HTTPException(status_code=400, detail="Maximum 10 schools can be compared") # Validate URN format for urn in urn_list: if not (100000 <= urn <= 999999): raise HTTPException(status_code=400, detail="Invalid URN format") except ValueError: raise HTTPException(status_code=400, detail="Invalid URN format") comparison_data = df[df["urn"].isin(urn_list)] if comparison_data.empty: raise HTTPException(status_code=404, detail="No schools found") # One session for all schools' supplementary blocks; failures degrade # to empty blocks rather than failing a working comparison (mirrors # the detail endpoint's defensive pattern). from . import database _EMPTY_SUPPLEMENTARY = { "ofsted": None, "census": None, "admissions": None, "admissions_history": [], "deprivation": None, } supplementary_by_urn: dict = {} census_benchmarks = None db = None try: db = database.SessionLocal() # One query per table for all schools, not ~5 queries per school. batch = get_supplementary_data_batch(db, urn_list) for urn in urn_list: supp = batch.get(urn, {}) supplementary_by_urn[urn] = { key: supp.get(key, default) for key, default in _EMPTY_SUPPLEMENTARY.items() } # Import-time census context benchmarks (fact_census_benchmarks); # absent mart → None, and compute_benchmarks leaves those fields null. try: from .models import CensusBenchmark rows = db.query(CensusBenchmark).all() by_phase = { r.phase: { "year": r.year, "fsm_pct": r.fsm_pct, "eal_pct": r.eal_pct, "median_pupils": r.median_pupils, } for r in rows if getattr(r, "phase", None) in ("primary", "secondary") } if by_phase: census_benchmarks = by_phase except Exception: # Missing mart (or a stubbed session in tests) must never break # the compare payload — and not every session has rollback(). try: db.rollback() except Exception: pass except Exception: supplementary_by_urn = {} finally: if db is not None: db.close() result = {} for urn in urn_list: school_data = comparison_data[comparison_data["urn"] == urn].sort_values("year") if not school_data.empty: latest = school_data.iloc[-1] result[str(urn)] = { "school_info": { "urn": urn, "school_name": latest.get("school_name", ""), "local_authority": latest.get("local_authority", ""), "school_type": latest.get("school_type", ""), "address": latest.get("address", ""), "phase": latest.get("phase", ""), "attainment_8_score": float(latest["attainment_8_score"]) if pd.notna(latest.get("attainment_8_score")) else None, "rwm_expected_pct": float(latest["rwm_expected_pct"]) if pd.notna(latest.get("rwm_expected_pct")) else None, # GIAS facts the compare "Who goes there" section needs # (same fields the detail endpoint exposes) "religious_denomination": convert_to_native(latest.get("religious_denomination")), "age_range": convert_to_native(latest.get("age_range")), "gender": convert_to_native(latest.get("gender")), # Needed by the admissions "What this means" copy: selective # schools get entrance-test framing, never the distance template. "admissions_policy": convert_to_native(latest.get("admissions_policy")), "has_sixth_form": convert_to_native(latest.get("has_sixth_form")), "capacity": convert_to_native(latest.get("capacity")), "gias_total_pupils": convert_to_native(latest.get("gias_total_pupils")), "trust_name": convert_to_native(latest.get("trust_name")), }, "yearly_data": clean_for_json(school_data), **supplementary_by_urn.get(urn, dict(_EMPTY_SUPPLEMENTARY)), } return { "comparison": result, # Official DfE anchors + computed state-school benchmarks so the # compare UI can label provenance correctly (spec §8.6). "national_averages": _national_averages_payload(df), "benchmarks": compute_benchmarks(df, census_benchmarks=census_benchmarks), } @app.get("/api/filters") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_filter_options(request: Request): """Get available filter options (local authorities, school types, years).""" df = load_school_data() if df.empty: return { "local_authorities": [], "school_types": [], "years": [], } # Phases: return values from data, ordered sensibly phases = clean_filter_values(df["phase"]) if "phase" in df.columns else [] secondary_df = df[df["attainment_8_score"].notna()] if "attainment_8_score" in df.columns else df.iloc[0:0] genders = clean_filter_values(secondary_df["gender"]) if "gender" in secondary_df.columns else [] admissions_policies = clean_filter_values(secondary_df["admissions_policy"]) if "admissions_policy" in secondary_df.columns else [] return { "local_authorities": clean_filter_values(df["local_authority"]) if "local_authority" in df.columns else [], "school_types": clean_filter_values(df["school_type"]) if "school_type" in df.columns else [], "years": sorted(df["year"].dropna().unique().tolist()), "phases": phases, "genders": genders, "admissions_policies": admissions_policies, } @app.get("/api/la-averages") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_la_averages(request: Request): """Get per-LA average Attainment 8 score for secondary schools in the latest year.""" df = load_school_data() if df.empty: return {"year": 0, "secondary": {"attainment_8_by_la": {}}} latest_year = int(df["year"].max()) sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()] la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict() return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}} _KS2_NATIONAL_METRICS = [ "rwm_expected_pct", "rwm_high_pct", "reading_expected_pct", "writing_expected_pct", "maths_expected_pct", # Per-subject higher-standard nationals: reading/maths reach the "higher # standard" in the tests; writing is teacher-assessed at "greater depth" # (writing_gd_pct). Needed so each SATs bar compares to its own benchmark. "reading_high_pct", "writing_gd_pct", "maths_high_pct", "gps_expected_pct", "gps_high_pct", "science_expected_pct", "reading_avg_score", "maths_avg_score", "gps_avg_score", "reading_progress", "writing_progress", "maths_progress", "overall_absence_pct", "persistent_absence_pct", "disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct", ] _KS4_NATIONAL_METRICS = [ "attainment_8_score", "progress_8_score", "english_maths_standard_pass_pct", "english_maths_strong_pass_pct", "ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct", "ebacc_avg_score", "gcse_grade_91_pct", ] def _national_averages_payload(df: pd.DataFrame) -> dict: """National-averages payload shared by /api/national-averages and /api/compare. Both series are persisted marts computed at import time: official DfE KS2 figures (fact_ks2_national_averages) and official DfE KS4 figures (fact_ks4_national_averages) — the API never aggregates the performance dataframe per request. If the KS4 mart hasn't been built yet, the secondary series is empty — never a computed stand-in, because the UI labels these figures as official DfE data. """ if df.empty: return {"primary": {}, "secondary": {}} latest_year = int(df["year"].max()) from . import database from .models import Ks2NationalAverage, Ks4NationalAverage def _row_metrics(row, metric_list): out = {} for col in metric_list: val = getattr(row, col, None) if val is not None: out[col] = val return out ks2_rows: list = [] ks4_rows: list = [] db = None try: db = database.SessionLocal() try: ks2_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all() except Exception: db.rollback() try: ks4_rows = db.query(Ks4NationalAverage).order_by(Ks4NationalAverage.year).all() except Exception: db.rollback() except Exception: pass finally: if db is not None: db.close() primary_by_year = {r.year: _row_metrics(r, _KS2_NATIONAL_METRICS) for r in ks2_rows} secondary_by_year = {r.year: _row_metrics(r, _KS4_NATIONAL_METRICS) for r in ks4_rows} all_years = sorted(set(primary_by_year) | set(secondary_by_year)) by_year = [ { "year": yr, "primary": primary_by_year.get(yr, {}), "secondary": secondary_by_year.get(yr, {}), } for yr in all_years ] latest_primary = next((e["primary"] for e in reversed(by_year) if e["primary"]), {}) latest_secondary = next((e["secondary"] for e in reversed(by_year) if e["secondary"]), {}) return { "year": latest_year, "primary": latest_primary, "secondary": latest_secondary, "by_year": by_year, } @app.get("/api/national-averages") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_national_averages(request: Request): """ National averages: official DfE KS2 figures per year plus computed KS4 averages, derived from the loaded DataFrame and the fact_ks2_national_averages mart. """ return _national_averages_payload(load_school_data()) @app.get("/api/metrics") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_available_metrics(request: Request): """ Get list of available performance metrics for schools. This is the single source of truth for metric definitions. Frontend should consume this to avoid duplication. """ df = load_school_data() available = [] for key, info in METRIC_DEFINITIONS.items(): if df.empty or key in df.columns: available.append({"key": key, **info}) return {"metrics": available} @app.get("/api/rankings") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_rankings( request: Request, metric: str = Query("rwm_expected_pct", description="Metric to rank by", max_length=50), year: Optional[int] = Query( None, description="Academic year code, e.g. 201819 (defaults to most recent)", ge=2000, le=210100, ), limit: int = Query(20, ge=1, le=100, description="Number of schools to return"), local_authority: Optional[str] = Query( None, description="Filter by local authority", max_length=100 ), phase: Optional[str] = Query( None, description="Filter by phase: primary or secondary", max_length=20 ), ): """Get school rankings by a specific metric.""" # Sanitize local authority input local_authority = sanitize_search_input(local_authority) # Validate metric name (only allow alphanumeric and underscore) if not re.match(r"^[a-z0-9_]+$", metric): raise HTTPException(status_code=400, detail="Invalid metric name") df = load_school_data() if df.empty: return {"metric": metric, "year": None, "rankings": [], "total": 0} if metric not in df.columns: raise HTTPException(status_code=400, detail=f"Metric '{metric}' not available") # Filter by year if year: df = df[df["year"] == year] else: # Use most recent year max_year = df["year"].max() df = df[df["year"] == max_year] # Filter by local authority if specified if local_authority: df = df[df["local_authority"].str.lower() == local_authority.lower()] # Filter by phase if phase == "primary" and "rwm_expected_pct" in df.columns: df = df[df["rwm_expected_pct"].notna()] elif phase == "secondary" and "attainment_8_score" in df.columns: df = df[df["attainment_8_score"].notna()] # Sort and rank (exclude rows with no data for this metric) df = df.dropna(subset=[metric]) total = len(df) # For progress scores, higher is better. For percentages, higher is also better. df = df.sort_values(metric, ascending=False).head(limit) # Return only relevant fields for rankings available_cols = [c for c in RANKING_COLUMNS if c in df.columns] df = df[available_cols].copy() # Surface the requested metric under a stable `value` key so the # frontend doesn't need to know each metric's column name. The raw # metric column is also kept in the row for callers that want it. df["value"] = df[metric] return { "metric": metric, "year": int(df["year"].iloc[0]) if not df.empty else None, "rankings": clean_for_json(df), "total": total, } @app.get("/api/places") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def list_places(request: Request): """Every published place. The sitemap and the link modules read this.""" registry = get_place_registry() return {"places": [ {"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)} for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)) ]} @app.get("/api/places/{kind}/{slug}") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_place(request: Request, kind: str, slug: str, phase: Optional[str] = None): """One place: its schools ranked, and its local averages.""" if kind not in VALID_PLACE_KINDS: raise HTTPException(status_code=404, detail="No such place") registry = get_place_registry() place = registry.get(f"{kind}:{slug}") if place is None: raise HTTPException(status_code=404, detail="No such place") df = load_latest_school_data() rows = df[df["urn"].isin(place.urns)] if phase: wanted = PHASE_GROUPS.get(phase.lower()) if wanted and "phase" in rows.columns: rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)] # The metric the page shows, and averages. metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct" # Alphabetical, not by score. A place page is read by someone looking for # a school they can name, and scanning for it is what the order should # serve. /rankings is where the league-table ordering lives, and it keeps # sorting by metric. if "school_name" in rows.columns: rows = rows.sort_values("school_name", key=lambda c: c.str.lower()) averages = { m: (None if m not in rows.columns or rows[m].dropna().empty else float(rows[m].dropna().mean())) for m in ("rwm_expected_pct", "attainment_8_score") } # dict.fromkeys, not a list: SCHOOL_COLUMNS already ends with latitude and # longitude, so concatenating them again selected each twice and pandas # dropped one of every duplicated pair with a "columns are not unique" # warning. Ordered de-duplication keeps the column order and the warning # cannot come back. # # nursery_provision and parliamentary_constituency are not in # SCHOOL_COLUMNS and the place table shows both. The `in rows.columns` # guard is what keeps a mart the pipeline has not rebuilt working: those # two are the optional GIAS columns data_loader degrades to NULL. cols = [c for c in dict.fromkeys( SCHOOL_COLUMNS + ["latitude", "longitude", "phase", "nursery_provision", "parliamentary_constituency", "rwm_expected_pct", "attainment_8_score", "total_pupils"]) if c in rows.columns] return { "place": {"kind": place.kind, "slug": place.slug, "name": place.name, "count": len(place.urns), "parent_authority": place.parent_authority, # Every authority the place meaningfully sits in. SW19 is # mostly Merton but partly Wandsworth; naming one asserts # something false. # # The slug is null where that authority has no page of its # own: City of London and the Isles of Scilly hold fewer # schools than the threshold. Naming them is still right; # linking them would be a 404. "authorities": [ {"name": name, "slug": (_slugify(name) if f"authority:{_slugify(name)}" in registry else None), "count": n} for name, n in place.authorities ], # Only phases that clear the threshold, so the page links # variants that exist rather than 404s. "phases": [ph for ph in ("primary", "secondary") if place.publishes_phase(ph)]}, "schools": clean_for_json(rows[cols]), "averages": averages, } # Two characters. One is not a query — it matches thousands of schools and the # response is useless, so it is not worth a round trip. SUGGEST_MIN_QUERY = 2 @app.get("/api/suggest") @limiter.limit("120/minute") async def suggest_schools( request: Request, q: str = Query("", max_length=100), limit: int = Query(8, ge=1, le=20), ): """School name suggestions, from Typesense alone. Deliberately not a mode of /api/schools: that path filters and sorts the full in-memory DataFrame, which is far too expensive to run per keystroke. Nothing here returns an error for ordinary input. A short query, no matches, or Typesense being unreachable are all 200 with an empty list — a dropdown that quietly does not appear is the right failure for a keystroke path, and there is no DataFrame fallback because the 25,000-row substring scan is precisely what this endpoint exists to avoid. 120/minute rather than the default 60: a 200 ms debounce makes typing legitimately bursty. """ query = q.strip() if len(query) < SUGGEST_MIN_QUERY: return {"suggestions": []} return {"suggestions": suggest_schools_typesense(query, limit)} @app.get("/api/flags") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_feature_flags(request: Request): """Every declared flag and its current value. Internal only. The Next proxy denies this path, because the response names every unreleased feature the codebase knows about — which is exactly what shipping dark is meant to keep quiet. """ return flags.all_flags() @app.get("/api/data-info") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_data_info(request: Request): """Get information about loaded data.""" # Get info directly from database db_info = get_db_info() if db_info["total_schools"] == 0: return { "status": "no_data", "message": "No data in marts. Run the annual EES pipeline to load KS2 data.", "data_source": "PostgreSQL", } # Also get DataFrame-based stats for backwards compatibility df = load_school_data() if df.empty: return { "status": "no_data", "message": "No data available", "data_source": "PostgreSQL", } years = [int(y) for y in sorted(df["year"].dropna().unique())] schools_per_year = { str(int(k)): int(v) for k, v in df.dropna(subset=["year"]).groupby("year")["urn"].nunique().to_dict().items() } la_counts = { str(k): int(v) for k, v in df["local_authority"].value_counts().to_dict().items() } return { "status": "loaded", "data_source": "PostgreSQL", "total_records": int(len(df)), "unique_schools": int(df["urn"].nunique()), "years_available": years, "schools_per_year": schools_per_year, "local_authorities": la_counts, } _publication_lock = asyncio.Lock() def _prepare_publication(df): if df.empty: raise ValueError("Refusing to publish an empty school dataset") if not {"urn", "year", "school_name"}.issubset(df.columns): raise ValueError("School dataset is missing required columns") if df["urn"].isna().any() or df.duplicated(["urn", "year"]).any(): raise ValueError("School dataset has missing URNs or duplicate school years") latest = build_latest_school_data(df) registry = build_place_registry(df) index = build_place_index(registry) sitemaps = build_sitemaps(df, registry) return df, latest, registry, index, sitemaps def _publish(prepared): # Called on the event loop with no await: routes cannot observe half a swap. # The application currently runs one worker; replicas require coordination. from . import data_loader global _place_registry, _place_index, _place_index_source, _sitemaps df, latest, registry, index, sitemaps = prepared data_loader._df_cache = df data_loader._df_latest_cache = latest _place_registry = registry _place_index = index _place_index_source = registry _sitemaps = sitemaps @app.post("/api/admin/reload") @limiter.limit("5/minute") async def reload_data(request: Request, _: bool = Depends(verify_admin_api_key)): """Validate a complete replacement before publishing it; retain data on failure.""" async with _publication_lock: try: df = await asyncio.to_thread(load_school_data_as_dataframe) prepared = await asyncio.to_thread(_prepare_publication, df) except Exception as exc: import logging logging.getLogger(__name__).exception("Dataset reload failed") raise HTTPException(status_code=503, detail="Dataset reload failed; previous data retained") from exc _publish(prepared) return {"status": "reloaded", "schools": len(prepared[1])} # ============================================================================= # SEO FILES # ============================================================================= @app.get("/favicon.svg") async def favicon(): """Serve favicon.""" return FileResponse(settings.frontend_dir / "favicon.svg", media_type="image/svg+xml") @app.get("/robots.txt") async def robots_txt(): """Serve robots.txt for search engine crawlers.""" return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain") def _serve_sitemap(name: str) -> Response: global _sitemaps if _sitemaps is None: try: _sitemaps = build_sitemaps() except Exception as e: raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}") if name not in _sitemaps: raise HTTPException(status_code=404, detail="No such sitemap") return Response(content=_sitemaps[name], media_type="application/xml") @app.get("/sitemap.xml") async def sitemap_xml(): """Serve the sitemap index.""" return _serve_sitemap("sitemap.xml") @app.get("/sitemaps/{name}") async def sitemap_child(name: str): """Serve a child sitemap (static.xml, or schools-N.xml).""" return _serve_sitemap(name) @app.post("/api/admin/regenerate-sitemap") @limiter.limit("10/minute") async def regenerate_sitemap( request: Request, _: bool = Depends(verify_admin_api_key), ): """Rebuild derived publication data without clearing the live registry.""" async with _publication_lock: try: prepared = await asyncio.to_thread(_prepare_publication, load_school_data()) except Exception as exc: raise HTTPException(status_code=503, detail="Sitemap rebuild failed; previous data retained") from exc _publish(prepared) n = sum(x.count("") for x in prepared[4].values()) return {"status": "ok", "urls": n, "sitemaps": len(prepared[4])} # Mount static files directly (must be after all routes to avoid catching API calls) if settings.frontend_dir.exists(): app.mount("/static", StaticFiles(directory=settings.frontend_dir), name="static") if __name__ == "__main__": import uvicorn uvicorn.run(app, host=settings.host, port=settings.port)