PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m12s
PR Checks / Backend Smoke (pull_request) Successful in 10s
PR Checks / Build Backend (no push) (pull_request) Successful in 16s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m21s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 11s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 19s
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1693 lines
67 KiB
Python
1693 lines
67 KiB
Python
"""
|
|
SchoolCompare.co.uk API
|
|
Serves primary and secondary school performance data for comparing schools.
|
|
Uses real data from UK Government Compare School Performance downloads.
|
|
"""
|
|
|
|
import hashlib
|
|
import re
|
|
import time
|
|
from contextlib import asynccontextmanager
|
|
from datetime import datetime, timezone
|
|
from typing import Optional
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
from fastapi import FastAPI, HTTPException, Query, Request, Depends, Header
|
|
from fastapi.middleware.cors import CORSMiddleware
|
|
from fastapi.middleware.gzip import GZipMiddleware
|
|
from fastapi.responses import FileResponse, JSONResponse, Response
|
|
from fastapi.staticfiles import StaticFiles
|
|
from slowapi import Limiter, _rate_limit_exceeded_handler
|
|
from slowapi.util import get_remote_address
|
|
from slowapi.errors import RateLimitExceeded
|
|
from starlette.middleware.base import BaseHTTPMiddleware
|
|
|
|
import asyncio
|
|
from .config import settings
|
|
from .data_loader import (
|
|
build_latest_school_data,
|
|
load_school_data_as_dataframe,
|
|
compute_benchmarks,
|
|
load_school_data,
|
|
load_latest_school_data,
|
|
geocode_single_postcode,
|
|
get_supplementary_data,
|
|
get_supplementary_data_batch,
|
|
search_schools_typesense,
|
|
suggest_schools_typesense,
|
|
)
|
|
from .data_loader import get_data_info as get_db_info
|
|
from . import flags
|
|
from .places import build_place_index, build_place_registry, places_for_urn
|
|
from .schemas import METRIC_DEFINITIONS, PHASE_GROUPS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
|
from .nearby_schools import select_nearby
|
|
from .utils import clean_for_json, convert_to_native
|
|
|
|
# Values to exclude from filter dropdowns (empty strings, non-applicable labels)
|
|
EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"}
|
|
|
|
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
|
|
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
|
|
BASE_URL = "https://www.schoolcompare.co.uk"
|
|
MAX_SLUG_LENGTH = 60
|
|
|
|
# In-memory sitemap cache: name -> XML. Populated on startup and by the admin
|
|
# regenerate endpoint after a pipeline run.
|
|
_sitemaps: dict[str, str] | None = None
|
|
|
|
# Built from the same DataFrame the sitemap uses, so places and sitemap can
|
|
# never describe different corpora. Reset by the same admin endpoint.
|
|
_place_registry: dict | None = None
|
|
# Cached beside the registry, and invalidated by identity against it — see
|
|
# get_place_index. Never cleared independently.
|
|
_place_index: dict | None = None
|
|
_place_index_source: dict | None = None
|
|
|
|
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
|
|
|
|
|
|
def _slugify(text: str) -> str:
|
|
text = text.lower()
|
|
text = re.sub(r"[^\w\s-]", "", text)
|
|
text = re.sub(r"\s+", "-", text)
|
|
text = re.sub(r"-+", "-", text)
|
|
return text.strip("-")
|
|
|
|
|
|
def _school_url(urn: int, school_name: str) -> str:
|
|
slug = _slugify(school_name)
|
|
if len(slug) > MAX_SLUG_LENGTH:
|
|
slug = slug[:MAX_SLUG_LENGTH].rstrip("-")
|
|
return f"/school/{urn}-{slug}"
|
|
|
|
|
|
# Routes worth submitting that are not a school page. /admissions was missing
|
|
# from the sitemap entirely despite being a static, indexable guide.
|
|
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
|
|
|
|
|
|
# A page has something a search result could state if any of these is present
|
|
# in any year. Shared by _has_publishable_data and the per-school check in
|
|
# _school_sitemap_rows so the two can never drift.
|
|
_PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade")
|
|
|
|
|
|
def _has_publishable_data(row) -> bool:
|
|
"""True when a school page has something a search result could state.
|
|
|
|
A school with no results in any year and no Ofsted grade renders an empty
|
|
page. Submitting it spends crawl budget and drags the corpus-wide quality
|
|
signal down, so it stays out of the sitemap. The page itself still resolves
|
|
for anyone who has the URL.
|
|
"""
|
|
for field in _PUBLISHABLE_FIELDS:
|
|
value = row.get(field)
|
|
if value is not None and not pd.isna(value):
|
|
return True
|
|
return False
|
|
|
|
|
|
def _url_element(loc: str, lastmod: str | None = None) -> str:
|
|
"""One <url> entry. No priority or changefreq, Google ignores both."""
|
|
body = f"<loc>{loc}</loc>"
|
|
if lastmod:
|
|
body += f"<lastmod>{lastmod}</lastmod>"
|
|
return f" <url>{body}</url>"
|
|
|
|
|
|
def _school_sitemap_rows(df) -> list[str]:
|
|
"""A <url> element per school that has something to show.
|
|
|
|
lastmod comes from the school's Ofsted date where there is one and is
|
|
omitted otherwise. An always-now lastmod is a claim Google learns to
|
|
distrust; an absent one honestly means "unknown".
|
|
"""
|
|
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
|
|
return []
|
|
|
|
rows: list[str] = []
|
|
seen: set[int] = set()
|
|
|
|
# Publishable is a property of the SCHOOL, not of its latest row.
|
|
#
|
|
# The first cut tested the latest year's row alone, which quietly dropped
|
|
# every school that has results in its history but a null row for the most
|
|
# recent year — a school that stopped reporting, or whose figures were
|
|
# suppressed for small-cohort disclosure. The Mallard Academy (150367) is
|
|
# the case that caught it: real KS2 results for 2015-16 through 2018-19,
|
|
# then null rows for 2022-23 onward. Its page shows all four years; the
|
|
# sitemap omitted it. Roughly 220 schools were affected.
|
|
publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
|
publishable: set[int] = (
|
|
set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int))
|
|
if publishable_cols else set()
|
|
)
|
|
|
|
# Latest row per URN first, so a school's most recent Ofsted date wins.
|
|
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
|
|
|
|
for _, row in ordered.iterrows():
|
|
urn = int(row["urn"])
|
|
if urn in seen:
|
|
continue
|
|
seen.add(urn)
|
|
if urn not in publishable:
|
|
continue
|
|
|
|
lastmod = None
|
|
ofsted_date = row.get("ofsted_date")
|
|
if ofsted_date is not None and not pd.isna(ofsted_date):
|
|
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
|
|
|
|
rows.append(_url_element(
|
|
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
|
|
|
|
return rows
|
|
|
|
|
|
# Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to
|
|
# scan by eye in Search Console, which is the point of splitting at all:
|
|
# coverage is reported per submitted sitemap, so one file per page family is
|
|
# what makes an indexation problem attributable to a family.
|
|
SITEMAP_CHUNK_SIZE = 10_000
|
|
|
|
# Children are served under /sitemaps/ because Next.js only treats a whole
|
|
# bracketed path segment as dynamic — a route folder named "sitemap-[...parts]"
|
|
# is read as a literal static segment and never matches.
|
|
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
|
|
|
|
|
def get_place_registry() -> dict:
|
|
"""The place registry, built once and cached for the process."""
|
|
global _place_registry
|
|
if _place_registry is None:
|
|
_place_registry = build_place_registry(load_school_data())
|
|
return _place_registry
|
|
|
|
|
|
def get_place_index() -> dict:
|
|
"""URN → its published places, cached against the registry it came from.
|
|
|
|
Invalidation is an identity check rather than a second flag to remember to
|
|
clear. Anything that drops `_place_registry` — the tests all do — gets a
|
|
fresh registry object here, which no longer matches the one the index was
|
|
built from, so the index rebuilds with it. A separate `_place_index = None`
|
|
would be one more thing to forget, and a stale reverse index is exactly the
|
|
bug that would put links to another dataset's places on a school page.
|
|
"""
|
|
global _place_index, _place_index_source
|
|
registry = get_place_registry()
|
|
if _place_index is None or _place_index_source is not registry:
|
|
_place_index = build_place_index(registry)
|
|
_place_index_source = registry
|
|
return _place_index
|
|
|
|
|
|
def _urlset(rows: list[str]) -> str:
|
|
return "\n".join([
|
|
'<?xml version="1.0" encoding="UTF-8"?>',
|
|
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
|
*rows,
|
|
"</urlset>",
|
|
])
|
|
|
|
|
|
def _place_url(place) -> str:
|
|
"""The canonical path for a place. Two namespaces, per the spec.
|
|
|
|
Towns and localities share /schools/[place]; authorities take their own
|
|
prefix because 67 town names collide with an authority name and neither
|
|
set contains the other.
|
|
"""
|
|
if place.kind == "authority":
|
|
return f"/schools/authority/{place.slug}"
|
|
if place.kind == "outcode":
|
|
return f"/schools/near/{place.slug}"
|
|
return f"/schools/{place.slug}"
|
|
|
|
|
|
def _places_payload(urn: int) -> list[dict]:
|
|
"""The published places containing this school, as the school page needs
|
|
them: a name to write in the link, a count so the anchor can say what it
|
|
leads to, and the canonical path.
|
|
|
|
`phases` carries the phase variants this school actually appears on, which
|
|
is usually one and is two for an all-through school — it is listed on both
|
|
pages, so there is no tie to break.
|
|
|
|
Membership is read straight from the registry's own `phase_urns` rather
|
|
than re-derived from the school's phase string. The registry is the one
|
|
place that decides which phases a place publishes and who is on them;
|
|
computing it a second time here is how a page comes to link a school to a
|
|
phase page that does not list it, or to a route that does not exist. That
|
|
is also why outcodes need no special case: they carry empty `phase_urns`,
|
|
so they report no phase links on their own.
|
|
"""
|
|
payload = []
|
|
for place in places_for_urn(get_place_index(), int(urn)):
|
|
phases = [
|
|
{
|
|
"phase": phase,
|
|
"count": len(phase_urns),
|
|
"url": f"{_place_url(place)}/{phase}",
|
|
}
|
|
for phase, phase_urns in sorted(place.phase_urns.items())
|
|
if int(urn) in phase_urns
|
|
]
|
|
payload.append({
|
|
"kind": place.kind,
|
|
"slug": place.slug,
|
|
"name": place.name,
|
|
"count": len(place.urns),
|
|
"url": _place_url(place),
|
|
"phases": phases,
|
|
})
|
|
return payload
|
|
|
|
|
|
def _nearby_schools_payload(urn: int) -> list[dict]:
|
|
"""The nearest eligible schools this page may offer, closest first.
|
|
|
|
Phase and reach are read from the school's own row inside select_nearby,
|
|
so nothing here can hand it a phase that disagrees with the data.
|
|
|
|
Wrapped: a failure in selection must never 500 a page that is otherwise
|
|
complete, which is the posture get_supplementary_data already takes. The
|
|
section simply does not render.
|
|
"""
|
|
try:
|
|
return select_nearby(load_latest_school_data(), int(urn))
|
|
except Exception:
|
|
import logging
|
|
|
|
logging.getLogger(__name__).exception(
|
|
"Nearby schools selection failed for urn=%s", urn
|
|
)
|
|
return []
|
|
|
|
|
|
def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
|
|
"""A <url> per place, plus a phase variant wherever that phase clears the
|
|
threshold on its own.
|
|
|
|
Phase is part of the query — "primary schools in beccles" — so each
|
|
variant is its own indexable page. Submitting only the bare place URL left
|
|
~950 of them reachable by nothing: absent from every sitemap, and not
|
|
linked from the place page either.
|
|
"""
|
|
rows: list[str] = []
|
|
if registry is None:
|
|
registry = get_place_registry()
|
|
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)):
|
|
if p.kind not in kinds:
|
|
continue
|
|
rows.append(_url_element(BASE_URL + _place_url(p)))
|
|
# Which phases a place publishes is the registry's decision alone —
|
|
# outcodes report none, because the spec gives them no phase route.
|
|
# Repeating that rule here was how the page and the sitemap came to
|
|
# disagree about which URLs exist.
|
|
for phase in ("primary", "secondary"):
|
|
if p.publishes_phase(phase):
|
|
rows.append(_url_element(f"{BASE_URL}{_place_url(p)}/{phase}"))
|
|
return rows
|
|
|
|
|
|
def build_sitemaps(df=None, registry=None) -> dict[str, str]:
|
|
"""Build the sitemap index and every child, keyed by name."""
|
|
if df is None:
|
|
df = load_school_data()
|
|
|
|
children: dict[str, str] = {
|
|
"static.xml": _urlset(
|
|
[_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]),
|
|
}
|
|
|
|
school_rows = _school_sitemap_rows(df)
|
|
# Always emit at least one school child, so the index shape is stable even
|
|
# on an empty database.
|
|
chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE]
|
|
for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
|
for n, chunk in enumerate(chunks, start=1):
|
|
children[f"schools-{n}.xml"] = _urlset(chunk)
|
|
|
|
# Separate children per family: Search Console reports coverage per
|
|
# submitted sitemap, which is how the location layer's indexation is
|
|
# measured apart from the school pages'.
|
|
for label, kinds in (("places", ("town", "locality", "authority")),
|
|
("outcodes", ("outcode",))):
|
|
rows = _place_sitemap_rows(kinds, registry)
|
|
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
|
|
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
|
for n, chunk in enumerate(chunks, start=1):
|
|
children[f"{label}-{n}.xml"] = _urlset(chunk)
|
|
|
|
# On a sitemap index, lastmod means "when this sitemap file last changed",
|
|
# so generation time is the correct value here — unlike on a <url>, where
|
|
# it would be a claim about content we cannot support.
|
|
generated = datetime.now(timezone.utc).date().isoformat()
|
|
index_rows = [
|
|
f" <sitemap><loc>{BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}</loc>"
|
|
f"<lastmod>{generated}</lastmod></sitemap>"
|
|
for name in children
|
|
]
|
|
index = "\n".join([
|
|
'<?xml version="1.0" encoding="UTF-8"?>',
|
|
'<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
|
*index_rows,
|
|
"</sitemapindex>",
|
|
])
|
|
return {**children, "sitemap.xml": index}
|
|
|
|
|
|
def build_sitemap() -> str:
|
|
"""The sitemap index. Kept for `lifespan` and the admin endpoint."""
|
|
return build_sitemaps()["sitemap.xml"]
|
|
|
|
|
|
def clean_filter_values(series: pd.Series) -> list[str]:
|
|
"""Return sorted unique values from a Series, excluding NaN and junk labels."""
|
|
return sorted(
|
|
v for v in series.dropna().unique().tolist()
|
|
if v not in EXCLUDED_FILTER_VALUES
|
|
)
|
|
|
|
|
|
# =============================================================================
|
|
# SECURITY MIDDLEWARE & HELPERS
|
|
# =============================================================================
|
|
|
|
def client_key(request: Request) -> str:
|
|
"""The rate-limit bucket: the real caller, not the proxy in front of them.
|
|
|
|
`get_remote_address` reads request.client.host. In staging and production
|
|
the backend has no published ports and sits on the internal network, so its
|
|
only caller is the Next proxy — meaning every browser user on the site
|
|
shared one bucket. Measured before this fix: 70 concurrent requests to
|
|
/api/schools returned 60 OK and 10 refused.
|
|
|
|
CF-Connecting-IP first, because Cloudflare (in front of both environments)
|
|
sets it on every origin request and *overwrites* any client-supplied value,
|
|
which a parsed X-Forwarded-For chain does not guarantee. The XFF fallback is
|
|
forgeable, but only by a caller already inside the Docker network, which is
|
|
the one place nothing untrusted can reach.
|
|
"""
|
|
cf = request.headers.get("cf-connecting-ip")
|
|
if cf:
|
|
return cf.strip()
|
|
xff = request.headers.get("x-forwarded-for")
|
|
if xff:
|
|
return xff.split(",")[0].strip()
|
|
return get_remote_address(request)
|
|
|
|
|
|
# Per-client limiter. Paired with the global ceiling below — the two do
|
|
# different jobs and neither substitutes for the other.
|
|
limiter = Limiter(key_func=client_key)
|
|
|
|
|
|
# --- The ceiling no header can raise ----------------------------------------
|
|
#
|
|
# client_key trusts CF-Connecting-IP, and nothing in this process can tell an
|
|
# edge-set header from an attacker-set one. That distinction can only be made
|
|
# at Cloudflare, with Authenticated Origin Pulls or an origin firewall. A
|
|
# caller reaching the origin directly could otherwise mint a fresh rate-limit
|
|
# bucket per request and evade per-client limits entirely — which would make
|
|
# correct keying a net regression against abuse, since the single shared bucket
|
|
# it replaced at least capped everyone at 60/minute together.
|
|
#
|
|
# So per-client limits give fairness, and this gives the origin a hard total.
|
|
# It does not make the header trustworthy; it bounds what trusting it can cost.
|
|
# The header problem itself is closed at Cloudflare, not here.
|
|
#
|
|
# [window_start_monotonic, count], or None before the first request. A fixed
|
|
# window is crude, which is right for a backstop: it has to be obviously
|
|
# correct rather than fair.
|
|
_global_window: Optional[list] = None
|
|
|
|
# The container healthcheck runs `curl http://localhost:80/api/data-info` from
|
|
# inside the container. Starving it would fail the check, restart the
|
|
# container, and turn a load spike into an outage loop — the ceiling exists to
|
|
# protect the origin, not to kill it.
|
|
_LOCAL_HOSTS = frozenset({"127.0.0.1", "::1", "localhost"})
|
|
|
|
|
|
def exempt_from_ceiling(request: Request) -> bool:
|
|
"""Whether the ceiling should ignore this request.
|
|
|
|
Its own function so the rule is testable without standing up a server —
|
|
and so the healthcheck exemption is somewhere a reader can find it.
|
|
"""
|
|
if not request.url.path.startswith("/api/"):
|
|
return True
|
|
# The peer address, never the Host header: Host is set by the caller and
|
|
# would hand every attacker an exemption.
|
|
return (request.client.host if request.client else "") in _LOCAL_HOSTS
|
|
|
|
|
|
class GlobalRateLimitMiddleware(BaseHTTPMiddleware):
|
|
"""A cap on total /api/ traffic, independent of any client identity."""
|
|
|
|
async def dispatch(self, request: Request, call_next):
|
|
global _global_window
|
|
|
|
if exempt_from_ceiling(request):
|
|
return await call_next(request)
|
|
|
|
now = time.monotonic()
|
|
# One event loop, and no await between the read and the write, so this
|
|
# sequence is atomic without a lock.
|
|
if _global_window is None or now - _global_window[0] >= 60:
|
|
_global_window = [now, 0]
|
|
_global_window[1] += 1
|
|
|
|
if _global_window[1] > settings.global_rate_limit_per_minute:
|
|
return JSONResponse(
|
|
# Distinguishable from slowapi's per-client 429: an operator
|
|
# reading logs has to be able to tell "one noisy client" from
|
|
# "the origin is saturated".
|
|
{"detail": "The service is at capacity. Please retry shortly."},
|
|
status_code=429,
|
|
headers={"Retry-After":
|
|
str(max(1, int(60 - (now - _global_window[0]))))},
|
|
)
|
|
return await call_next(request)
|
|
|
|
|
|
class SecurityHeadersMiddleware(BaseHTTPMiddleware):
|
|
"""Add security headers to all responses."""
|
|
|
|
async def dispatch(self, request: Request, call_next):
|
|
response = await call_next(request)
|
|
|
|
# Prevent clickjacking
|
|
response.headers["X-Frame-Options"] = "DENY"
|
|
|
|
# Prevent MIME type sniffing
|
|
response.headers["X-Content-Type-Options"] = "nosniff"
|
|
|
|
# XSS Protection (legacy browsers)
|
|
response.headers["X-XSS-Protection"] = "1; mode=block"
|
|
|
|
# Referrer policy
|
|
response.headers["Referrer-Policy"] = "strict-origin-when-cross-origin"
|
|
|
|
# Permissions policy (restrict browser features)
|
|
response.headers["Permissions-Policy"] = (
|
|
"geolocation=(), microphone=(), camera=(), payment=()"
|
|
)
|
|
|
|
# Content Security Policy
|
|
response.headers["Content-Security-Policy"] = (
|
|
"default-src 'self'; "
|
|
"script-src 'self' 'unsafe-inline' https://cdn.jsdelivr.net https://unpkg.com https://analytics.schoolcompare.co.uk; "
|
|
"style-src 'self' 'unsafe-inline' https://fonts.googleapis.com https://cdn.jsdelivr.net https://unpkg.com; "
|
|
"font-src 'self' https://fonts.gstatic.com; "
|
|
"img-src 'self' data: https://*.tile.openstreetmap.org https://unpkg.com; "
|
|
"connect-src 'self' https://cdn.jsdelivr.net https://*.tile.openstreetmap.org https://unpkg.com https://analytics.schoolcompare.co.uk; "
|
|
"frame-ancestors 'none'; "
|
|
"base-uri 'self'; "
|
|
"form-action 'self' https://formsubmit.co;"
|
|
)
|
|
|
|
# HSTS (only enable if using HTTPS in production)
|
|
response.headers["Strict-Transport-Security"] = (
|
|
"max-age=31536000; includeSubDomains"
|
|
)
|
|
|
|
return response
|
|
|
|
|
|
# Per-path Cache-Control rules. Keys are matched as path prefixes (longest wins).
|
|
# Values: (max_age, s_maxage, stale_while_revalidate)
|
|
CACHE_RULES: list[tuple[str, tuple[int, int, int]]] = [
|
|
("/api/filters", (300, 86400, 604800)),
|
|
("/api/metrics", (300, 86400, 604800)),
|
|
("/api/national-averages", (300, 86400, 604800)),
|
|
("/api/la-averages", (300, 86400, 604800)),
|
|
("/api/data-info", (300, 86400, 604800)),
|
|
("/api/schools/", (300, 3600, 86400)), # /api/schools/{urn}
|
|
("/api/rankings", (60, 600, 3600)),
|
|
("/api/compare", (60, 600, 3600)),
|
|
("/api/suggest", (60, 3600, 86400)), # autosuggest
|
|
("/api/schools", (30, 300, 1800)), # search list
|
|
]
|
|
|
|
|
|
def _cache_control_for_path(path: str) -> Optional[str]:
|
|
# Longest-prefix match
|
|
best: Optional[tuple[int, tuple[int, int, int]]] = None
|
|
for prefix, vals in CACHE_RULES:
|
|
if path.startswith(prefix) and (best is None or len(prefix) > best[0]):
|
|
best = (len(prefix), vals)
|
|
if best is None:
|
|
return None
|
|
max_age, s_maxage, swr = best[1]
|
|
return f"public, max-age={max_age}, s-maxage={s_maxage}, stale-while-revalidate={swr}"
|
|
|
|
|
|
class CacheAndETagMiddleware(BaseHTTPMiddleware):
|
|
"""Set Cache-Control on cacheable API responses and serve 304s via ETag."""
|
|
|
|
async def dispatch(self, request: Request, call_next):
|
|
response = await call_next(request)
|
|
|
|
# Only cache GETs that succeeded.
|
|
if request.method != "GET" or response.status_code != 200:
|
|
return response
|
|
|
|
cache_header = _cache_control_for_path(request.url.path)
|
|
if cache_header is None:
|
|
return response
|
|
|
|
# Drain body so we can hash it for ETag.
|
|
body_chunks = []
|
|
async for chunk in response.body_iterator:
|
|
body_chunks.append(chunk)
|
|
body = b"".join(body_chunks)
|
|
|
|
etag = '"' + hashlib.md5(body).hexdigest() + '"'
|
|
headers = dict(response.headers)
|
|
headers["Cache-Control"] = cache_header
|
|
headers["ETag"] = etag
|
|
headers["Vary"] = ", ".join(filter(None, [headers.get("Vary"), "Accept-Encoding"]))
|
|
|
|
inm = request.headers.get("if-none-match")
|
|
if inm and inm == etag:
|
|
# Strip content headers on 304.
|
|
for h in ("Content-Length", "content-length", "Content-Type", "content-type"):
|
|
headers.pop(h, None)
|
|
return Response(status_code=304, headers=headers)
|
|
|
|
return Response(content=body, status_code=200, headers=headers, media_type=response.media_type)
|
|
|
|
|
|
class RequestSizeLimitMiddleware(BaseHTTPMiddleware):
|
|
"""Limit request body size to prevent DoS attacks."""
|
|
|
|
async def dispatch(self, request: Request, call_next):
|
|
content_length = request.headers.get("content-length")
|
|
if content_length:
|
|
if int(content_length) > settings.max_request_size:
|
|
return Response(
|
|
content="Request too large",
|
|
status_code=413,
|
|
)
|
|
return await call_next(request)
|
|
|
|
|
|
def verify_admin_api_key(x_api_key: str = Header(None)) -> bool:
|
|
"""Verify admin API key for protected endpoints."""
|
|
if not x_api_key or x_api_key != settings.admin_api_key:
|
|
raise HTTPException(
|
|
status_code=401,
|
|
detail="Invalid or missing API key",
|
|
headers={"WWW-Authenticate": "ApiKey"},
|
|
)
|
|
return True
|
|
|
|
|
|
# Input validation helpers
|
|
def sanitize_search_input(value: Optional[str], max_length: int = 100) -> Optional[str]:
|
|
"""Sanitize search input to prevent injection attacks."""
|
|
if value is None:
|
|
return None
|
|
# Strip whitespace and limit length
|
|
value = value.strip()[:max_length]
|
|
# Remove potentially dangerous characters (allow alphanumeric, spaces, common punctuation)
|
|
value = re.sub(r"[^\w\s\-\',\.]", "", value)
|
|
return value if value else None
|
|
|
|
|
|
def validate_postcode(postcode: Optional[str]) -> Optional[str]:
|
|
"""Validate and normalize UK postcode format."""
|
|
if not postcode:
|
|
return None
|
|
postcode = postcode.strip().upper()
|
|
# UK postcode pattern
|
|
pattern = r"^[A-Z]{1,2}[0-9][A-Z0-9]?\s*[0-9][A-Z]{2}$"
|
|
if not re.match(pattern, postcode):
|
|
return None
|
|
return postcode
|
|
|
|
|
|
@asynccontextmanager
|
|
async def lifespan(app: FastAPI):
|
|
"""Application lifespan - startup and shutdown events."""
|
|
global _sitemaps
|
|
flags.init()
|
|
print("Loading school data from marts...")
|
|
df = load_school_data()
|
|
if df.empty:
|
|
print("Warning: No data in marts. Run the annual EES pipeline to populate KS2 data.")
|
|
else:
|
|
print(f"Data loaded successfully: {len(df)} records.")
|
|
# Pre-compute the latest-year snapshot so the first search request is fast
|
|
await asyncio.to_thread(load_latest_school_data)
|
|
try:
|
|
_sitemaps = build_sitemaps()
|
|
n = sum(x.count("<url>") for x in _sitemaps.values())
|
|
print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.")
|
|
except Exception as e:
|
|
print(f"Warning: sitemap build failed on startup: {e}")
|
|
|
|
yield
|
|
|
|
print("Shutting down...")
|
|
|
|
|
|
app = FastAPI(
|
|
title="SchoolCompare API",
|
|
description="API for comparing primary and secondary school performance data - schoolcompare.co.uk",
|
|
version="2.0.0",
|
|
lifespan=lifespan,
|
|
# Disable docs in production for security
|
|
docs_url="/docs" if settings.debug else None,
|
|
redoc_url="/redoc" if settings.debug else None,
|
|
openapi_url="/openapi.json" if settings.debug else None,
|
|
)
|
|
|
|
# Add rate limiter
|
|
app.state.limiter = limiter
|
|
app.add_exception_handler(RateLimitExceeded, _rate_limit_exceeded_handler)
|
|
|
|
# Middleware (Starlette runs the last-added middleware first on the way out,
|
|
# so list outermost-last: GZip wraps everything and compresses the final body).
|
|
app.add_middleware(CacheAndETagMiddleware)
|
|
app.add_middleware(SecurityHeadersMiddleware)
|
|
app.add_middleware(RequestSizeLimitMiddleware)
|
|
app.add_middleware(GZipMiddleware, minimum_size=512)
|
|
# Added last, so it is outermost and refuses before anything downstream does
|
|
# work. A ceiling that only applies after the expensive part has run is not a
|
|
# ceiling.
|
|
app.add_middleware(GlobalRateLimitMiddleware)
|
|
|
|
# CORS middleware - restricted for production
|
|
app.add_middleware(
|
|
CORSMiddleware,
|
|
allow_origins=settings.allowed_origins,
|
|
allow_credentials=False, # Don't allow credentials unless needed
|
|
allow_methods=["GET", "POST"], # Only allow needed methods
|
|
allow_headers=["Content-Type", "X-API-Key"], # Only allow needed headers
|
|
)
|
|
|
|
|
|
@app.get("/")
|
|
async def root():
|
|
"""Serve the frontend."""
|
|
return FileResponse(settings.frontend_dir / "index.html")
|
|
|
|
|
|
@app.get("/compare")
|
|
async def serve_compare():
|
|
"""Serve the frontend for /compare route (SPA routing)."""
|
|
return FileResponse(settings.frontend_dir / "index.html")
|
|
|
|
|
|
@app.get("/rankings")
|
|
async def serve_rankings():
|
|
"""Serve the frontend for /rankings route (SPA routing)."""
|
|
return FileResponse(settings.frontend_dir / "index.html")
|
|
|
|
|
|
@app.get("/api/config")
|
|
async def get_config():
|
|
"""Return public configuration for the frontend."""
|
|
return {
|
|
"ga_measurement_id": settings.ga_measurement_id
|
|
}
|
|
|
|
|
|
@app.get("/api/release")
|
|
async def release_identity():
|
|
import json
|
|
from pathlib import Path
|
|
path = Path(__file__).with_name("build-info.json")
|
|
identity = json.loads(path.read_text()) if path.exists() else {"sha": "development", "build_id": "development"}
|
|
return JSONResponse(identity, headers={"Cache-Control": "no-store"})
|
|
|
|
|
|
@app.get("/api/schools")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_schools(
|
|
request: Request,
|
|
search: Optional[str] = Query(None, description="Search by school name", max_length=100),
|
|
local_authority: Optional[str] = Query(
|
|
None, description="Filter by local authority", max_length=100
|
|
),
|
|
school_type: Optional[str] = Query(None, description="Filter by school type", max_length=100),
|
|
phase: Optional[str] = Query(None, description="Filter by phase: primary, secondary, all-through", max_length=50),
|
|
postcode: Optional[str] = Query(None, description="Search near postcode", max_length=10),
|
|
radius: float = Query(5.0, ge=0.1, le=5, description="Search radius in miles"),
|
|
page: int = Query(1, ge=1, le=1000, description="Page number"),
|
|
page_size: int = Query(25, ge=1, le=500, description="Results per page"),
|
|
gender: Optional[str] = Query(None, description="Filter by gender (Mixed/Boys/Girls)", max_length=50),
|
|
admissions_policy: Optional[str] = Query(None, description="Filter by admissions policy", max_length=100),
|
|
has_sixth_form: Optional[str] = Query(None, description="Filter by sixth form presence: yes/no", max_length=3),
|
|
):
|
|
"""
|
|
Get list of schools with pagination.
|
|
|
|
Returns paginated results with total count for efficient loading.
|
|
Supports location-based search using postcode and phase filtering.
|
|
"""
|
|
# Sanitize inputs
|
|
search = sanitize_search_input(search)
|
|
local_authority = sanitize_search_input(local_authority)
|
|
school_type = sanitize_search_input(school_type)
|
|
phase = sanitize_search_input(phase)
|
|
postcode = validate_postcode(postcode)
|
|
|
|
# Load the pre-computed latest-year snapshot (cached after first request / startup).
|
|
# This avoids rebuilding the expensive groupby + prev-year merge on every search.
|
|
df_latest = load_latest_school_data()
|
|
|
|
if df_latest.empty:
|
|
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
|
|
|
|
# Use configured default if not specified
|
|
if page_size is None:
|
|
page_size = settings.default_page_size
|
|
|
|
# Phase filter — uses PHASE_GROUPS so all-through/middle schools appear
|
|
# in the correct phase(s) rather than being invisible to both filters.
|
|
if phase:
|
|
phase_lower = phase.lower().replace("_", "-")
|
|
allowed = PHASE_GROUPS.get(phase_lower)
|
|
if allowed:
|
|
df_latest = df_latest[df_latest["phase"].str.lower().isin(allowed)]
|
|
|
|
# Secondary-specific filters (after phase filter)
|
|
if gender:
|
|
df_latest = df_latest[df_latest["gender"].str.lower() == gender.lower()]
|
|
if admissions_policy:
|
|
df_latest = df_latest[df_latest["admissions_policy"].str.lower() == admissions_policy.lower()]
|
|
# GIAS OfficialSixthForm flag (dim_school.has_sixth_form). NULL (flag not
|
|
# yet populated by the pipeline) is treated as "no sixth form".
|
|
if has_sixth_form in ("yes", "no"):
|
|
if "has_sixth_form" in df_latest.columns:
|
|
flag = df_latest["has_sixth_form"].eq(True)
|
|
else: # Defensive fallback only — data_loader now always synthesizes
|
|
# has_sixth_form as NULL when the DB predates the pipeline re-run,
|
|
# so this branch shouldn't normally trigger. Falls back to age
|
|
# range if the column is somehow absent anyway.
|
|
flag = df_latest["age_range"].str.contains("18", na=False)
|
|
df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag]
|
|
|
|
# Include key result metrics for display on cards
|
|
location_cols = ["latitude", "longitude"]
|
|
result_cols = [
|
|
"phase",
|
|
"year",
|
|
"rwm_expected_pct",
|
|
"rwm_high_pct",
|
|
"prev_rwm_expected_pct",
|
|
"prev_attainment_8_score",
|
|
"reading_expected_pct",
|
|
"writing_expected_pct",
|
|
"maths_expected_pct",
|
|
"total_pupils",
|
|
"attainment_8_score",
|
|
"english_maths_standard_pass_pct",
|
|
]
|
|
available_cols = [
|
|
c
|
|
for c in SCHOOL_COLUMNS + location_cols + result_cols
|
|
if c in df_latest.columns
|
|
]
|
|
# fact_performance guarantees one row per (urn, year); df_latest has one row per urn.
|
|
schools_df = df_latest[available_cols]
|
|
|
|
# Location-based search (uses pre-geocoded data from database)
|
|
search_coords = None
|
|
if postcode:
|
|
# Offload the synchronous HTTP call to a thread so the event loop stays free
|
|
coords = await asyncio.to_thread(geocode_single_postcode, postcode)
|
|
if coords:
|
|
search_coords = coords
|
|
schools_df = schools_df.copy()
|
|
|
|
# Filter by distance using pre-geocoded lat/long from database
|
|
# Use vectorized haversine calculation for better performance
|
|
lat1, lon1 = search_coords
|
|
|
|
# Handle potential duplicate columns by taking first occurrence
|
|
lat_col = schools_df.loc[:, "latitude"]
|
|
lon_col = schools_df.loc[:, "longitude"]
|
|
if isinstance(lat_col, pd.DataFrame):
|
|
lat_col = lat_col.iloc[:, 0]
|
|
if isinstance(lon_col, pd.DataFrame):
|
|
lon_col = lon_col.iloc[:, 0]
|
|
|
|
lat2 = lat_col.values
|
|
lon2 = lon_col.values
|
|
|
|
# Vectorized haversine formula
|
|
R = 3959 # Earth's radius in miles
|
|
lat1_rad = np.radians(lat1)
|
|
lat2_rad = np.radians(lat2)
|
|
dlat = np.radians(lat2 - lat1)
|
|
dlon = np.radians(lon2 - lon1)
|
|
|
|
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_rad) * np.cos(lat2_rad) * np.sin(dlon / 2) ** 2
|
|
c = 2 * np.arctan2(np.sqrt(a), np.sqrt(1 - a))
|
|
distances = R * c
|
|
|
|
# Handle missing coordinates
|
|
has_coords = ~(pd.isna(lat_col) | pd.isna(lon_col))
|
|
distances = np.where(has_coords.values, distances, float("inf"))
|
|
schools_df["distance"] = distances
|
|
schools_df = schools_df[schools_df["distance"] <= radius]
|
|
schools_df = schools_df.sort_values("distance")
|
|
|
|
# Apply filters
|
|
if search:
|
|
ts_urns = await asyncio.to_thread(search_schools_typesense, search)
|
|
if ts_urns is not None:
|
|
urn_order = {urn: i for i, urn in enumerate(ts_urns)}
|
|
schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy()
|
|
schools_df["_ts_rank"] = schools_df["urn"].map(urn_order)
|
|
schools_df = schools_df.sort_values("_ts_rank").drop(columns=["_ts_rank"])
|
|
else:
|
|
# Fallback: Typesense unavailable, use substring match
|
|
search_lower = search.lower()
|
|
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False, regex=False)
|
|
if "address" in schools_df.columns:
|
|
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False, regex=False)
|
|
schools_df = schools_df[mask]
|
|
|
|
if local_authority:
|
|
schools_df = schools_df[
|
|
schools_df["local_authority"].str.lower() == local_authority.lower()
|
|
]
|
|
|
|
if school_type:
|
|
schools_df = schools_df[
|
|
schools_df["school_type"].str.lower() == school_type.lower()
|
|
]
|
|
|
|
# Compute result-scoped filter values (before pagination).
|
|
# Gender and admissions are secondary-only filters — scope them to schools
|
|
# with KS4 data so they don't appear for purely primary result sets.
|
|
_sec_mask = schools_df["attainment_8_score"].notna() if "attainment_8_score" in schools_df.columns else pd.Series(False, index=schools_df.index)
|
|
result_filters = {
|
|
"local_authorities": clean_filter_values(schools_df["local_authority"]) if "local_authority" in schools_df.columns else [],
|
|
"school_types": clean_filter_values(schools_df["school_type"]) if "school_type" in schools_df.columns else [],
|
|
"phases": clean_filter_values(schools_df["phase"]) if "phase" in schools_df.columns else [],
|
|
"genders": clean_filter_values(schools_df.loc[_sec_mask, "gender"]) if "gender" in schools_df.columns and _sec_mask.any() else [],
|
|
"admissions_policies": clean_filter_values(schools_df.loc[_sec_mask, "admissions_policy"]) if "admissions_policy" in schools_df.columns and _sec_mask.any() else [],
|
|
}
|
|
|
|
# Pagination
|
|
total = len(schools_df)
|
|
start_idx = (page - 1) * page_size
|
|
end_idx = start_idx + page_size
|
|
schools_df = schools_df.iloc[start_idx:end_idx]
|
|
|
|
return {
|
|
"schools": clean_for_json(schools_df),
|
|
"total": total,
|
|
"page": page,
|
|
"page_size": page_size,
|
|
"total_pages": (total + page_size - 1) // page_size if page_size > 0 else 0,
|
|
"result_filters": result_filters,
|
|
"location_info": {
|
|
"postcode": postcode,
|
|
"radius": radius * 1.60934, # Convert miles to km for frontend display
|
|
"coordinates": [search_coords[0], search_coords[1]]
|
|
}
|
|
if search_coords
|
|
else None,
|
|
}
|
|
|
|
|
|
@app.get("/api/schools/{urn}")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_school_details(request: Request, urn: int):
|
|
"""Get detailed performance data for a specific school across all years."""
|
|
# Validate URN range (UK school URNs are 6 digits)
|
|
if not (100000 <= urn <= 999999):
|
|
raise HTTPException(status_code=400, detail="Invalid URN format")
|
|
|
|
df = load_school_data()
|
|
|
|
if df.empty:
|
|
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
|
|
|
|
school_data = df[df["urn"] == urn]
|
|
|
|
if school_data.empty:
|
|
raise HTTPException(status_code=404, detail="School not found")
|
|
|
|
# Sort by year
|
|
school_data = school_data.sort_values("year")
|
|
|
|
# Get latest info for the school
|
|
latest = school_data.iloc[-1]
|
|
|
|
# Fetch supplementary data (Ofsted, admissions, etc.)
|
|
from .database import SessionLocal
|
|
supplementary = {}
|
|
try:
|
|
db = SessionLocal()
|
|
supplementary = get_supplementary_data(db, urn)
|
|
db.close()
|
|
except Exception:
|
|
pass
|
|
|
|
# Schools with no performance rows (post-16 institutions, PRUs, new
|
|
# schools) carry NaN in every LEFT-JOINed numeric column; NaN reaching
|
|
# JSONResponse raises ValueError, so school_info needs the same
|
|
# conversion yearly_data gets from clean_for_json.
|
|
school_info = {
|
|
k: convert_to_native(v)
|
|
for k, v in {
|
|
"urn": urn,
|
|
"school_name": latest.get("school_name", ""),
|
|
"local_authority": latest.get("local_authority", ""),
|
|
"school_type": latest.get("school_type", ""),
|
|
"address": latest.get("address", ""),
|
|
"religious_denomination": latest.get("religious_denomination", ""),
|
|
"age_range": latest.get("age_range", ""),
|
|
"has_sixth_form": latest.get("has_sixth_form"),
|
|
"nursery_provision": latest.get("nursery_provision"),
|
|
"status": latest.get("status"),
|
|
"latitude": latest.get("latitude"),
|
|
"longitude": latest.get("longitude"),
|
|
"phase": latest.get("phase"),
|
|
# GIAS fields
|
|
"website": latest.get("website"),
|
|
"telephone": latest.get("telephone"),
|
|
"headteacher_name": latest.get("headteacher_name"),
|
|
"capacity": latest.get("capacity"),
|
|
"total_pupils": latest.get("gias_total_pupils"),
|
|
"trust_name": latest.get("trust_name"),
|
|
"gender": latest.get("gender"),
|
|
"county": latest.get("county"),
|
|
"parliamentary_constituency": latest.get("parliamentary_constituency"),
|
|
}.items()
|
|
}
|
|
|
|
return {
|
|
"school_info": school_info,
|
|
# Where this school sits in the location layer, for the page's link
|
|
# module and breadcrumb. Derived from the same registry the place
|
|
# pages and the sitemap use, so a link is never offered for a page
|
|
# that does not exist. Empty is a valid answer: a school whose town
|
|
# and authority both fall below the publish threshold has nowhere to
|
|
# point, and the page renders without the module.
|
|
"places": _places_payload(urn),
|
|
# The nearest eligible schools, closest first. Always present on a
|
|
# build with this code; the frontend treats absent and empty
|
|
# identically, which is what lets the two images deploy independently.
|
|
"nearby_schools": _nearby_schools_payload(urn),
|
|
"yearly_data": clean_for_json(school_data),
|
|
# Supplementary data (null if not yet populated by Kestra)
|
|
"ofsted": supplementary.get("ofsted"),
|
|
"census": supplementary.get("census"),
|
|
"admissions": supplementary.get("admissions"),
|
|
"admissions_history": supplementary.get("admissions_history") or [],
|
|
# Behind a flag, and withheld at the source rather than rendered-but-
|
|
# hidden: this endpoint is public and unauthenticated, so a field left
|
|
# in the payload is a published field. The key is absent, not null —
|
|
# null would state that this school has no cut-off, which is a
|
|
# different claim from "we are not publishing cut-offs".
|
|
**({"admission_distance": supplementary.get("admission_distance")}
|
|
if flags.is_enabled("admission_distance") else {}),
|
|
"sen_detail": supplementary.get("sen_detail"),
|
|
"phonics": supplementary.get("phonics"),
|
|
"deprivation": supplementary.get("deprivation"),
|
|
"finance": supplementary.get("finance"),
|
|
"destinations": supplementary.get("destinations"),
|
|
}
|
|
|
|
|
|
@app.get("/api/compare")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def compare_schools(
|
|
request: Request,
|
|
urns: str = Query(..., description="Comma-separated URNs", max_length=100)
|
|
):
|
|
"""Compare multiple schools side by side."""
|
|
df = load_school_data()
|
|
|
|
if df.empty:
|
|
raise HTTPException(status_code=404, detail="No data available")
|
|
|
|
try:
|
|
urn_list = [int(u.strip()) for u in urns.split(",")]
|
|
# Limit number of schools to compare
|
|
if len(urn_list) > 10:
|
|
raise HTTPException(status_code=400, detail="Maximum 10 schools can be compared")
|
|
# Validate URN format
|
|
for urn in urn_list:
|
|
if not (100000 <= urn <= 999999):
|
|
raise HTTPException(status_code=400, detail="Invalid URN format")
|
|
except ValueError:
|
|
raise HTTPException(status_code=400, detail="Invalid URN format")
|
|
|
|
comparison_data = df[df["urn"].isin(urn_list)]
|
|
|
|
if comparison_data.empty:
|
|
raise HTTPException(status_code=404, detail="No schools found")
|
|
|
|
# One session for all schools' supplementary blocks; failures degrade
|
|
# to empty blocks rather than failing a working comparison (mirrors
|
|
# the detail endpoint's defensive pattern).
|
|
from . import database
|
|
|
|
_EMPTY_SUPPLEMENTARY = {
|
|
"ofsted": None,
|
|
"census": None,
|
|
"admissions": None,
|
|
"admissions_history": [],
|
|
"deprivation": None,
|
|
}
|
|
supplementary_by_urn: dict = {}
|
|
census_benchmarks = None
|
|
db = None
|
|
try:
|
|
db = database.SessionLocal()
|
|
# One query per table for all schools, not ~5 queries per school.
|
|
batch = get_supplementary_data_batch(db, urn_list)
|
|
for urn in urn_list:
|
|
supp = batch.get(urn, {})
|
|
supplementary_by_urn[urn] = {
|
|
key: supp.get(key, default)
|
|
for key, default in _EMPTY_SUPPLEMENTARY.items()
|
|
}
|
|
# Import-time census context benchmarks (fact_census_benchmarks);
|
|
# absent mart → None, and compute_benchmarks leaves those fields null.
|
|
try:
|
|
from .models import CensusBenchmark
|
|
|
|
rows = db.query(CensusBenchmark).all()
|
|
by_phase = {
|
|
r.phase: {
|
|
"year": r.year,
|
|
"fsm_pct": r.fsm_pct,
|
|
"eal_pct": r.eal_pct,
|
|
"median_pupils": r.median_pupils,
|
|
}
|
|
for r in rows
|
|
if getattr(r, "phase", None) in ("primary", "secondary")
|
|
}
|
|
if by_phase:
|
|
census_benchmarks = by_phase
|
|
except Exception:
|
|
# Missing mart (or a stubbed session in tests) must never break
|
|
# the compare payload — and not every session has rollback().
|
|
try:
|
|
db.rollback()
|
|
except Exception:
|
|
pass
|
|
except Exception:
|
|
supplementary_by_urn = {}
|
|
finally:
|
|
if db is not None:
|
|
db.close()
|
|
|
|
result = {}
|
|
for urn in urn_list:
|
|
school_data = comparison_data[comparison_data["urn"] == urn].sort_values("year")
|
|
if not school_data.empty:
|
|
latest = school_data.iloc[-1]
|
|
result[str(urn)] = {
|
|
"school_info": {
|
|
"urn": urn,
|
|
"school_name": latest.get("school_name", ""),
|
|
"local_authority": latest.get("local_authority", ""),
|
|
"school_type": latest.get("school_type", ""),
|
|
"address": latest.get("address", ""),
|
|
"phase": latest.get("phase", ""),
|
|
"attainment_8_score": float(latest["attainment_8_score"]) if pd.notna(latest.get("attainment_8_score")) else None,
|
|
"rwm_expected_pct": float(latest["rwm_expected_pct"]) if pd.notna(latest.get("rwm_expected_pct")) else None,
|
|
# GIAS facts the compare "Who goes there" section needs
|
|
# (same fields the detail endpoint exposes)
|
|
"religious_denomination": convert_to_native(latest.get("religious_denomination")),
|
|
"age_range": convert_to_native(latest.get("age_range")),
|
|
"gender": convert_to_native(latest.get("gender")),
|
|
# Needed by the admissions "What this means" copy: selective
|
|
# schools get entrance-test framing, never the distance template.
|
|
"admissions_policy": convert_to_native(latest.get("admissions_policy")),
|
|
"has_sixth_form": convert_to_native(latest.get("has_sixth_form")),
|
|
"capacity": convert_to_native(latest.get("capacity")),
|
|
"gias_total_pupils": convert_to_native(latest.get("gias_total_pupils")),
|
|
"trust_name": convert_to_native(latest.get("trust_name")),
|
|
},
|
|
"yearly_data": clean_for_json(school_data),
|
|
**supplementary_by_urn.get(urn, dict(_EMPTY_SUPPLEMENTARY)),
|
|
}
|
|
|
|
return {
|
|
"comparison": result,
|
|
# Official DfE anchors + computed state-school benchmarks so the
|
|
# compare UI can label provenance correctly (spec §8.6).
|
|
"national_averages": _national_averages_payload(df),
|
|
"benchmarks": compute_benchmarks(df, census_benchmarks=census_benchmarks),
|
|
}
|
|
|
|
|
|
@app.get("/api/filters")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_filter_options(request: Request):
|
|
"""Get available filter options (local authorities, school types, years)."""
|
|
df = load_school_data()
|
|
|
|
if df.empty:
|
|
return {
|
|
"local_authorities": [],
|
|
"school_types": [],
|
|
"years": [],
|
|
}
|
|
|
|
# Phases: return values from data, ordered sensibly
|
|
phases = clean_filter_values(df["phase"]) if "phase" in df.columns else []
|
|
|
|
secondary_df = df[df["attainment_8_score"].notna()] if "attainment_8_score" in df.columns else df.iloc[0:0]
|
|
genders = clean_filter_values(secondary_df["gender"]) if "gender" in secondary_df.columns else []
|
|
admissions_policies = clean_filter_values(secondary_df["admissions_policy"]) if "admissions_policy" in secondary_df.columns else []
|
|
|
|
return {
|
|
"local_authorities": clean_filter_values(df["local_authority"]) if "local_authority" in df.columns else [],
|
|
"school_types": clean_filter_values(df["school_type"]) if "school_type" in df.columns else [],
|
|
"years": sorted(df["year"].dropna().unique().tolist()),
|
|
"phases": phases,
|
|
"genders": genders,
|
|
"admissions_policies": admissions_policies,
|
|
}
|
|
|
|
|
|
@app.get("/api/la-averages")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_la_averages(request: Request):
|
|
"""Get per-LA average Attainment 8 score for secondary schools in the latest year."""
|
|
df = load_school_data()
|
|
if df.empty:
|
|
return {"year": 0, "secondary": {"attainment_8_by_la": {}}}
|
|
latest_year = int(df["year"].max())
|
|
sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()]
|
|
la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict()
|
|
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
|
|
|
|
|
_KS2_NATIONAL_METRICS = [
|
|
"rwm_expected_pct", "rwm_high_pct",
|
|
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
|
|
# Per-subject higher-standard nationals: reading/maths reach the "higher
|
|
# standard" in the tests; writing is teacher-assessed at "greater depth"
|
|
# (writing_gd_pct). Needed so each SATs bar compares to its own benchmark.
|
|
"reading_high_pct", "writing_gd_pct", "maths_high_pct",
|
|
"gps_expected_pct", "gps_high_pct", "science_expected_pct",
|
|
"reading_avg_score", "maths_avg_score", "gps_avg_score",
|
|
"reading_progress", "writing_progress", "maths_progress",
|
|
"overall_absence_pct", "persistent_absence_pct",
|
|
"disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct",
|
|
]
|
|
_KS4_NATIONAL_METRICS = [
|
|
"attainment_8_score", "progress_8_score",
|
|
"english_maths_standard_pass_pct", "english_maths_strong_pass_pct",
|
|
"ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct",
|
|
"ebacc_avg_score", "gcse_grade_91_pct",
|
|
]
|
|
|
|
|
|
def _national_averages_payload(df: pd.DataFrame) -> dict:
|
|
"""National-averages payload shared by /api/national-averages and
|
|
/api/compare.
|
|
|
|
Both series are persisted marts computed at import time: official DfE
|
|
KS2 figures (fact_ks2_national_averages) and official DfE KS4 figures
|
|
(fact_ks4_national_averages) — the API never aggregates the performance
|
|
dataframe per request. If the KS4 mart hasn't been built yet, the
|
|
secondary series is empty — never a computed stand-in, because the UI
|
|
labels these figures as official DfE data.
|
|
"""
|
|
if df.empty:
|
|
return {"primary": {}, "secondary": {}}
|
|
|
|
latest_year = int(df["year"].max())
|
|
|
|
from . import database
|
|
from .models import Ks2NationalAverage, Ks4NationalAverage
|
|
|
|
def _row_metrics(row, metric_list):
|
|
out = {}
|
|
for col in metric_list:
|
|
val = getattr(row, col, None)
|
|
if val is not None:
|
|
out[col] = val
|
|
return out
|
|
|
|
ks2_rows: list = []
|
|
ks4_rows: list = []
|
|
db = None
|
|
try:
|
|
db = database.SessionLocal()
|
|
try:
|
|
ks2_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
|
|
except Exception:
|
|
db.rollback()
|
|
try:
|
|
ks4_rows = db.query(Ks4NationalAverage).order_by(Ks4NationalAverage.year).all()
|
|
except Exception:
|
|
db.rollback()
|
|
except Exception:
|
|
pass
|
|
finally:
|
|
if db is not None:
|
|
db.close()
|
|
|
|
primary_by_year = {r.year: _row_metrics(r, _KS2_NATIONAL_METRICS) for r in ks2_rows}
|
|
secondary_by_year = {r.year: _row_metrics(r, _KS4_NATIONAL_METRICS) for r in ks4_rows}
|
|
|
|
all_years = sorted(set(primary_by_year) | set(secondary_by_year))
|
|
by_year = [
|
|
{
|
|
"year": yr,
|
|
"primary": primary_by_year.get(yr, {}),
|
|
"secondary": secondary_by_year.get(yr, {}),
|
|
}
|
|
for yr in all_years
|
|
]
|
|
|
|
latest_primary = next((e["primary"] for e in reversed(by_year) if e["primary"]), {})
|
|
latest_secondary = next((e["secondary"] for e in reversed(by_year) if e["secondary"]), {})
|
|
|
|
return {
|
|
"year": latest_year,
|
|
"primary": latest_primary,
|
|
"secondary": latest_secondary,
|
|
"by_year": by_year,
|
|
}
|
|
|
|
|
|
@app.get("/api/national-averages")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_national_averages(request: Request):
|
|
"""
|
|
National averages: official DfE KS2 figures per year plus computed
|
|
KS4 averages, derived from the loaded DataFrame and the
|
|
fact_ks2_national_averages mart.
|
|
"""
|
|
return _national_averages_payload(load_school_data())
|
|
|
|
|
|
@app.get("/api/metrics")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_available_metrics(request: Request):
|
|
"""
|
|
Get list of available performance metrics for schools.
|
|
|
|
This is the single source of truth for metric definitions.
|
|
Frontend should consume this to avoid duplication.
|
|
"""
|
|
df = load_school_data()
|
|
|
|
available = []
|
|
for key, info in METRIC_DEFINITIONS.items():
|
|
if df.empty or key in df.columns:
|
|
available.append({"key": key, **info})
|
|
|
|
return {"metrics": available}
|
|
|
|
|
|
@app.get("/api/rankings")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_rankings(
|
|
request: Request,
|
|
metric: str = Query("rwm_expected_pct", description="Metric to rank by", max_length=50),
|
|
year: Optional[int] = Query(
|
|
None,
|
|
description="Academic year code, e.g. 201819 (defaults to most recent)",
|
|
ge=2000,
|
|
le=210100,
|
|
),
|
|
limit: int = Query(20, ge=1, le=100, description="Number of schools to return"),
|
|
local_authority: Optional[str] = Query(
|
|
None, description="Filter by local authority", max_length=100
|
|
),
|
|
phase: Optional[str] = Query(
|
|
None, description="Filter by phase: primary or secondary", max_length=20
|
|
),
|
|
):
|
|
"""Get school rankings by a specific metric."""
|
|
# Sanitize local authority input
|
|
local_authority = sanitize_search_input(local_authority)
|
|
|
|
# Validate metric name (only allow alphanumeric and underscore)
|
|
if not re.match(r"^[a-z0-9_]+$", metric):
|
|
raise HTTPException(status_code=400, detail="Invalid metric name")
|
|
|
|
df = load_school_data()
|
|
|
|
if df.empty:
|
|
return {"metric": metric, "year": None, "rankings": [], "total": 0}
|
|
|
|
if metric not in df.columns:
|
|
raise HTTPException(status_code=400, detail=f"Metric '{metric}' not available")
|
|
|
|
# Filter by year
|
|
if year:
|
|
df = df[df["year"] == year]
|
|
else:
|
|
# Use most recent year
|
|
max_year = df["year"].max()
|
|
df = df[df["year"] == max_year]
|
|
|
|
# Filter by local authority if specified
|
|
if local_authority:
|
|
df = df[df["local_authority"].str.lower() == local_authority.lower()]
|
|
|
|
# Filter by phase
|
|
if phase == "primary" and "rwm_expected_pct" in df.columns:
|
|
df = df[df["rwm_expected_pct"].notna()]
|
|
elif phase == "secondary" and "attainment_8_score" in df.columns:
|
|
df = df[df["attainment_8_score"].notna()]
|
|
|
|
# Sort and rank (exclude rows with no data for this metric)
|
|
df = df.dropna(subset=[metric])
|
|
total = len(df)
|
|
|
|
# For progress scores, higher is better. For percentages, higher is also better.
|
|
df = df.sort_values(metric, ascending=False).head(limit)
|
|
|
|
# Return only relevant fields for rankings
|
|
available_cols = [c for c in RANKING_COLUMNS if c in df.columns]
|
|
df = df[available_cols].copy()
|
|
|
|
# Surface the requested metric under a stable `value` key so the
|
|
# frontend doesn't need to know each metric's column name. The raw
|
|
# metric column is also kept in the row for callers that want it.
|
|
df["value"] = df[metric]
|
|
|
|
return {
|
|
"metric": metric,
|
|
"year": int(df["year"].iloc[0]) if not df.empty else None,
|
|
"rankings": clean_for_json(df),
|
|
"total": total,
|
|
}
|
|
|
|
|
|
@app.get("/api/places")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def list_places(request: Request):
|
|
"""Every published place. The sitemap and the link modules read this."""
|
|
registry = get_place_registry()
|
|
return {"places": [
|
|
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
|
|
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
|
|
]}
|
|
|
|
|
|
@app.get("/api/places/{kind}/{slug}")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_place(request: Request, kind: str, slug: str,
|
|
phase: Optional[str] = None):
|
|
"""One place: its schools ranked, and its local averages."""
|
|
if kind not in VALID_PLACE_KINDS:
|
|
raise HTTPException(status_code=404, detail="No such place")
|
|
|
|
registry = get_place_registry()
|
|
place = registry.get(f"{kind}:{slug}")
|
|
if place is None:
|
|
raise HTTPException(status_code=404, detail="No such place")
|
|
|
|
df = load_latest_school_data()
|
|
rows = df[df["urn"].isin(place.urns)]
|
|
|
|
if phase:
|
|
wanted = PHASE_GROUPS.get(phase.lower())
|
|
if wanted and "phase" in rows.columns:
|
|
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
|
|
|
|
# The metric the page shows, and averages.
|
|
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
|
|
|
|
# Alphabetical, not by score. A place page is read by someone looking for
|
|
# a school they can name, and scanning for it is what the order should
|
|
# serve. /rankings is where the league-table ordering lives, and it keeps
|
|
# sorting by metric.
|
|
if "school_name" in rows.columns:
|
|
rows = rows.sort_values("school_name", key=lambda c: c.str.lower())
|
|
|
|
averages = {
|
|
m: (None if m not in rows.columns or rows[m].dropna().empty
|
|
else float(rows[m].dropna().mean()))
|
|
for m in ("rwm_expected_pct", "attainment_8_score")
|
|
}
|
|
|
|
# dict.fromkeys, not a list: SCHOOL_COLUMNS already ends with latitude and
|
|
# longitude, so concatenating them again selected each twice and pandas
|
|
# dropped one of every duplicated pair with a "columns are not unique"
|
|
# warning. Ordered de-duplication keeps the column order and the warning
|
|
# cannot come back.
|
|
#
|
|
# nursery_provision and parliamentary_constituency are not in
|
|
# SCHOOL_COLUMNS and the place table shows both. The `in rows.columns`
|
|
# guard is what keeps a mart the pipeline has not rebuilt working: those
|
|
# two are the optional GIAS columns data_loader degrades to NULL.
|
|
cols = [c for c in dict.fromkeys(
|
|
SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
|
|
"nursery_provision",
|
|
"parliamentary_constituency",
|
|
"rwm_expected_pct", "attainment_8_score",
|
|
"total_pupils"])
|
|
if c in rows.columns]
|
|
|
|
return {
|
|
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
|
|
"count": len(place.urns),
|
|
"parent_authority": place.parent_authority,
|
|
# Every authority the place meaningfully sits in. SW19 is
|
|
# mostly Merton but partly Wandsworth; naming one asserts
|
|
# something false.
|
|
#
|
|
# The slug is null where that authority has no page of its
|
|
# own: City of London and the Isles of Scilly hold fewer
|
|
# schools than the threshold. Naming them is still right;
|
|
# linking them would be a 404.
|
|
"authorities": [
|
|
{"name": name,
|
|
"slug": (_slugify(name)
|
|
if f"authority:{_slugify(name)}" in registry
|
|
else None),
|
|
"count": n}
|
|
for name, n in place.authorities
|
|
],
|
|
# Only phases that clear the threshold, so the page links
|
|
# variants that exist rather than 404s.
|
|
"phases": [ph for ph in ("primary", "secondary")
|
|
if place.publishes_phase(ph)]},
|
|
"schools": clean_for_json(rows[cols]),
|
|
"averages": averages,
|
|
}
|
|
|
|
|
|
# Two characters. One is not a query — it matches thousands of schools and the
|
|
# response is useless, so it is not worth a round trip.
|
|
SUGGEST_MIN_QUERY = 2
|
|
|
|
|
|
@app.get("/api/suggest")
|
|
@limiter.limit("120/minute")
|
|
async def suggest_schools(
|
|
request: Request,
|
|
q: str = Query("", max_length=100),
|
|
limit: int = Query(8, ge=1, le=20),
|
|
):
|
|
"""School name suggestions, from Typesense alone.
|
|
|
|
Deliberately not a mode of /api/schools: that path filters and sorts the
|
|
full in-memory DataFrame, which is far too expensive to run per keystroke.
|
|
|
|
Nothing here returns an error for ordinary input. A short query, no
|
|
matches, or Typesense being unreachable are all 200 with an empty list —
|
|
a dropdown that quietly does not appear is the right failure for a
|
|
keystroke path, and there is no DataFrame fallback because the 25,000-row
|
|
substring scan is precisely what this endpoint exists to avoid.
|
|
|
|
120/minute rather than the default 60: a 200 ms debounce makes typing
|
|
legitimately bursty.
|
|
"""
|
|
query = q.strip()
|
|
if len(query) < SUGGEST_MIN_QUERY:
|
|
return {"suggestions": []}
|
|
return {"suggestions": suggest_schools_typesense(query, limit)}
|
|
|
|
|
|
@app.get("/api/flags")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_feature_flags(request: Request):
|
|
"""Every declared flag and its current value.
|
|
|
|
Internal only. The Next proxy denies this path, because the response names
|
|
every unreleased feature the codebase knows about — which is exactly what
|
|
shipping dark is meant to keep quiet.
|
|
"""
|
|
return flags.all_flags()
|
|
|
|
|
|
@app.get("/api/data-info")
|
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
async def get_data_info(request: Request):
|
|
"""Get information about loaded data."""
|
|
# Get info directly from database
|
|
db_info = get_db_info()
|
|
|
|
if db_info["total_schools"] == 0:
|
|
return {
|
|
"status": "no_data",
|
|
"message": "No data in marts. Run the annual EES pipeline to load KS2 data.",
|
|
"data_source": "PostgreSQL",
|
|
}
|
|
|
|
# Also get DataFrame-based stats for backwards compatibility
|
|
df = load_school_data()
|
|
|
|
if df.empty:
|
|
return {
|
|
"status": "no_data",
|
|
"message": "No data available",
|
|
"data_source": "PostgreSQL",
|
|
}
|
|
|
|
years = [int(y) for y in sorted(df["year"].dropna().unique())]
|
|
schools_per_year = {
|
|
str(int(k)): int(v)
|
|
for k, v in df.dropna(subset=["year"]).groupby("year")["urn"].nunique().to_dict().items()
|
|
}
|
|
la_counts = {
|
|
str(k): int(v)
|
|
for k, v in df["local_authority"].value_counts().to_dict().items()
|
|
}
|
|
|
|
return {
|
|
"status": "loaded",
|
|
"data_source": "PostgreSQL",
|
|
"total_records": int(len(df)),
|
|
"unique_schools": int(df["urn"].nunique()),
|
|
"years_available": years,
|
|
"schools_per_year": schools_per_year,
|
|
"local_authorities": la_counts,
|
|
}
|
|
|
|
|
|
_publication_lock = asyncio.Lock()
|
|
|
|
|
|
def _prepare_publication(df):
|
|
if df.empty:
|
|
raise ValueError("Refusing to publish an empty school dataset")
|
|
if not {"urn", "year", "school_name"}.issubset(df.columns):
|
|
raise ValueError("School dataset is missing required columns")
|
|
if df["urn"].isna().any() or df.duplicated(["urn", "year"]).any():
|
|
raise ValueError("School dataset has missing URNs or duplicate school years")
|
|
latest = build_latest_school_data(df)
|
|
registry = build_place_registry(df)
|
|
index = build_place_index(registry)
|
|
sitemaps = build_sitemaps(df, registry)
|
|
return df, latest, registry, index, sitemaps
|
|
|
|
|
|
def _publish(prepared):
|
|
# Called on the event loop with no await: routes cannot observe half a swap.
|
|
# The application currently runs one worker; replicas require coordination.
|
|
from . import data_loader
|
|
global _place_registry, _place_index, _place_index_source, _sitemaps
|
|
df, latest, registry, index, sitemaps = prepared
|
|
data_loader._df_cache = df
|
|
data_loader._df_latest_cache = latest
|
|
_place_registry = registry
|
|
_place_index = index
|
|
_place_index_source = registry
|
|
_sitemaps = sitemaps
|
|
|
|
|
|
@app.post("/api/admin/reload")
|
|
@limiter.limit("5/minute")
|
|
async def reload_data(request: Request, _: bool = Depends(verify_admin_api_key)):
|
|
"""Validate a complete replacement before publishing it; retain data on failure."""
|
|
async with _publication_lock:
|
|
try:
|
|
df = await asyncio.to_thread(load_school_data_as_dataframe)
|
|
prepared = await asyncio.to_thread(_prepare_publication, df)
|
|
except Exception as exc:
|
|
import logging
|
|
logging.getLogger(__name__).exception("Dataset reload failed")
|
|
raise HTTPException(status_code=503, detail="Dataset reload failed; previous data retained") from exc
|
|
_publish(prepared)
|
|
return {"status": "reloaded", "schools": len(prepared[1])}
|
|
|
|
|
|
|
|
|
|
# =============================================================================
|
|
# SEO FILES
|
|
# =============================================================================
|
|
|
|
|
|
@app.get("/favicon.svg")
|
|
async def favicon():
|
|
"""Serve favicon."""
|
|
return FileResponse(settings.frontend_dir / "favicon.svg", media_type="image/svg+xml")
|
|
|
|
|
|
@app.get("/robots.txt")
|
|
async def robots_txt():
|
|
"""Serve robots.txt for search engine crawlers."""
|
|
return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain")
|
|
|
|
|
|
def _serve_sitemap(name: str) -> Response:
|
|
global _sitemaps
|
|
if _sitemaps is None:
|
|
try:
|
|
_sitemaps = build_sitemaps()
|
|
except Exception as e:
|
|
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
|
|
if name not in _sitemaps:
|
|
raise HTTPException(status_code=404, detail="No such sitemap")
|
|
return Response(content=_sitemaps[name], media_type="application/xml")
|
|
|
|
|
|
@app.get("/sitemap.xml")
|
|
async def sitemap_xml():
|
|
"""Serve the sitemap index."""
|
|
return _serve_sitemap("sitemap.xml")
|
|
|
|
|
|
@app.get("/sitemaps/{name}")
|
|
async def sitemap_child(name: str):
|
|
"""Serve a child sitemap (static.xml, or schools-N.xml)."""
|
|
return _serve_sitemap(name)
|
|
|
|
|
|
@app.post("/api/admin/regenerate-sitemap")
|
|
@limiter.limit("10/minute")
|
|
async def regenerate_sitemap(
|
|
request: Request,
|
|
_: bool = Depends(verify_admin_api_key),
|
|
):
|
|
"""Rebuild derived publication data without clearing the live registry."""
|
|
async with _publication_lock:
|
|
try:
|
|
prepared = await asyncio.to_thread(_prepare_publication, load_school_data())
|
|
except Exception as exc:
|
|
raise HTTPException(status_code=503, detail="Sitemap rebuild failed; previous data retained") from exc
|
|
_publish(prepared)
|
|
n = sum(x.count("<url>") for x in prepared[4].values())
|
|
return {"status": "ok", "urls": n, "sitemaps": len(prepared[4])}
|
|
|
|
|
|
|
|
# Mount static files directly (must be after all routes to avoid catching API calls)
|
|
if settings.frontend_dir.exists():
|
|
app.mount("/static", StaticFiles(directory=settings.frontend_dir), name="static")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import uvicorn
|
|
|
|
uvicorn.run(app, host=settings.host, port=settings.port)
|