diff --git a/backend/app.py b/backend/app.py index de856b2..4ac9986 100644 --- a/backend/app.py +++ b/backend/app.py @@ -1273,17 +1273,63 @@ async def get_filter_options(request: Request): } +def _la_averages_payload(df: pd.DataFrame) -> dict: + """Per-LA Attainment 8 for the "vs LA avg" comparison: DfE's own LA + averages (fact_ks4_la_averages, all state-funded schools), never a mean of + the dataframe. That mean counted independent and special schools and put + most LAs about 7 points low (audit H2). + + The year is the latest with any school Attainment 8, so the average and + the scores set against it are the same year. Figures are keyed by our LA + name through the LA code. An LA without a DfE figure for that year (City of + London) is absent; no figures for the year, or no mart, give an empty map, + so rows show no comparison, never another year's figure. + """ + empty = {"year": 0, "secondary": {"attainment_8_by_la": {}}} + if df.empty or "attainment_8_score" not in df.columns: + return empty + scored = df[df["attainment_8_score"].notna()] + if scored.empty: + return empty + year = int(scored["year"].max()) + + la = (df[["local_authority_code", "local_authority"]] + .dropna() + .drop_duplicates("local_authority_code")) + name_by_code = {int(code): name for code, name in + zip(la["local_authority_code"], la["local_authority"])} + + from . import database + from .models import Ks4LaAverage + + rows: list = [] + db = None + try: + db = database.SessionLocal() + rows = db.query(Ks4LaAverage).filter(Ks4LaAverage.year == year).all() + except Exception: + import logging + logging.getLogger(__name__).warning( + "DfE LA averages unavailable for %s", year, exc_info=True) + if db is not None: + db.rollback() + finally: + if db is not None: + db.close() + + by_la = { + name_by_code[row.la_code]: row.attainment_8_score + for row in rows + if row.attainment_8_score is not None and row.la_code in name_by_code + } + return {"year": year, "secondary": {"attainment_8_by_la": by_la}} + + @app.get("/api/la-averages") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_la_averages(request: Request): - """Get per-LA average Attainment 8 score for secondary schools in the latest year.""" - df = load_school_data() - if df.empty: - return {"year": 0, "secondary": {"attainment_8_by_la": {}}} - latest_year = int(df["year"].max()) - sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()] - la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict() - return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}} + """DfE's per-LA Attainment 8 averages for the latest year with results.""" + return _la_averages_payload(load_school_data()) _KS2_NATIONAL_METRICS = [ diff --git a/backend/models.py b/backend/models.py index 2fb6f74..b30c5e6 100644 --- a/backend/models.py +++ b/backend/models.py @@ -344,6 +344,18 @@ class Ks4NationalAverage(Base): gcse_grade_91_pct = Column(Float) +class Ks4LaAverage(Base): + """Official DfE KS4 local-authority averages (all state-funded schools) — + one row per academic year and LA. la_code is the GIAS LA code.""" + __tablename__ = "fact_ks4_la_averages" + __table_args__ = MARTS + + year = Column(Integer, primary_key=True) + la_code = Column(Integer, primary_key=True) + la_name = Column(String) + attainment_8_score = Column(Float) + + class Ks2NationalAverage(Base): """Official DfE KS2 national headline averages — one row per academic year.""" __tablename__ = "fact_ks2_national_averages" diff --git a/backend/tests/test_la_averages_marts.py b/backend/tests/test_la_averages_marts.py new file mode 100644 index 0000000..115a83f --- /dev/null +++ b/backend/tests/test_la_averages_marts.py @@ -0,0 +1,119 @@ +"""/api/la-averages serves DfE's own LA averages (fact_ks4_la_averages). + +It used to average the dataframe: every school with an Attainment 8, +independent and special schools included. Kensington and Chelsea came out at +35.2 against DfE's 54.5, and most LAs about 7 points low (audit H2). +""" + +import numpy as np +import pandas as pd +import pytest +from fastapi.testclient import TestClient + +LATEST = 202425 + + +def _df(): + return pd.DataFrame([ + # A state school and an independent: their mean, 37.65, is not DfE's figure. + dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=54.9), + dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=20.4), + dict(year=LATEST, local_authority="Bristol, City of", local_authority_code=801, attainment_8_score=45.0), + dict(year=LATEST, local_authority="West Sussex", local_authority_code=938, attainment_8_score=48.0), + # DfE publishes no LA figure for City of London. + dict(year=LATEST, local_authority="City of London", local_authority_code=201, attainment_8_score=30.0), + # A newer year with primary results only. + dict(year=202526, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=np.nan), + ]) + + +class _Row: + def __init__(self, year, la_code, la_name, attainment_8_score): + self.year = year + self.la_code = la_code + self.la_name = la_name + self.attainment_8_score = attainment_8_score + + +class _StubSession: + rows = [ + _Row(LATEST, 207, "Kensington and Chelsea", 54.5), + _Row(LATEST, 801, "Bristol City", 46.3), # DfE's spelling, not ours + _Row(LATEST, 938, "West Sussex", None), # suppressed + _Row(LATEST, 330, "Birmingham", 44.0), # no school of ours there + _Row(202324, 207, "Kensington and Chelsea", 54.5), + ] + + def query(self, model): + assert model.__name__ == "Ks4LaAverage" + return self + + def filter(self, condition): + # The payload filters on year == ; apply it as Postgres would. + self._year = condition.right.value + return self + + def all(self): + return [r for r in self.rows if r.year == self._year] + + def rollback(self): + pass + + def close(self): + pass + + +class _OldYearOnly(_StubSession): + rows = [_Row(202324, 207, "Kensington and Chelsea", 54.5)] + + +class _NoMart(_StubSession): + def all(self): + raise RuntimeError('relation "marts.fact_ks4_la_averages" does not exist') + + +@pytest.fixture() +def payload(monkeypatch): + from backend import app as app_module + from backend import database as database_module + + def _run(session_cls): + monkeypatch.setattr(database_module, "SessionLocal", session_cls) + return app_module._la_averages_payload(_df()) + + return _run + + +def test_serves_dfe_figures_keyed_by_our_la_names(payload): + out = payload(_StubSession) + assert out["year"] == LATEST + assert out["secondary"]["attainment_8_by_la"] == { + "Kensington and Chelsea": 54.5, + "Bristol, City of": 46.3, + } + + +def test_an_la_without_a_dfe_figure_is_absent(payload): + by_la = payload(_StubSession)["secondary"]["attainment_8_by_la"] + assert "City of London" not in by_la + assert "West Sussex" not in by_la + assert None not in by_la.values() + + +def test_no_dfe_figures_for_the_year_give_an_empty_map(payload): + assert payload(_OldYearOnly) == {"year": LATEST, "secondary": {"attainment_8_by_la": {}}} + + +def test_a_missing_mart_gives_an_empty_map(payload): + assert payload(_NoMart)["secondary"]["attainment_8_by_la"] == {} + + +def test_the_endpoint_serves_dfe_figures(monkeypatch): + from backend import app as app_module + from backend import database as database_module + + monkeypatch.setattr(app_module, "load_school_data", _df) + monkeypatch.setattr(database_module, "SessionLocal", _StubSession) + resp = TestClient(app_module.app).get("/api/la-averages") + assert resp.status_code == 200, resp.text + assert resp.json()["secondary"]["attainment_8_by_la"]["Kensington and Chelsea"] == 54.5