fix(api): serve DfE's LA averages for "vs LA avg" (H2)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 12:35:11 +01:00
1 parent c26b65246f
commit e659867590
3 files changed
+185 -8

No files matched your search

+54 -8
View File
@@ -1273,17 +1273,63 @@ async def get_filter_options(request: Request):
}
def _la_averages_payload(df: pd.DataFrame) -> dict:
"""Per-LA Attainment 8 for the "vs LA avg" comparison: DfE's own LA
averages (fact_ks4_la_averages, all state-funded schools), never a mean of
the dataframe. That mean counted independent and special schools and put
most LAs about 7 points low (audit H2).
The year is the latest with any school Attainment 8, so the average and
the scores set against it are the same year. Figures are keyed by our LA
name through the LA code. An LA without a DfE figure for that year (City of
London) is absent; no figures for the year, or no mart, give an empty map,
so rows show no comparison, never another year's figure.
"""
empty = {"year": 0, "secondary": {"attainment_8_by_la": {}}}
if df.empty or "attainment_8_score" not in df.columns:
return empty
scored = df[df["attainment_8_score"].notna()]
if scored.empty:
return empty
year = int(scored["year"].max())
la = (df[["local_authority_code", "local_authority"]]
.dropna()
.drop_duplicates("local_authority_code"))
name_by_code = {int(code): name for code, name in
zip(la["local_authority_code"], la["local_authority"])}
from . import database
from .models import Ks4LaAverage
rows: list = []
db = None
try:
db = database.SessionLocal()
rows = db.query(Ks4LaAverage).filter(Ks4LaAverage.year == year).all()
except Exception:
import logging
logging.getLogger(__name__).warning(
"DfE LA averages unavailable for %s", year, exc_info=True)
if db is not None:
db.rollback()
finally:
if db is not None:
db.close()
by_la = {
name_by_code[row.la_code]: row.attainment_8_score
for row in rows
if row.attainment_8_score is not None and row.la_code in name_by_code
}
return {"year": year, "secondary": {"attainment_8_by_la": by_la}}
@app.get("/api/la-averages")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_la_averages(request: Request):
"""Get per-LA average Attainment 8 score for secondary schools in the latest year."""
df = load_school_data()
if df.empty:
return {"year": 0, "secondary": {"attainment_8_by_la": {}}}
latest_year = int(df["year"].max())
sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()]
la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict()
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
"""DfE's per-LA Attainment 8 averages for the latest year with results."""
return _la_averages_payload(load_school_data())
_KS2_NATIONAL_METRICS = [
+12
View File
@@ -344,6 +344,18 @@ class Ks4NationalAverage(Base):
gcse_grade_91_pct = Column(Float)
class Ks4LaAverage(Base):
"""Official DfE KS4 local-authority averages (all state-funded schools) —
one row per academic year and LA. la_code is the GIAS LA code."""
__tablename__ = "fact_ks4_la_averages"
__table_args__ = MARTS
year = Column(Integer, primary_key=True)
la_code = Column(Integer, primary_key=True)
la_name = Column(String)
attainment_8_score = Column(Float)
class Ks2NationalAverage(Base):
"""Official DfE KS2 national headline averages — one row per academic year."""
__tablename__ = "fact_ks2_national_averages"
+119
View File
@@ -0,0 +1,119 @@
"""/api/la-averages serves DfE's own LA averages (fact_ks4_la_averages).
It used to average the dataframe: every school with an Attainment 8,
independent and special schools included. Kensington and Chelsea came out at
35.2 against DfE's 54.5, and most LAs about 7 points low (audit H2).
"""
import numpy as np
import pandas as pd
import pytest
from fastapi.testclient import TestClient
LATEST = 202425
def _df():
return pd.DataFrame([
# A state school and an independent: their mean, 37.65, is not DfE's figure.
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=54.9),
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=20.4),
dict(year=LATEST, local_authority="Bristol, City of", local_authority_code=801, attainment_8_score=45.0),
dict(year=LATEST, local_authority="West Sussex", local_authority_code=938, attainment_8_score=48.0),
# DfE publishes no LA figure for City of London.
dict(year=LATEST, local_authority="City of London", local_authority_code=201, attainment_8_score=30.0),
# A newer year with primary results only.
dict(year=202526, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=np.nan),
])
class _Row:
def __init__(self, year, la_code, la_name, attainment_8_score):
self.year = year
self.la_code = la_code
self.la_name = la_name
self.attainment_8_score = attainment_8_score
class _StubSession:
rows = [
_Row(LATEST, 207, "Kensington and Chelsea", 54.5),
_Row(LATEST, 801, "Bristol City", 46.3), # DfE's spelling, not ours
_Row(LATEST, 938, "West Sussex", None), # suppressed
_Row(LATEST, 330, "Birmingham", 44.0), # no school of ours there
_Row(202324, 207, "Kensington and Chelsea", 54.5),
]
def query(self, model):
assert model.__name__ == "Ks4LaAverage"
return self
def filter(self, condition):
# The payload filters on year == <year>; apply it as Postgres would.
self._year = condition.right.value
return self
def all(self):
return [r for r in self.rows if r.year == self._year]
def rollback(self):
pass
def close(self):
pass
class _OldYearOnly(_StubSession):
rows = [_Row(202324, 207, "Kensington and Chelsea", 54.5)]
class _NoMart(_StubSession):
def all(self):
raise RuntimeError('relation "marts.fact_ks4_la_averages" does not exist')
@pytest.fixture()
def payload(monkeypatch):
from backend import app as app_module
from backend import database as database_module
def _run(session_cls):
monkeypatch.setattr(database_module, "SessionLocal", session_cls)
return app_module._la_averages_payload(_df())
return _run
def test_serves_dfe_figures_keyed_by_our_la_names(payload):
out = payload(_StubSession)
assert out["year"] == LATEST
assert out["secondary"]["attainment_8_by_la"] == {
"Kensington and Chelsea": 54.5,
"Bristol, City of": 46.3,
}
def test_an_la_without_a_dfe_figure_is_absent(payload):
by_la = payload(_StubSession)["secondary"]["attainment_8_by_la"]
assert "City of London" not in by_la
assert "West Sussex" not in by_la
assert None not in by_la.values()
def test_no_dfe_figures_for_the_year_give_an_empty_map(payload):
assert payload(_OldYearOnly) == {"year": LATEST, "secondary": {"attainment_8_by_la": {}}}
def test_a_missing_mart_gives_an_empty_map(payload):
assert payload(_NoMart)["secondary"]["attainment_8_by_la"] == {}
def test_the_endpoint_serves_dfe_figures(monkeypatch):
from backend import app as app_module
from backend import database as database_module
monkeypatch.setattr(app_module, "load_school_data", _df)
monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
resp = TestClient(app_module.app).get("/api/la-averages")
assert resp.status_code == 200, resp.text
assert resp.json()["secondary"]["attainment_8_by_la"]["Kensington and Chelsea"] == 54.5