fix(api): serve DfE's LA averages for "vs LA avg" (H2)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
c26b65246f
commit
e659867590
3 files changed
+185
-8
No files matched your search
+54
-8
@@ -1273,17 +1273,63 @@ async def get_filter_options(request: Request):
|
||||
}
|
||||
|
||||
|
||||
def _la_averages_payload(df: pd.DataFrame) -> dict:
|
||||
"""Per-LA Attainment 8 for the "vs LA avg" comparison: DfE's own LA
|
||||
averages (fact_ks4_la_averages, all state-funded schools), never a mean of
|
||||
the dataframe. That mean counted independent and special schools and put
|
||||
most LAs about 7 points low (audit H2).
|
||||
|
||||
The year is the latest with any school Attainment 8, so the average and
|
||||
the scores set against it are the same year. Figures are keyed by our LA
|
||||
name through the LA code. An LA without a DfE figure for that year (City of
|
||||
London) is absent; no figures for the year, or no mart, give an empty map,
|
||||
so rows show no comparison, never another year's figure.
|
||||
"""
|
||||
empty = {"year": 0, "secondary": {"attainment_8_by_la": {}}}
|
||||
if df.empty or "attainment_8_score" not in df.columns:
|
||||
return empty
|
||||
scored = df[df["attainment_8_score"].notna()]
|
||||
if scored.empty:
|
||||
return empty
|
||||
year = int(scored["year"].max())
|
||||
|
||||
la = (df[["local_authority_code", "local_authority"]]
|
||||
.dropna()
|
||||
.drop_duplicates("local_authority_code"))
|
||||
name_by_code = {int(code): name for code, name in
|
||||
zip(la["local_authority_code"], la["local_authority"])}
|
||||
|
||||
from . import database
|
||||
from .models import Ks4LaAverage
|
||||
|
||||
rows: list = []
|
||||
db = None
|
||||
try:
|
||||
db = database.SessionLocal()
|
||||
rows = db.query(Ks4LaAverage).filter(Ks4LaAverage.year == year).all()
|
||||
except Exception:
|
||||
import logging
|
||||
logging.getLogger(__name__).warning(
|
||||
"DfE LA averages unavailable for %s", year, exc_info=True)
|
||||
if db is not None:
|
||||
db.rollback()
|
||||
finally:
|
||||
if db is not None:
|
||||
db.close()
|
||||
|
||||
by_la = {
|
||||
name_by_code[row.la_code]: row.attainment_8_score
|
||||
for row in rows
|
||||
if row.attainment_8_score is not None and row.la_code in name_by_code
|
||||
}
|
||||
return {"year": year, "secondary": {"attainment_8_by_la": by_la}}
|
||||
|
||||
|
||||
@app.get("/api/la-averages")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_la_averages(request: Request):
|
||||
"""Get per-LA average Attainment 8 score for secondary schools in the latest year."""
|
||||
df = load_school_data()
|
||||
if df.empty:
|
||||
return {"year": 0, "secondary": {"attainment_8_by_la": {}}}
|
||||
latest_year = int(df["year"].max())
|
||||
sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()]
|
||||
la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict()
|
||||
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
||||
"""DfE's per-LA Attainment 8 averages for the latest year with results."""
|
||||
return _la_averages_payload(load_school_data())
|
||||
|
||||
|
||||
_KS2_NATIONAL_METRICS = [
|
||||
|
||||
@@ -344,6 +344,18 @@ class Ks4NationalAverage(Base):
|
||||
gcse_grade_91_pct = Column(Float)
|
||||
|
||||
|
||||
class Ks4LaAverage(Base):
|
||||
"""Official DfE KS4 local-authority averages (all state-funded schools) —
|
||||
one row per academic year and LA. la_code is the GIAS LA code."""
|
||||
__tablename__ = "fact_ks4_la_averages"
|
||||
__table_args__ = MARTS
|
||||
|
||||
year = Column(Integer, primary_key=True)
|
||||
la_code = Column(Integer, primary_key=True)
|
||||
la_name = Column(String)
|
||||
attainment_8_score = Column(Float)
|
||||
|
||||
|
||||
class Ks2NationalAverage(Base):
|
||||
"""Official DfE KS2 national headline averages — one row per academic year."""
|
||||
__tablename__ = "fact_ks2_national_averages"
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
"""/api/la-averages serves DfE's own LA averages (fact_ks4_la_averages).
|
||||
|
||||
It used to average the dataframe: every school with an Attainment 8,
|
||||
independent and special schools included. Kensington and Chelsea came out at
|
||||
35.2 against DfE's 54.5, and most LAs about 7 points low (audit H2).
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
LATEST = 202425
|
||||
|
||||
|
||||
def _df():
|
||||
return pd.DataFrame([
|
||||
# A state school and an independent: their mean, 37.65, is not DfE's figure.
|
||||
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=54.9),
|
||||
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=20.4),
|
||||
dict(year=LATEST, local_authority="Bristol, City of", local_authority_code=801, attainment_8_score=45.0),
|
||||
dict(year=LATEST, local_authority="West Sussex", local_authority_code=938, attainment_8_score=48.0),
|
||||
# DfE publishes no LA figure for City of London.
|
||||
dict(year=LATEST, local_authority="City of London", local_authority_code=201, attainment_8_score=30.0),
|
||||
# A newer year with primary results only.
|
||||
dict(year=202526, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=np.nan),
|
||||
])
|
||||
|
||||
|
||||
class _Row:
|
||||
def __init__(self, year, la_code, la_name, attainment_8_score):
|
||||
self.year = year
|
||||
self.la_code = la_code
|
||||
self.la_name = la_name
|
||||
self.attainment_8_score = attainment_8_score
|
||||
|
||||
|
||||
class _StubSession:
|
||||
rows = [
|
||||
_Row(LATEST, 207, "Kensington and Chelsea", 54.5),
|
||||
_Row(LATEST, 801, "Bristol City", 46.3), # DfE's spelling, not ours
|
||||
_Row(LATEST, 938, "West Sussex", None), # suppressed
|
||||
_Row(LATEST, 330, "Birmingham", 44.0), # no school of ours there
|
||||
_Row(202324, 207, "Kensington and Chelsea", 54.5),
|
||||
]
|
||||
|
||||
def query(self, model):
|
||||
assert model.__name__ == "Ks4LaAverage"
|
||||
return self
|
||||
|
||||
def filter(self, condition):
|
||||
# The payload filters on year == <year>; apply it as Postgres would.
|
||||
self._year = condition.right.value
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
return [r for r in self.rows if r.year == self._year]
|
||||
|
||||
def rollback(self):
|
||||
pass
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
|
||||
class _OldYearOnly(_StubSession):
|
||||
rows = [_Row(202324, 207, "Kensington and Chelsea", 54.5)]
|
||||
|
||||
|
||||
class _NoMart(_StubSession):
|
||||
def all(self):
|
||||
raise RuntimeError('relation "marts.fact_ks4_la_averages" does not exist')
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def payload(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend import database as database_module
|
||||
|
||||
def _run(session_cls):
|
||||
monkeypatch.setattr(database_module, "SessionLocal", session_cls)
|
||||
return app_module._la_averages_payload(_df())
|
||||
|
||||
return _run
|
||||
|
||||
|
||||
def test_serves_dfe_figures_keyed_by_our_la_names(payload):
|
||||
out = payload(_StubSession)
|
||||
assert out["year"] == LATEST
|
||||
assert out["secondary"]["attainment_8_by_la"] == {
|
||||
"Kensington and Chelsea": 54.5,
|
||||
"Bristol, City of": 46.3,
|
||||
}
|
||||
|
||||
|
||||
def test_an_la_without_a_dfe_figure_is_absent(payload):
|
||||
by_la = payload(_StubSession)["secondary"]["attainment_8_by_la"]
|
||||
assert "City of London" not in by_la
|
||||
assert "West Sussex" not in by_la
|
||||
assert None not in by_la.values()
|
||||
|
||||
|
||||
def test_no_dfe_figures_for_the_year_give_an_empty_map(payload):
|
||||
assert payload(_OldYearOnly) == {"year": LATEST, "secondary": {"attainment_8_by_la": {}}}
|
||||
|
||||
|
||||
def test_a_missing_mart_gives_an_empty_map(payload):
|
||||
assert payload(_NoMart)["secondary"]["attainment_8_by_la"] == {}
|
||||
|
||||
|
||||
def test_the_endpoint_serves_dfe_figures(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend import database as database_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
|
||||
resp = TestClient(app_module.app).get("/api/la-averages")
|
||||
assert resp.status_code == 200, resp.text
|
||||
assert resp.json()["secondary"]["attainment_8_by_la"]["Kensington and Chelsea"] == 54.5
|
||||
Reference in new issue
Block a user