perf(api): persist KS4 national averages as a mart; stop per-request aggregation

fact_ks4_national_averages is computed once at dbt build time (covered by
the EES DAG's stg_ees_ks4+ selector). _national_averages_payload now reads
both national-averages marts instead of scanning the performance dataframe
per year on every /api/compare request (~250ms saved per call). Fallback
for the deploy-before-DAG window computes the latest year only.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0146VHeLAWjDVE2B5uU67jCB
This commit is contained in:
Tudor
2026-07-14 13:05:03 +01:00
co-authored by Claude Fable 5
parent 9990f540f7
commit 52f8994401
5 changed files with 219 additions and 70 deletions
+71 -63
View File
@@ -772,14 +772,7 @@ async def get_la_averages(request: Request):
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}} return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
def _national_averages_payload(df: pd.DataFrame) -> dict: _KS2_NATIONAL_METRICS = [
"""National-averages payload shared by /api/national-averages and
/api/compare. Official DfE KS2 figures come from the mart table;
KS4 figures are computed from our dataset (no DfE dataset yet)."""
if df.empty:
return {"primary": {}, "secondary": {}}
ks2_metrics = [
"rwm_expected_pct", "rwm_high_pct", "rwm_expected_pct", "rwm_high_pct",
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct", "reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
"gps_expected_pct", "gps_high_pct", "science_expected_pct", "gps_expected_pct", "gps_high_pct", "science_expected_pct",
@@ -787,78 +780,93 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
"reading_progress", "writing_progress", "maths_progress", "reading_progress", "writing_progress", "maths_progress",
"overall_absence_pct", "persistent_absence_pct", "overall_absence_pct", "persistent_absence_pct",
"disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct", "disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct",
] ]
ks4_metrics = [ _KS4_NATIONAL_METRICS = [
"attainment_8_score", "progress_8_score", "attainment_8_score", "progress_8_score",
"english_maths_standard_pass_pct", "english_maths_strong_pass_pct", "english_maths_standard_pass_pct", "english_maths_strong_pass_pct",
"ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct", "ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct",
"ebacc_avg_score", "gcse_grade_91_pct", "ebacc_avg_score", "gcse_grade_91_pct",
] ]
def _means(sub_df, metric_list):
out = {} def _national_averages_payload(df: pd.DataFrame) -> dict:
for col in metric_list: """National-averages payload shared by /api/national-averages and
if col in sub_df.columns: /api/compare.
val = sub_df[col].dropna()
if len(val) > 0: Both series are persisted marts computed at import time: official DfE
out[col] = round(float(val.mean()), 2) KS2 figures (fact_ks2_national_averages) and dataset-computed KS4
return out averages (fact_ks4_national_averages) — the API never aggregates the
performance dataframe per request. If the KS4 mart hasn't been built
yet (deploy lands before the next DAG run), fall back to computing the
latest year only — a single-year scan, never the historical loop.
"""
if df.empty:
return {"primary": {}, "secondary": {}}
latest_year = int(df["year"].max()) latest_year = int(df["year"].max())
df_latest = df[df["year"] == latest_year]
# Primary: schools where KS2 data is non-null
primary_df = df_latest[df_latest["rwm_expected_pct"].notna()]
# Secondary: schools where KS4 data is non-null
secondary_df = df_latest[df_latest["attainment_8_score"].notna()]
latest_primary = _means(primary_df, ks2_metrics)
latest_secondary = _means(secondary_df, ks4_metrics)
# Per-year KS2 primary averages: use official DfE figures from the mart table.
# Per-year KS4 secondary averages: computed from our dataset (no DfE dataset yet).
from . import database from . import database
from .models import Ks2NationalAverage from .models import Ks2NationalAverage, Ks4NationalAverage
by_year = [] def _row_metrics(row, metric_list):
out = {}
for col in metric_list:
val = getattr(row, col, None)
if val is not None:
out[col] = val
return out
ks2_rows: list = []
ks4_rows: list = []
db = None db = None
try: try:
db = database.SessionLocal() db = database.SessionLocal()
nat_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all() try:
# Build a lookup of computed secondary averages per year as fallback ks2_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
secondary_by_year = {} except Exception:
for yr in sorted(df["year"].dropna().unique()): db.rollback()
yr = int(yr) try:
df_yr = df[df["year"] == yr] ks4_rows = db.query(Ks4NationalAverage).order_by(Ks4NationalAverage.year).all()
secondary_by_year[yr] = _means( except Exception:
df_yr[df_yr["attainment_8_score"].notna()], ks4_metrics db.rollback()
) except Exception:
# Merge: official KS2 figures + computed KS4 figures per year pass
ks2_years = {r.year for r in nat_rows}
all_years = sorted(ks2_years | set(secondary_by_year.keys()))
nat_lookup = {r.year: r for r in nat_rows}
for yr in all_years:
primary_yr: dict = {}
if yr in nat_lookup:
r = nat_lookup[yr]
for col in ks2_metrics:
val = getattr(r, col, None)
if val is not None:
primary_yr[col] = val
by_year.append({
"year": yr,
"primary": primary_yr,
"secondary": secondary_by_year.get(yr, {}),
})
finally: finally:
if db is not None: if db is not None:
db.close() db.close()
# Update latest_primary with official DfE figure for the latest year if available primary_by_year = {r.year: _row_metrics(r, _KS2_NATIONAL_METRICS) for r in ks2_rows}
if by_year: secondary_by_year = {r.year: _row_metrics(r, _KS4_NATIONAL_METRICS) for r in ks4_rows}
latest_official = next((e["primary"] for e in reversed(by_year) if e["primary"]), None)
if latest_official: if not any(secondary_by_year.values()):
latest_primary = latest_official # KS4 mart missing/empty: compute the latest year only.
df_latest = df[df["year"] == latest_year]
sec = (
df_latest[df_latest["attainment_8_score"].notna()]
if "attainment_8_score" in df_latest.columns
else df_latest.iloc[0:0]
)
vals = {}
for col in _KS4_NATIONAL_METRICS:
if col in sec.columns:
v = sec[col].dropna()
if len(v) > 0:
vals[col] = round(float(v.mean()), 2)
if vals:
secondary_by_year[latest_year] = vals
all_years = sorted(set(primary_by_year) | set(secondary_by_year))
by_year = [
{
"year": yr,
"primary": primary_by_year.get(yr, {}),
"secondary": secondary_by_year.get(yr, {}),
}
for yr in all_years
]
latest_primary = next((e["primary"] for e in reversed(by_year) if e["primary"]), {})
latest_secondary = next((e["secondary"] for e in reversed(by_year) if e["secondary"]), {})
return { return {
"year": latest_year, "year": latest_year,
+17
View File
@@ -231,6 +231,23 @@ class FactFinance(Base):
premises_cost_pct = Column(Float) premises_cost_pct = Column(Float)
class Ks4NationalAverage(Base):
"""Computed national KS4 averages (from our dataset) — one row per year."""
__tablename__ = "fact_ks4_national_averages"
__table_args__ = MARTS
year = Column(Integer, primary_key=True)
attainment_8_score = Column(Float)
progress_8_score = Column(Float)
english_maths_standard_pass_pct = Column(Float)
english_maths_strong_pass_pct = Column(Float)
ebacc_entry_pct = Column(Float)
ebacc_standard_pass_pct = Column(Float)
ebacc_strong_pass_pct = Column(Float)
ebacc_avg_score = Column(Float)
gcse_grade_91_pct = Column(Float)
class Ks2NationalAverage(Base): class Ks2NationalAverage(Base):
"""Official DfE KS2 national headline averages — one row per academic year.""" """Official DfE KS2 national headline averages — one row per academic year."""
__tablename__ = "fact_ks2_national_averages" __tablename__ = "fact_ks2_national_averages"
@@ -0,0 +1,93 @@
"""_national_averages_payload reads persisted marts (computed at import
time) — it must never loop the dataframe per year. The only dataframe work
allowed is the single-latest-year KS4 fallback for the window between a
deploy and the next DAG run."""
import numpy as np
import pandas as pd
import pytest
LATEST = 202425
def _df():
return pd.DataFrame(
[
dict(year=202324, attainment_8_score=40.0, rwm_expected_pct=np.nan),
dict(year=LATEST, attainment_8_score=50.0, rwm_expected_pct=np.nan),
dict(year=LATEST, attainment_8_score=30.0, rwm_expected_pct=np.nan),
dict(year=LATEST, attainment_8_score=np.nan, rwm_expected_pct=80.0),
]
)
class _Ks2Row:
year = LATEST
rwm_expected_pct = 62.1
gps_expected_pct = 72.0
class _Ks4Row:
year = LATEST
attainment_8_score = 46.5
progress_8_score = -0.02
class _StubSession:
"""Returns KS2 rows for the first query and KS4 rows for the second —
mirroring the payload's query order."""
def __init__(self):
self.calls = 0
def query(self, model):
self._model = model.__name__
return self
def order_by(self, *a):
return self
def all(self):
return [_Ks2Row()] if self._model == "Ks2NationalAverage" else [_Ks4Row()]
def close(self):
pass
class _Ks4MissingSession(_StubSession):
def all(self):
if self._model == "Ks4NationalAverage":
raise RuntimeError("relation does not exist")
return [_Ks2Row()]
def rollback(self):
pass
@pytest.fixture()
def payload(monkeypatch):
from backend import app as app_module
from backend import database as database_module
def _run(session_cls):
monkeypatch.setattr(database_module, "SessionLocal", session_cls)
return app_module._national_averages_payload(_df())
return _run
def test_ks4_averages_come_from_the_mart_not_the_dataframe(payload):
body = payload(_StubSession)
# Mart value (46.5), NOT the dataframe mean of (50+30)/2 = 40.0
assert body["secondary"]["attainment_8_score"] == 46.5
assert body["primary"]["rwm_expected_pct"] == 62.1
assert body["by_year"][-1]["secondary"]["progress_8_score"] == -0.02
def test_missing_ks4_mart_falls_back_to_latest_year_only(payload):
body = payload(_Ks4MissingSession)
# Fallback computes the latest year from the df: mean(50, 30) = 40.0
assert body["secondary"]["attainment_8_score"] == 40.0
# ...and only the latest year — no historical KS4 loop
ks4_years = [e["year"] for e in body["by_year"] if e["secondary"]]
assert ks4_years == [LATEST]
@@ -160,6 +160,12 @@ models:
- name: year - name: year
tests: [not_null, unique] tests: [not_null, unique]
- name: fact_ks4_national_averages
description: Computed national KS4 averages (means across state schools in our dataset — not official DfE figures) — one row per academic year
columns:
- name: year
tests: [not_null, unique]
- name: fact_deprivation - name: fact_deprivation
description: IDACI deprivation index — one row per URN description: IDACI deprivation index — one row per URN
columns: columns:
@@ -0,0 +1,25 @@
{{ config(materialized='table') }}
-- Mart: Computed national KS4 averages — one row per academic year.
-- Unlike fact_ks2_national_averages (official DfE figures), DfE publishes no
-- KS4 national-headline dataset we ingest yet, so these are means computed
-- across the state schools in our dataset. Computed once at build time so the
-- API never has to aggregate the full performance table per request.
-- Semantics match the API's previous per-request computation: rows where
-- attainment_8_score is non-null; per-column means ignore NULLs.
select
year,
round(avg(attainment_8_score)::numeric, 2) as attainment_8_score,
round(avg(progress_8_score)::numeric, 2) as progress_8_score,
round(avg(english_maths_standard_pass_pct)::numeric, 2) as english_maths_standard_pass_pct,
round(avg(english_maths_strong_pass_pct)::numeric, 2) as english_maths_strong_pass_pct,
round(avg(ebacc_entry_pct)::numeric, 2) as ebacc_entry_pct,
round(avg(ebacc_standard_pass_pct)::numeric, 2) as ebacc_standard_pass_pct,
round(avg(ebacc_strong_pass_pct)::numeric, 2) as ebacc_strong_pass_pct,
round(avg(ebacc_avg_score)::numeric, 2) as ebacc_avg_score,
round(avg(gcse_grade_91_pct)::numeric, 2) as gcse_grade_91_pct
from {{ ref('fact_ks4_performance') }}
where attainment_8_score is not null
group by year
order by year