fix(data): official DfE KS4 national headline averages; drop mislabelled computed means
New ees_ks4_national stream ingests the EES 'National characteristics summary data' series (England, state-funded, all pupils). The old mart's unweighted school means were 7-15 points off every headline measure and produced an impossible national Progress 8 (-0.27). The API's computed fallback is gone too: the footnote calls these figures official, so an unbuilt mart now yields an empty series, never a stand-in. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0146VHeLAWjDVE2B5uU67jCB
This commit is contained in:
+5
-22
@@ -823,11 +823,11 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
|
||||
/api/compare.
|
||||
|
||||
Both series are persisted marts computed at import time: official DfE
|
||||
KS2 figures (fact_ks2_national_averages) and dataset-computed KS4
|
||||
averages (fact_ks4_national_averages) — the API never aggregates the
|
||||
performance dataframe per request. If the KS4 mart hasn't been built
|
||||
yet (deploy lands before the next DAG run), fall back to computing the
|
||||
latest year only — a single-year scan, never the historical loop.
|
||||
KS2 figures (fact_ks2_national_averages) and official DfE KS4 figures
|
||||
(fact_ks4_national_averages) — the API never aggregates the performance
|
||||
dataframe per request. If the KS4 mart hasn't been built yet, the
|
||||
secondary series is empty — never a computed stand-in, because the UI
|
||||
labels these figures as official DfE data.
|
||||
"""
|
||||
if df.empty:
|
||||
return {"primary": {}, "secondary": {}}
|
||||
@@ -867,23 +867,6 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
|
||||
primary_by_year = {r.year: _row_metrics(r, _KS2_NATIONAL_METRICS) for r in ks2_rows}
|
||||
secondary_by_year = {r.year: _row_metrics(r, _KS4_NATIONAL_METRICS) for r in ks4_rows}
|
||||
|
||||
if not any(secondary_by_year.values()):
|
||||
# KS4 mart missing/empty: compute the latest year only.
|
||||
df_latest = df[df["year"] == latest_year]
|
||||
sec = (
|
||||
df_latest[df_latest["attainment_8_score"].notna()]
|
||||
if "attainment_8_score" in df_latest.columns
|
||||
else df_latest.iloc[0:0]
|
||||
)
|
||||
vals = {}
|
||||
for col in _KS4_NATIONAL_METRICS:
|
||||
if col in sec.columns:
|
||||
v = sec[col].dropna()
|
||||
if len(v) > 0:
|
||||
vals[col] = round(float(v.mean()), 2)
|
||||
if vals:
|
||||
secondary_by_year[latest_year] = vals
|
||||
|
||||
all_years = sorted(set(primary_by_year) | set(secondary_by_year))
|
||||
by_year = [
|
||||
{
|
||||
|
||||
+4
-1
@@ -251,7 +251,10 @@ class CensusBenchmark(Base):
|
||||
|
||||
|
||||
class Ks4NationalAverage(Base):
|
||||
"""Computed national KS4 averages (from our dataset) — one row per year."""
|
||||
"""Official DfE KS4 national headline averages — one row per academic year.
|
||||
|
||||
gcse_grade_91_pct has no official national series and is always NULL.
|
||||
"""
|
||||
__tablename__ = "fact_ks4_national_averages"
|
||||
__table_args__ = MARTS
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""_national_averages_payload reads persisted marts (computed at import
|
||||
time) — it must never loop the dataframe per year. The only dataframe work
|
||||
allowed is the single-latest-year KS4 fallback for the window between a
|
||||
deploy and the next DAG run."""
|
||||
time) — it must never aggregate the dataframe. Both marts hold OFFICIAL
|
||||
DfE figures, so a missing KS4 mart yields an empty secondary series —
|
||||
never a computed stand-in the UI would mislabel as official."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -84,10 +84,11 @@ def test_ks4_averages_come_from_the_mart_not_the_dataframe(payload):
|
||||
assert body["by_year"][-1]["secondary"]["progress_8_score"] == -0.02
|
||||
|
||||
|
||||
def test_missing_ks4_mart_falls_back_to_latest_year_only(payload):
|
||||
def test_ks4_secondary_empty_when_mart_missing(payload):
|
||||
# No computed stand-in: the UI labels national figures as official DfE
|
||||
# data, so an empty mart must yield an empty secondary series.
|
||||
body = payload(_Ks4MissingSession)
|
||||
# Fallback computes the latest year from the df: mean(50, 30) = 40.0
|
||||
assert body["secondary"]["attainment_8_score"] == 40.0
|
||||
# ...and only the latest year — no historical KS4 loop
|
||||
ks4_years = [e["year"] for e in body["by_year"] if e["secondary"]]
|
||||
assert ks4_years == [LATEST]
|
||||
assert body["secondary"] == {}
|
||||
assert all(not e["secondary"] for e in body["by_year"])
|
||||
# The KS2 series is unaffected.
|
||||
assert body["primary"]["rwm_expected_pct"] == 62.1
|
||||
|
||||
Reference in New Issue
Block a user