feat(ees): keep DfE's KS4 LA averages from the summary data set (H2)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 10:46:53 +01:00
1 parent 967b1f0eed
commit dbb60acff3
3 files changed
+177 -49

No files matched your search

@@ -0,0 +1,56 @@
"""DfE's KS4 "summary, all state-funded" data set (Key stage 4 performance).
One CSV holds England, regional and local-authority headline rows for every
year since 2018/19. The England stream (ees_ks4_national) and the LA stream
(ees_ks4_la) both read it. Its LA rows match DfE's published performance-table
LA averages exactly (audit H2). Suppressed values ('z', 'x') become NULL in
dbt; Progress 8 is 'z' in years with no KS2 baseline (2024/25): DfE policy,
not missing data.
Free of the Singer SDK so CI's pytest can load it.
"""
from __future__ import annotations
import pandas as pd
KS4_SUMMARY_CSV_URL = (
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
)
# CSV column → Singer field: the same 8 headline measures at every level.
KS4_HEADLINE_COL_MAP = {
"attainment8_average": "attainment_8_score",
"progress8_average": "progress_8_score",
"engmath_94_percent": "english_maths_standard_pass_pct",
"engmath_95_percent": "english_maths_strong_pass_pct",
"ebacc_entering_percent": "ebacc_entry_pct",
"ebacc_94_percent": "ebacc_standard_pass_pct",
"ebacc_95_percent": "ebacc_strong_pass_pct",
"ebacc_aps_average": "ebacc_avg_score",
}
def headline_rows(df: pd.DataFrame, geographic_level: str) -> pd.DataFrame:
"""All-pupil rows for all state-funded schools at one geographic level
("National" or "Local authority"). Column names are lower-cased first;
a filter column the file lacks is not applied."""
df = df.copy()
df.columns = [c.strip().lower() for c in df.columns]
for col, want in (
("geographic_level", geographic_level),
("establishment_type_group", "All state-funded"),
("breakdown_topic", "Total"),
("breakdown", "Total"),
):
if col in df.columns:
df = df[df[col].str.strip().str.lower() == want.lower()]
return df
def headline_record(row: pd.Series, keys: tuple[str, ...]) -> dict[str, str]:
"""A Singer record: the identifying columns, then the headline measures."""
record = {key: str(row.get(key, "")).strip() for key in keys}
for csv_col, field in KS4_HEADLINE_COL_MAP.items():
record[field] = str(row.get(csv_col, "")).strip()
return record
@@ -18,6 +18,12 @@ from singer_sdk import Stream, Tap
from singer_sdk import typing as th
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
from tap_uk_ees.ks4_summary import (
KS4_HEADLINE_COL_MAP,
KS4_SUMMARY_CSV_URL,
headline_record,
headline_rows,
)
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
CONTENT_API_BASE = (
@@ -571,37 +577,23 @@ class EESKs2NationalStream(Stream):
yield record
# ── KS4 National Headlines (national level only — one row per year) ──────────
# Dataset: "National characteristics summary data" (Key stage 4 performance).
# Official England state-funded headline measures, 2018/19 → latest.
# Suppressed values ('z', 'x') → NULL downstream. Progress 8 is legitimately
# absent in years with no KS2 baseline (e.g. 2024/25) — that is DfE policy,
# not missing data.
# ── KS4 National and LA Headlines (one data set, two streams) ────────────────
# DfE's "summary, all state-funded" data set: England, regional and LA rows,
# 2018/19 → latest. URL, measures and filters: ks4_summary.py.
_KS4_NATIONAL_CSV_URL = (
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
)
def _read_ks4_summary(logger):
"""Download DfE's KS4 summary data set."""
import pandas as pd
_KS4_NATIONAL_COL_MAP = {
"attainment8_average": "attainment_8_score",
"progress8_average": "progress_8_score",
"engmath_94_percent": "english_maths_standard_pass_pct",
"engmath_95_percent": "english_maths_strong_pass_pct",
"ebacc_entering_percent": "ebacc_entry_pct",
"ebacc_94_percent": "ebacc_standard_pass_pct",
"ebacc_95_percent": "ebacc_strong_pass_pct",
"ebacc_aps_average": "ebacc_avg_score",
}
logger.info("Downloading KS4 summary data set: %s", KS4_SUMMARY_CSV_URL)
resp = requests.get(KS4_SUMMARY_CSV_URL, timeout=60)
resp.raise_for_status()
return pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
class EESKs4NationalStream(Stream):
"""National KS4 headline averages — one row per academic year.
Filters to geographic_level == 'National', establishment_type_group ==
'All state-funded', breakdown_topic == 'Total', breakdown == 'Total'
so only the England-wide all-pupils row per year is emitted.
"""
"""National KS4 headline averages — one row per academic year (England,
all state-funded schools, all pupils)."""
name = "ees_ks4_national"
primary_keys = ["time_period"]
@@ -609,34 +601,41 @@ class EESKs4NationalStream(Stream):
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
*[th.Property(out, th.StringType) for out in _KS4_NATIONAL_COL_MAP.values()],
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
).to_dict()
def get_records(self, context):
import pandas as pd
self.logger.info("Downloading KS4 national headlines: %s", _KS4_NATIONAL_CSV_URL)
resp = requests.get(_KS4_NATIONAL_CSV_URL, timeout=60)
resp.raise_for_status()
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
df.columns = [c.strip().lower() for c in df.columns]
for col, want in [
("geographic_level", "national"),
("establishment_type_group", "all state-funded"),
("breakdown_topic", "total"),
("breakdown", "total"),
]:
if col in df.columns:
df = df[df[col].str.strip().str.lower() == want]
df = headline_rows(_read_ks4_summary(self.logger), "National")
self.logger.info("Emitting %d national KS4 rows", len(df))
for _, row in df.iterrows():
record = {"time_period": row.get("time_period", "").strip()}
for csv_col, field in _KS4_NATIONAL_COL_MAP.items():
record[field] = row.get(csv_col, "").strip()
yield record
yield headline_record(row, ("time_period",))
class EESKs4LaStream(Stream):
"""DfE's KS4 local-authority averages — one row per academic year and LA
(all state-funded schools, all pupils), from the same data set as
ees_ks4_national. They match DfE's published performance-table LA averages
and replace a mean the API took over every school, independent and special
included (audit H2). old_la_code is the GIAS LA code: schools join on it,
not on the name."""
name = "ees_ks4_la"
primary_keys = ["time_period", "old_la_code"]
replication_key = None
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
th.Property("old_la_code", th.StringType, required=True),
th.Property("new_la_code", th.StringType),
th.Property("la_name", th.StringType),
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
).to_dict()
def get_records(self, context):
df = headline_rows(_read_ks4_summary(self.logger), "Local authority")
self.logger.info("Emitting %d LA KS4 rows", len(df))
for _, row in df.iterrows():
yield headline_record(row, ("time_period", "old_la_code", "new_la_code", "la_name"))
# ── Legacy KS2 (pre-COVID wide format from DfE performance tables) ────────────
@@ -979,6 +978,7 @@ class TapUKEES(Tap):
LegacyKS4Stream(self),
EESKs2NationalStream(self),
EESKs4NationalStream(self),
EESKs4LaStream(self),
]
+72
View File
@@ -0,0 +1,72 @@
"""DfE's KS4 "summary, all state-funded" data set holds England, regional and
LA rows. The England stream kept only the England row; its LA rows match
DfE's published LA averages exactly, where the API's own mean was 7 points
low (audit H2). Values below are DfE's for Kensington and Chelsea (207) and
Wandsworth (212).
"""
import importlib.util
import io
from pathlib import Path
import pandas as pd
import pytest
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees'
/ 'tap_uk_ees' / 'ks4_summary.py')
CSV = """time_period,geographic_level,old_la_code,new_la_code,la_name,establishment_type_group,breakdown_topic,breakdown,attainment8_average,progress8_average,engmath_94_percent,engmath_95_percent,ebacc_entering_percent,ebacc_94_percent,ebacc_95_percent,ebacc_aps_average
202425,National,,,,All state-funded,Total,Total,46.1,z,64.5,45.7,40.5,26.9,17.7,4.1
202425,National,,,,All state-funded,Sex,Boys,44.1,z,62.0,43.0,38.0,24.0,16.0,3.9
202425,Regional,,,,All state-funded,Total,Total,47.2,z,66.0,47.0,41.0,28.0,18.0,4.2
202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,z,77,61.4,45.6,32,26.6,4.89
202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Sex,Girls,57.0,z,80,64.0,48.0,35,28.0,5.1
202324,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,0.29,76,60.0,44.0,31,25.0,4.8
202425,Local authority,212,E09000032,Wandsworth,All state-funded,Total,Total,51.8,z,72,55.0,50.0,33,24.0,4.6
202425,Local authority,212,E09000032,Wandsworth,Academies and free schools,Total,Total,52.0,z,73,56.0,51.0,34,25.0,4.7
"""
LA_KEYS = ('time_period', 'old_la_code', 'new_la_code', 'la_name')
def _df():
return pd.read_csv(io.StringIO(CSV), dtype=str, keep_default_na=False)
@pytest.fixture
def summary():
spec = importlib.util.spec_from_file_location('ks4_summary', MODULE)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def test_la_rows_are_one_per_la_and_year(summary):
rows = summary.headline_rows(_df(), 'Local authority')
assert sorted(zip(rows['time_period'], rows['old_la_code'])) == [
('202324', '207'), ('202425', '207'), ('202425', '212')]
def test_an_la_record_carries_codes_name_and_the_headline_measures(summary):
rows = summary.headline_rows(_df(), 'Local authority')
row = rows[(rows['time_period'] == '202425') & (rows['old_la_code'] == '207')].iloc[0]
assert summary.headline_record(row, LA_KEYS) == {
'time_period': '202425', 'old_la_code': '207', 'new_la_code': 'E09000020',
'la_name': 'Kensington and Chelsea',
'attainment_8_score': '54.5', 'progress_8_score': 'z',
'english_maths_standard_pass_pct': '77', 'english_maths_strong_pass_pct': '61.4',
'ebacc_entry_pct': '45.6', 'ebacc_standard_pass_pct': '32',
'ebacc_strong_pass_pct': '26.6', 'ebacc_avg_score': '4.89',
}
def test_the_england_rows_are_one_per_year(summary):
rows = summary.headline_rows(_df(), 'National')
assert list(rows['time_period']) == ['202425']
assert summary.headline_record(rows.iloc[0], ('time_period',))['attainment_8_score'] == '46.1'
def test_column_names_and_labels_match_whatever_their_case(summary):
df = _df()
df.columns = [c.upper() for c in df.columns]
df['GEOGRAPHIC_LEVEL'] = df['GEOGRAPHIC_LEVEL'].str.upper()
assert len(summary.headline_rows(df, 'Local authority')) == 3