feat(ees): keep DfE's KS4 LA averages from the summary data set (H2)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 10:46:53 +01:00
1 parent 967b1f0eed
commit dbb60acff3
3 files changed
+177 -49

No files matched your search

@@ -0,0 +1,56 @@
"""DfE's KS4 "summary, all state-funded" data set (Key stage 4 performance).
One CSV holds England, regional and local-authority headline rows for every
year since 2018/19. The England stream (ees_ks4_national) and the LA stream
(ees_ks4_la) both read it. Its LA rows match DfE's published performance-table
LA averages exactly (audit H2). Suppressed values ('z', 'x') become NULL in
dbt; Progress 8 is 'z' in years with no KS2 baseline (2024/25): DfE policy,
not missing data.
Free of the Singer SDK so CI's pytest can load it.
"""
from __future__ import annotations
import pandas as pd
KS4_SUMMARY_CSV_URL = (
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
)
# CSV column → Singer field: the same 8 headline measures at every level.
KS4_HEADLINE_COL_MAP = {
"attainment8_average": "attainment_8_score",
"progress8_average": "progress_8_score",
"engmath_94_percent": "english_maths_standard_pass_pct",
"engmath_95_percent": "english_maths_strong_pass_pct",
"ebacc_entering_percent": "ebacc_entry_pct",
"ebacc_94_percent": "ebacc_standard_pass_pct",
"ebacc_95_percent": "ebacc_strong_pass_pct",
"ebacc_aps_average": "ebacc_avg_score",
}
def headline_rows(df: pd.DataFrame, geographic_level: str) -> pd.DataFrame:
"""All-pupil rows for all state-funded schools at one geographic level
("National" or "Local authority"). Column names are lower-cased first;
a filter column the file lacks is not applied."""
df = df.copy()
df.columns = [c.strip().lower() for c in df.columns]
for col, want in (
("geographic_level", geographic_level),
("establishment_type_group", "All state-funded"),
("breakdown_topic", "Total"),
("breakdown", "Total"),
):
if col in df.columns:
df = df[df[col].str.strip().str.lower() == want.lower()]
return df
def headline_record(row: pd.Series, keys: tuple[str, ...]) -> dict[str, str]:
"""A Singer record: the identifying columns, then the headline measures."""
record = {key: str(row.get(key, "")).strip() for key in keys}
for csv_col, field in KS4_HEADLINE_COL_MAP.items():
record[field] = str(row.get(csv_col, "")).strip()
return record
@@ -18,6 +18,12 @@ from singer_sdk import Stream, Tap
from singer_sdk import typing as th
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
from tap_uk_ees.ks4_summary import (
KS4_HEADLINE_COL_MAP,
KS4_SUMMARY_CSV_URL,
headline_record,
headline_rows,
)
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
CONTENT_API_BASE = (
@@ -571,37 +577,23 @@ class EESKs2NationalStream(Stream):
yield record
# ── KS4 National Headlines (national level only — one row per year) ──────────
# Dataset: "National characteristics summary data" (Key stage 4 performance).
# Official England state-funded headline measures, 2018/19 → latest.
# Suppressed values ('z', 'x') → NULL downstream. Progress 8 is legitimately
# absent in years with no KS2 baseline (e.g. 2024/25) — that is DfE policy,
# not missing data.
# ── KS4 National and LA Headlines (one data set, two streams) ────────────────
# DfE's "summary, all state-funded" data set: England, regional and LA rows,
# 2018/19 → latest. URL, measures and filters: ks4_summary.py.
_KS4_NATIONAL_CSV_URL = (
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
)
def _read_ks4_summary(logger):
"""Download DfE's KS4 summary data set."""
import pandas as pd
_KS4_NATIONAL_COL_MAP = {
"attainment8_average": "attainment_8_score",
"progress8_average": "progress_8_score",
"engmath_94_percent": "english_maths_standard_pass_pct",
"engmath_95_percent": "english_maths_strong_pass_pct",
"ebacc_entering_percent": "ebacc_entry_pct",
"ebacc_94_percent": "ebacc_standard_pass_pct",
"ebacc_95_percent": "ebacc_strong_pass_pct",
"ebacc_aps_average": "ebacc_avg_score",
}
logger.info("Downloading KS4 summary data set: %s", KS4_SUMMARY_CSV_URL)
resp = requests.get(KS4_SUMMARY_CSV_URL, timeout=60)
resp.raise_for_status()
return pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
class EESKs4NationalStream(Stream):
"""National KS4 headline averages — one row per academic year.
Filters to geographic_level == 'National', establishment_type_group ==
'All state-funded', breakdown_topic == 'Total', breakdown == 'Total'
so only the England-wide all-pupils row per year is emitted.
"""
"""National KS4 headline averages — one row per academic year (England,
all state-funded schools, all pupils)."""
name = "ees_ks4_national"
primary_keys = ["time_period"]
@@ -609,34 +601,41 @@ class EESKs4NationalStream(Stream):
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
*[th.Property(out, th.StringType) for out in _KS4_NATIONAL_COL_MAP.values()],
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
).to_dict()
def get_records(self, context):
import pandas as pd
self.logger.info("Downloading KS4 national headlines: %s", _KS4_NATIONAL_CSV_URL)
resp = requests.get(_KS4_NATIONAL_CSV_URL, timeout=60)
resp.raise_for_status()
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
df.columns = [c.strip().lower() for c in df.columns]
for col, want in [
("geographic_level", "national"),
("establishment_type_group", "all state-funded"),
("breakdown_topic", "total"),
("breakdown", "total"),
]:
if col in df.columns:
df = df[df[col].str.strip().str.lower() == want]
df = headline_rows(_read_ks4_summary(self.logger), "National")
self.logger.info("Emitting %d national KS4 rows", len(df))
for _, row in df.iterrows():
record = {"time_period": row.get("time_period", "").strip()}
for csv_col, field in _KS4_NATIONAL_COL_MAP.items():
record[field] = row.get(csv_col, "").strip()
yield record
yield headline_record(row, ("time_period",))
class EESKs4LaStream(Stream):
"""DfE's KS4 local-authority averages — one row per academic year and LA
(all state-funded schools, all pupils), from the same data set as
ees_ks4_national. They match DfE's published performance-table LA averages
and replace a mean the API took over every school, independent and special
included (audit H2). old_la_code is the GIAS LA code: schools join on it,
not on the name."""
name = "ees_ks4_la"
primary_keys = ["time_period", "old_la_code"]
replication_key = None
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
th.Property("old_la_code", th.StringType, required=True),
th.Property("new_la_code", th.StringType),
th.Property("la_name", th.StringType),
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
).to_dict()
def get_records(self, context):
df = headline_rows(_read_ks4_summary(self.logger), "Local authority")
self.logger.info("Emitting %d LA KS4 rows", len(df))
for _, row in df.iterrows():
yield headline_record(row, ("time_period", "old_la_code", "new_la_code", "la_name"))
# ── Legacy KS2 (pre-COVID wide format from DfE performance tables) ────────────
@@ -979,6 +978,7 @@ class TapUKEES(Tap):
LegacyKS4Stream(self),
EESKs2NationalStream(self),
EESKs4NationalStream(self),
EESKs4LaStream(self),
]