From dbb60acff327e6423cad14bf5f61c6eb05f064df Mon Sep 17 00:00:00 2001 From: Tudor Date: Tue, 6 Oct 2026 10:46:53 +0100 Subject: [PATCH] feat(ees): keep DfE's KS4 LA averages from the summary data set (H2) Co-Authored-By: Claude Opus 5.5 --- .../tap-uk-ees/tap_uk_ees/ks4_summary.py | 56 +++++++++++ .../extractors/tap-uk-ees/tap_uk_ees/tap.py | 98 +++++++++---------- pipeline/tests/test_ees_ks4_summary.py | 72 ++++++++++++++ 3 files changed, 177 insertions(+), 49 deletions(-) create mode 100644 pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_summary.py create mode 100644 pipeline/tests/test_ees_ks4_summary.py diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_summary.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_summary.py new file mode 100644 index 0000000..3ef7eec --- /dev/null +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_summary.py @@ -0,0 +1,56 @@ +"""DfE's KS4 "summary, all state-funded" data set (Key stage 4 performance). + +One CSV holds England, regional and local-authority headline rows for every +year since 2018/19. The England stream (ees_ks4_national) and the LA stream +(ees_ks4_la) both read it. Its LA rows match DfE's published performance-table +LA averages exactly (audit H2). Suppressed values ('z', 'x') become NULL in +dbt; Progress 8 is 'z' in years with no KS2 baseline (2024/25): DfE policy, +not missing data. + +Free of the Singer SDK so CI's pytest can load it. +""" +from __future__ import annotations + +import pandas as pd + +KS4_SUMMARY_CSV_URL = ( + "https://explore-education-statistics.service.gov.uk/data-catalogue/" + "data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv" +) + +# CSV column → Singer field: the same 8 headline measures at every level. +KS4_HEADLINE_COL_MAP = { + "attainment8_average": "attainment_8_score", + "progress8_average": "progress_8_score", + "engmath_94_percent": "english_maths_standard_pass_pct", + "engmath_95_percent": "english_maths_strong_pass_pct", + "ebacc_entering_percent": "ebacc_entry_pct", + "ebacc_94_percent": "ebacc_standard_pass_pct", + "ebacc_95_percent": "ebacc_strong_pass_pct", + "ebacc_aps_average": "ebacc_avg_score", +} + + +def headline_rows(df: pd.DataFrame, geographic_level: str) -> pd.DataFrame: + """All-pupil rows for all state-funded schools at one geographic level + ("National" or "Local authority"). Column names are lower-cased first; + a filter column the file lacks is not applied.""" + df = df.copy() + df.columns = [c.strip().lower() for c in df.columns] + for col, want in ( + ("geographic_level", geographic_level), + ("establishment_type_group", "All state-funded"), + ("breakdown_topic", "Total"), + ("breakdown", "Total"), + ): + if col in df.columns: + df = df[df[col].str.strip().str.lower() == want.lower()] + return df + + +def headline_record(row: pd.Series, keys: tuple[str, ...]) -> dict[str, str]: + """A Singer record: the identifying columns, then the headline measures.""" + record = {key: str(row.get(key, "")).strip() for key in keys} + for csv_col, field in KS4_HEADLINE_COL_MAP.items(): + record[field] = str(row.get(csv_col, "")).strip() + return record diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py index cb7816a..d550b80 100644 --- a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py @@ -18,6 +18,12 @@ from singer_sdk import Stream, Tap from singer_sdk import typing as th from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES +from tap_uk_ees.ks4_summary import ( + KS4_HEADLINE_COL_MAP, + KS4_SUMMARY_CSV_URL, + headline_record, + headline_rows, +) from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in CONTENT_API_BASE = ( @@ -571,37 +577,23 @@ class EESKs2NationalStream(Stream): yield record -# ── KS4 National Headlines (national level only — one row per year) ────────── -# Dataset: "National characteristics summary data" (Key stage 4 performance). -# Official England state-funded headline measures, 2018/19 → latest. -# Suppressed values ('z', 'x') → NULL downstream. Progress 8 is legitimately -# absent in years with no KS2 baseline (e.g. 2024/25) — that is DfE policy, -# not missing data. +# ── KS4 National and LA Headlines (one data set, two streams) ──────────────── +# DfE's "summary, all state-funded" data set: England, regional and LA rows, +# 2018/19 → latest. URL, measures and filters: ks4_summary.py. -_KS4_NATIONAL_CSV_URL = ( - "https://explore-education-statistics.service.gov.uk/data-catalogue/" - "data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv" -) +def _read_ks4_summary(logger): + """Download DfE's KS4 summary data set.""" + import pandas as pd -_KS4_NATIONAL_COL_MAP = { - "attainment8_average": "attainment_8_score", - "progress8_average": "progress_8_score", - "engmath_94_percent": "english_maths_standard_pass_pct", - "engmath_95_percent": "english_maths_strong_pass_pct", - "ebacc_entering_percent": "ebacc_entry_pct", - "ebacc_94_percent": "ebacc_standard_pass_pct", - "ebacc_95_percent": "ebacc_strong_pass_pct", - "ebacc_aps_average": "ebacc_avg_score", -} + logger.info("Downloading KS4 summary data set: %s", KS4_SUMMARY_CSV_URL) + resp = requests.get(KS4_SUMMARY_CSV_URL, timeout=60) + resp.raise_for_status() + return pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False) class EESKs4NationalStream(Stream): - """National KS4 headline averages — one row per academic year. - - Filters to geographic_level == 'National', establishment_type_group == - 'All state-funded', breakdown_topic == 'Total', breakdown == 'Total' - so only the England-wide all-pupils row per year is emitted. - """ + """National KS4 headline averages — one row per academic year (England, + all state-funded schools, all pupils).""" name = "ees_ks4_national" primary_keys = ["time_period"] @@ -609,34 +601,41 @@ class EESKs4NationalStream(Stream): schema = th.PropertiesList( th.Property("time_period", th.StringType, required=True), - *[th.Property(out, th.StringType) for out in _KS4_NATIONAL_COL_MAP.values()], + *[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()], ).to_dict() def get_records(self, context): - import pandas as pd - - self.logger.info("Downloading KS4 national headlines: %s", _KS4_NATIONAL_CSV_URL) - resp = requests.get(_KS4_NATIONAL_CSV_URL, timeout=60) - resp.raise_for_status() - - df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False) - df.columns = [c.strip().lower() for c in df.columns] - - for col, want in [ - ("geographic_level", "national"), - ("establishment_type_group", "all state-funded"), - ("breakdown_topic", "total"), - ("breakdown", "total"), - ]: - if col in df.columns: - df = df[df[col].str.strip().str.lower() == want] - + df = headline_rows(_read_ks4_summary(self.logger), "National") self.logger.info("Emitting %d national KS4 rows", len(df)) for _, row in df.iterrows(): - record = {"time_period": row.get("time_period", "").strip()} - for csv_col, field in _KS4_NATIONAL_COL_MAP.items(): - record[field] = row.get(csv_col, "").strip() - yield record + yield headline_record(row, ("time_period",)) + + +class EESKs4LaStream(Stream): + """DfE's KS4 local-authority averages — one row per academic year and LA + (all state-funded schools, all pupils), from the same data set as + ees_ks4_national. They match DfE's published performance-table LA averages + and replace a mean the API took over every school, independent and special + included (audit H2). old_la_code is the GIAS LA code: schools join on it, + not on the name.""" + + name = "ees_ks4_la" + primary_keys = ["time_period", "old_la_code"] + replication_key = None + + schema = th.PropertiesList( + th.Property("time_period", th.StringType, required=True), + th.Property("old_la_code", th.StringType, required=True), + th.Property("new_la_code", th.StringType), + th.Property("la_name", th.StringType), + *[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()], + ).to_dict() + + def get_records(self, context): + df = headline_rows(_read_ks4_summary(self.logger), "Local authority") + self.logger.info("Emitting %d LA KS4 rows", len(df)) + for _, row in df.iterrows(): + yield headline_record(row, ("time_period", "old_la_code", "new_la_code", "la_name")) # ── Legacy KS2 (pre-COVID wide format from DfE performance tables) ──────────── @@ -979,6 +978,7 @@ class TapUKEES(Tap): LegacyKS4Stream(self), EESKs2NationalStream(self), EESKs4NationalStream(self), + EESKs4LaStream(self), ] diff --git a/pipeline/tests/test_ees_ks4_summary.py b/pipeline/tests/test_ees_ks4_summary.py new file mode 100644 index 0000000..da8a1a9 --- /dev/null +++ b/pipeline/tests/test_ees_ks4_summary.py @@ -0,0 +1,72 @@ +"""DfE's KS4 "summary, all state-funded" data set holds England, regional and +LA rows. The England stream kept only the England row; its LA rows match +DfE's published LA averages exactly, where the API's own mean was 7 points +low (audit H2). Values below are DfE's for Kensington and Chelsea (207) and +Wandsworth (212). +""" +import importlib.util +import io +from pathlib import Path + +import pandas as pd +import pytest + +MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees' + / 'tap_uk_ees' / 'ks4_summary.py') + +CSV = """time_period,geographic_level,old_la_code,new_la_code,la_name,establishment_type_group,breakdown_topic,breakdown,attainment8_average,progress8_average,engmath_94_percent,engmath_95_percent,ebacc_entering_percent,ebacc_94_percent,ebacc_95_percent,ebacc_aps_average +202425,National,,,,All state-funded,Total,Total,46.1,z,64.5,45.7,40.5,26.9,17.7,4.1 +202425,National,,,,All state-funded,Sex,Boys,44.1,z,62.0,43.0,38.0,24.0,16.0,3.9 +202425,Regional,,,,All state-funded,Total,Total,47.2,z,66.0,47.0,41.0,28.0,18.0,4.2 +202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,z,77,61.4,45.6,32,26.6,4.89 +202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Sex,Girls,57.0,z,80,64.0,48.0,35,28.0,5.1 +202324,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,0.29,76,60.0,44.0,31,25.0,4.8 +202425,Local authority,212,E09000032,Wandsworth,All state-funded,Total,Total,51.8,z,72,55.0,50.0,33,24.0,4.6 +202425,Local authority,212,E09000032,Wandsworth,Academies and free schools,Total,Total,52.0,z,73,56.0,51.0,34,25.0,4.7 +""" + +LA_KEYS = ('time_period', 'old_la_code', 'new_la_code', 'la_name') + + +def _df(): + return pd.read_csv(io.StringIO(CSV), dtype=str, keep_default_na=False) + + +@pytest.fixture +def summary(): + spec = importlib.util.spec_from_file_location('ks4_summary', MODULE) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_la_rows_are_one_per_la_and_year(summary): + rows = summary.headline_rows(_df(), 'Local authority') + assert sorted(zip(rows['time_period'], rows['old_la_code'])) == [ + ('202324', '207'), ('202425', '207'), ('202425', '212')] + + +def test_an_la_record_carries_codes_name_and_the_headline_measures(summary): + rows = summary.headline_rows(_df(), 'Local authority') + row = rows[(rows['time_period'] == '202425') & (rows['old_la_code'] == '207')].iloc[0] + assert summary.headline_record(row, LA_KEYS) == { + 'time_period': '202425', 'old_la_code': '207', 'new_la_code': 'E09000020', + 'la_name': 'Kensington and Chelsea', + 'attainment_8_score': '54.5', 'progress_8_score': 'z', + 'english_maths_standard_pass_pct': '77', 'english_maths_strong_pass_pct': '61.4', + 'ebacc_entry_pct': '45.6', 'ebacc_standard_pass_pct': '32', + 'ebacc_strong_pass_pct': '26.6', 'ebacc_avg_score': '4.89', + } + + +def test_the_england_rows_are_one_per_year(summary): + rows = summary.headline_rows(_df(), 'National') + assert list(rows['time_period']) == ['202425'] + assert summary.headline_record(rows.iloc[0], ('time_period',))['attainment_8_score'] == '46.1' + + +def test_column_names_and_labels_match_whatever_their_case(summary): + df = _df() + df.columns = [c.upper() for c in df.columns] + df['GEOGRAPHIC_LEVEL'] = df['GEOGRAPHIC_LEVEL'].str.upper() + assert len(summary.headline_rows(df, 'Local authority')) == 3