diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_info.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_info.py new file mode 100644 index 0000000..eff6b94 --- /dev/null +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/ks4_info.py @@ -0,0 +1,47 @@ +"""KS4 school information: the fields the stream declares, and the older +names DfE used for them. + +2023/24 information exists only in the 2023/24 release, whose +202324_information_about_schools_final.csv (re-issued 10 March 2026) uses the +older names. Without the renames every 2023/24 field loaded as null (audit C2). +Newer files contain none of the older names, so the renames leave them alone. + +Free of the Singer SDK so CI's pytest can load it. +""" + +# Declared Singer fields besides the required time_period and school_urn. +KS4_INFO_FIELDS = ( + "school_laestab", + "school_name", + "establishment_type_group", + "reldenom", + "admpol_pt", + "egender", + "agerange", + "allks_pupil_count", + "allks_boys_count", + "allks_girls_count", + "endks4_pupil_count", + "ks2_scaledscore_average", + "sen_with_ehcp_pupil_percent", + "sen_pupil_percent", + "sen_no_ehcp_pupil_percent", + "attainment8_diffn", + "progress8_diffn", + "progress8_banding", +) + +# 2023/24 column name → declared field. +KS4_INFO_RENAMES = { + "t_allks_pupils": "allks_pupil_count", + "t_allks_boys": "allks_boys_count", + "t_allks_girls": "allks_girls_count", + "t_pupils": "endks4_pupil_count", + "avg_ks2_scaledscore": "ks2_scaledscore_average", + "pt_sen_with_ehcp": "sen_with_ehcp_pupil_percent", + "pt_sen": "sen_pupil_percent", + "pt_sen_no_ehcp": "sen_no_ehcp_pupil_percent", + "diffn_att8": "attainment8_diffn", + "diffn_p8mea": "progress8_diffn", + "p8_banding": "progress8_banding", +} diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py index 1b399df..cb7816a 100644 --- a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py @@ -17,6 +17,7 @@ import requests from singer_sdk import Stream, Tap from singer_sdk import typing as th +from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in CONTENT_API_BASE = ( @@ -330,34 +331,20 @@ class EESKS4PerformanceStream(EESDatasetStream): # ── KS4 Information (wide format: one row per school, context/demographics) ── -# File: 202425_information_about_schools_provisional.csv (38 cols) +# Files: 202425_information_about_schools_final.csv (38 cols, current names); +# 202324_information_about_schools_final.csv (60 cols, older names — the only +# source of 2023/24 information). Field list and renames: ks4_info.py. class EESKS4InfoStream(EESDatasetStream): name = "ees_ks4_info" primary_keys = ["school_urn", "time_period"] _publication_slug = "key-stage-4-performance" _target_filename = "information_about_schools" + _column_renames = KS4_INFO_RENAMES schema = th.PropertiesList( th.Property("time_period", th.StringType, required=True), th.Property("school_urn", th.StringType, required=True), - th.Property("school_laestab", th.StringType), - th.Property("school_name", th.StringType), - th.Property("establishment_type_group", th.StringType), - th.Property("reldenom", th.StringType), - th.Property("admpol_pt", th.StringType), - th.Property("egender", th.StringType), - th.Property("agerange", th.StringType), - th.Property("allks_pupil_count", th.StringType), - th.Property("allks_boys_count", th.StringType), - th.Property("allks_girls_count", th.StringType), - th.Property("endks4_pupil_count", th.StringType), - th.Property("ks2_scaledscore_average", th.StringType), - th.Property("sen_with_ehcp_pupil_percent", th.StringType), - th.Property("sen_pupil_percent", th.StringType), - th.Property("sen_no_ehcp_pupil_percent", th.StringType), - th.Property("attainment8_diffn", th.StringType), - th.Property("progress8_diffn", th.StringType), - th.Property("progress8_banding", th.StringType), + *[th.Property(field, th.StringType) for field in KS4_INFO_FIELDS], ).to_dict() diff --git a/pipeline/tests/test_ees_ks4_info_names.py b/pipeline/tests/test_ees_ks4_info_names.py new file mode 100644 index 0000000..ccdb102 --- /dev/null +++ b/pipeline/tests/test_ees_ks4_info_names.py @@ -0,0 +1,67 @@ +"""KS4 school information for 2023/24 exists only in the 2023/24 release, +whose file uses DfE's older column names. The stream declared only the newer +ones, so every 2023/24 field loaded as null (audit C2). The headers below are +DfE's, copied from 202324_information_about_schools_final.csv and +202425_information_about_schools_final.csv. +""" +import importlib.util +from pathlib import Path + +import pytest + +MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees' + / 'tap_uk_ees' / 'ks4_info.py') + +HEADER_2023_24 = ( + 'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name', + 'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name', + 'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name', + 'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange', + 't_allks_pupils', 't_allks_boys', 't_allks_girls', 't_pupils', 't_boys', 'pt_boys', + 't_girls', 'pt_girls', 'avg_ks2_scaledscore', 't_prior_lo', 'pt_prior_lo', 't_prior_av', + 'pt_prior_av', 't_prior_hi', 'pt_prior_hi', 't_disadvantaged', 'pt_disadvantaged', + 't_not_disadvantaged', 'pt_not_disadvantaged', 't_language_not_english', + 'pt_language_not_english', 't_language_english', 'pt_language_english', + 't_language_unknown', 'pt_language_unknown', 't_not_mobile', 'pt_not_mobile', + 't_sen_with_ehcp', 'pt_sen_with_ehcp', 't_sen', 'pt_sen', 't_sen_no_ehcp', + 'pt_sen_no_ehcp', 'diffn_att8', 'diffn_p8mea', 'p8_banding', +) + +HEADER_2024_25 = ( + 'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name', + 'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name', + 'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name', + 'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange', + 'allks_pupil_count', 'allks_boys_count', 'allks_girls_count', 'endks4_pupil_count', + 'ks2_scaledscore_average', 'sen_with_ehcp_pupil_count', 'sen_with_ehcp_pupil_percent', + 'sen_pupil_count', 'sen_pupil_percent', 'sen_no_ehcp_pupil_count', + 'sen_no_ehcp_pupil_percent', 'attainment8_diffn', 'progress8_diffn', 'progress8_banding', +) + + +@pytest.fixture +def ks4_info(): + spec = importlib.util.spec_from_file_location('ks4_info', MODULE) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_every_declared_field_is_in_the_2023_24_file_once_renamed(ks4_info): + renamed = {ks4_info.KS4_INFO_RENAMES.get(c, c) for c in HEADER_2023_24} + assert set(ks4_info.KS4_INFO_FIELDS) - renamed == set() + + +def test_every_declared_field_is_in_the_2024_25_file(ks4_info): + assert set(ks4_info.KS4_INFO_FIELDS) - set(HEADER_2024_25) == set() + + +def test_the_renames_cannot_collide_with_either_file(ks4_info): + # An old name in the current file, or a new name already in the old file, + # would let a rename overwrite a real column. + assert set(ks4_info.KS4_INFO_RENAMES) & set(HEADER_2024_25) == set() + assert set(ks4_info.KS4_INFO_RENAMES.values()) & set(HEADER_2023_24) == set() + + +def test_every_rename_names_a_declared_field(ks4_info): + assert set(ks4_info.KS4_INFO_RENAMES.values()) <= set(ks4_info.KS4_INFO_FIELDS)