fix(ees): read 2023/24 KS4 school information under DfE's older names (C2)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 10:45:57 +01:00
1 parent 4db1131d0f
commit 967b1f0eed
3 files changed
+120 -19

No files matched your search

@@ -0,0 +1,47 @@
"""KS4 school information: the fields the stream declares, and the older
names DfE used for them.
2023/24 information exists only in the 2023/24 release, whose
202324_information_about_schools_final.csv (re-issued 10 March 2026) uses the
older names. Without the renames every 2023/24 field loaded as null (audit C2).
Newer files contain none of the older names, so the renames leave them alone.
Free of the Singer SDK so CI's pytest can load it.
"""
# Declared Singer fields besides the required time_period and school_urn.
KS4_INFO_FIELDS = (
"school_laestab",
"school_name",
"establishment_type_group",
"reldenom",
"admpol_pt",
"egender",
"agerange",
"allks_pupil_count",
"allks_boys_count",
"allks_girls_count",
"endks4_pupil_count",
"ks2_scaledscore_average",
"sen_with_ehcp_pupil_percent",
"sen_pupil_percent",
"sen_no_ehcp_pupil_percent",
"attainment8_diffn",
"progress8_diffn",
"progress8_banding",
)
# 2023/24 column name → declared field.
KS4_INFO_RENAMES = {
"t_allks_pupils": "allks_pupil_count",
"t_allks_boys": "allks_boys_count",
"t_allks_girls": "allks_girls_count",
"t_pupils": "endks4_pupil_count",
"avg_ks2_scaledscore": "ks2_scaledscore_average",
"pt_sen_with_ehcp": "sen_with_ehcp_pupil_percent",
"pt_sen": "sen_pupil_percent",
"pt_sen_no_ehcp": "sen_no_ehcp_pupil_percent",
"diffn_att8": "attainment8_diffn",
"diffn_p8mea": "progress8_diffn",
"p8_banding": "progress8_banding",
}
@@ -17,6 +17,7 @@ import requests
from singer_sdk import Stream, Tap
from singer_sdk import typing as th
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
CONTENT_API_BASE = (
@@ -330,34 +331,20 @@ class EESKS4PerformanceStream(EESDatasetStream):
# ── KS4 Information (wide format: one row per school, context/demographics) ──
# File: 202425_information_about_schools_provisional.csv (38 cols)
# Files: 202425_information_about_schools_final.csv (38 cols, current names);
# 202324_information_about_schools_final.csv (60 cols, older names — the only
# source of 2023/24 information). Field list and renames: ks4_info.py.
class EESKS4InfoStream(EESDatasetStream):
name = "ees_ks4_info"
primary_keys = ["school_urn", "time_period"]
_publication_slug = "key-stage-4-performance"
_target_filename = "information_about_schools"
_column_renames = KS4_INFO_RENAMES
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
th.Property("school_urn", th.StringType, required=True),
th.Property("school_laestab", th.StringType),
th.Property("school_name", th.StringType),
th.Property("establishment_type_group", th.StringType),
th.Property("reldenom", th.StringType),
th.Property("admpol_pt", th.StringType),
th.Property("egender", th.StringType),
th.Property("agerange", th.StringType),
th.Property("allks_pupil_count", th.StringType),
th.Property("allks_boys_count", th.StringType),
th.Property("allks_girls_count", th.StringType),
th.Property("endks4_pupil_count", th.StringType),
th.Property("ks2_scaledscore_average", th.StringType),
th.Property("sen_with_ehcp_pupil_percent", th.StringType),
th.Property("sen_pupil_percent", th.StringType),
th.Property("sen_no_ehcp_pupil_percent", th.StringType),
th.Property("attainment8_diffn", th.StringType),
th.Property("progress8_diffn", th.StringType),
th.Property("progress8_banding", th.StringType),
*[th.Property(field, th.StringType) for field in KS4_INFO_FIELDS],
).to_dict()
+67
View File
@@ -0,0 +1,67 @@
"""KS4 school information for 2023/24 exists only in the 2023/24 release,
whose file uses DfE's older column names. The stream declared only the newer
ones, so every 2023/24 field loaded as null (audit C2). The headers below are
DfE's, copied from 202324_information_about_schools_final.csv and
202425_information_about_schools_final.csv.
"""
import importlib.util
from pathlib import Path
import pytest
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees'
/ 'tap_uk_ees' / 'ks4_info.py')
HEADER_2023_24 = (
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
't_allks_pupils', 't_allks_boys', 't_allks_girls', 't_pupils', 't_boys', 'pt_boys',
't_girls', 'pt_girls', 'avg_ks2_scaledscore', 't_prior_lo', 'pt_prior_lo', 't_prior_av',
'pt_prior_av', 't_prior_hi', 'pt_prior_hi', 't_disadvantaged', 'pt_disadvantaged',
't_not_disadvantaged', 'pt_not_disadvantaged', 't_language_not_english',
'pt_language_not_english', 't_language_english', 'pt_language_english',
't_language_unknown', 'pt_language_unknown', 't_not_mobile', 'pt_not_mobile',
't_sen_with_ehcp', 'pt_sen_with_ehcp', 't_sen', 'pt_sen', 't_sen_no_ehcp',
'pt_sen_no_ehcp', 'diffn_att8', 'diffn_p8mea', 'p8_banding',
)
HEADER_2024_25 = (
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
'allks_pupil_count', 'allks_boys_count', 'allks_girls_count', 'endks4_pupil_count',
'ks2_scaledscore_average', 'sen_with_ehcp_pupil_count', 'sen_with_ehcp_pupil_percent',
'sen_pupil_count', 'sen_pupil_percent', 'sen_no_ehcp_pupil_count',
'sen_no_ehcp_pupil_percent', 'attainment8_diffn', 'progress8_diffn', 'progress8_banding',
)
@pytest.fixture
def ks4_info():
spec = importlib.util.spec_from_file_location('ks4_info', MODULE)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def test_every_declared_field_is_in_the_2023_24_file_once_renamed(ks4_info):
renamed = {ks4_info.KS4_INFO_RENAMES.get(c, c) for c in HEADER_2023_24}
assert set(ks4_info.KS4_INFO_FIELDS) - renamed == set()
def test_every_declared_field_is_in_the_2024_25_file(ks4_info):
assert set(ks4_info.KS4_INFO_FIELDS) - set(HEADER_2024_25) == set()
def test_the_renames_cannot_collide_with_either_file(ks4_info):
# An old name in the current file, or a new name already in the old file,
# would let a rename overwrite a real column.
assert set(ks4_info.KS4_INFO_RENAMES) & set(HEADER_2024_25) == set()
assert set(ks4_info.KS4_INFO_RENAMES.values()) & set(HEADER_2023_24) == set()
def test_every_rename_names_a_declared_field(ks4_info):
assert set(ks4_info.KS4_INFO_RENAMES.values()) <= set(ks4_info.KS4_INFO_FIELDS)