fix(ees): read 2023/24 KS4 school information under DfE's older names (C2)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
4db1131d0f
commit
967b1f0eed
3 files changed
+120
-19
No files matched your search
@@ -0,0 +1,47 @@
|
||||
"""KS4 school information: the fields the stream declares, and the older
|
||||
names DfE used for them.
|
||||
|
||||
2023/24 information exists only in the 2023/24 release, whose
|
||||
202324_information_about_schools_final.csv (re-issued 10 March 2026) uses the
|
||||
older names. Without the renames every 2023/24 field loaded as null (audit C2).
|
||||
Newer files contain none of the older names, so the renames leave them alone.
|
||||
|
||||
Free of the Singer SDK so CI's pytest can load it.
|
||||
"""
|
||||
|
||||
# Declared Singer fields besides the required time_period and school_urn.
|
||||
KS4_INFO_FIELDS = (
|
||||
"school_laestab",
|
||||
"school_name",
|
||||
"establishment_type_group",
|
||||
"reldenom",
|
||||
"admpol_pt",
|
||||
"egender",
|
||||
"agerange",
|
||||
"allks_pupil_count",
|
||||
"allks_boys_count",
|
||||
"allks_girls_count",
|
||||
"endks4_pupil_count",
|
||||
"ks2_scaledscore_average",
|
||||
"sen_with_ehcp_pupil_percent",
|
||||
"sen_pupil_percent",
|
||||
"sen_no_ehcp_pupil_percent",
|
||||
"attainment8_diffn",
|
||||
"progress8_diffn",
|
||||
"progress8_banding",
|
||||
)
|
||||
|
||||
# 2023/24 column name → declared field.
|
||||
KS4_INFO_RENAMES = {
|
||||
"t_allks_pupils": "allks_pupil_count",
|
||||
"t_allks_boys": "allks_boys_count",
|
||||
"t_allks_girls": "allks_girls_count",
|
||||
"t_pupils": "endks4_pupil_count",
|
||||
"avg_ks2_scaledscore": "ks2_scaledscore_average",
|
||||
"pt_sen_with_ehcp": "sen_with_ehcp_pupil_percent",
|
||||
"pt_sen": "sen_pupil_percent",
|
||||
"pt_sen_no_ehcp": "sen_no_ehcp_pupil_percent",
|
||||
"diffn_att8": "attainment8_diffn",
|
||||
"diffn_p8mea": "progress8_diffn",
|
||||
"p8_banding": "progress8_banding",
|
||||
}
|
||||
@@ -17,6 +17,7 @@ import requests
|
||||
from singer_sdk import Stream, Tap
|
||||
from singer_sdk import typing as th
|
||||
|
||||
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
|
||||
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
|
||||
|
||||
CONTENT_API_BASE = (
|
||||
@@ -330,34 +331,20 @@ class EESKS4PerformanceStream(EESDatasetStream):
|
||||
|
||||
|
||||
# ── KS4 Information (wide format: one row per school, context/demographics) ──
|
||||
# File: 202425_information_about_schools_provisional.csv (38 cols)
|
||||
# Files: 202425_information_about_schools_final.csv (38 cols, current names);
|
||||
# 202324_information_about_schools_final.csv (60 cols, older names — the only
|
||||
# source of 2023/24 information). Field list and renames: ks4_info.py.
|
||||
|
||||
class EESKS4InfoStream(EESDatasetStream):
|
||||
name = "ees_ks4_info"
|
||||
primary_keys = ["school_urn", "time_period"]
|
||||
_publication_slug = "key-stage-4-performance"
|
||||
_target_filename = "information_about_schools"
|
||||
_column_renames = KS4_INFO_RENAMES
|
||||
schema = th.PropertiesList(
|
||||
th.Property("time_period", th.StringType, required=True),
|
||||
th.Property("school_urn", th.StringType, required=True),
|
||||
th.Property("school_laestab", th.StringType),
|
||||
th.Property("school_name", th.StringType),
|
||||
th.Property("establishment_type_group", th.StringType),
|
||||
th.Property("reldenom", th.StringType),
|
||||
th.Property("admpol_pt", th.StringType),
|
||||
th.Property("egender", th.StringType),
|
||||
th.Property("agerange", th.StringType),
|
||||
th.Property("allks_pupil_count", th.StringType),
|
||||
th.Property("allks_boys_count", th.StringType),
|
||||
th.Property("allks_girls_count", th.StringType),
|
||||
th.Property("endks4_pupil_count", th.StringType),
|
||||
th.Property("ks2_scaledscore_average", th.StringType),
|
||||
th.Property("sen_with_ehcp_pupil_percent", th.StringType),
|
||||
th.Property("sen_pupil_percent", th.StringType),
|
||||
th.Property("sen_no_ehcp_pupil_percent", th.StringType),
|
||||
th.Property("attainment8_diffn", th.StringType),
|
||||
th.Property("progress8_diffn", th.StringType),
|
||||
th.Property("progress8_banding", th.StringType),
|
||||
*[th.Property(field, th.StringType) for field in KS4_INFO_FIELDS],
|
||||
).to_dict()
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
"""KS4 school information for 2023/24 exists only in the 2023/24 release,
|
||||
whose file uses DfE's older column names. The stream declared only the newer
|
||||
ones, so every 2023/24 field loaded as null (audit C2). The headers below are
|
||||
DfE's, copied from 202324_information_about_schools_final.csv and
|
||||
202425_information_about_schools_final.csv.
|
||||
"""
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees'
|
||||
/ 'tap_uk_ees' / 'ks4_info.py')
|
||||
|
||||
HEADER_2023_24 = (
|
||||
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
|
||||
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
|
||||
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
|
||||
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
|
||||
't_allks_pupils', 't_allks_boys', 't_allks_girls', 't_pupils', 't_boys', 'pt_boys',
|
||||
't_girls', 'pt_girls', 'avg_ks2_scaledscore', 't_prior_lo', 'pt_prior_lo', 't_prior_av',
|
||||
'pt_prior_av', 't_prior_hi', 'pt_prior_hi', 't_disadvantaged', 'pt_disadvantaged',
|
||||
't_not_disadvantaged', 'pt_not_disadvantaged', 't_language_not_english',
|
||||
'pt_language_not_english', 't_language_english', 'pt_language_english',
|
||||
't_language_unknown', 'pt_language_unknown', 't_not_mobile', 'pt_not_mobile',
|
||||
't_sen_with_ehcp', 'pt_sen_with_ehcp', 't_sen', 'pt_sen', 't_sen_no_ehcp',
|
||||
'pt_sen_no_ehcp', 'diffn_att8', 'diffn_p8mea', 'p8_banding',
|
||||
)
|
||||
|
||||
HEADER_2024_25 = (
|
||||
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
|
||||
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
|
||||
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
|
||||
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
|
||||
'allks_pupil_count', 'allks_boys_count', 'allks_girls_count', 'endks4_pupil_count',
|
||||
'ks2_scaledscore_average', 'sen_with_ehcp_pupil_count', 'sen_with_ehcp_pupil_percent',
|
||||
'sen_pupil_count', 'sen_pupil_percent', 'sen_no_ehcp_pupil_count',
|
||||
'sen_no_ehcp_pupil_percent', 'attainment8_diffn', 'progress8_diffn', 'progress8_banding',
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ks4_info():
|
||||
spec = importlib.util.spec_from_file_location('ks4_info', MODULE)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def test_every_declared_field_is_in_the_2023_24_file_once_renamed(ks4_info):
|
||||
renamed = {ks4_info.KS4_INFO_RENAMES.get(c, c) for c in HEADER_2023_24}
|
||||
assert set(ks4_info.KS4_INFO_FIELDS) - renamed == set()
|
||||
|
||||
|
||||
def test_every_declared_field_is_in_the_2024_25_file(ks4_info):
|
||||
assert set(ks4_info.KS4_INFO_FIELDS) - set(HEADER_2024_25) == set()
|
||||
|
||||
|
||||
def test_the_renames_cannot_collide_with_either_file(ks4_info):
|
||||
# An old name in the current file, or a new name already in the old file,
|
||||
# would let a rename overwrite a real column.
|
||||
assert set(ks4_info.KS4_INFO_RENAMES) & set(HEADER_2024_25) == set()
|
||||
assert set(ks4_info.KS4_INFO_RENAMES.values()) & set(HEADER_2023_24) == set()
|
||||
|
||||
|
||||
def test_every_rename_names_a_declared_field(ks4_info):
|
||||
assert set(ks4_info.KS4_INFO_RENAMES.values()) <= set(ks4_info.KS4_INFO_FIELDS)
|
||||
Reference in new issue
Block a user