fix(ees): read 2023/24 KS4 school information under DfE's older names (C2)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
4db1131d0f
commit
967b1f0eed
3 files changed
+120
-19
No files matched your search
@@ -0,0 +1,67 @@
|
||||
"""KS4 school information for 2023/24 exists only in the 2023/24 release,
|
||||
whose file uses DfE's older column names. The stream declared only the newer
|
||||
ones, so every 2023/24 field loaded as null (audit C2). The headers below are
|
||||
DfE's, copied from 202324_information_about_schools_final.csv and
|
||||
202425_information_about_schools_final.csv.
|
||||
"""
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees'
|
||||
/ 'tap_uk_ees' / 'ks4_info.py')
|
||||
|
||||
HEADER_2023_24 = (
|
||||
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
|
||||
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
|
||||
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
|
||||
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
|
||||
't_allks_pupils', 't_allks_boys', 't_allks_girls', 't_pupils', 't_boys', 'pt_boys',
|
||||
't_girls', 'pt_girls', 'avg_ks2_scaledscore', 't_prior_lo', 'pt_prior_lo', 't_prior_av',
|
||||
'pt_prior_av', 't_prior_hi', 'pt_prior_hi', 't_disadvantaged', 'pt_disadvantaged',
|
||||
't_not_disadvantaged', 'pt_not_disadvantaged', 't_language_not_english',
|
||||
'pt_language_not_english', 't_language_english', 'pt_language_english',
|
||||
't_language_unknown', 'pt_language_unknown', 't_not_mobile', 'pt_not_mobile',
|
||||
't_sen_with_ehcp', 'pt_sen_with_ehcp', 't_sen', 'pt_sen', 't_sen_no_ehcp',
|
||||
'pt_sen_no_ehcp', 'diffn_att8', 'diffn_p8mea', 'p8_banding',
|
||||
)
|
||||
|
||||
HEADER_2024_25 = (
|
||||
'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name',
|
||||
'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name',
|
||||
'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name',
|
||||
'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange',
|
||||
'allks_pupil_count', 'allks_boys_count', 'allks_girls_count', 'endks4_pupil_count',
|
||||
'ks2_scaledscore_average', 'sen_with_ehcp_pupil_count', 'sen_with_ehcp_pupil_percent',
|
||||
'sen_pupil_count', 'sen_pupil_percent', 'sen_no_ehcp_pupil_count',
|
||||
'sen_no_ehcp_pupil_percent', 'attainment8_diffn', 'progress8_diffn', 'progress8_banding',
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ks4_info():
|
||||
spec = importlib.util.spec_from_file_location('ks4_info', MODULE)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def test_every_declared_field_is_in_the_2023_24_file_once_renamed(ks4_info):
|
||||
renamed = {ks4_info.KS4_INFO_RENAMES.get(c, c) for c in HEADER_2023_24}
|
||||
assert set(ks4_info.KS4_INFO_FIELDS) - renamed == set()
|
||||
|
||||
|
||||
def test_every_declared_field_is_in_the_2024_25_file(ks4_info):
|
||||
assert set(ks4_info.KS4_INFO_FIELDS) - set(HEADER_2024_25) == set()
|
||||
|
||||
|
||||
def test_the_renames_cannot_collide_with_either_file(ks4_info):
|
||||
# An old name in the current file, or a new name already in the old file,
|
||||
# would let a rename overwrite a real column.
|
||||
assert set(ks4_info.KS4_INFO_RENAMES) & set(HEADER_2024_25) == set()
|
||||
assert set(ks4_info.KS4_INFO_RENAMES.values()) & set(HEADER_2023_24) == set()
|
||||
|
||||
|
||||
def test_every_rename_names_a_declared_field(ks4_info):
|
||||
assert set(ks4_info.KS4_INFO_RENAMES.values()) <= set(ks4_info.KS4_INFO_FIELDS)
|
||||
Reference in new issue
Block a user