"""KS4 school information for 2023/24 exists only in the 2023/24 release, whose file uses DfE's older column names. The stream declared only the newer ones, so every 2023/24 field loaded as null (audit C2). The headers below are DfE's, copied from 202324_information_about_schools_final.csv and 202425_information_about_schools_final.csv. """ import importlib.util from pathlib import Path import pytest MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees' / 'tap_uk_ees' / 'ks4_info.py') HEADER_2023_24 = ( 'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name', 'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name', 'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name', 'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange', 't_allks_pupils', 't_allks_boys', 't_allks_girls', 't_pupils', 't_boys', 'pt_boys', 't_girls', 'pt_girls', 'avg_ks2_scaledscore', 't_prior_lo', 'pt_prior_lo', 't_prior_av', 'pt_prior_av', 't_prior_hi', 'pt_prior_hi', 't_disadvantaged', 'pt_disadvantaged', 't_not_disadvantaged', 'pt_not_disadvantaged', 't_language_not_english', 'pt_language_not_english', 't_language_english', 'pt_language_english', 't_language_unknown', 'pt_language_unknown', 't_not_mobile', 'pt_not_mobile', 't_sen_with_ehcp', 'pt_sen_with_ehcp', 't_sen', 'pt_sen', 't_sen_no_ehcp', 'pt_sen_no_ehcp', 'diffn_att8', 'diffn_p8mea', 'p8_banding', ) HEADER_2024_25 = ( 'time_period', 'time_identifier', 'geographic_level', 'country_code', 'country_name', 'school_laestab', 'school_urn', 'school_name', 'old_la_code', 'new_la_code', 'la_name', 'version', 'establishment_type_group', 'full_address', 'telnum', 'pcon_code', 'pcon_name', 'contflag', 'iclose', 'reldenom', 'admpol_pt', 'egender', 'feeder', 'agerange', 'allks_pupil_count', 'allks_boys_count', 'allks_girls_count', 'endks4_pupil_count', 'ks2_scaledscore_average', 'sen_with_ehcp_pupil_count', 'sen_with_ehcp_pupil_percent', 'sen_pupil_count', 'sen_pupil_percent', 'sen_no_ehcp_pupil_count', 'sen_no_ehcp_pupil_percent', 'attainment8_diffn', 'progress8_diffn', 'progress8_banding', ) @pytest.fixture def ks4_info(): spec = importlib.util.spec_from_file_location('ks4_info', MODULE) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) return module def test_every_declared_field_is_in_the_2023_24_file_once_renamed(ks4_info): renamed = {ks4_info.KS4_INFO_RENAMES.get(c, c) for c in HEADER_2023_24} assert set(ks4_info.KS4_INFO_FIELDS) - renamed == set() def test_every_declared_field_is_in_the_2024_25_file(ks4_info): assert set(ks4_info.KS4_INFO_FIELDS) - set(HEADER_2024_25) == set() def test_the_renames_cannot_collide_with_either_file(ks4_info): # An old name in the current file, or a new name already in the old file, # would let a rename overwrite a real column. assert set(ks4_info.KS4_INFO_RENAMES) & set(HEADER_2024_25) == set() assert set(ks4_info.KS4_INFO_RENAMES.values()) & set(HEADER_2023_24) == set() def test_every_rename_names_a_declared_field(ks4_info): assert set(ks4_info.KS4_INFO_RENAMES.values()) <= set(ks4_info.KS4_INFO_FIELDS)