"""DfE's KS4 "summary, all state-funded" data set holds England, regional and LA rows. The England stream kept only the England row; its LA rows match DfE's published LA averages exactly, where the API's own mean was 7 points low (audit H2). Values below are DfE's for Kensington and Chelsea (207) and Wandsworth (212). """ import importlib.util import io from pathlib import Path import pandas as pd import pytest MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-ees' / 'tap_uk_ees' / 'ks4_summary.py') CSV = """time_period,geographic_level,old_la_code,new_la_code,la_name,establishment_type_group,breakdown_topic,breakdown,attainment8_average,progress8_average,engmath_94_percent,engmath_95_percent,ebacc_entering_percent,ebacc_94_percent,ebacc_95_percent,ebacc_aps_average 202425,National,,,,All state-funded,Total,Total,46.1,z,64.5,45.7,40.5,26.9,17.7,4.1 202425,National,,,,All state-funded,Sex,Boys,44.1,z,62.0,43.0,38.0,24.0,16.0,3.9 202425,Regional,,,,All state-funded,Total,Total,47.2,z,66.0,47.0,41.0,28.0,18.0,4.2 202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,z,77,61.4,45.6,32,26.6,4.89 202425,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Sex,Girls,57.0,z,80,64.0,48.0,35,28.0,5.1 202324,Local authority,207,E09000020,Kensington and Chelsea,All state-funded,Total,Total,54.5,0.29,76,60.0,44.0,31,25.0,4.8 202425,Local authority,212,E09000032,Wandsworth,All state-funded,Total,Total,51.8,z,72,55.0,50.0,33,24.0,4.6 202425,Local authority,212,E09000032,Wandsworth,Academies and free schools,Total,Total,52.0,z,73,56.0,51.0,34,25.0,4.7 """ LA_KEYS = ('time_period', 'old_la_code', 'new_la_code', 'la_name') def _df(): return pd.read_csv(io.StringIO(CSV), dtype=str, keep_default_na=False) @pytest.fixture def summary(): spec = importlib.util.spec_from_file_location('ks4_summary', MODULE) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) return module def test_la_rows_are_one_per_la_and_year(summary): rows = summary.headline_rows(_df(), 'Local authority') assert sorted(zip(rows['time_period'], rows['old_la_code'])) == [ ('202324', '207'), ('202425', '207'), ('202425', '212')] def test_an_la_record_carries_codes_name_and_the_headline_measures(summary): rows = summary.headline_rows(_df(), 'Local authority') row = rows[(rows['time_period'] == '202425') & (rows['old_la_code'] == '207')].iloc[0] assert summary.headline_record(row, LA_KEYS) == { 'time_period': '202425', 'old_la_code': '207', 'new_la_code': 'E09000020', 'la_name': 'Kensington and Chelsea', 'attainment_8_score': '54.5', 'progress_8_score': 'z', 'english_maths_standard_pass_pct': '77', 'english_maths_strong_pass_pct': '61.4', 'ebacc_entry_pct': '45.6', 'ebacc_standard_pass_pct': '32', 'ebacc_strong_pass_pct': '26.6', 'ebacc_avg_score': '4.89', } def test_the_england_rows_are_one_per_year(summary): rows = summary.headline_rows(_df(), 'National') assert list(rows['time_period']) == ['202425'] assert summary.headline_record(rows.iloc[0], ('time_period',))['attainment_8_score'] == '46.1' def test_column_names_and_labels_match_whatever_their_case(summary): df = _df() df.columns = [c.upper() for c in df.columns] df['GEOGRAPHIC_LEVEL'] = df['GEOGRAPHIC_LEVEL'].str.upper() assert len(summary.headline_rows(df, 'Local authority')) == 3