fix(ees): read 2023/24 KS4 school information under DfE's older names (C2)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 10:45:57 +01:00
1 parent 4db1131d0f
commit 967b1f0eed
3 files changed
+120 -19

No files matched your search

@@ -0,0 +1,47 @@
"""KS4 school information: the fields the stream declares, and the older
names DfE used for them.
2023/24 information exists only in the 2023/24 release, whose
202324_information_about_schools_final.csv (re-issued 10 March 2026) uses the
older names. Without the renames every 2023/24 field loaded as null (audit C2).
Newer files contain none of the older names, so the renames leave them alone.
Free of the Singer SDK so CI's pytest can load it.
"""
# Declared Singer fields besides the required time_period and school_urn.
KS4_INFO_FIELDS = (
"school_laestab",
"school_name",
"establishment_type_group",
"reldenom",
"admpol_pt",
"egender",
"agerange",
"allks_pupil_count",
"allks_boys_count",
"allks_girls_count",
"endks4_pupil_count",
"ks2_scaledscore_average",
"sen_with_ehcp_pupil_percent",
"sen_pupil_percent",
"sen_no_ehcp_pupil_percent",
"attainment8_diffn",
"progress8_diffn",
"progress8_banding",
)
# 2023/24 column name → declared field.
KS4_INFO_RENAMES = {
"t_allks_pupils": "allks_pupil_count",
"t_allks_boys": "allks_boys_count",
"t_allks_girls": "allks_girls_count",
"t_pupils": "endks4_pupil_count",
"avg_ks2_scaledscore": "ks2_scaledscore_average",
"pt_sen_with_ehcp": "sen_with_ehcp_pupil_percent",
"pt_sen": "sen_pupil_percent",
"pt_sen_no_ehcp": "sen_no_ehcp_pupil_percent",
"diffn_att8": "attainment8_diffn",
"diffn_p8mea": "progress8_diffn",
"p8_banding": "progress8_banding",
}
@@ -17,6 +17,7 @@ import requests
from singer_sdk import Stream, Tap
from singer_sdk import typing as th
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
CONTENT_API_BASE = (
@@ -330,34 +331,20 @@ class EESKS4PerformanceStream(EESDatasetStream):
# ── KS4 Information (wide format: one row per school, context/demographics) ──
# File: 202425_information_about_schools_provisional.csv (38 cols)
# Files: 202425_information_about_schools_final.csv (38 cols, current names);
# 202324_information_about_schools_final.csv (60 cols, older names — the only
# source of 2023/24 information). Field list and renames: ks4_info.py.
class EESKS4InfoStream(EESDatasetStream):
name = "ees_ks4_info"
primary_keys = ["school_urn", "time_period"]
_publication_slug = "key-stage-4-performance"
_target_filename = "information_about_schools"
_column_renames = KS4_INFO_RENAMES
schema = th.PropertiesList(
th.Property("time_period", th.StringType, required=True),
th.Property("school_urn", th.StringType, required=True),
th.Property("school_laestab", th.StringType),
th.Property("school_name", th.StringType),
th.Property("establishment_type_group", th.StringType),
th.Property("reldenom", th.StringType),
th.Property("admpol_pt", th.StringType),
th.Property("egender", th.StringType),
th.Property("agerange", th.StringType),
th.Property("allks_pupil_count", th.StringType),
th.Property("allks_boys_count", th.StringType),
th.Property("allks_girls_count", th.StringType),
th.Property("endks4_pupil_count", th.StringType),
th.Property("ks2_scaledscore_average", th.StringType),
th.Property("sen_with_ehcp_pupil_percent", th.StringType),
th.Property("sen_pupil_percent", th.StringType),
th.Property("sen_no_ehcp_pupil_percent", th.StringType),
th.Property("attainment8_diffn", th.StringType),
th.Property("progress8_diffn", th.StringType),
th.Property("progress8_banding", th.StringType),
*[th.Property(field, th.StringType) for field in KS4_INFO_FIELDS],
).to_dict()