feat(ees): keep DfE's KS4 LA averages from the summary data set (H2)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
967b1f0eed
commit
dbb60acff3
3 files changed
+177
-49
No files matched your search
@@ -0,0 +1,56 @@
|
||||
"""DfE's KS4 "summary, all state-funded" data set (Key stage 4 performance).
|
||||
|
||||
One CSV holds England, regional and local-authority headline rows for every
|
||||
year since 2018/19. The England stream (ees_ks4_national) and the LA stream
|
||||
(ees_ks4_la) both read it. Its LA rows match DfE's published performance-table
|
||||
LA averages exactly (audit H2). Suppressed values ('z', 'x') become NULL in
|
||||
dbt; Progress 8 is 'z' in years with no KS2 baseline (2024/25): DfE policy,
|
||||
not missing data.
|
||||
|
||||
Free of the Singer SDK so CI's pytest can load it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
|
||||
KS4_SUMMARY_CSV_URL = (
|
||||
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
|
||||
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
|
||||
)
|
||||
|
||||
# CSV column → Singer field: the same 8 headline measures at every level.
|
||||
KS4_HEADLINE_COL_MAP = {
|
||||
"attainment8_average": "attainment_8_score",
|
||||
"progress8_average": "progress_8_score",
|
||||
"engmath_94_percent": "english_maths_standard_pass_pct",
|
||||
"engmath_95_percent": "english_maths_strong_pass_pct",
|
||||
"ebacc_entering_percent": "ebacc_entry_pct",
|
||||
"ebacc_94_percent": "ebacc_standard_pass_pct",
|
||||
"ebacc_95_percent": "ebacc_strong_pass_pct",
|
||||
"ebacc_aps_average": "ebacc_avg_score",
|
||||
}
|
||||
|
||||
|
||||
def headline_rows(df: pd.DataFrame, geographic_level: str) -> pd.DataFrame:
|
||||
"""All-pupil rows for all state-funded schools at one geographic level
|
||||
("National" or "Local authority"). Column names are lower-cased first;
|
||||
a filter column the file lacks is not applied."""
|
||||
df = df.copy()
|
||||
df.columns = [c.strip().lower() for c in df.columns]
|
||||
for col, want in (
|
||||
("geographic_level", geographic_level),
|
||||
("establishment_type_group", "All state-funded"),
|
||||
("breakdown_topic", "Total"),
|
||||
("breakdown", "Total"),
|
||||
):
|
||||
if col in df.columns:
|
||||
df = df[df[col].str.strip().str.lower() == want.lower()]
|
||||
return df
|
||||
|
||||
|
||||
def headline_record(row: pd.Series, keys: tuple[str, ...]) -> dict[str, str]:
|
||||
"""A Singer record: the identifying columns, then the headline measures."""
|
||||
record = {key: str(row.get(key, "")).strip() for key in keys}
|
||||
for csv_col, field in KS4_HEADLINE_COL_MAP.items():
|
||||
record[field] = str(row.get(csv_col, "")).strip()
|
||||
return record
|
||||
@@ -18,6 +18,12 @@ from singer_sdk import Stream, Tap
|
||||
from singer_sdk import typing as th
|
||||
|
||||
from tap_uk_ees.ks4_info import KS4_INFO_FIELDS, KS4_INFO_RENAMES
|
||||
from tap_uk_ees.ks4_summary import (
|
||||
KS4_HEADLINE_COL_MAP,
|
||||
KS4_SUMMARY_CSV_URL,
|
||||
headline_record,
|
||||
headline_rows,
|
||||
)
|
||||
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
|
||||
|
||||
CONTENT_API_BASE = (
|
||||
@@ -571,37 +577,23 @@ class EESKs2NationalStream(Stream):
|
||||
yield record
|
||||
|
||||
|
||||
# ── KS4 National Headlines (national level only — one row per year) ──────────
|
||||
# Dataset: "National characteristics summary data" (Key stage 4 performance).
|
||||
# Official England state-funded headline measures, 2018/19 → latest.
|
||||
# Suppressed values ('z', 'x') → NULL downstream. Progress 8 is legitimately
|
||||
# absent in years with no KS2 baseline (e.g. 2024/25) — that is DfE policy,
|
||||
# not missing data.
|
||||
# ── KS4 National and LA Headlines (one data set, two streams) ────────────────
|
||||
# DfE's "summary, all state-funded" data set: England, regional and LA rows,
|
||||
# 2018/19 → latest. URL, measures and filters: ks4_summary.py.
|
||||
|
||||
_KS4_NATIONAL_CSV_URL = (
|
||||
"https://explore-education-statistics.service.gov.uk/data-catalogue/"
|
||||
"data-set/1b649e16-01e8-435b-a814-56be2faf9054/csv"
|
||||
)
|
||||
def _read_ks4_summary(logger):
|
||||
"""Download DfE's KS4 summary data set."""
|
||||
import pandas as pd
|
||||
|
||||
_KS4_NATIONAL_COL_MAP = {
|
||||
"attainment8_average": "attainment_8_score",
|
||||
"progress8_average": "progress_8_score",
|
||||
"engmath_94_percent": "english_maths_standard_pass_pct",
|
||||
"engmath_95_percent": "english_maths_strong_pass_pct",
|
||||
"ebacc_entering_percent": "ebacc_entry_pct",
|
||||
"ebacc_94_percent": "ebacc_standard_pass_pct",
|
||||
"ebacc_95_percent": "ebacc_strong_pass_pct",
|
||||
"ebacc_aps_average": "ebacc_avg_score",
|
||||
}
|
||||
logger.info("Downloading KS4 summary data set: %s", KS4_SUMMARY_CSV_URL)
|
||||
resp = requests.get(KS4_SUMMARY_CSV_URL, timeout=60)
|
||||
resp.raise_for_status()
|
||||
return pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
|
||||
|
||||
|
||||
class EESKs4NationalStream(Stream):
|
||||
"""National KS4 headline averages — one row per academic year.
|
||||
|
||||
Filters to geographic_level == 'National', establishment_type_group ==
|
||||
'All state-funded', breakdown_topic == 'Total', breakdown == 'Total'
|
||||
so only the England-wide all-pupils row per year is emitted.
|
||||
"""
|
||||
"""National KS4 headline averages — one row per academic year (England,
|
||||
all state-funded schools, all pupils)."""
|
||||
|
||||
name = "ees_ks4_national"
|
||||
primary_keys = ["time_period"]
|
||||
@@ -609,34 +601,41 @@ class EESKs4NationalStream(Stream):
|
||||
|
||||
schema = th.PropertiesList(
|
||||
th.Property("time_period", th.StringType, required=True),
|
||||
*[th.Property(out, th.StringType) for out in _KS4_NATIONAL_COL_MAP.values()],
|
||||
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
|
||||
).to_dict()
|
||||
|
||||
def get_records(self, context):
|
||||
import pandas as pd
|
||||
|
||||
self.logger.info("Downloading KS4 national headlines: %s", _KS4_NATIONAL_CSV_URL)
|
||||
resp = requests.get(_KS4_NATIONAL_CSV_URL, timeout=60)
|
||||
resp.raise_for_status()
|
||||
|
||||
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
|
||||
df.columns = [c.strip().lower() for c in df.columns]
|
||||
|
||||
for col, want in [
|
||||
("geographic_level", "national"),
|
||||
("establishment_type_group", "all state-funded"),
|
||||
("breakdown_topic", "total"),
|
||||
("breakdown", "total"),
|
||||
]:
|
||||
if col in df.columns:
|
||||
df = df[df[col].str.strip().str.lower() == want]
|
||||
|
||||
df = headline_rows(_read_ks4_summary(self.logger), "National")
|
||||
self.logger.info("Emitting %d national KS4 rows", len(df))
|
||||
for _, row in df.iterrows():
|
||||
record = {"time_period": row.get("time_period", "").strip()}
|
||||
for csv_col, field in _KS4_NATIONAL_COL_MAP.items():
|
||||
record[field] = row.get(csv_col, "").strip()
|
||||
yield record
|
||||
yield headline_record(row, ("time_period",))
|
||||
|
||||
|
||||
class EESKs4LaStream(Stream):
|
||||
"""DfE's KS4 local-authority averages — one row per academic year and LA
|
||||
(all state-funded schools, all pupils), from the same data set as
|
||||
ees_ks4_national. They match DfE's published performance-table LA averages
|
||||
and replace a mean the API took over every school, independent and special
|
||||
included (audit H2). old_la_code is the GIAS LA code: schools join on it,
|
||||
not on the name."""
|
||||
|
||||
name = "ees_ks4_la"
|
||||
primary_keys = ["time_period", "old_la_code"]
|
||||
replication_key = None
|
||||
|
||||
schema = th.PropertiesList(
|
||||
th.Property("time_period", th.StringType, required=True),
|
||||
th.Property("old_la_code", th.StringType, required=True),
|
||||
th.Property("new_la_code", th.StringType),
|
||||
th.Property("la_name", th.StringType),
|
||||
*[th.Property(out, th.StringType) for out in KS4_HEADLINE_COL_MAP.values()],
|
||||
).to_dict()
|
||||
|
||||
def get_records(self, context):
|
||||
df = headline_rows(_read_ks4_summary(self.logger), "Local authority")
|
||||
self.logger.info("Emitting %d LA KS4 rows", len(df))
|
||||
for _, row in df.iterrows():
|
||||
yield headline_record(row, ("time_period", "old_la_code", "new_la_code", "la_name"))
|
||||
|
||||
|
||||
# ── Legacy KS2 (pre-COVID wide format from DfE performance tables) ────────────
|
||||
@@ -979,6 +978,7 @@ class TapUKEES(Tap):
|
||||
LegacyKS4Stream(self),
|
||||
EESKs2NationalStream(self),
|
||||
EESKs4NationalStream(self),
|
||||
EESKs4LaStream(self),
|
||||
]
|
||||
|
||||
|
||||
|
||||
Reference in new issue
Block a user