diff --git a/backend/app.py b/backend/app.py index a513aea..7836ca7 100644 --- a/backend/app.py +++ b/backend/app.py @@ -919,6 +919,7 @@ async def get_school_details(request: Request, urn: int): "phonics": supplementary.get("phonics"), "deprivation": supplementary.get("deprivation"), "finance": supplementary.get("finance"), + "destinations": supplementary.get("destinations"), } diff --git a/backend/data_loader.py b/backend/data_loader.py index 8aff4df..a176fe4 100644 --- a/backend/data_loader.py +++ b/backend/data_loader.py @@ -20,6 +20,7 @@ from .models import ( DimSchool, DimLocation, KS2Performance, FactOfstedInspection, FactAdmissions, FactAdmissionDistance, FactDeprivation, FactFinance, FactPupilCharacteristics, + FactKs4Destinations, FactKs5Destinations, ) from .ofsted_codes import ofsted_page_url, report_card_labels from .schemas import SCHOOL_TYPE_MAP @@ -816,6 +817,73 @@ def _finance_dict(f) -> dict: } +# Destination measures that are totals DfE published itself, rather than one of +# the categories that partition the cohort. +_AGGREGATE_MEASURES = {"agg_sustained_education", "agg_sustained_all"} + + +def _format_cohort_year(year) -> str | None: + """202223 -> '2022/23'. + + The section has to date its own cohort. Destination measures run about two + GCSE years behind the results shown above them on the same page, so an + undated figure reads as stale data rather than as a different question. + """ + if not year: + return None + text = str(year) + if len(text) == 6: + return f"{text[:4]}/{text[4:6]}" + if len(text) == 8: + return f"{text[:4]}/{text[6:8]}" + return text + + +def _destinations_block(rows: list) -> dict | None: + """Shape destination rows for one phase into the API's block. + + Carries `status` through untouched and emits no computed totals. The only + aggregates present are ones DfE published itself; whether showing one is + safe depends on how many of its components are suppressed, which the + frontend decides (lib/destinations.ts, rule R2). + + Deliberately does NOT compute a residual, a "remaining pupils" figure, or + any total that would close a gap left by a suppressed category — the + categories sum to the cohort, so such a figure names the withheld cell. + """ + if not rows: + return None + + years = [r["year"] for r in rows if r.get("year") is not None] + if not years: + return None + latest_year = max(years) + rows = [r for r in rows if r.get("year") == latest_year] + + groups: dict = {} + for row in rows: + group = groups.setdefault( + row["pupil_group"], + {"cohort": row.get("cohort_pupils"), "categories": [], "aggregates": {}}, + ) + measure = row["destination_measure"] + cell = { + "category": measure, + "pupils": row.get("pupils"), + "percentage": row.get("percentage"), + "status": row.get("status"), + } + if measure in _AGGREGATE_MEASURES: + group["aggregates"][measure[len("agg_"):]] = cell + else: + group["categories"].append(cell) + + if not groups: + return None + + return {"cohort_year": _format_cohort_year(latest_year), "groups": groups} + + def _empty_supplementary() -> dict: return { "ofsted": None, @@ -827,6 +895,7 @@ def _empty_supplementary() -> dict: "phonics": None, "deprivation": None, "finance": None, + "destinations": None, } @@ -954,6 +1023,38 @@ def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict: result[f.urn]["finance"] = _finance_dict(f) _safe(_finance) + # Destinations — KS4 and 16-18. Both marts are long-format, so every row + # for a URN is collected and _destinations_block picks the latest year and + # shapes the pupil groups. A phase with no rows serialises as null rather + # than an empty shell, so the frontend renders nothing rather than an empty + # section. + def _destinations(): + from collections import defaultdict + + def _collect(model): + per_urn = defaultdict(list) + for r in db.query(model).filter(model.urn.in_(urns)).all(): + per_urn[r.urn].append({ + "year": r.year, + "pupil_group": r.pupil_group, + "destination_measure": r.destination_measure, + "cohort_pupils": r.cohort_pupils, + "pupils": r.pupils, + "percentage": r.percentage, + "status": r.status, + }) + return per_urn + + ks4_rows = _collect(FactKs4Destinations) + ks5_rows = _collect(FactKs5Destinations) + for urn in urns: + ks4 = _destinations_block(ks4_rows.get(urn, [])) + ks5 = _destinations_block(ks5_rows.get(urn, [])) + result[urn]["destinations"] = ( + {"ks4": ks4, "ks5": ks5} if (ks4 or ks5) else None + ) + _safe(_destinations) + return result diff --git a/backend/models.py b/backend/models.py index b68b342..d090d8f 100644 --- a/backend/models.py +++ b/backend/models.py @@ -321,3 +321,48 @@ class Ks2NationalAverage(Base): gps_high_pct = Column(Float) gps_avg_score = Column(Float) science_expected_pct = Column(Float) + + +class FactKs4Destinations(Base): + """KS4 leavers destinations — one row per URN, year, pupil group, measure. + + Long format rather than wide because pupil_group is a real third dimension. + `status` is load-bearing: 'suppressed' means DfE withheld a figure it + considered disclosive and the page must print "withheld"; 'not_applicable' + means the measure does not apply and the page must print nothing. `pupils` + is null for both, so collapsing status to a null check loses the + difference — and the categories sum to the cohort, so a consumer that + treats a withheld cell as zero republishes what DfE hid. + """ + __tablename__ = "fact_ks4_destinations" + __table_args__ = ( + Index("ix_ks4_dest_urn_year", "urn", "year"), + MARTS, + ) + + urn = Column(Integer, primary_key=True) + year = Column(Integer, primary_key=True) + pupil_group = Column(String(20), primary_key=True) + destination_measure = Column(String(40), primary_key=True) + cohort_pupils = Column(Integer) + pupils = Column(Integer) + percentage = Column(Float) + status = Column(String(20)) + + +class FactKs5Destinations(Base): + """16-18 study leavers destinations — same grain as FactKs4Destinations.""" + __tablename__ = "fact_ks5_destinations" + __table_args__ = ( + Index("ix_ks5_dest_urn_year", "urn", "year"), + MARTS, + ) + + urn = Column(Integer, primary_key=True) + year = Column(Integer, primary_key=True) + pupil_group = Column(String(20), primary_key=True) + destination_measure = Column(String(40), primary_key=True) + cohort_pupils = Column(Integer) + pupils = Column(Integer) + percentage = Column(Float) + status = Column(String(20)) diff --git a/backend/tests/test_destinations_api.py b/backend/tests/test_destinations_api.py new file mode 100644 index 0000000..a2a2cd9 --- /dev/null +++ b/backend/tests/test_destinations_api.py @@ -0,0 +1,112 @@ +"""The destinations serialiser's contract: it carries suppression through, and +never emits a total that closes a gap left by a suppressed category. + +The destination categories sum to the cohort, so an aggregate that happens to +equal the residual names the withheld figure exactly. See +docs/superpowers/specs/2026-08-28-destination-measures-design.md. +""" + +from backend.data_loader import _destinations_block, _format_cohort_year + + +def _row(group, measure, pupils, status, cohort=180, percentage=None, year=202223): + return { + "pupil_group": group, + "destination_measure": measure, + "pupils": pupils, + "percentage": percentage, + "status": status, + "cohort_pupils": cohort, + "year": year, + } + + +def test_suppressed_category_serialises_as_suppressed_with_null_pupils(): + rows = [ + _row("all", "school_sixth_form", 75, "published", percentage=41.7), + _row("all", "sixth_form_college", None, "suppressed"), + ] + block = _destinations_block(rows) + cats = {c["category"]: c for c in block["groups"]["all"]["categories"]} + assert cats["sixth_form_college"]["status"] == "suppressed" + assert cats["sixth_form_college"]["pupils"] is None + assert cats["sixth_form_college"]["percentage"] is None + + +def test_published_category_keeps_its_figures(): + block = _destinations_block([ + _row("all", "school_sixth_form", 75, "published", percentage=41.7), + ]) + cat = block["groups"]["all"]["categories"][0] + assert cat["pupils"] == 75 + assert cat["percentage"] == 41.7 + assert cat["status"] == "published" + + +def test_no_closing_total_is_emitted_for_a_partially_suppressed_group(): + rows = [ + _row("all", "school_sixth_form", 75, "published", percentage=41.7), + _row("all", "sixth_form_college", None, "suppressed"), + _row("all", "further_education", 61, "published", percentage=33.9), + _row("all", "apprenticeship", 8, "published", percentage=4.4), + _row("all", "employment", 6, "published", percentage=3.3), + _row("all", "not_sustained", 5, "published", percentage=2.8), + _row("all", "not_captured", 4, "published", percentage=2.2), + ] + block = _destinations_block(rows) + group = block["groups"]["all"] + published = sum(c["pupils"] for c in group["categories"] if c["pupils"] is not None) + residual = group["cohort"] - published + for value in group["aggregates"].values(): + if value is None or value.get("pupils") is None: + continue + assert value["pupils"] != residual, ( + "an aggregate equal to the residual identifies the suppressed cell" + ) + + +def test_aggregates_are_separated_from_categories(): + rows = [ + _row("all", "school_sixth_form", 75, "published", percentage=41.7), + _row("all", "agg_sustained_all", 171, "published", percentage=95.0), + ] + block = _destinations_block(rows) + group = block["groups"]["all"] + assert [c["category"] for c in group["categories"]] == ["school_sixth_form"] + assert group["aggregates"]["sustained_all"]["pupils"] == 171 + + +def test_only_the_latest_year_is_served(): + rows = [ + _row("all", "school_sixth_form", 60, "published", year=202122), + _row("all", "school_sixth_form", 75, "published", year=202223), + ] + block = _destinations_block(rows) + assert block["cohort_year"] == "2022/23" + assert len(block["groups"]["all"]["categories"]) == 1 + assert block["groups"]["all"]["categories"][0]["pupils"] == 75 + + +def test_all_three_pupil_groups_are_carried(): + rows = [ + _row("all", "school_sixth_form", 75, "published"), + _row("disadvantaged", "school_sixth_form", 17, "published", cohort=62), + _row("other", "school_sixth_form", 58, "published", cohort=118), + ] + block = _destinations_block(rows) + assert set(block["groups"]) == {"all", "disadvantaged", "other"} + assert block["groups"]["disadvantaged"]["cohort"] == 62 + + +def test_cohort_year_is_reported_so_the_page_can_date_itself(): + block = _destinations_block([_row("all", "school_sixth_form", 75, "published")]) + assert block["cohort_year"] == "2022/23" + + +def test_format_cohort_year_handles_the_six_digit_form(): + assert _format_cohort_year(202223) == "2022/23" + assert _format_cohort_year(None) is None + + +def test_empty_rows_yield_none_not_an_empty_shell(): + assert _destinations_block([]) is None diff --git a/backend/tests/test_supplementary_batch.py b/backend/tests/test_supplementary_batch.py index 5fb6cdd..7813e7f 100644 --- a/backend/tests/test_supplementary_batch.py +++ b/backend/tests/test_supplementary_batch.py @@ -132,16 +132,23 @@ def test_one_query_per_table_and_latest_row_per_urn(): "FactPupilCharacteristics": [], "FactDeprivation": [], "FactFinance": [], + "FactKs4Destinations": [], + "FactKs5Destinations": [], } session = _FakeSession(rows) out = get_supplementary_data_batch(session, [1, 2]) - # Exactly one query per table — six total, regardless of two URNs. + # Exactly one query per table — eight total, regardless of two URNs. assert sorted(session.queries) == [ "FactAdmissionDistance", "FactAdmissions", "FactDeprivation", - "FactFinance", "FactOfstedInspection", "FactPupilCharacteristics", + "FactFinance", "FactKs4Destinations", "FactKs5Destinations", + "FactOfstedInspection", "FactPupilCharacteristics", ] + # A school with no destination rows gets null, not an empty shell — the + # frontend renders the section from the block's presence. + assert out[1]["destinations"] is None + # Latest Ofsted kept per URN assert out[1]["ofsted"]["overall_effectiveness"] == 2 assert out[2]["ofsted"]["overall_effectiveness"] == 1