From 9cc87c41bb60b2cfc769991140d1d1c17cbf711a Mon Sep 17 00:00:00 2001 From: Tudor Date: Sat, 22 Aug 2026 00:14:31 +0100 Subject: [PATCH] fix(places): a phase page needs results, not merely publishable schools MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Asked where schools with no results should sit in an alphabetical list, and found that some pages were almost entirely made of them. The per-phase threshold counted schools that were publishable — a result OR an Ofsted grade — while a phase page exists for its results column. /schools/kent/primary published with none of its five rows carrying a result; Minehead had one of seven, Buntingford one of five. Forty-four phase pages were majority-blank. It is the same rule as "no page without a local average", which was written into the spec as a thin-page control and never extended per phase. The threshold now counts schools with a result for that phase. It gates whether the page exists; it does not filter rows — a page that publishes still lists every school of the phase, because someone looking up a school by name has to find it whether or not it published results. 126 of 1,012 variant pages stop publishing: 62 primary, 64 secondary. Every one of them was a table with too little in it to be worth a page. The ordering itself is unchanged: pure A-Z, blanks interleaved. A school sits where its name says it does, and at roughly a tenth of rows that reads fine. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- backend/places.py | 37 +++++++++++++++++++++++++++-- backend/tests/test_places.py | 46 ++++++++++++++++++++++++++++++++++++ 2 files changed, 81 insertions(+), 2 deletions(-) diff --git a/backend/places.py b/backend/places.py index 870f779..aae7fdd 100644 --- a/backend/places.py +++ b/backend/places.py @@ -59,18 +59,51 @@ def _publishable_urns(df) -> set[int]: return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int)) +# The measure a phase page is built around. A page with no results in this +# column has nothing a list of school names does not already give. +_PHASE_METRIC = { + "primary": "rwm_expected_pct", + "secondary": "attainment_8_score", +} + + def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]: - """URNs per phase. All-through schools count toward both, matching the - PHASE_GROUPS mapping the search filters already use.""" + """URNs per phase, counting only schools with a result for that phase. + + Not merely "publishable". A school with an Ofsted grade and no results is + worth a page of its own and belongs in the place list, but it cannot + populate a phase page's results column — and the threshold is there to ask + whether that column will have anything in it. + + Counting publishable schools instead let /schools/kent/primary publish + with none of its five rows carrying a result, and left 44 phase pages + majority-blank. It is the same rule as "no page without a local average", + which was never extended per phase. + + All-through schools count toward both phases, matching the PHASE_GROUPS + mapping the search filters already use. + """ from backend.app import PHASE_GROUPS if "phase" not in group.columns: return {} lowered = group["phase"].fillna("").str.lower() + out: dict[str, tuple[int, ...]] = {} for phase in ("primary", "secondary"): wanted = PHASE_GROUPS.get(phase, set()) subset = group[lowered.isin(wanted)] + + # The page lists every school of the phase; the threshold counts only + # those carrying a result, so a mostly-empty table never publishes. + metric = _PHASE_METRIC[phase] + with_result = ( + {int(u) for u in subset.loc[subset[metric].notna(), "urn"]} + if metric in subset.columns else set() + ) + if len(with_result & publishable) < MIN_SCHOOLS: + continue + urns = tuple(sorted({int(u) for u in subset["urn"]} & publishable)) if urns: out[phase] = urns diff --git a/backend/tests/test_places.py b/backend/tests/test_places.py index 6f3be5e..521c536 100644 --- a/backend/tests/test_places.py +++ b/backend/tests/test_places.py @@ -343,3 +343,49 @@ def test_the_merged_place_takes_its_most_common_spelling(): + _town(5, "NEWCASTLE-UNDER-LYME", "Staffordshire", start=400000)) reg = build_place_registry(_df(rows)) assert reg["town:newcastle-under-lyme"].name == "Newcastle-under-Lyme" + + +def test_a_phase_page_needs_results_not_merely_publishable_schools(): + """/schools/kent/primary published with none of its five rows scored. + + The threshold counted schools that were publishable — a result OR an + Ofsted grade — while the page exists for its results column. Forty-four + phase pages were majority-blank; one had no results at all. + """ + rows = _town(MIN_SCHOOLS, "Kent", "Kent") + for r in rows: + r["rwm_expected_pct"] = np.nan # Ofsted only, no results + reg = build_place_registry(_df(rows)) + + assert "town:kent" in reg # the place still publishes + assert not reg["town:kent"].publishes_phase("primary") + + +def test_a_phase_page_publishes_once_enough_schools_carry_a_result(): + rows = _town(MIN_SCHOOLS, "Beccles", "Suffolk") + reg = build_place_registry(_df(rows)) + assert reg["town:beccles"].publishes_phase("primary") + + +def test_a_publishing_phase_page_still_lists_its_unscored_schools(): + """The threshold gates whether the page exists; it does not filter rows. + + A parent looking up a school by name has to find it whether or not it + published results. + """ + scored = _town(MIN_SCHOOLS, "Beccles", "Suffolk", start=300000) + unscored = _town(2, "Beccles", "Suffolk", start=400000) + for r in unscored: + r["rwm_expected_pct"] = np.nan + reg = build_place_registry(_df(scored + unscored)) + + place = reg["town:beccles"] + assert place.publishes_phase("primary") + assert len(place.phase_urns["primary"]) == MIN_SCHOOLS + 2 + + +def test_the_secondary_threshold_counts_its_own_metric(): + # A town full of scored primaries must not thereby publish a secondary page. + rows = _town(MIN_SCHOOLS, "Brentwood", "Essex") + reg = build_place_registry(_df(rows)) + assert not reg["town:brentwood"].publishes_phase("secondary")