fix(seo): a school is publishable on any year's results, not the latest
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m2s
PR Checks / Backend Smoke (pull_request) Successful in 8s
PR Checks / Build Backend (no push) (pull_request) Successful in 16s
PR Checks / Build Frontend (no push) (pull_request) Successful in 44s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 53s

_school_sitemap_rows tested only the latest year's row, which quietly dropped
every school with results in its history but a null row for the most recent
year — a school that stopped reporting, or whose figures were suppressed for
small-cohort disclosure.

The Mallard Academy (150367) is the case that caught it: real KS2 results for
2015-16 through 2018-19, then null rows from 2022-23 on. Its detail page shows
all four years; the sitemap omitted it. Sampling 40 of the 2,206 excluded
schools found 4 like this, so roughly 220 real pages were being withheld.

Publishable is now a property of the school, computed across every row, while
lastmod still comes from the latest row so the most recent Ofsted date wins.
The field list is a module constant shared with _has_publishable_data so the
two checks cannot drift.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-21 00:04:32 +01:00
1 parent bb81337aba
commit 07c97a46c5
2 files changed
+64 -2

No files matched your search

+23 -2
View File
@@ -79,6 +79,12 @@ def _school_url(urn: int, school_name: str) -> str:
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions") STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
# A page has something a search result could state if any of these is present
# in any year. Shared by _has_publishable_data and the per-school check in
# _school_sitemap_rows so the two can never drift.
_PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade")
def _has_publishable_data(row) -> bool: def _has_publishable_data(row) -> bool:
"""True when a school page has something a search result could state. """True when a school page has something a search result could state.
@@ -87,7 +93,7 @@ def _has_publishable_data(row) -> bool:
signal down, so it stays out of the sitemap. The page itself still resolves signal down, so it stays out of the sitemap. The page itself still resolves
for anyone who has the URL. for anyone who has the URL.
""" """
for field in ("rwm_expected_pct", "attainment_8_score", "ofsted_grade"): for field in _PUBLISHABLE_FIELDS:
value = row.get(field) value = row.get(field)
if value is not None and not pd.isna(value): if value is not None and not pd.isna(value):
return True return True
@@ -115,6 +121,21 @@ def _school_sitemap_rows(df) -> list[str]:
rows: list[str] = [] rows: list[str] = []
seen: set[int] = set() seen: set[int] = set()
# Publishable is a property of the SCHOOL, not of its latest row.
#
# The first cut tested the latest year's row alone, which quietly dropped
# every school that has results in its history but a null row for the most
# recent year — a school that stopped reporting, or whose figures were
# suppressed for small-cohort disclosure. The Mallard Academy (150367) is
# the case that caught it: real KS2 results for 2015-16 through 2018-19,
# then null rows for 2022-23 onward. Its page shows all four years; the
# sitemap omitted it. Roughly 220 schools were affected.
publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
publishable: set[int] = (
set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int))
if publishable_cols else set()
)
# Latest row per URN first, so a school's most recent Ofsted date wins. # Latest row per URN first, so a school's most recent Ofsted date wins.
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
@@ -123,7 +144,7 @@ def _school_sitemap_rows(df) -> list[str]:
if urn in seen: if urn in seen:
continue continue
seen.add(urn) seen.add(urn)
if not _has_publishable_data(row): if urn not in publishable:
continue continue
lastmod = None lastmod = None
+41
View File
@@ -176,3 +176,44 @@ def test_children_are_chunked_under_the_limit(monkeypatch):
def test_build_sitemap_still_returns_the_index(sitemap): def test_build_sitemap_still_returns_the_index(sitemap):
# lifespan and the admin endpoint call build_sitemap(); keep it working. # lifespan and the admin endpoint call build_sitemap(); keep it working.
assert "<sitemapindex" in sitemap assert "<sitemapindex" in sitemap
def test_school_with_results_in_an_earlier_year_is_still_listed(monkeypatch):
"""Regression: The Mallard Academy (150367).
Real KS2 results 2015-16 to 2018-19, then null rows from 2022-23 onward
because the school stopped reporting. The first cut tested the latest
year's row alone and dropped it, along with ~220 others, even though its
detail page shows all four years of results.
"""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "urn": 150367,
"school_name": "Mallard Academy"}
df = _pd.DataFrame([
{**base, "year": 201819, "rwm_expected_pct": 67.0},
{**base, "year": 202324, "rwm_expected_pct": np.nan},
{**base, "year": 202425, "rwm_expected_pct": np.nan},
])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "/school/150367-mallard-academy" in xml
def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
"""The fix must not turn into "list everything"."""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "rwm_expected_pct": np.nan,
"urn": 100002, "school_name": "Ghost Primary"}
df = _pd.DataFrame([{**base, "year": y} for y in (202324, 202425)])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]