Files
school_compare/backend/tests/test_sitemap.py
T
TudorandClaude Opus 5 07c97a46c5
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m2s
PR Checks / Backend Smoke (pull_request) Successful in 8s
PR Checks / Build Backend (no push) (pull_request) Successful in 16s
PR Checks / Build Frontend (no push) (pull_request) Successful in 44s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 53s
fix(seo): a school is publishable on any year's results, not the latest
_school_sitemap_rows tested only the latest year's row, which quietly dropped
every school with results in its history but a null row for the most recent
year — a school that stopped reporting, or whose figures were suppressed for
small-cohort disclosure.

The Mallard Academy (150367) is the case that caught it: real KS2 results for
2015-16 through 2018-19, then null rows from 2022-23 on. Its detail page shows
all four years; the sitemap omitted it. Sampling 40 of the 2,206 excluded
schools found 4 like this, so roughly 220 real pages were being withheld.

Publishable is now a property of the school, computed across every row, while
lastmod still comes from the latest row so the most recent Ofsted date wins.
The field list is a module constant shared with _has_publishable_data so the
two checks cannot drift.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
2026-08-21 00:04:32 +01:00

220 lines
7.8 KiB
Python

"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
The sitemap is built from the in-memory school DataFrame, so these inject a
small frame via monkeypatch rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
def _schools_df() -> pd.DataFrame:
"""Two schools: one with results, one with neither results nor Ofsted."""
base = {
"local_authority": "Testshire",
"school_type": "Academy",
"phase": "Primary",
"year": 202425,
"ofsted_date": None,
}
return pd.DataFrame(
[
{**base, "urn": 100001, "school_name": "Alpha Primary",
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
"ofsted_grade": 2.0},
{**base, "urn": 100002, "school_name": "Ghost Primary",
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
"ofsted_grade": np.nan},
]
)
@pytest.fixture()
def sitemap(monkeypatch) -> str:
"""The sitemap index."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemap()
@pytest.fixture()
def schools_child(monkeypatch) -> str:
"""The first school child sitemap, where school URLs actually live."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["schools-1.xml"]
@pytest.fixture()
def static_child(monkeypatch) -> str:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["static.xml"]
def test_every_loc_uses_the_www_host(sitemaps):
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
# Checked across every file, index included, not just one.
for name, xml in sitemaps.items():
assert "https://www.schoolcompare.co.uk" in xml, name
assert "https://schoolcompare.co.uk" not in xml, name
def test_school_with_results_is_listed(schools_child):
assert "/school/100001-alpha-primary" in schools_child
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
# Nothing for a search result to say about it. Submitting it spends crawl
# budget and drags the corpus-wide quality signal down.
#
# Asserted against the child, not the index: the index carries no school
# URLs at all, so it would pass this trivially and prove nothing.
assert "/school/100002" not in schools_child
def test_no_invented_priority_or_changefreq(sitemaps):
# Google ignores both. They were noise dressed as signal.
for name, xml in sitemaps.items():
assert "<priority>" not in xml, name
assert "<changefreq>" not in xml, name
def test_ofsted_date_becomes_lastmod(monkeypatch):
from backend import app as app_module
import datetime
def _df():
base = _schools_df()
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
return base
monkeypatch.setattr(app_module, "load_school_data", _df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>2024-03-14</lastmod>" in xml
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
# An always-now lastmod is a claim Google learns to distrust. Absent
# honestly means unknown.
from backend import app as app_module
def _df():
df = _schools_df()
df["ofsted_date"] = None
return df
monkeypatch.setattr(app_module, "load_school_data", _df)
# The child only. The index legitimately carries a lastmod, because there
# it means "when this sitemap file changed", which we do know.
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>" not in xml
def test_static_routes_are_listed(static_child):
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
@pytest.fixture()
def sitemaps(monkeypatch) -> dict:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()
def test_index_lists_each_child(sitemaps):
index = sitemaps["sitemap.xml"]
assert "<sitemapindex" in index
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
def test_index_carries_no_url_elements(sitemaps):
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
assert "<url>" not in sitemaps["sitemap.xml"]
def test_index_does_not_list_itself(sitemaps):
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
def test_static_child_holds_the_static_routes(sitemaps):
static = sitemaps["static.xml"]
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
def test_school_child_holds_the_schools(sitemaps):
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
def test_children_are_chunked_under_the_limit(monkeypatch):
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
# small enough to eyeball in Search Console.
from backend import app as app_module
import pandas as _pd
rows = [
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": None,
"ofsted_grade": 2.0, "ofsted_date": None}
for i in range(10_001)
]
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
maps = app_module.build_sitemaps()
assert maps["schools-1.xml"].count("<url>") == 10_000
assert maps["schools-2.xml"].count("<url>") == 1
def test_build_sitemap_still_returns_the_index(sitemap):
# lifespan and the admin endpoint call build_sitemap(); keep it working.
assert "<sitemapindex" in sitemap
def test_school_with_results_in_an_earlier_year_is_still_listed(monkeypatch):
"""Regression: The Mallard Academy (150367).
Real KS2 results 2015-16 to 2018-19, then null rows from 2022-23 onward
because the school stopped reporting. The first cut tested the latest
year's row alone and dropped it, along with ~220 others, even though its
detail page shows all four years of results.
"""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "urn": 150367,
"school_name": "Mallard Academy"}
df = _pd.DataFrame([
{**base, "year": 201819, "rwm_expected_pct": 67.0},
{**base, "year": 202324, "rwm_expected_pct": np.nan},
{**base, "year": 202425, "rwm_expected_pct": np.nan},
])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "/school/150367-mallard-academy" in xml
def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
"""The fix must not turn into "list everything"."""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "rwm_expected_pct": np.nan,
"urn": 100002, "school_name": "Ghost Primary"}
df = _pd.DataFrame([{**base, "year": y} for y in (202324, 202425)])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]