PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m4s
PR Checks / Backend Smoke (pull_request) Successful in 7s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 44s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 2m2s
Search Console reports coverage per submitted sitemap, so one file per page family is what will make W2's location pages measurable when they land. The index's lastmod is generation time, which is the correct semantic there — unlike on a <url>, where it would be a claim we cannot support. Children sit under /sitemaps/ because Next only treats a whole bracketed path segment as dynamic; a route folder named sitemap-[...parts] would be read as a literal static segment and never match. Confirmed by the build output, which lists /sitemaps/[...parts] as a dynamic route. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
179 lines
6.0 KiB
Python
179 lines
6.0 KiB
Python
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
|
|
|
|
The sitemap is built from the in-memory school DataFrame, so these inject a
|
|
small frame via monkeypatch rather than touching a database.
|
|
"""
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
import pytest
|
|
|
|
|
|
def _schools_df() -> pd.DataFrame:
|
|
"""Two schools: one with results, one with neither results nor Ofsted."""
|
|
base = {
|
|
"local_authority": "Testshire",
|
|
"school_type": "Academy",
|
|
"phase": "Primary",
|
|
"year": 202425,
|
|
"ofsted_date": None,
|
|
}
|
|
return pd.DataFrame(
|
|
[
|
|
{**base, "urn": 100001, "school_name": "Alpha Primary",
|
|
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
|
|
"ofsted_grade": 2.0},
|
|
{**base, "urn": 100002, "school_name": "Ghost Primary",
|
|
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
|
|
"ofsted_grade": np.nan},
|
|
]
|
|
)
|
|
|
|
|
|
@pytest.fixture()
|
|
def sitemap(monkeypatch) -> str:
|
|
"""The sitemap index."""
|
|
from backend import app as app_module
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
return app_module.build_sitemap()
|
|
|
|
|
|
@pytest.fixture()
|
|
def schools_child(monkeypatch) -> str:
|
|
"""The first school child sitemap, where school URLs actually live."""
|
|
from backend import app as app_module
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
return app_module.build_sitemaps()["schools-1.xml"]
|
|
|
|
|
|
@pytest.fixture()
|
|
def static_child(monkeypatch) -> str:
|
|
from backend import app as app_module
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
return app_module.build_sitemaps()["static.xml"]
|
|
|
|
|
|
def test_every_loc_uses_the_www_host(sitemaps):
|
|
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
|
# Checked across every file, index included, not just one.
|
|
for name, xml in sitemaps.items():
|
|
assert "https://www.schoolcompare.co.uk" in xml, name
|
|
assert "https://schoolcompare.co.uk" not in xml, name
|
|
|
|
|
|
def test_school_with_results_is_listed(schools_child):
|
|
assert "/school/100001-alpha-primary" in schools_child
|
|
|
|
|
|
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
|
|
# Nothing for a search result to say about it. Submitting it spends crawl
|
|
# budget and drags the corpus-wide quality signal down.
|
|
#
|
|
# Asserted against the child, not the index: the index carries no school
|
|
# URLs at all, so it would pass this trivially and prove nothing.
|
|
assert "/school/100002" not in schools_child
|
|
|
|
|
|
def test_no_invented_priority_or_changefreq(sitemaps):
|
|
# Google ignores both. They were noise dressed as signal.
|
|
for name, xml in sitemaps.items():
|
|
assert "<priority>" not in xml, name
|
|
assert "<changefreq>" not in xml, name
|
|
|
|
|
|
def test_ofsted_date_becomes_lastmod(monkeypatch):
|
|
from backend import app as app_module
|
|
import datetime
|
|
|
|
def _df():
|
|
base = _schools_df()
|
|
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
|
|
return base
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _df)
|
|
xml = app_module.build_sitemaps()["schools-1.xml"]
|
|
assert "<lastmod>2024-03-14</lastmod>" in xml
|
|
|
|
|
|
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
|
|
# An always-now lastmod is a claim Google learns to distrust. Absent
|
|
# honestly means unknown.
|
|
from backend import app as app_module
|
|
|
|
def _df():
|
|
df = _schools_df()
|
|
df["ofsted_date"] = None
|
|
return df
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _df)
|
|
# The child only. The index legitimately carries a lastmod, because there
|
|
# it means "when this sitemap file changed", which we do know.
|
|
xml = app_module.build_sitemaps()["schools-1.xml"]
|
|
assert "<lastmod>" not in xml
|
|
|
|
|
|
def test_static_routes_are_listed(static_child):
|
|
for path in ("/", "/rankings", "/compare", "/admissions"):
|
|
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
|
|
|
|
|
|
@pytest.fixture()
|
|
def sitemaps(monkeypatch) -> dict:
|
|
from backend import app as app_module
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
return app_module.build_sitemaps()
|
|
|
|
|
|
def test_index_lists_each_child(sitemaps):
|
|
index = sitemaps["sitemap.xml"]
|
|
assert "<sitemapindex" in index
|
|
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
|
|
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
|
|
|
|
|
|
def test_index_carries_no_url_elements(sitemaps):
|
|
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
|
|
assert "<url>" not in sitemaps["sitemap.xml"]
|
|
|
|
|
|
def test_index_does_not_list_itself(sitemaps):
|
|
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
|
|
|
|
|
|
def test_static_child_holds_the_static_routes(sitemaps):
|
|
static = sitemaps["static.xml"]
|
|
for path in ("/", "/rankings", "/compare", "/admissions"):
|
|
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
|
|
|
|
|
|
def test_school_child_holds_the_schools(sitemaps):
|
|
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
|
|
|
|
|
|
def test_children_are_chunked_under_the_limit(monkeypatch):
|
|
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
|
|
# small enough to eyeball in Search Console.
|
|
from backend import app as app_module
|
|
import pandas as _pd
|
|
|
|
rows = [
|
|
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
|
|
"rwm_expected_pct": 60.0, "attainment_8_score": None,
|
|
"ofsted_grade": 2.0, "ofsted_date": None}
|
|
for i in range(10_001)
|
|
]
|
|
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
|
|
|
|
maps = app_module.build_sitemaps()
|
|
assert maps["schools-1.xml"].count("<url>") == 10_000
|
|
assert maps["schools-2.xml"].count("<url>") == 1
|
|
|
|
|
|
def test_build_sitemap_still_returns_the_index(sitemap):
|
|
# lifespan and the admin endpoint call build_sitemap(); keep it working.
|
|
assert "<sitemapindex" in sitemap
|