2026-08-20 22:10:07 +01:00
|
|
|
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
|
|
|
|
|
|
|
|
|
|
The sitemap is built from the in-memory school DataFrame, so these inject a
|
|
|
|
|
small frame via monkeypatch rather than touching a database.
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
import numpy as np
|
|
|
|
|
import pandas as pd
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _schools_df() -> pd.DataFrame:
|
|
|
|
|
"""Two schools: one with results, one with neither results nor Ofsted."""
|
|
|
|
|
base = {
|
|
|
|
|
"local_authority": "Testshire",
|
|
|
|
|
"school_type": "Academy",
|
|
|
|
|
"phase": "Primary",
|
|
|
|
|
"year": 202425,
|
|
|
|
|
"ofsted_date": None,
|
|
|
|
|
}
|
|
|
|
|
return pd.DataFrame(
|
|
|
|
|
[
|
|
|
|
|
{**base, "urn": 100001, "school_name": "Alpha Primary",
|
|
|
|
|
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
|
|
|
|
|
"ofsted_grade": 2.0},
|
|
|
|
|
{**base, "urn": 100002, "school_name": "Ghost Primary",
|
|
|
|
|
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
|
|
|
|
|
"ofsted_grade": np.nan},
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.fixture()
|
|
|
|
|
def sitemap(monkeypatch) -> str:
|
2026-08-20 22:14:32 +01:00
|
|
|
"""The sitemap index."""
|
2026-08-20 22:10:07 +01:00
|
|
|
from backend import app as app_module
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
|
|
|
return app_module.build_sitemap()
|
|
|
|
|
|
|
|
|
|
|
2026-08-20 22:14:32 +01:00
|
|
|
@pytest.fixture()
|
|
|
|
|
def schools_child(monkeypatch) -> str:
|
|
|
|
|
"""The first school child sitemap, where school URLs actually live."""
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
|
|
|
return app_module.build_sitemaps()["schools-1.xml"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.fixture()
|
|
|
|
|
def static_child(monkeypatch) -> str:
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
|
|
|
return app_module.build_sitemaps()["static.xml"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_every_loc_uses_the_www_host(sitemaps):
|
2026-08-20 22:10:07 +01:00
|
|
|
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
2026-08-20 22:14:32 +01:00
|
|
|
# Checked across every file, index included, not just one.
|
|
|
|
|
for name, xml in sitemaps.items():
|
|
|
|
|
assert "https://www.schoolcompare.co.uk" in xml, name
|
|
|
|
|
assert "https://schoolcompare.co.uk" not in xml, name
|
2026-08-20 22:12:20 +01:00
|
|
|
|
|
|
|
|
|
2026-08-20 22:14:32 +01:00
|
|
|
def test_school_with_results_is_listed(schools_child):
|
|
|
|
|
assert "/school/100001-alpha-primary" in schools_child
|
2026-08-20 22:12:20 +01:00
|
|
|
|
|
|
|
|
|
2026-08-20 22:14:32 +01:00
|
|
|
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
|
2026-08-20 22:12:20 +01:00
|
|
|
# Nothing for a search result to say about it. Submitting it spends crawl
|
|
|
|
|
# budget and drags the corpus-wide quality signal down.
|
2026-08-20 22:14:32 +01:00
|
|
|
#
|
|
|
|
|
# Asserted against the child, not the index: the index carries no school
|
|
|
|
|
# URLs at all, so it would pass this trivially and prove nothing.
|
|
|
|
|
assert "/school/100002" not in schools_child
|
2026-08-20 22:12:20 +01:00
|
|
|
|
|
|
|
|
|
2026-08-20 22:14:32 +01:00
|
|
|
def test_no_invented_priority_or_changefreq(sitemaps):
|
2026-08-20 22:12:20 +01:00
|
|
|
# Google ignores both. They were noise dressed as signal.
|
2026-08-20 22:14:32 +01:00
|
|
|
for name, xml in sitemaps.items():
|
|
|
|
|
assert "<priority>" not in xml, name
|
|
|
|
|
assert "<changefreq>" not in xml, name
|
2026-08-20 22:12:20 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_ofsted_date_becomes_lastmod(monkeypatch):
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
import datetime
|
|
|
|
|
|
|
|
|
|
def _df():
|
|
|
|
|
base = _schools_df()
|
|
|
|
|
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
|
|
|
|
|
return base
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _df)
|
2026-08-20 22:14:32 +01:00
|
|
|
xml = app_module.build_sitemaps()["schools-1.xml"]
|
2026-08-20 22:12:20 +01:00
|
|
|
assert "<lastmod>2024-03-14</lastmod>" in xml
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
|
|
|
|
|
# An always-now lastmod is a claim Google learns to distrust. Absent
|
|
|
|
|
# honestly means unknown.
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
|
|
|
|
|
def _df():
|
|
|
|
|
df = _schools_df()
|
|
|
|
|
df["ofsted_date"] = None
|
|
|
|
|
return df
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _df)
|
2026-08-20 22:14:32 +01:00
|
|
|
# The child only. The index legitimately carries a lastmod, because there
|
|
|
|
|
# it means "when this sitemap file changed", which we do know.
|
|
|
|
|
xml = app_module.build_sitemaps()["schools-1.xml"]
|
2026-08-20 22:12:20 +01:00
|
|
|
assert "<lastmod>" not in xml
|
|
|
|
|
|
|
|
|
|
|
2026-08-20 22:14:32 +01:00
|
|
|
def test_static_routes_are_listed(static_child):
|
2026-08-20 22:12:20 +01:00
|
|
|
for path in ("/", "/rankings", "/compare", "/admissions"):
|
2026-08-20 22:14:32 +01:00
|
|
|
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.fixture()
|
|
|
|
|
def sitemaps(monkeypatch) -> dict:
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
|
|
|
return app_module.build_sitemaps()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_index_lists_each_child(sitemaps):
|
|
|
|
|
index = sitemaps["sitemap.xml"]
|
|
|
|
|
assert "<sitemapindex" in index
|
|
|
|
|
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
|
|
|
|
|
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_index_carries_no_url_elements(sitemaps):
|
|
|
|
|
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
|
|
|
|
|
assert "<url>" not in sitemaps["sitemap.xml"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_index_does_not_list_itself(sitemaps):
|
|
|
|
|
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_static_child_holds_the_static_routes(sitemaps):
|
|
|
|
|
static = sitemaps["static.xml"]
|
|
|
|
|
for path in ("/", "/rankings", "/compare", "/admissions"):
|
|
|
|
|
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_school_child_holds_the_schools(sitemaps):
|
|
|
|
|
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_children_are_chunked_under_the_limit(monkeypatch):
|
|
|
|
|
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
|
|
|
|
|
# small enough to eyeball in Search Console.
|
|
|
|
|
from backend import app as app_module
|
|
|
|
|
import pandas as _pd
|
|
|
|
|
|
|
|
|
|
rows = [
|
|
|
|
|
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
|
|
|
|
|
"rwm_expected_pct": 60.0, "attainment_8_score": None,
|
|
|
|
|
"ofsted_grade": 2.0, "ofsted_date": None}
|
|
|
|
|
for i in range(10_001)
|
|
|
|
|
]
|
|
|
|
|
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
|
|
|
|
|
|
|
|
|
|
maps = app_module.build_sitemaps()
|
|
|
|
|
assert maps["schools-1.xml"].count("<url>") == 10_000
|
|
|
|
|
assert maps["schools-2.xml"].count("<url>") == 1
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_build_sitemap_still_returns_the_index(sitemap):
|
|
|
|
|
# lifespan and the admin endpoint call build_sitemap(); keep it working.
|
|
|
|
|
assert "<sitemapindex" in sitemap
|