"""Tests for sitemap generation (spec 2026-08-20, workstream W1). The sitemap is built from the in-memory school DataFrame, so these inject a small frame via monkeypatch rather than touching a database. """ import numpy as np import pandas as pd import pytest def _schools_df() -> pd.DataFrame: """Two schools: one with results, one with neither results nor Ofsted.""" base = { "local_authority": "Testshire", "school_type": "Academy", "phase": "Primary", "year": 202425, "ofsted_date": None, } return pd.DataFrame( [ {**base, "urn": 100001, "school_name": "Alpha Primary", "rwm_expected_pct": 62.0, "attainment_8_score": np.nan, "ofsted_grade": 2.0}, {**base, "urn": 100002, "school_name": "Ghost Primary", "rwm_expected_pct": np.nan, "attainment_8_score": np.nan, "ofsted_grade": np.nan}, ] ) @pytest.fixture() def sitemap(monkeypatch) -> str: """The sitemap index.""" from backend import app as app_module monkeypatch.setattr(app_module, "load_school_data", _schools_df) return app_module.build_sitemap() @pytest.fixture() def schools_child(monkeypatch) -> str: """The first school child sitemap, where school URLs actually live.""" from backend import app as app_module monkeypatch.setattr(app_module, "load_school_data", _schools_df) return app_module.build_sitemaps()["schools-1.xml"] @pytest.fixture() def static_child(monkeypatch) -> str: from backend import app as app_module monkeypatch.setattr(app_module, "load_school_data", _schools_df) return app_module.build_sitemaps()["static.xml"] def test_every_loc_uses_the_www_host(sitemaps): # The apex 301s to www. A that redirects burns a crawl per URL. # Checked across every file, index included, not just one. for name, xml in sitemaps.items(): assert "https://www.schoolcompare.co.uk" in xml, name assert "https://schoolcompare.co.uk" not in xml, name def test_school_with_results_is_listed(schools_child): assert "/school/100001-alpha-primary" in schools_child def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child): # Nothing for a search result to say about it. Submitting it spends crawl # budget and drags the corpus-wide quality signal down. # # Asserted against the child, not the index: the index carries no school # URLs at all, so it would pass this trivially and prove nothing. assert "/school/100002" not in schools_child def test_no_invented_priority_or_changefreq(sitemaps): # Google ignores both. They were noise dressed as signal. for name, xml in sitemaps.items(): assert "" not in xml, name assert "" not in xml, name def test_ofsted_date_becomes_lastmod(monkeypatch): from backend import app as app_module import datetime def _df(): base = _schools_df() base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14) return base monkeypatch.setattr(app_module, "load_school_data", _df) xml = app_module.build_sitemaps()["schools-1.xml"] assert "2024-03-14" in xml def test_no_lastmod_invented_when_date_unknown(monkeypatch): # An always-now lastmod is a claim Google learns to distrust. Absent # honestly means unknown. from backend import app as app_module def _df(): df = _schools_df() df["ofsted_date"] = None return df monkeypatch.setattr(app_module, "load_school_data", _df) # The child only. The index legitimately carries a lastmod, because there # it means "when this sitemap file changed", which we do know. xml = app_module.build_sitemaps()["schools-1.xml"] assert "" not in xml def test_static_routes_are_listed(static_child): for path in ("/", "/rankings", "/compare", "/admissions"): assert f"https://www.schoolcompare.co.uk{path}" in static_child @pytest.fixture() def sitemaps(monkeypatch) -> dict: from backend import app as app_module monkeypatch.setattr(app_module, "load_school_data", _schools_df) return app_module.build_sitemaps() def test_index_lists_each_child(sitemaps): index = sitemaps["sitemap.xml"] assert " entries only; mixing in is invalid. assert "" not in sitemaps["sitemap.xml"] def test_index_does_not_list_itself(sitemaps): assert "https://www.schoolcompare.co.uk/sitemap.xml" not in sitemaps["sitemap.xml"] def test_static_child_holds_the_static_routes(sitemaps): static = sitemaps["static.xml"] for path in ("/", "/rankings", "/compare", "/admissions"): assert f"https://www.schoolcompare.co.uk{path}" in static def test_school_child_holds_the_schools(sitemaps): assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"] def test_children_are_chunked_under_the_limit(monkeypatch): # Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays # small enough to eyeball in Search Console. from backend import app as app_module import pandas as _pd rows = [ {"urn": 200000 + i, "school_name": f"School {i}", "year": 202425, "rwm_expected_pct": 60.0, "attainment_8_score": None, "ofsted_grade": 2.0, "ofsted_date": None} for i in range(10_001) ] monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows)) maps = app_module.build_sitemaps() assert maps["schools-1.xml"].count("") == 10_000 assert maps["schools-2.xml"].count("") == 1 def test_build_sitemap_still_returns_the_index(sitemap): # lifespan and the admin endpoint call build_sitemap(); keep it working. assert "