feat(places): submit place and outcode sitemaps

Separate children per family so Search Console reports the location layer's
indexation apart from the school pages' — which is the point of the index
built in W1, and the number the stop condition watches.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-21 18:13:55 +01:00
1 parent c5af476213
commit 42138fc402
3 files changed
+87 -2

No files matched your search

+34
View File
@@ -194,6 +194,29 @@ def _urlset(rows: list[str]) -> str:
])
def _place_url(place) -> str:
"""The canonical path for a place. Two namespaces, per the spec.
Towns and localities share /schools/[place]; authorities take their own
prefix because 67 town names collide with an authority name and neither
set contains the other.
"""
if place.kind == "authority":
return f"/schools/authority/{place.slug}"
if place.kind == "outcode":
return f"/schools/near/{place.slug}"
return f"/schools/{place.slug}"
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
return [
_url_element(BASE_URL + _place_url(p))
for p in sorted(get_place_registry().values(),
key=lambda p: (p.kind, p.slug))
if p.kind in kinds
]
def build_sitemaps() -> dict[str, str]:
"""Build the sitemap index and every child, keyed by name."""
df = load_school_data()
@@ -211,6 +234,17 @@ def build_sitemaps() -> dict[str, str]:
for n, chunk in enumerate(chunks, start=1):
children[f"schools-{n}.xml"] = _urlset(chunk)
# Separate children per family: Search Console reports coverage per
# submitted sitemap, which is how the location layer's indexation is
# measured apart from the school pages'.
for label, kinds in (("places", ("town", "locality", "authority")),
("outcodes", ("outcode",))):
rows = _place_sitemap_rows(kinds)
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1):
children[f"{label}-{n}.xml"] = _urlset(chunk)
# On a sitemap index, lastmod means "when this sitemap file last changed",
# so generation time is the correct value here — unlike on a <url>, where
# it would be a claim about content we cannot support.
+52 -1
View File
@@ -59,9 +59,14 @@ def static_child(monkeypatch) -> str:
def test_every_loc_uses_the_www_host(sitemaps):
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
# Checked across every file, index included, not just one.
#
# A child can legitimately be empty — this fixture holds two schools and no
# town clearing the threshold — so the presence check applies only to files
# that carry URLs. The absence check applies to all of them.
for name, xml in sitemaps.items():
assert "https://www.schoolcompare.co.uk" in xml, name
assert "https://schoolcompare.co.uk" not in xml, name
if "<loc>" in xml:
assert "https://www.schoolcompare.co.uk" in xml, name
def test_school_with_results_is_listed(schools_child):
@@ -217,3 +222,49 @@ def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]
def _places_df() -> pd.DataFrame:
base = {
"local_authority": "Essex", "school_type": "Academy",
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
"ofsted_date": None, "attainment_8_score": np.nan,
"town": "Brentwood", "postcode": "CM13 1AA",
}
return pd.DataFrame([
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
"rwm_expected_pct": 60.0}
for i in range(6)
])
@pytest.fixture()
def place_sitemaps(monkeypatch) -> dict:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _places_df)
monkeypatch.setattr(app_module, "_place_registry", None)
return app_module.build_sitemaps()
def test_place_children_are_listed_in_the_index(place_sitemaps):
index = place_sitemaps["sitemap.xml"]
assert "/sitemaps/places-1.xml" in index
assert "/sitemaps/outcodes-1.xml" in index
def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps):
xml = place_sitemaps["places-1.xml"]
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood</loc>" in xml
assert "<loc>https://www.schoolcompare.co.uk/schools/authority/essex</loc>" in xml
def test_outcode_urls_live_in_their_own_child(place_sitemaps):
assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"]
assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"]
def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps):
for name in ("places-1.xml", "outcodes-1.xml"):
assert "<priority>" not in place_sitemaps[name]
assert "<changefreq>" not in place_sitemaps[name]
+1 -1
View File
@@ -9,7 +9,7 @@ export const runtime = 'nodejs';
* validated here rather than passed through, so this route cannot be used to
* reach arbitrary backend paths.
*/
const CHILD = /^(static|schools-\d+)\.xml$/;
const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/;
export async function GET(
_request: Request,