From 42138fc4027085602f854ed65a80bbdb961ea3a5 Mon Sep 17 00:00:00 2001 From: Tudor Date: Fri, 21 Aug 2026 18:13:55 +0100 Subject: [PATCH] feat(places): submit place and outcode sitemaps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Separate children per family so Search Console reports the location layer's indexation apart from the school pages' — which is the point of the index built in W1, and the number the stop condition watches. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- backend/app.py | 34 +++++++++++++ backend/tests/test_sitemap.py | 53 ++++++++++++++++++++- nextjs-app/app/sitemaps/[...parts]/route.ts | 2 +- 3 files changed, 87 insertions(+), 2 deletions(-) diff --git a/backend/app.py b/backend/app.py index ce6d339..934f0e4 100644 --- a/backend/app.py +++ b/backend/app.py @@ -194,6 +194,29 @@ def _urlset(rows: list[str]) -> str: ]) +def _place_url(place) -> str: + """The canonical path for a place. Two namespaces, per the spec. + + Towns and localities share /schools/[place]; authorities take their own + prefix because 67 town names collide with an authority name and neither + set contains the other. + """ + if place.kind == "authority": + return f"/schools/authority/{place.slug}" + if place.kind == "outcode": + return f"/schools/near/{place.slug}" + return f"/schools/{place.slug}" + + +def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]: + return [ + _url_element(BASE_URL + _place_url(p)) + for p in sorted(get_place_registry().values(), + key=lambda p: (p.kind, p.slug)) + if p.kind in kinds + ] + + def build_sitemaps() -> dict[str, str]: """Build the sitemap index and every child, keyed by name.""" df = load_school_data() @@ -211,6 +234,17 @@ def build_sitemaps() -> dict[str, str]: for n, chunk in enumerate(chunks, start=1): children[f"schools-{n}.xml"] = _urlset(chunk) + # Separate children per family: Search Console reports coverage per + # submitted sitemap, which is how the location layer's indexation is + # measured apart from the school pages'. + for label, kinds in (("places", ("town", "locality", "authority")), + ("outcodes", ("outcode",))): + rows = _place_sitemap_rows(kinds) + chunks = [rows[i:i + SITEMAP_CHUNK_SIZE] + for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]] + for n, chunk in enumerate(chunks, start=1): + children[f"{label}-{n}.xml"] = _urlset(chunk) + # On a sitemap index, lastmod means "when this sitemap file last changed", # so generation time is the correct value here — unlike on a , where # it would be a claim about content we cannot support. diff --git a/backend/tests/test_sitemap.py b/backend/tests/test_sitemap.py index c3d908a..21d44ff 100644 --- a/backend/tests/test_sitemap.py +++ b/backend/tests/test_sitemap.py @@ -59,9 +59,14 @@ def static_child(monkeypatch) -> str: def test_every_loc_uses_the_www_host(sitemaps): # The apex 301s to www. A that redirects burns a crawl per URL. # Checked across every file, index included, not just one. + # + # A child can legitimately be empty — this fixture holds two schools and no + # town clearing the threshold — so the presence check applies only to files + # that carry URLs. The absence check applies to all of them. for name, xml in sitemaps.items(): - assert "https://www.schoolcompare.co.uk" in xml, name assert "https://schoolcompare.co.uk" not in xml, name + if "" in xml: + assert "https://www.schoolcompare.co.uk" in xml, name def test_school_with_results_is_listed(schools_child): @@ -217,3 +222,49 @@ def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch): monkeypatch.setattr(app_module, "load_school_data", lambda: df) assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"] + + +def _places_df() -> pd.DataFrame: + base = { + "local_authority": "Essex", "school_type": "Academy", + "phase": "Primary", "year": 202425, "ofsted_grade": 2.0, + "ofsted_date": None, "attainment_8_score": np.nan, + "town": "Brentwood", "postcode": "CM13 1AA", + } + return pd.DataFrame([ + {**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}", + "rwm_expected_pct": 60.0} + for i in range(6) + ]) + + +@pytest.fixture() +def place_sitemaps(monkeypatch) -> dict: + from backend import app as app_module + + monkeypatch.setattr(app_module, "load_school_data", _places_df) + monkeypatch.setattr(app_module, "_place_registry", None) + return app_module.build_sitemaps() + + +def test_place_children_are_listed_in_the_index(place_sitemaps): + index = place_sitemaps["sitemap.xml"] + assert "/sitemaps/places-1.xml" in index + assert "/sitemaps/outcodes-1.xml" in index + + +def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps): + xml = place_sitemaps["places-1.xml"] + assert "https://www.schoolcompare.co.uk/schools/brentwood" in xml + assert "https://www.schoolcompare.co.uk/schools/authority/essex" in xml + + +def test_outcode_urls_live_in_their_own_child(place_sitemaps): + assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"] + assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"] + + +def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps): + for name in ("places-1.xml", "outcodes-1.xml"): + assert "" not in place_sitemaps[name] + assert "" not in place_sitemaps[name] diff --git a/nextjs-app/app/sitemaps/[...parts]/route.ts b/nextjs-app/app/sitemaps/[...parts]/route.ts index c540b82..8e71cfe 100644 --- a/nextjs-app/app/sitemaps/[...parts]/route.ts +++ b/nextjs-app/app/sitemaps/[...parts]/route.ts @@ -9,7 +9,7 @@ export const runtime = 'nodejs'; * validated here rather than passed through, so this route cannot be used to * reach arbitrary backend paths. */ -const CHILD = /^(static|schools-\d+)\.xml$/; +const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/; export async function GET( _request: Request,