fix(seo): crawl hygiene and a per-family sitemap index (W1) #108
No files matched your search
+65
-29
@@ -72,41 +72,77 @@ def _school_url(urn: int, school_name: str) -> str:
|
||||
return f"/school/{urn}-{slug}"
|
||||
|
||||
|
||||
# Routes worth submitting that are not a school page. /admissions was missing
|
||||
# from the sitemap entirely despite being a static, indexable guide.
|
||||
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
|
||||
|
||||
|
||||
def _has_publishable_data(row) -> bool:
|
||||
"""True when a school page has something a search result could state.
|
||||
|
||||
A school with no results in any year and no Ofsted grade renders an empty
|
||||
page. Submitting it spends crawl budget and drags the corpus-wide quality
|
||||
signal down, so it stays out of the sitemap. The page itself still resolves
|
||||
for anyone who has the URL.
|
||||
"""
|
||||
for field in ("rwm_expected_pct", "attainment_8_score", "ofsted_grade"):
|
||||
value = row.get(field)
|
||||
if value is not None and not pd.isna(value):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _url_element(loc: str, lastmod: str | None = None) -> str:
|
||||
"""One <url> entry. No priority or changefreq — Google ignores both."""
|
||||
body = f"<loc>{loc}</loc>"
|
||||
if lastmod:
|
||||
body += f"<lastmod>{lastmod}</lastmod>"
|
||||
return f" <url>{body}</url>"
|
||||
|
||||
|
||||
def _school_sitemap_rows(df) -> list[str]:
|
||||
"""A <url> element per school that has something to show.
|
||||
|
||||
lastmod comes from the school's Ofsted date where there is one and is
|
||||
omitted otherwise. An always-now lastmod is a claim Google learns to
|
||||
distrust; an absent one honestly means "unknown".
|
||||
"""
|
||||
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
|
||||
return []
|
||||
|
||||
rows: list[str] = []
|
||||
seen: set[int] = set()
|
||||
|
||||
# Latest row per URN first, so a school's most recent Ofsted date wins.
|
||||
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
|
||||
|
||||
for _, row in ordered.iterrows():
|
||||
urn = int(row["urn"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
if not _has_publishable_data(row):
|
||||
continue
|
||||
|
||||
lastmod = None
|
||||
ofsted_date = row.get("ofsted_date")
|
||||
if ofsted_date is not None and not pd.isna(ofsted_date):
|
||||
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
|
||||
|
||||
rows.append(_url_element(
|
||||
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
|
||||
|
||||
return rows
|
||||
|
||||
|
||||
def build_sitemap() -> str:
|
||||
"""Generate sitemap XML from in-memory school data. Returns the XML string."""
|
||||
df = load_school_data()
|
||||
|
||||
static_urls = [
|
||||
(BASE_URL + "/", "daily", "1.0"),
|
||||
(BASE_URL + "/rankings", "weekly", "0.8"),
|
||||
(BASE_URL + "/compare", "weekly", "0.8"),
|
||||
]
|
||||
|
||||
lines = ['<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
|
||||
|
||||
for url, freq, priority in static_urls:
|
||||
lines.append(
|
||||
f" <url><loc>{url}</loc>"
|
||||
f"<changefreq>{freq}</changefreq>"
|
||||
f"<priority>{priority}</priority></url>"
|
||||
)
|
||||
|
||||
if not df.empty and "urn" in df.columns and "school_name" in df.columns:
|
||||
seen = set()
|
||||
for _, row in df[["urn", "school_name"]].drop_duplicates(subset="urn").iterrows():
|
||||
urn = int(row["urn"])
|
||||
name = str(row["school_name"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
path = _school_url(urn, name)
|
||||
lines.append(
|
||||
f" <url><loc>{BASE_URL}{path}</loc>"
|
||||
f"<changefreq>monthly</changefreq>"
|
||||
f"<priority>0.6</priority></url>"
|
||||
)
|
||||
|
||||
lines.extend(_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS)
|
||||
lines.extend(_school_sitemap_rows(df))
|
||||
lines.append("</urlset>")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
@@ -42,3 +42,53 @@ def test_every_loc_uses_the_www_host(sitemap):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
assert "https://www.schoolcompare.co.uk" in sitemap
|
||||
assert "https://schoolcompare.co.uk" not in sitemap
|
||||
|
||||
|
||||
def test_school_with_results_is_listed(sitemap):
|
||||
assert "/school/100001-alpha-primary" in sitemap
|
||||
|
||||
|
||||
def test_school_with_no_results_and_no_ofsted_is_omitted(sitemap):
|
||||
# Nothing for a search result to say about it. Submitting it spends crawl
|
||||
# budget and drags the corpus-wide quality signal down.
|
||||
assert "/school/100002" not in sitemap
|
||||
|
||||
|
||||
def test_no_invented_priority_or_changefreq(sitemap):
|
||||
# Google ignores both. They were noise dressed as signal.
|
||||
assert "<priority>" not in sitemap
|
||||
assert "<changefreq>" not in sitemap
|
||||
|
||||
|
||||
def test_ofsted_date_becomes_lastmod(monkeypatch):
|
||||
from backend import app as app_module
|
||||
import datetime
|
||||
|
||||
def _df():
|
||||
base = _schools_df()
|
||||
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
|
||||
return base
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
xml = app_module.build_sitemap()
|
||||
assert "<lastmod>2024-03-14</lastmod>" in xml
|
||||
|
||||
|
||||
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
|
||||
# An always-now lastmod is a claim Google learns to distrust. Absent
|
||||
# honestly means unknown.
|
||||
from backend import app as app_module
|
||||
|
||||
def _df():
|
||||
df = _schools_df()
|
||||
df["ofsted_date"] = None
|
||||
return df
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
xml = app_module.build_sitemap()
|
||||
assert "<lastmod>" not in xml
|
||||
|
||||
|
||||
def test_static_routes_are_listed(sitemap):
|
||||
for path in ("/", "/rankings", "/compare", "/admissions"):
|
||||
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in sitemap
|
||||
Reference in new issue
Block a user