Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b34511e459 | ||
|
|
1fc1e07d21 |
No files matched your search
+51
-152
@@ -7,7 +7,6 @@ Uses real data from UK Government Compare School Performance downloads.
|
||||
import hashlib
|
||||
import re
|
||||
from contextlib import asynccontextmanager
|
||||
from datetime import datetime, timezone
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
@@ -49,14 +48,11 @@ PHASE_GROUPS: dict[str, set[str]] = {
|
||||
"all-through": {"all-through"},
|
||||
}
|
||||
|
||||
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
|
||||
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
|
||||
BASE_URL = "https://www.schoolcompare.co.uk"
|
||||
BASE_URL = "https://schoolcompare.co.uk"
|
||||
MAX_SLUG_LENGTH = 60
|
||||
|
||||
# In-memory sitemap cache: name -> XML. Populated on startup and by the admin
|
||||
# regenerate endpoint after a pipeline run.
|
||||
_sitemaps: dict[str, str] | None = None
|
||||
# In-memory sitemap cache
|
||||
_sitemap_xml: str | None = None
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
@@ -74,128 +70,43 @@ def _school_url(urn: int, school_name: str) -> str:
|
||||
return f"/school/{urn}-{slug}"
|
||||
|
||||
|
||||
# Routes worth submitting that are not a school page. /admissions was missing
|
||||
# from the sitemap entirely despite being a static, indexable guide.
|
||||
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
|
||||
|
||||
|
||||
def _has_publishable_data(row) -> bool:
|
||||
"""True when a school page has something a search result could state.
|
||||
|
||||
A school with no results in any year and no Ofsted grade renders an empty
|
||||
page. Submitting it spends crawl budget and drags the corpus-wide quality
|
||||
signal down, so it stays out of the sitemap. The page itself still resolves
|
||||
for anyone who has the URL.
|
||||
"""
|
||||
for field in ("rwm_expected_pct", "attainment_8_score", "ofsted_grade"):
|
||||
value = row.get(field)
|
||||
if value is not None and not pd.isna(value):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _url_element(loc: str, lastmod: str | None = None) -> str:
|
||||
"""One <url> entry. No priority or changefreq — Google ignores both."""
|
||||
body = f"<loc>{loc}</loc>"
|
||||
if lastmod:
|
||||
body += f"<lastmod>{lastmod}</lastmod>"
|
||||
return f" <url>{body}</url>"
|
||||
|
||||
|
||||
def _school_sitemap_rows(df) -> list[str]:
|
||||
"""A <url> element per school that has something to show.
|
||||
|
||||
lastmod comes from the school's Ofsted date where there is one and is
|
||||
omitted otherwise. An always-now lastmod is a claim Google learns to
|
||||
distrust; an absent one honestly means "unknown".
|
||||
"""
|
||||
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
|
||||
return []
|
||||
|
||||
rows: list[str] = []
|
||||
seen: set[int] = set()
|
||||
|
||||
# Latest row per URN first, so a school's most recent Ofsted date wins.
|
||||
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
|
||||
|
||||
for _, row in ordered.iterrows():
|
||||
urn = int(row["urn"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
if not _has_publishable_data(row):
|
||||
continue
|
||||
|
||||
lastmod = None
|
||||
ofsted_date = row.get("ofsted_date")
|
||||
if ofsted_date is not None and not pd.isna(ofsted_date):
|
||||
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
|
||||
|
||||
rows.append(_url_element(
|
||||
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
|
||||
|
||||
return rows
|
||||
|
||||
|
||||
# Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to
|
||||
# scan by eye in Search Console, which is the point of splitting at all:
|
||||
# coverage is reported per submitted sitemap, so one file per page family is
|
||||
# what makes an indexation problem attributable to a family.
|
||||
SITEMAP_CHUNK_SIZE = 10_000
|
||||
|
||||
# Children are served under /sitemaps/ because Next.js only treats a whole
|
||||
# bracketed path segment as dynamic — a route folder named "sitemap-[...parts]"
|
||||
# is read as a literal static segment and never matches.
|
||||
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
||||
|
||||
|
||||
def _urlset(rows: list[str]) -> str:
|
||||
return "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
||||
*rows,
|
||||
"</urlset>",
|
||||
])
|
||||
|
||||
|
||||
def build_sitemaps() -> dict[str, str]:
|
||||
"""Build the sitemap index and every child, keyed by name."""
|
||||
def build_sitemap() -> str:
|
||||
"""Generate sitemap XML from in-memory school data. Returns the XML string."""
|
||||
df = load_school_data()
|
||||
|
||||
children: dict[str, str] = {
|
||||
"static.xml": _urlset(
|
||||
[_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]),
|
||||
}
|
||||
|
||||
school_rows = _school_sitemap_rows(df)
|
||||
# Always emit at least one school child, so the index shape is stable even
|
||||
# on an empty database.
|
||||
chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE]
|
||||
for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"schools-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# On a sitemap index, lastmod means "when this sitemap file last changed",
|
||||
# so generation time is the correct value here — unlike on a <url>, where
|
||||
# it would be a claim about content we cannot support.
|
||||
generated = datetime.now(timezone.utc).date().isoformat()
|
||||
index_rows = [
|
||||
f" <sitemap><loc>{BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}</loc>"
|
||||
f"<lastmod>{generated}</lastmod></sitemap>"
|
||||
for name in children
|
||||
static_urls = [
|
||||
(BASE_URL + "/", "daily", "1.0"),
|
||||
(BASE_URL + "/rankings", "weekly", "0.8"),
|
||||
(BASE_URL + "/compare", "weekly", "0.8"),
|
||||
]
|
||||
index = "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
|
||||
*index_rows,
|
||||
"</sitemapindex>",
|
||||
])
|
||||
return {**children, "sitemap.xml": index}
|
||||
|
||||
lines = ['<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
|
||||
|
||||
def build_sitemap() -> str:
|
||||
"""The sitemap index. Kept for `lifespan` and the admin endpoint."""
|
||||
return build_sitemaps()["sitemap.xml"]
|
||||
for url, freq, priority in static_urls:
|
||||
lines.append(
|
||||
f" <url><loc>{url}</loc>"
|
||||
f"<changefreq>{freq}</changefreq>"
|
||||
f"<priority>{priority}</priority></url>"
|
||||
)
|
||||
|
||||
if not df.empty and "urn" in df.columns and "school_name" in df.columns:
|
||||
seen = set()
|
||||
for _, row in df[["urn", "school_name"]].drop_duplicates(subset="urn").iterrows():
|
||||
urn = int(row["urn"])
|
||||
name = str(row["school_name"])
|
||||
if urn in seen:
|
||||
continue
|
||||
seen.add(urn)
|
||||
path = _school_url(urn, name)
|
||||
lines.append(
|
||||
f" <url><loc>{BASE_URL}{path}</loc>"
|
||||
f"<changefreq>monthly</changefreq>"
|
||||
f"<priority>0.6</priority></url>"
|
||||
)
|
||||
|
||||
lines.append("</urlset>")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def clean_filter_values(series: pd.Series) -> list[str]:
|
||||
@@ -373,7 +284,7 @@ def validate_postcode(postcode: Optional[str]) -> Optional[str]:
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
"""Application lifespan - startup and shutdown events."""
|
||||
global _sitemaps
|
||||
global _sitemap_xml
|
||||
print("Loading school data from marts...")
|
||||
df = load_school_data()
|
||||
if df.empty:
|
||||
@@ -383,9 +294,9 @@ async def lifespan(app: FastAPI):
|
||||
# Pre-compute the latest-year snapshot so the first search request is fast
|
||||
await asyncio.to_thread(load_latest_school_data)
|
||||
try:
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.")
|
||||
_sitemap_xml = build_sitemap()
|
||||
n = _sitemap_xml.count("<url>")
|
||||
print(f"Sitemap built: {n} URLs.")
|
||||
except Exception as e:
|
||||
print(f"Warning: sitemap build failed on startup: {e}")
|
||||
|
||||
@@ -1176,28 +1087,16 @@ async def robots_txt():
|
||||
return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain")
|
||||
|
||||
|
||||
def _serve_sitemap(name: str) -> Response:
|
||||
global _sitemaps
|
||||
if _sitemaps is None:
|
||||
try:
|
||||
_sitemaps = build_sitemaps()
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
|
||||
if name not in _sitemaps:
|
||||
raise HTTPException(status_code=404, detail="No such sitemap")
|
||||
return Response(content=_sitemaps[name], media_type="application/xml")
|
||||
|
||||
|
||||
@app.get("/sitemap.xml")
|
||||
async def sitemap_xml():
|
||||
"""Serve the sitemap index."""
|
||||
return _serve_sitemap("sitemap.xml")
|
||||
|
||||
|
||||
@app.get("/sitemaps/{name}")
|
||||
async def sitemap_child(name: str):
|
||||
"""Serve a child sitemap (static.xml, or schools-N.xml)."""
|
||||
return _serve_sitemap(name)
|
||||
"""Serve sitemap.xml for search engine indexing."""
|
||||
global _sitemap_xml
|
||||
if _sitemap_xml is None:
|
||||
try:
|
||||
_sitemap_xml = build_sitemap()
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
|
||||
return Response(content=_sitemap_xml, media_type="application/xml")
|
||||
|
||||
|
||||
@app.post("/api/admin/regenerate-sitemap")
|
||||
@@ -1207,10 +1106,10 @@ async def regenerate_sitemap(
|
||||
_: bool = Depends(verify_admin_api_key),
|
||||
):
|
||||
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
|
||||
global _sitemaps
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
|
||||
global _sitemap_xml
|
||||
_sitemap_xml = build_sitemap()
|
||||
n = _sitemap_xml.count("<url>")
|
||||
return {"status": "ok", "urls": n}
|
||||
|
||||
|
||||
# Mount static files directly (must be after all routes to avoid catching API calls)
|
||||
|
||||
@@ -1,178 +0,0 @@
|
||||
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
|
||||
|
||||
The sitemap is built from the in-memory school DataFrame, so these inject a
|
||||
small frame via monkeypatch rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
"""Two schools: one with results, one with neither results nor Ofsted."""
|
||||
base = {
|
||||
"local_authority": "Testshire",
|
||||
"school_type": "Academy",
|
||||
"phase": "Primary",
|
||||
"year": 202425,
|
||||
"ofsted_date": None,
|
||||
}
|
||||
return pd.DataFrame(
|
||||
[
|
||||
{**base, "urn": 100001, "school_name": "Alpha Primary",
|
||||
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": 2.0},
|
||||
{**base, "urn": 100002, "school_name": "Ghost Primary",
|
||||
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": np.nan},
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sitemap(monkeypatch) -> str:
|
||||
"""The sitemap index."""
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemap()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def schools_child(monkeypatch) -> str:
|
||||
"""The first school child sitemap, where school URLs actually live."""
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()["schools-1.xml"]
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def static_child(monkeypatch) -> str:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()["static.xml"]
|
||||
|
||||
|
||||
def test_every_loc_uses_the_www_host(sitemaps):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
# Checked across every file, index included, not just one.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
assert "https://schoolcompare.co.uk" not in xml, name
|
||||
|
||||
|
||||
def test_school_with_results_is_listed(schools_child):
|
||||
assert "/school/100001-alpha-primary" in schools_child
|
||||
|
||||
|
||||
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
|
||||
# Nothing for a search result to say about it. Submitting it spends crawl
|
||||
# budget and drags the corpus-wide quality signal down.
|
||||
#
|
||||
# Asserted against the child, not the index: the index carries no school
|
||||
# URLs at all, so it would pass this trivially and prove nothing.
|
||||
assert "/school/100002" not in schools_child
|
||||
|
||||
|
||||
def test_no_invented_priority_or_changefreq(sitemaps):
|
||||
# Google ignores both. They were noise dressed as signal.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "<priority>" not in xml, name
|
||||
assert "<changefreq>" not in xml, name
|
||||
|
||||
|
||||
def test_ofsted_date_becomes_lastmod(monkeypatch):
|
||||
from backend import app as app_module
|
||||
import datetime
|
||||
|
||||
def _df():
|
||||
base = _schools_df()
|
||||
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
|
||||
return base
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "<lastmod>2024-03-14</lastmod>" in xml
|
||||
|
||||
|
||||
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
|
||||
# An always-now lastmod is a claim Google learns to distrust. Absent
|
||||
# honestly means unknown.
|
||||
from backend import app as app_module
|
||||
|
||||
def _df():
|
||||
df = _schools_df()
|
||||
df["ofsted_date"] = None
|
||||
return df
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
# The child only. The index legitimately carries a lastmod, because there
|
||||
# it means "when this sitemap file changed", which we do know.
|
||||
xml = app_module.build_sitemaps()["schools-1.xml"]
|
||||
assert "<lastmod>" not in xml
|
||||
|
||||
|
||||
def test_static_routes_are_listed(static_child):
|
||||
for path in ("/", "/rankings", "/compare", "/admissions"):
|
||||
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sitemaps(monkeypatch) -> dict:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemaps()
|
||||
|
||||
|
||||
def test_index_lists_each_child(sitemaps):
|
||||
index = sitemaps["sitemap.xml"]
|
||||
assert "<sitemapindex" in index
|
||||
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
|
||||
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
|
||||
|
||||
|
||||
def test_index_carries_no_url_elements(sitemaps):
|
||||
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
|
||||
assert "<url>" not in sitemaps["sitemap.xml"]
|
||||
|
||||
|
||||
def test_index_does_not_list_itself(sitemaps):
|
||||
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
|
||||
|
||||
|
||||
def test_static_child_holds_the_static_routes(sitemaps):
|
||||
static = sitemaps["static.xml"]
|
||||
for path in ("/", "/rankings", "/compare", "/admissions"):
|
||||
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
|
||||
|
||||
|
||||
def test_school_child_holds_the_schools(sitemaps):
|
||||
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
|
||||
|
||||
|
||||
def test_children_are_chunked_under_the_limit(monkeypatch):
|
||||
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
|
||||
# small enough to eyeball in Search Console.
|
||||
from backend import app as app_module
|
||||
import pandas as _pd
|
||||
|
||||
rows = [
|
||||
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": None,
|
||||
"ofsted_grade": 2.0, "ofsted_date": None}
|
||||
for i in range(10_001)
|
||||
]
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
|
||||
|
||||
maps = app_module.build_sitemaps()
|
||||
assert maps["schools-1.xml"].count("<url>") == 10_000
|
||||
assert maps["schools-2.xml"].count("<url>") == 1
|
||||
|
||||
|
||||
def test_build_sitemap_still_returns_the_index(sitemap):
|
||||
# lifespan and the admin endpoint call build_sitemap(); keep it working.
|
||||
assert "<sitemapindex" in sitemap
|
||||
@@ -0,0 +1,85 @@
|
||||
# Remote branch cleanup, 2026-08-20
|
||||
# Restore any branch with: git push origin <sha>:refs/heads/<name>
|
||||
|
||||
## Deleted: fully merged into main (content is in main)
|
||||
75677f4252b759ef895e7d5f7c19f8f1745bdb59 add-contact-form-footer
|
||||
fa1abff642683dfd26ba88a295a0a6710d77147a chore/byline-removal-and-audit-figure
|
||||
95081d38bdf87764ef5d298676c25fae4cd3b792 chore/remove-parent-view
|
||||
6877abedebfc1c7d95f1f6ebe68945c62be528ab chore/staged-prod-promotion
|
||||
090d5f7bec824e083d3252e2c6e636686016304d ci/frontend-checks-speedup
|
||||
8c3a5cc4e9f551f0190d85357ce7741ad87f3a4c design/cohort-identity
|
||||
955659580067ca8b81bd77e31e4e2554103f094d feat/allow-analytics-iframe-embed
|
||||
6828f6cd4417284ea3eb6f088fa20945b8b40ed3 feat/compare-chips-two-per-row
|
||||
d5cd0abfee226885119665da5d2aa8288b59217f feat/compare-data-foundation
|
||||
6dd9b04b50bee146682da87efad8fc8b526251c5 feat/compare-frontend-rebuild
|
||||
96d5fcf5b07b6f175b48e9b20fcb765a320a907f feat/detail-header-details-reveal
|
||||
f1388ff5bd0af1409823a1e047b7ba84246a0f70 feat/gias-sixth-form-flag
|
||||
eddf74745f86c9c6d9eb07d867246ff9cc90dc20 feat/hero-artwork-v2
|
||||
3015c37bac6dc28db58c80fdb9235942c25b83d7 feat/hero-byline
|
||||
8e763e39d17a4964cf558e51c03f044186371f6d feat/info-popover-tooltip
|
||||
88c653215d520ab6e902c9de55bf27d86eeb90c3 feat/last-distance-offered
|
||||
a72323874f7aebdb2e64b6d64a5febd61152d09d feat/last-distance-offered-full
|
||||
c9a1892bfb0370e0672e5849cba294ddfabe3c65 feat/latest-cutoff-only
|
||||
4e8df006d75d8be2a1d8529ddf855c445854cba1 feat/near-me-by-search
|
||||
45ab479062c6a1639facad636fc0cc0cf0fd9155 feat/proposed-to-close-schools
|
||||
3bf2e8f262cbe058fda6de6f8ea3e050224a51b9 feat/school-detail-visualisations
|
||||
1f80571b1ff217dc92b660a936c02b5f49d07f0b feat/umami-heatmap-recorder
|
||||
609bb923d96aa5730131463ea35d5efdd404bf96 feature/ingest-independent-schools
|
||||
94151c58ea38a9256d66d15161293486a500c7a2 fix/admissions-section-height
|
||||
79246edc22961c2beb3520437e9d064d3b809d10 fix/annual-dag-ks4-national-selector
|
||||
59ac9c10b97e0ef1143f57fea06b324e72ac3d4a fix/chart-marker-contrast
|
||||
9f8dba227c95706ca3527bd48d381e7622cc0a5e fix/compare-chart-refetch-resilience
|
||||
e74d3882ce78a141fa1a57daa3102d7a58852dc3 fix/compare-expert-fixes
|
||||
80176cac4db4820e76ea2a156c7a2974bee2f203 fix/compare-final-review-mustfix
|
||||
f579630fab6c456e26a6984a3e8eebdbe3184202 fix/compare-mockup-drift
|
||||
dc85254ad2ddf134b4434065d4762cef374d2220 fix/compare-null-year-blanks-chart
|
||||
43a2c4a6bc539b621f31655aec05ef319a25f343 fix/compare-refresh-and-fetch
|
||||
d677b5453365b72c81d6df2de62b1fa0d05d684d fix/daily-dag-cache-invalidation
|
||||
f6bb037c471553e8195b5a8b147467ce0d07a688 fix/detail-all-through
|
||||
17bd4d5a5eb0b12ca79b97db587f14f7d671e85b fix/detail-chart-truthfulness
|
||||
4e6be0ce65647410b4ff74f8b763207c28920c26 fix/detail-inclusion-admissions
|
||||
fdda52ff0af3fae03a4b059a655973cdbb269f91 fix/detail-ofsted-correctness
|
||||
e36125b24aba254a8d15c5c33b4d2a296e691995 fix/detail-provenance-anchoring
|
||||
b31e71ac884df6507f567adc046c9fc52d9d310a fix/detail-report-card-render-date
|
||||
32f8a02862be6a1d49f4c3928b17fcf15d4c94cc fix/detail-trend-chart-taller
|
||||
e65688d600a86818fe21ae4c61ba27e5b6ec8d7c fix/e2e-brand-assertions
|
||||
3adea73ee04cdedfab54b0351878b297f72756ad fix/e2e-compare-chips-phase
|
||||
06e4898c30feaedc471f97aba28ddb0d61379f4d fix/e2e-compare-samephase
|
||||
9abd020967670a855e80fe5a908c8048a3aa9f14 fix/e2e-distance-locator
|
||||
acec8135e1ec7c9c3c255e5b23733a0ef862b590 fix/e2e-rankings-year-pick
|
||||
5944d88f0b1517ef1ef56af1b62270d2cf28e717 fix/expert-signoff-mustfixes
|
||||
2433101fa08be5df6f170d41790512ffe823d33e fix/font-cascade-and-map-palette
|
||||
74ca76d150deec6725259d9637ea86d7bb90c683 fix/gias-legacy-fallback
|
||||
bdaa05cd542f563ef74c45307cd8f7fc465193c9 fix/hero-fallback-and-sharp
|
||||
d52d384cf23d282b44e9251176f8f3402d808600 fix/hero-map-ios-fullscreen
|
||||
4043270a77bbe4edb18207fa5f1d94d4747fe12f fix/hero-mobile-and-wording
|
||||
22e9eb2d48b0d6623e88fd67cb6ca8e5e4074583 fix/homepage-education-accuracy
|
||||
8d50afef1e8a2b621b7344eadf475b0d609ad7a2 fix/leaflet-specificity-and-font-assertion
|
||||
b2b2cad5acf534ae7a667d3fb2be15efff37c4a7 fix/list-map-report-card-signal
|
||||
dc21e80a5e9eebd13aaab84735642eb77cef35e6 fix/mobile-cell-name-size
|
||||
e5f7f4c959f024c073472122d858333a0f24866c fix/mobile-compare-polish
|
||||
a00cbe916182d1e04661c750a1ce2e7ad27a68ae fix/mobile-sort-select-overflow
|
||||
2fd997bfe640c419a6713e85df008463de7f56e8 fix/modal-keyboard-viewport
|
||||
3e7705756776a0c44d27966dfc972023f1f38b69 fix/ofsted-link-text
|
||||
ce422e64363e2b03c186ef316832b6de15502d67 fix/promote-status-token
|
||||
4522cbf64560db1e1cd519e119aa466b42cd16a1 fix/proposed-to-close-copy
|
||||
15da060e4af37fbae919e0edf25f108266de2585 fix/rankings-admissions-accuracy
|
||||
6c872ce726f210433354ca38dc5314bc6a467534 fix/rankings-year-validation
|
||||
1c1df7796194af3d47f8e5ac0a0fbe6f323700e7 fix/report-card-chip-alignment
|
||||
b2dc4d0779ced02709429afaa5985acf4abda794 fix/results-map-ios-fullscreen
|
||||
95a5783da1fc994df76cb97238b55596dce4cd8f fix/runtime-api-proxy
|
||||
8a9ba30cc24e29653a72c03c0e817684b7db7c07 fix/sats-per-level-national
|
||||
536832a524fc4f9ed858f047e00b9a4d9d429c46 fix/school-detail-nan-500
|
||||
fef83b3bf244a9bf3cb4afa75dbc433d8725014f fix/schoolbar-sticky-offset
|
||||
b0c5b6bb57c879477da12c37b15cff950c29ebd3 fix/secondary-anchors-button-affordance
|
||||
261403bcd2b01aa4f26ee212e26302fc0f769bf9 fix/special-note-full-width
|
||||
ea5249a2ea6faf6bfa5a1522644387ba59770ec8 fix/standardise-distance-units
|
||||
e4565e9f158721d4df6b918f2b065de82851f8d9 fix/trends-chart-height
|
||||
3aad5101a842539105022f9059e85c224fc5973a fix/welsh-establishment-leak
|
||||
315f1feede70bdf3101d2fdd305b3d06e037fdac perf/batch-supplementary
|
||||
d2dc78aeb599e16b7ef5019be2b360b08df463bc perf/compare-loading
|
||||
e098ad4bd1130152705788d4773837b7d1e7112e perf/server-client-split
|
||||
|
||||
## Deleted: superseded by PR #110 (content preserved on feat/seo-crawl-hygiene-main)
|
||||
786ec80dd4de4a3cb674a89e35b3b5e461639289 feat/seo-crawl-hygiene
|
||||
a5ac0bcd1b37bc10dcbce88f8601d01bf7b3eaaf feat/england-only-corpus
|
||||
+28
-92
@@ -1587,111 +1587,47 @@ test('a Welsh school URL 404s while an English one still resolves', async ({ pag
|
||||
expect(welsh?.status(), 'a Welsh school should no longer resolve').toBe(404);
|
||||
});
|
||||
|
||||
async function sitemapChildren(page: Page): Promise<string[]> {
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
const res = await page.request.get('/sitemap.xml');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const index = await res.text();
|
||||
expect(index).toContain('<sitemapindex');
|
||||
return [...index.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1]);
|
||||
}
|
||||
const xml = await res.text();
|
||||
|
||||
test('the sitemap index names children that all resolve', async ({ page }) => {
|
||||
const index = await (await page.request.get('/sitemap.xml')).text();
|
||||
// An index holds <sitemap> entries only; mixing in <url> is invalid.
|
||||
expect(index).not.toContain('<url>');
|
||||
const urlCount = (xml.match(/<url>/g) ?? []).length;
|
||||
expect(urlCount, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
|
||||
const locs = await sitemapChildren(page);
|
||||
expect(locs.length).toBeGreaterThanOrEqual(2);
|
||||
|
||||
for (const loc of locs) {
|
||||
expect(loc.startsWith('https://www.schoolcompare.co.uk/sitemaps/')).toBeTruthy();
|
||||
const child = await page.request.get(new URL(loc).pathname);
|
||||
expect(child.ok(), `${loc} should resolve`).toBeTruthy();
|
||||
expect(await child.text()).toContain('<urlset');
|
||||
}
|
||||
});
|
||||
|
||||
test('the sitemap submits no Welsh or overseas school', async ({ page }) => {
|
||||
const locs = await sitemapChildren(page);
|
||||
|
||||
let total = 0;
|
||||
for (const loc of locs) {
|
||||
const xml = await (await page.request.get(new URL(loc).pathname)).text();
|
||||
total += (xml.match(/<url>/g) ?? []).length;
|
||||
// 401559 (Adamsdown, Cardiff) and 402426 (ACT Schools, Cardiff) were both
|
||||
// submitted before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
}
|
||||
expect(total, 'sitemap looks empty or truncated').toBeGreaterThan(1000);
|
||||
});
|
||||
|
||||
test('the sitemap invents no priority or changefreq', async ({ page }) => {
|
||||
const [first] = await sitemapChildren(page);
|
||||
expect(first).toBeTruthy();
|
||||
|
||||
const xml = await (await page.request.get(new URL(first).pathname)).text();
|
||||
// Google ignores both. They were noise dressed as signal.
|
||||
expect(xml).not.toContain('<priority>');
|
||||
expect(xml).not.toContain('<changefreq>');
|
||||
// 401559 (Cardiff) and 402426 (ACT Schools, Cardiff) were both submitted
|
||||
// before the England-only filter landed.
|
||||
expect(xml).not.toContain('/school/401559');
|
||||
expect(xml).not.toContain('/school/402426');
|
||||
});
|
||||
|
||||
/*
|
||||
* Canonical URLs (spec 2026-08-20, W1).
|
||||
* Staging must not be indexable (spec 2026-08-20, W1 hygiene).
|
||||
*
|
||||
* Every indexable route declares exactly one canonical, on the www host, with
|
||||
* no query string. The homepage's eleven search params filter a result set
|
||||
* rather than making a new document, so they all collapse onto "/".
|
||||
* These journeys only ever run against staging — deploy.yml passes
|
||||
* STAGING_BASE_URL, and promote.yml only smoke-polls production without
|
||||
* Playwright — so asserting the noindex header here is safe.
|
||||
*/
|
||||
const CANONICAL_ROUTES: Array<[string, string]> = [
|
||||
['/', 'https://www.schoolcompare.co.uk/'],
|
||||
['/rankings', 'https://www.schoolcompare.co.uk/rankings'],
|
||||
['/admissions', 'https://www.schoolcompare.co.uk/admissions'],
|
||||
];
|
||||
test('staging answers noindex, and stays crawlable so the noindex is seen', async ({ page }) => {
|
||||
const res = await page.request.get('/');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
|
||||
for (const [path, expected] of CANONICAL_ROUTES) {
|
||||
test(`${path} declares exactly one canonical, on the www host`, async ({ page }) => {
|
||||
await page.goto(path);
|
||||
const hrefs = await page.locator('link[rel="canonical"]').evaluateAll(
|
||||
(els) => els.map((e) => e.getAttribute('href')));
|
||||
expect(hrefs, `${path} should declare one canonical`).toHaveLength(1);
|
||||
expect(hrefs[0]).toBe(expected);
|
||||
});
|
||||
}
|
||||
const tag = res.headers()['x-robots-tag'];
|
||||
expect(tag, 'staging must send X-Robots-Tag').toBeTruthy();
|
||||
expect(tag).toContain('noindex');
|
||||
|
||||
test('a filtered homepage still canonicalises to the bare root', async ({ page }) => {
|
||||
await page.goto('/?search=primary&phase=primary&sort=name&page=2');
|
||||
const href = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(href).toBe('https://www.schoolcompare.co.uk/');
|
||||
// The other half, and the reason this is one test rather than two: a
|
||||
// Disallow would stop Google fetching the page at all, so it would never
|
||||
// see the noindex above. The two only work together.
|
||||
const robots = await (await page.request.get('/robots.txt')).text();
|
||||
expect(robots).not.toMatch(/^\s*Disallow:\s*\/\s*$/mi);
|
||||
});
|
||||
|
||||
test('a school page canonicalises to its own slug on the www host', async ({ page }) => {
|
||||
const res = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const [first] = (await res.json()).schools ?? [];
|
||||
test('a school page on staging is noindexed too, not just the homepage', async ({ page }) => {
|
||||
const list = await page.request.get('/api/schools?search=primary&per_page=1');
|
||||
const [first] = (await list.json()).schools ?? [];
|
||||
expect(first, 'no school available').toBeTruthy();
|
||||
|
||||
await page.goto(`/school/${first.urn}-x`);
|
||||
const href = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(href).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\/\d+-/);
|
||||
});
|
||||
|
||||
test('a bare /compare is indexable, a parameterised one is not', async ({ page }) => {
|
||||
await page.goto('/compare');
|
||||
await expect(page.locator('meta[name="robots"]')).toHaveCount(0);
|
||||
|
||||
const [a, b] = await twoPrimaryUrns(page);
|
||||
await page.goto(`/compare?urns=${a},${b}`);
|
||||
const robots = await page.locator('meta[name="robots"]').first()
|
||||
.getAttribute('content');
|
||||
expect(robots).toContain('noindex');
|
||||
expect(robots).toContain('follow');
|
||||
|
||||
// noindex but follow: the links out to each school page still count, so the
|
||||
// canonical must still be present and point at the bare path.
|
||||
const canonical = await page.locator('link[rel="canonical"]').first()
|
||||
.getAttribute('href');
|
||||
expect(canonical).toBe('https://www.schoolcompare.co.uk/compare');
|
||||
const res = await page.request.get(`/school/${first.urn}-x`);
|
||||
expect(res.headers()['x-robots-tag']).toContain('noindex');
|
||||
});
|
||||
@@ -1,53 +0,0 @@
|
||||
import { metadata as homeMetadata } from '@/app/page';
|
||||
import { metadata as rankingsMetadata } from '@/app/rankings/page';
|
||||
import { metadata as admissionsMetadata } from '@/app/admissions/page';
|
||||
import { generateMetadata as compareMetadata } from '@/app/compare/page';
|
||||
|
||||
describe('canonical URLs', () => {
|
||||
it('the homepage canonicalises to the bare root', () => {
|
||||
// page.tsx reads eleven search params. Without this, every filter
|
||||
// combination is a crawlable near-duplicate of the one page we want to
|
||||
// rank for "compare schools".
|
||||
expect(homeMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
|
||||
it('rankings canonicalises to the bare path', () => {
|
||||
expect(rankingsMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('admissions canonicalises to the bare path', () => {
|
||||
expect(admissionsMetadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/admissions');
|
||||
});
|
||||
});
|
||||
|
||||
describe('/compare indexability', () => {
|
||||
it('the bare compare page is indexable and canonical to itself', async () => {
|
||||
// This is the landing page for the "compare schools" head term.
|
||||
const meta = await compareMetadata({ searchParams: Promise.resolve({}) });
|
||||
expect(meta.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/compare');
|
||||
expect(meta.robots).toBeUndefined();
|
||||
});
|
||||
|
||||
it('a comparison of specific schools is noindex, follow', async () => {
|
||||
// ~317 million pairs before triples. Indexing the parameter space would
|
||||
// swamp everything else in the corpus.
|
||||
const meta = await compareMetadata({
|
||||
searchParams: Promise.resolve({ urns: '100001,100002' }),
|
||||
});
|
||||
expect(meta.robots).toEqual({ index: false, follow: true });
|
||||
});
|
||||
|
||||
it('a parameterised comparison still canonicalises to the bare path', async () => {
|
||||
// follow:true plus a canonical means the outbound links to each school
|
||||
// page still pass value even though this URL is not indexed.
|
||||
const meta = await compareMetadata({
|
||||
searchParams: Promise.resolve({ urns: '100001,100002' }),
|
||||
});
|
||||
expect(meta.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/compare');
|
||||
});
|
||||
});
|
||||
@@ -1,27 +0,0 @@
|
||||
import { SITE_URL, absoluteUrl } from '@/lib/site';
|
||||
|
||||
describe('SITE_URL', () => {
|
||||
it('is the www host, which is the one that serves a 200', () => {
|
||||
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
|
||||
// is a wasted signal, so every absolute URL we emit must already be www.
|
||||
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
|
||||
});
|
||||
|
||||
it('has no trailing slash, so joins never double up', () => {
|
||||
expect(SITE_URL.endsWith('/')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('absoluteUrl', () => {
|
||||
it('joins a rooted path', () => {
|
||||
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('joins a path missing its leading slash', () => {
|
||||
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('maps the site root to a bare trailing slash', () => {
|
||||
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
});
|
||||
@@ -1,4 +1,3 @@
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
import { AdmissionsView } from '@/components/AdmissionsView';
|
||||
|
||||
@@ -8,7 +7,6 @@ export const metadata: Metadata = {
|
||||
title: 'School Admissions Guide',
|
||||
description:
|
||||
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
|
||||
alternates: { canonical: absoluteUrl('/admissions') },
|
||||
};
|
||||
|
||||
export default function AdmissionsPage() {
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
|
||||
import { fetchComparison, fetchMetrics } from '@/lib/api';
|
||||
import { ComparisonView } from '@/components/ComparisonView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
interface ComparePageProps {
|
||||
@@ -15,33 +14,13 @@ interface ComparePageProps {
|
||||
}>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Indexability depends on the query string, so this cannot be a static export.
|
||||
*
|
||||
* Bare /compare is the landing page for the "compare schools" head term and
|
||||
* stays indexable. /compare?urns=… is an unbounded parameter space — 25,193
|
||||
* schools make ~317 million pairs — so it goes noindex. It stays `follow` and
|
||||
* keeps a canonical to the bare path, so the links out to each school page
|
||||
* still count.
|
||||
*/
|
||||
export async function generateMetadata(
|
||||
{ searchParams }: ComparePageProps,
|
||||
): Promise<Metadata> {
|
||||
const { urns } = await searchParams;
|
||||
|
||||
const base: Metadata = {
|
||||
title: 'Compare Schools',
|
||||
description:
|
||||
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
|
||||
keywords:
|
||||
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
|
||||
alternates: { canonical: absoluteUrl('/compare') },
|
||||
};
|
||||
|
||||
if (!urns) return base;
|
||||
|
||||
return { ...base, robots: { index: false, follow: true } };
|
||||
}
|
||||
export const metadata: Metadata = {
|
||||
title: 'Compare Schools',
|
||||
description:
|
||||
'Compare schools in England side by side — Ofsted inspections, KS2 and GCSE results against the England average, admissions odds and school community.',
|
||||
keywords:
|
||||
'school comparison, compare schools, Ofsted comparison, school admissions, KS2 comparison, primary school performance',
|
||||
};
|
||||
|
||||
// Dynamic via searchParams; remove force-dynamic so internal data fetches
|
||||
// can still use Next.js's per-call revalidate cache.
|
||||
|
||||
@@ -5,7 +5,6 @@ import { Navigation } from '@/components/Navigation';
|
||||
import { Footer } from '@/components/Footer';
|
||||
import { ComparisonToast } from '@/components/ComparisonToast';
|
||||
import { ComparisonProvider } from '@/context/ComparisonProvider';
|
||||
import { SITE_URL } from '@/lib/site';
|
||||
import './globals.css';
|
||||
|
||||
// Manrope carries headings and key messaging — the guideline's "friendly,
|
||||
@@ -58,12 +57,12 @@ export const metadata: Metadata = {
|
||||
// No `icons` key on purpose: setting it here would override the file
|
||||
// conventions. app/icon.svg and app/apple-icon.tsx are the source, and
|
||||
// app/opengraph-image.tsx supplies og:image and twitter:image.
|
||||
metadataBase: new URL(SITE_URL),
|
||||
metadataBase: new URL('https://schoolcompare.co.uk'),
|
||||
openGraph: {
|
||||
type: 'website',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
url: SITE_URL,
|
||||
url: 'https://schoolcompare.co.uk',
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
|
||||
@@ -3,7 +3,6 @@
|
||||
* Main landing page with school search and browsing
|
||||
*/
|
||||
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
import { fetchSchools, fetchFilters, fetchDataInfo } from '@/lib/api';
|
||||
import { formatAcademicYear } from '@/lib/utils';
|
||||
@@ -36,10 +35,6 @@ interface HomePageProps {
|
||||
export const metadata: Metadata = {
|
||||
title: { absolute: 'schoolcompare | Compare every school in England' },
|
||||
description: 'Search and compare school performance across England',
|
||||
// This page reads eleven search params. They filter a result set; they do
|
||||
// not make a new document. Collapsing every combination onto "/" stops the
|
||||
// homepage competing with itself for its own head terms.
|
||||
alternates: { canonical: absoluteUrl('/') },
|
||||
};
|
||||
|
||||
// The page reads searchParams, which makes rendering dynamic by default.
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
|
||||
import { fetchRankings, fetchFilters, fetchMetrics } from '@/lib/api';
|
||||
import { RankingsView } from '@/components/RankingsView';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
interface RankingsPageProps {
|
||||
@@ -21,9 +20,6 @@ export const metadata: Metadata = {
|
||||
title: 'School Rankings',
|
||||
description: 'Top-ranked schools by SATs and GCSE performance across England',
|
||||
keywords: 'school rankings, top schools, best schools, KS2 rankings, KS4 rankings, school league tables',
|
||||
// Param forms (?metric=&local_authority=&year=&phase=) collapse here for
|
||||
// now. W3 replaces them with real indexable paths.
|
||||
alternates: { canonical: absoluteUrl('/rankings') },
|
||||
};
|
||||
|
||||
// Dynamic via searchParams; remove force-dynamic so internal data fetches
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
*/
|
||||
|
||||
import { MetadataRoute } from 'next';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
export default function robots(): MetadataRoute.Robots {
|
||||
return {
|
||||
@@ -15,6 +14,6 @@ export default function robots(): MetadataRoute.Robots {
|
||||
disallow: ['/api/', '/_next/'],
|
||||
},
|
||||
],
|
||||
sitemap: absoluteUrl('/sitemap.xml'),
|
||||
sitemap: 'https://schoolcompare.co.uk/sitemap.xml',
|
||||
};
|
||||
}
|
||||
@@ -15,7 +15,6 @@ import {
|
||||
} from '@/lib/schoolSections';
|
||||
import { parseSchoolSlug, schoolUrl } from '@/lib/utils';
|
||||
import type { NationalAverages } from '@/lib/types';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
/**
|
||||
@@ -98,7 +97,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
title,
|
||||
description,
|
||||
type: 'website',
|
||||
url: absoluteUrl(canonicalPath),
|
||||
url: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
@@ -107,7 +106,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
description,
|
||||
},
|
||||
alternates: {
|
||||
canonical: absoluteUrl(canonicalPath),
|
||||
canonical: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
},
|
||||
};
|
||||
} catch {
|
||||
|
||||
@@ -1,8 +1,32 @@
|
||||
import { proxySitemap } from '@/lib/sitemapProxy';
|
||||
/**
|
||||
* Runtime proxy for /sitemap.xml → the FastAPI backend's generated sitemap.
|
||||
*
|
||||
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
|
||||
* baking the backend host into the build, so one image works in every
|
||||
* environment. robots.ts points crawlers here.
|
||||
*/
|
||||
|
||||
import { NextResponse } from 'next/server';
|
||||
|
||||
export const dynamic = 'force-dynamic';
|
||||
export const runtime = 'nodejs';
|
||||
|
||||
function backendOrigin(): string {
|
||||
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
||||
return base.replace(/\/api$/, '');
|
||||
}
|
||||
|
||||
export async function GET() {
|
||||
return proxySitemap('/sitemap.xml');
|
||||
let upstream: Response;
|
||||
try {
|
||||
upstream = await fetch(`${backendOrigin()}/sitemap.xml`, { cache: 'no-store' });
|
||||
} catch {
|
||||
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
|
||||
}
|
||||
|
||||
const body = await upstream.text();
|
||||
return new NextResponse(body, {
|
||||
status: upstream.status,
|
||||
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
|
||||
});
|
||||
}
|
||||
@@ -1,24 +0,0 @@
|
||||
import { NextResponse } from 'next/server';
|
||||
import { proxySitemap } from '@/lib/sitemapProxy';
|
||||
|
||||
export const dynamic = 'force-dynamic';
|
||||
export const runtime = 'nodejs';
|
||||
|
||||
/**
|
||||
* Children are /sitemaps/static.xml and /sitemaps/schools-{n}.xml. The name is
|
||||
* validated here rather than passed through, so this route cannot be used to
|
||||
* reach arbitrary backend paths.
|
||||
*/
|
||||
const CHILD = /^(static|schools-\d+)\.xml$/;
|
||||
|
||||
export async function GET(
|
||||
_request: Request,
|
||||
{ params }: { params: Promise<{ parts: string[] }> },
|
||||
) {
|
||||
const { parts } = await params;
|
||||
const name = parts.join('/');
|
||||
if (!CHILD.test(name)) {
|
||||
return new NextResponse('Not found', { status: 404 });
|
||||
}
|
||||
return proxySitemap(`/sitemaps/${name}`);
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
/**
|
||||
* The one place the site's absolute origin is written down.
|
||||
*
|
||||
* The apex domain 301s to www at Cloudflare, so www is the host that actually
|
||||
* serves a 200. Canonicals, og:url, sitemap <loc> entries and the robots.txt
|
||||
* Sitemap: line must all agree with it — a canonical pointing at a redirect
|
||||
* makes Google resolve the hop before it can consolidate the signal.
|
||||
*
|
||||
* backend/app.py holds the same value as BASE_URL for the sitemap. The two are
|
||||
* asserted against each other by the e2e journeys rather than shared at build
|
||||
* time, because the backend and frontend ship as separate images.
|
||||
*/
|
||||
export const SITE_URL = 'https://www.schoolcompare.co.uk';
|
||||
|
||||
/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */
|
||||
export function absoluteUrl(path: string): string {
|
||||
const rooted = path.startsWith('/') ? path : `/${path}`;
|
||||
return `${SITE_URL}${rooted}`;
|
||||
}
|
||||
@@ -1,29 +0,0 @@
|
||||
/**
|
||||
* Runtime proxy for the sitemap family → the FastAPI backend.
|
||||
*
|
||||
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
|
||||
* baking the backend host into the build, so one image works in every
|
||||
* environment. robots.ts points crawlers at /sitemap.xml, which is the index;
|
||||
* the index names children under /sitemaps/, which land on the same proxy.
|
||||
*/
|
||||
import { NextResponse } from 'next/server';
|
||||
|
||||
function backendOrigin(): string {
|
||||
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
||||
return base.replace(/\/api$/, '');
|
||||
}
|
||||
|
||||
export async function proxySitemap(path: string): Promise<NextResponse> {
|
||||
let upstream: Response;
|
||||
try {
|
||||
upstream = await fetch(`${backendOrigin()}${path}`, { cache: 'no-store' });
|
||||
} catch {
|
||||
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
|
||||
}
|
||||
|
||||
const body = await upstream.text();
|
||||
return new NextResponse(body, {
|
||||
status: upstream.status,
|
||||
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
|
||||
});
|
||||
}
|
||||
@@ -55,6 +55,37 @@ const nextConfig = {
|
||||
// Headers for caching and security
|
||||
async headers() {
|
||||
return [
|
||||
{
|
||||
/*
|
||||
* Keep non-production hosts out of the index.
|
||||
*
|
||||
* Staging serves the same image as production off stx., so without
|
||||
* this it is a full crawlable duplicate of the site.
|
||||
*
|
||||
* X-Robots-Tag, NOT a robots.txt Disallow. Disallow blocks crawling,
|
||||
* which is not the same as blocking indexing — a disallowed URL can
|
||||
* still be indexed from external links, and worse, blocking the crawl
|
||||
* means Google never fetches the page and never sees a noindex at all.
|
||||
* Staging therefore stays crawlable and answers "noindex" when crawled.
|
||||
*
|
||||
* Matched on the staging host explicitly rather than "any host that is
|
||||
* not production". The inverted form is tempting because it would cover
|
||||
* future environments automatically, but its failure mode is
|
||||
* deindexing production if the Host header ever arrives rewritten by a
|
||||
* proxy. This form's failure mode is a new environment being indexable
|
||||
* until someone adds it here — recoverable, where the other is not.
|
||||
*
|
||||
* Any new non-production hostname must be added to this list.
|
||||
*/
|
||||
source: '/:path*',
|
||||
has: [{ type: 'host', value: 'stx.schoolcompare.co.uk' }],
|
||||
headers: [
|
||||
{
|
||||
key: 'X-Robots-Tag',
|
||||
value: 'noindex, nofollow',
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
source: '/:path*',
|
||||
headers: [
|
||||
|
||||
Reference in new issue
Block a user