Compare commits

..
Author SHA1 Message Date
TudorandClaude Opus 5 914b429a28 fix(home): art-direct the hero's fallback path, and declare sharp
Two findings from review on #93, both verified before fixing.

<img src> cannot vary by viewport, so it was always the wide desktop crop. A
browser taking neither AVIF nor WebP therefore fell through to the desktop
frame on a phone and lost the schoolhouse — the exact failure the two-crop
<picture> exists to prevent, surviving in the one path nobody looks at. The
band JPEG the build script already emitted was never referenced, which was the
tell. It now backs a <source media> placed after the modern formats, so they
still win wherever they are supported.

Verified by stripping the AVIF and WebP <source>s at runtime and letting
<picture> re-resolve, which is what an old browser actually sees:

  phone    hero-band-500.avif  →  hero-band-700.jpg   (band crop, school kept)
  desktop  hero-wide-1672.avif →  hero-wide-1200.jpg

sharp was not declared: it arrives transitively from next@16.1.6, so the
documented regeneration command works today and breaks on a Next upgrade or a
clean install that resolves differently. Declared in devDependencies for the
same reason next.config.js already declares its traced font files rather than
trusting the tracer to keep finding them.

The third finding — that the hero's licence is marked unconfirmed in
CREDITS.md while the artwork ships — is accurate and deliberate. It is the
owner's to answer; recording it as unknown is the point of the file.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-14 21:46:04 +01:00
242 changed files with 512 additions and 36979 deletions

No files matched your search

-6
View File
@@ -5,9 +5,3 @@ __pycache__/
pipeline/transform/target/
pipeline/transform/logs/
pipeline/transform/.user.yml
# Playwright MCP scratch output (screenshots, console logs, page snapshots)
.playwright-mcp/
# Playwright run artefacts written when the suite is run from the repo root
test-results/
+54 -561
View File
@@ -6,9 +6,7 @@ Uses real data from UK Government Compare School Performance downloads.
import hashlib
import re
import time
from contextlib import asynccontextmanager
from datetime import datetime, timezone
from typing import Optional
import numpy as np
@@ -16,7 +14,7 @@ import pandas as pd
from fastapi import FastAPI, HTTPException, Query, Request, Depends, Header
from fastapi.middleware.cors import CORSMiddleware
from fastapi.middleware.gzip import GZipMiddleware
from fastapi.responses import FileResponse, JSONResponse, Response
from fastapi.responses import FileResponse, Response
from fastapi.staticfiles import StaticFiles
from slowapi import Limiter, _rate_limit_exceeded_handler
from slowapi.util import get_remote_address
@@ -34,11 +32,8 @@ from .data_loader import (
get_supplementary_data,
get_supplementary_data_batch,
search_schools_typesense,
suggest_schools_typesense,
)
from .data_loader import get_data_info as get_db_info
from . import flags
from .places import build_place_index, build_place_registry, places_for_urn
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
from .utils import clean_for_json, convert_to_native
@@ -53,24 +48,11 @@ PHASE_GROUPS: dict[str, set[str]] = {
"all-through": {"all-through"},
}
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
BASE_URL = "https://www.schoolcompare.co.uk"
BASE_URL = "https://schoolcompare.co.uk"
MAX_SLUG_LENGTH = 60
# In-memory sitemap cache: name -> XML. Populated on startup and by the admin
# regenerate endpoint after a pipeline run.
_sitemaps: dict[str, str] | None = None
# Built from the same DataFrame the sitemap uses, so places and sitemap can
# never describe different corpora. Reset by the same admin endpoint.
_place_registry: dict | None = None
# Cached beside the registry, and invalidated by identity against it — see
# get_place_index. Never cleared independently.
_place_index: dict | None = None
_place_index_source: dict | None = None
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
# In-memory sitemap cache
_sitemap_xml: str | None = None
def _slugify(text: str) -> str:
@@ -88,263 +70,43 @@ def _school_url(urn: int, school_name: str) -> str:
return f"/school/{urn}-{slug}"
# Routes worth submitting that are not a school page. /admissions was missing
# from the sitemap entirely despite being a static, indexable guide.
STATIC_SITEMAP_PATHS = ("/", "/rankings", "/compare", "/admissions")
# A page has something a search result could state if any of these is present
# in any year. Shared by _has_publishable_data and the per-school check in
# _school_sitemap_rows so the two can never drift.
_PUBLISHABLE_FIELDS = ("rwm_expected_pct", "attainment_8_score", "ofsted_grade")
def _has_publishable_data(row) -> bool:
"""True when a school page has something a search result could state.
A school with no results in any year and no Ofsted grade renders an empty
page. Submitting it spends crawl budget and drags the corpus-wide quality
signal down, so it stays out of the sitemap. The page itself still resolves
for anyone who has the URL.
"""
for field in _PUBLISHABLE_FIELDS:
value = row.get(field)
if value is not None and not pd.isna(value):
return True
return False
def _url_element(loc: str, lastmod: str | None = None) -> str:
"""One <url> entry. No priority or changefreq — Google ignores both."""
body = f"<loc>{loc}</loc>"
if lastmod:
body += f"<lastmod>{lastmod}</lastmod>"
return f" <url>{body}</url>"
def _school_sitemap_rows(df) -> list[str]:
"""A <url> element per school that has something to show.
lastmod comes from the school's Ofsted date where there is one and is
omitted otherwise. An always-now lastmod is a claim Google learns to
distrust; an absent one honestly means "unknown".
"""
if df.empty or "urn" not in df.columns or "school_name" not in df.columns:
return []
rows: list[str] = []
seen: set[int] = set()
# Publishable is a property of the SCHOOL, not of its latest row.
#
# The first cut tested the latest year's row alone, which quietly dropped
# every school that has results in its history but a null row for the most
# recent year — a school that stopped reporting, or whose figures were
# suppressed for small-cohort disclosure. The Mallard Academy (150367) is
# the case that caught it: real KS2 results for 2015-16 through 2018-19,
# then null rows for 2022-23 onward. Its page shows all four years; the
# sitemap omitted it. Roughly 220 schools were affected.
publishable_cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
publishable: set[int] = (
set(df.loc[df[publishable_cols].notna().any(axis=1), "urn"].astype(int))
if publishable_cols else set()
)
# Latest row per URN first, so a school's most recent Ofsted date wins.
ordered = df.sort_values("year", ascending=False) if "year" in df.columns else df
for _, row in ordered.iterrows():
urn = int(row["urn"])
if urn in seen:
continue
seen.add(urn)
if urn not in publishable:
continue
lastmod = None
ofsted_date = row.get("ofsted_date")
if ofsted_date is not None and not pd.isna(ofsted_date):
lastmod = pd.Timestamp(ofsted_date).date().isoformat()
rows.append(_url_element(
BASE_URL + _school_url(urn, str(row["school_name"])), lastmod))
return rows
# Sitemaps cap at 50,000 URLs per file. 10,000 keeps a child small enough to
# scan by eye in Search Console, which is the point of splitting at all:
# coverage is reported per submitted sitemap, so one file per page family is
# what makes an indexation problem attributable to a family.
SITEMAP_CHUNK_SIZE = 10_000
# Children are served under /sitemaps/ because Next.js only treats a whole
# bracketed path segment as dynamic — a route folder named "sitemap-[...parts]"
# is read as a literal static segment and never matches.
SITEMAP_CHILD_PREFIX = "/sitemaps"
def get_place_registry() -> dict:
"""The place registry, built once and cached for the process."""
global _place_registry
if _place_registry is None:
_place_registry = build_place_registry(load_school_data())
return _place_registry
def get_place_index() -> dict:
"""URN → its published places, cached against the registry it came from.
Invalidation is an identity check rather than a second flag to remember to
clear. Anything that drops `_place_registry` — the tests all do — gets a
fresh registry object here, which no longer matches the one the index was
built from, so the index rebuilds with it. A separate `_place_index = None`
would be one more thing to forget, and a stale reverse index is exactly the
bug that would put links to another dataset's places on a school page.
"""
global _place_index, _place_index_source
registry = get_place_registry()
if _place_index is None or _place_index_source is not registry:
_place_index = build_place_index(registry)
_place_index_source = registry
return _place_index
def _urlset(rows: list[str]) -> str:
return "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
*rows,
"</urlset>",
])
def _place_url(place) -> str:
"""The canonical path for a place. Two namespaces, per the spec.
Towns and localities share /schools/[place]; authorities take their own
prefix because 67 town names collide with an authority name and neither
set contains the other.
"""
if place.kind == "authority":
return f"/schools/authority/{place.slug}"
if place.kind == "outcode":
return f"/schools/near/{place.slug}"
return f"/schools/{place.slug}"
def _places_payload(urn: int) -> list[dict]:
"""The published places containing this school, as the school page needs
them: a name to write in the link, a count so the anchor can say what it
leads to, and the canonical path.
`phases` carries the phase variants this school actually appears on, which
is usually one and is two for an all-through school — it is listed on both
pages, so there is no tie to break.
Membership is read straight from the registry's own `phase_urns` rather
than re-derived from the school's phase string. The registry is the one
place that decides which phases a place publishes and who is on them;
computing it a second time here is how a page comes to link a school to a
phase page that does not list it, or to a route that does not exist. That
is also why outcodes need no special case: they carry empty `phase_urns`,
so they report no phase links on their own.
"""
payload = []
for place in places_for_urn(get_place_index(), int(urn)):
phases = [
{
"phase": phase,
"count": len(phase_urns),
"url": f"{_place_url(place)}/{phase}",
}
for phase, phase_urns in sorted(place.phase_urns.items())
if int(urn) in phase_urns
]
payload.append({
"kind": place.kind,
"slug": place.slug,
"name": place.name,
"count": len(place.urns),
"url": _place_url(place),
"phases": phases,
})
return payload
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
"""A <url> per place, plus a phase variant wherever that phase clears the
threshold on its own.
Phase is part of the query — "primary schools in beccles" — so each
variant is its own indexable page. Submitting only the bare place URL left
~950 of them reachable by nothing: absent from every sitemap, and not
linked from the place page either.
"""
rows: list[str] = []
for p in sorted(get_place_registry().values(), key=lambda p: (p.kind, p.slug)):
if p.kind not in kinds:
continue
rows.append(_url_element(BASE_URL + _place_url(p)))
# Which phases a place publishes is the registry's decision alone —
# outcodes report none, because the spec gives them no phase route.
# Repeating that rule here was how the page and the sitemap came to
# disagree about which URLs exist.
for phase in ("primary", "secondary"):
if p.publishes_phase(phase):
rows.append(_url_element(f"{BASE_URL}{_place_url(p)}/{phase}"))
return rows
def build_sitemaps() -> dict[str, str]:
"""Build the sitemap index and every child, keyed by name."""
def build_sitemap() -> str:
"""Generate sitemap XML from in-memory school data. Returns the XML string."""
df = load_school_data()
children: dict[str, str] = {
"static.xml": _urlset(
[_url_element(BASE_URL + path) for path in STATIC_SITEMAP_PATHS]),
}
school_rows = _school_sitemap_rows(df)
# Always emit at least one school child, so the index shape is stable even
# on an empty database.
chunks = [school_rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(school_rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1):
children[f"schools-{n}.xml"] = _urlset(chunk)
# Separate children per family: Search Console reports coverage per
# submitted sitemap, which is how the location layer's indexation is
# measured apart from the school pages'.
for label, kinds in (("places", ("town", "locality", "authority")),
("outcodes", ("outcode",))):
rows = _place_sitemap_rows(kinds)
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1):
children[f"{label}-{n}.xml"] = _urlset(chunk)
# On a sitemap index, lastmod means "when this sitemap file last changed",
# so generation time is the correct value here — unlike on a <url>, where
# it would be a claim about content we cannot support.
generated = datetime.now(timezone.utc).date().isoformat()
index_rows = [
f" <sitemap><loc>{BASE_URL}{SITEMAP_CHILD_PREFIX}/{name}</loc>"
f"<lastmod>{generated}</lastmod></sitemap>"
for name in children
static_urls = [
(BASE_URL + "/", "daily", "1.0"),
(BASE_URL + "/rankings", "weekly", "0.8"),
(BASE_URL + "/compare", "weekly", "0.8"),
]
index = "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
'<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">',
*index_rows,
"</sitemapindex>",
])
return {**children, "sitemap.xml": index}
lines = ['<?xml version="1.0" encoding="UTF-8"?>',
'<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">']
def build_sitemap() -> str:
"""The sitemap index. Kept for `lifespan` and the admin endpoint."""
return build_sitemaps()["sitemap.xml"]
for url, freq, priority in static_urls:
lines.append(
f" <url><loc>{url}</loc>"
f"<changefreq>{freq}</changefreq>"
f"<priority>{priority}</priority></url>"
)
if not df.empty and "urn" in df.columns and "school_name" in df.columns:
seen = set()
for _, row in df[["urn", "school_name"]].drop_duplicates(subset="urn").iterrows():
urn = int(row["urn"])
name = str(row["school_name"])
if urn in seen:
continue
seen.add(urn)
path = _school_url(urn, name)
lines.append(
f" <url><loc>{BASE_URL}{path}</loc>"
f"<changefreq>monthly</changefreq>"
f"<priority>0.6</priority></url>"
)
lines.append("</urlset>")
return "\n".join(lines)
def clean_filter_values(series: pd.Series) -> list[str]:
@@ -359,101 +121,8 @@ def clean_filter_values(series: pd.Series) -> list[str]:
# SECURITY MIDDLEWARE & HELPERS
# =============================================================================
def client_key(request: Request) -> str:
"""The rate-limit bucket: the real caller, not the proxy in front of them.
`get_remote_address` reads request.client.host. In staging and production
the backend has no published ports and sits on the internal network, so its
only caller is the Next proxy — meaning every browser user on the site
shared one bucket. Measured before this fix: 70 concurrent requests to
/api/schools returned 60 OK and 10 refused.
CF-Connecting-IP first, because Cloudflare (in front of both environments)
sets it on every origin request and *overwrites* any client-supplied value,
which a parsed X-Forwarded-For chain does not guarantee. The XFF fallback is
forgeable, but only by a caller already inside the Docker network, which is
the one place nothing untrusted can reach.
"""
cf = request.headers.get("cf-connecting-ip")
if cf:
return cf.strip()
xff = request.headers.get("x-forwarded-for")
if xff:
return xff.split(",")[0].strip()
return get_remote_address(request)
# Per-client limiter. Paired with the global ceiling below — the two do
# different jobs and neither substitutes for the other.
limiter = Limiter(key_func=client_key)
# --- The ceiling no header can raise ----------------------------------------
#
# client_key trusts CF-Connecting-IP, and nothing in this process can tell an
# edge-set header from an attacker-set one. That distinction can only be made
# at Cloudflare, with Authenticated Origin Pulls or an origin firewall. A
# caller reaching the origin directly could otherwise mint a fresh rate-limit
# bucket per request and evade per-client limits entirely — which would make
# correct keying a net regression against abuse, since the single shared bucket
# it replaced at least capped everyone at 60/minute together.
#
# So per-client limits give fairness, and this gives the origin a hard total.
# It does not make the header trustworthy; it bounds what trusting it can cost.
# The header problem itself is closed at Cloudflare, not here.
#
# [window_start_monotonic, count], or None before the first request. A fixed
# window is crude, which is right for a backstop: it has to be obviously
# correct rather than fair.
_global_window: Optional[list] = None
# The container healthcheck runs `curl http://localhost:80/api/data-info` from
# inside the container. Starving it would fail the check, restart the
# container, and turn a load spike into an outage loop — the ceiling exists to
# protect the origin, not to kill it.
_LOCAL_HOSTS = frozenset({"127.0.0.1", "::1", "localhost"})
def exempt_from_ceiling(request: Request) -> bool:
"""Whether the ceiling should ignore this request.
Its own function so the rule is testable without standing up a server —
and so the healthcheck exemption is somewhere a reader can find it.
"""
if not request.url.path.startswith("/api/"):
return True
# The peer address, never the Host header: Host is set by the caller and
# would hand every attacker an exemption.
return (request.client.host if request.client else "") in _LOCAL_HOSTS
class GlobalRateLimitMiddleware(BaseHTTPMiddleware):
"""A cap on total /api/ traffic, independent of any client identity."""
async def dispatch(self, request: Request, call_next):
global _global_window
if exempt_from_ceiling(request):
return await call_next(request)
now = time.monotonic()
# One event loop, and no await between the read and the write, so this
# sequence is atomic without a lock.
if _global_window is None or now - _global_window[0] >= 60:
_global_window = [now, 0]
_global_window[1] += 1
if _global_window[1] > settings.global_rate_limit_per_minute:
return JSONResponse(
# Distinguishable from slowapi's per-client 429: an operator
# reading logs has to be able to tell "one noisy client" from
# "the origin is saturated".
{"detail": "The service is at capacity. Please retry shortly."},
status_code=429,
headers={"Retry-After":
str(max(1, int(60 - (now - _global_window[0]))))},
)
return await call_next(request)
# Rate limiter
limiter = Limiter(key_func=get_remote_address)
class SecurityHeadersMiddleware(BaseHTTPMiddleware):
@@ -511,7 +180,6 @@ CACHE_RULES: list[tuple[str, tuple[int, int, int]]] = [
("/api/schools/", (300, 3600, 86400)), # /api/schools/{urn}
("/api/rankings", (60, 600, 3600)),
("/api/compare", (60, 600, 3600)),
("/api/suggest", (60, 3600, 86400)), # autosuggest
("/api/schools", (30, 300, 1800)), # search list
]
@@ -616,8 +284,7 @@ def validate_postcode(postcode: Optional[str]) -> Optional[str]:
@asynccontextmanager
async def lifespan(app: FastAPI):
"""Application lifespan - startup and shutdown events."""
global _sitemaps
flags.init()
global _sitemap_xml
print("Loading school data from marts...")
df = load_school_data()
if df.empty:
@@ -627,9 +294,9 @@ async def lifespan(app: FastAPI):
# Pre-compute the latest-year snapshot so the first search request is fast
await asyncio.to_thread(load_latest_school_data)
try:
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
print(f"Sitemaps built: {len(_sitemaps)} files, {n} URLs.")
_sitemap_xml = build_sitemap()
n = _sitemap_xml.count("<url>")
print(f"Sitemap built: {n} URLs.")
except Exception as e:
print(f"Warning: sitemap build failed on startup: {e}")
@@ -659,10 +326,6 @@ app.add_middleware(CacheAndETagMiddleware)
app.add_middleware(SecurityHeadersMiddleware)
app.add_middleware(RequestSizeLimitMiddleware)
app.add_middleware(GZipMiddleware, minimum_size=512)
# Added last, so it is outermost and refuses before anything downstream does
# work. A ceiling that only applies after the expensive part has run is not a
# ceiling.
app.add_middleware(GlobalRateLimitMiddleware)
# CORS middleware - restricted for production
app.add_middleware(
@@ -963,31 +626,16 @@ async def get_school_details(request: Request, urn: int):
return {
"school_info": school_info,
# Where this school sits in the location layer, for the page's link
# module and breadcrumb. Derived from the same registry the place
# pages and the sitemap use, so a link is never offered for a page
# that does not exist. Empty is a valid answer: a school whose town
# and authority both fall below the publish threshold has nowhere to
# point, and the page renders without the module.
"places": _places_payload(urn),
"yearly_data": clean_for_json(school_data),
# Supplementary data (null if not yet populated by Kestra)
"ofsted": supplementary.get("ofsted"),
"census": supplementary.get("census"),
"admissions": supplementary.get("admissions"),
"admissions_history": supplementary.get("admissions_history") or [],
# Behind a flag, and withheld at the source rather than rendered-but-
# hidden: this endpoint is public and unauthenticated, so a field left
# in the payload is a published field. The key is absent, not null —
# null would state that this school has no cut-off, which is a
# different claim from "we are not publishing cut-offs".
**({"admission_distance": supplementary.get("admission_distance")}
if flags.is_enabled("admission_distance") else {}),
"sen_detail": supplementary.get("sen_detail"),
"phonics": supplementary.get("phonics"),
"deprivation": supplementary.get("deprivation"),
"finance": supplementary.get("finance"),
"destinations": supplementary.get("destinations"),
}
@@ -1358,145 +1006,6 @@ async def get_rankings(
}
@app.get("/api/places")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def list_places(request: Request):
"""Every published place. The sitemap and the link modules read this."""
registry = get_place_registry()
return {"places": [
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
]}
@app.get("/api/places/{kind}/{slug}")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_place(request: Request, kind: str, slug: str,
phase: Optional[str] = None):
"""One place: its schools ranked, and its local averages."""
if kind not in VALID_PLACE_KINDS:
raise HTTPException(status_code=404, detail="No such place")
registry = get_place_registry()
place = registry.get(f"{kind}:{slug}")
if place is None:
raise HTTPException(status_code=404, detail="No such place")
df = load_latest_school_data()
rows = df[df["urn"].isin(place.urns)]
if phase:
wanted = PHASE_GROUPS.get(phase.lower())
if wanted and "phase" in rows.columns:
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
# The metric the page shows, and averages.
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
# Alphabetical, not by score. A place page is read by someone looking for
# a school they can name, and scanning for it is what the order should
# serve. /rankings is where the league-table ordering lives, and it keeps
# sorting by metric.
if "school_name" in rows.columns:
rows = rows.sort_values("school_name", key=lambda c: c.str.lower())
averages = {
m: (None if m not in rows.columns or rows[m].dropna().empty
else float(rows[m].dropna().mean()))
for m in ("rwm_expected_pct", "attainment_8_score")
}
# dict.fromkeys, not a list: SCHOOL_COLUMNS already ends with latitude and
# longitude, so concatenating them again selected each twice and pandas
# dropped one of every duplicated pair with a "columns are not unique"
# warning. Ordered de-duplication keeps the column order and the warning
# cannot come back.
#
# nursery_provision and parliamentary_constituency are not in
# SCHOOL_COLUMNS and the place table shows both. The `in rows.columns`
# guard is what keeps a mart the pipeline has not rebuilt working: those
# two are the optional GIAS columns data_loader degrades to NULL.
cols = [c for c in dict.fromkeys(
SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
"nursery_provision",
"parliamentary_constituency",
"rwm_expected_pct", "attainment_8_score",
"total_pupils"])
if c in rows.columns]
return {
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
"count": len(place.urns),
"parent_authority": place.parent_authority,
# Every authority the place meaningfully sits in. SW19 is
# mostly Merton but partly Wandsworth; naming one asserts
# something false.
#
# The slug is null where that authority has no page of its
# own: City of London and the Isles of Scilly hold fewer
# schools than the threshold. Naming them is still right;
# linking them would be a 404.
"authorities": [
{"name": name,
"slug": (_slugify(name)
if f"authority:{_slugify(name)}" in registry
else None),
"count": n}
for name, n in place.authorities
],
# Only phases that clear the threshold, so the page links
# variants that exist rather than 404s.
"phases": [ph for ph in ("primary", "secondary")
if place.publishes_phase(ph)]},
"schools": clean_for_json(rows[cols]),
"averages": averages,
}
# Two characters. One is not a query — it matches thousands of schools and the
# response is useless, so it is not worth a round trip.
SUGGEST_MIN_QUERY = 2
@app.get("/api/suggest")
@limiter.limit("120/minute")
async def suggest_schools(
request: Request,
q: str = Query("", max_length=100),
limit: int = Query(8, ge=1, le=20),
):
"""School name suggestions, from Typesense alone.
Deliberately not a mode of /api/schools: that path filters and sorts the
full in-memory DataFrame, which is far too expensive to run per keystroke.
Nothing here returns an error for ordinary input. A short query, no
matches, or Typesense being unreachable are all 200 with an empty list —
a dropdown that quietly does not appear is the right failure for a
keystroke path, and there is no DataFrame fallback because the 25,000-row
substring scan is precisely what this endpoint exists to avoid.
120/minute rather than the default 60: a 200 ms debounce makes typing
legitimately bursty.
"""
query = q.strip()
if len(query) < SUGGEST_MIN_QUERY:
return {"suggestions": []}
return {"suggestions": suggest_schools_typesense(query, limit)}
@app.get("/api/flags")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_feature_flags(request: Request):
"""Every declared flag and its current value.
Internal only. The Next proxy denies this path, because the response names
every unreleased feature the codebase knows about — which is exactly what
shipping dark is meant to keep quiet.
"""
return flags.all_flags()
@app.get("/api/data-info")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_data_info(request: Request):
@@ -1577,28 +1086,16 @@ async def robots_txt():
return FileResponse(settings.frontend_dir / "robots.txt", media_type="text/plain")
def _serve_sitemap(name: str) -> Response:
global _sitemaps
if _sitemaps is None:
try:
_sitemaps = build_sitemaps()
except Exception as e:
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
if name not in _sitemaps:
raise HTTPException(status_code=404, detail="No such sitemap")
return Response(content=_sitemaps[name], media_type="application/xml")
@app.get("/sitemap.xml")
async def sitemap_xml():
"""Serve the sitemap index."""
return _serve_sitemap("sitemap.xml")
@app.get("/sitemaps/{name}")
async def sitemap_child(name: str):
"""Serve a child sitemap (static.xml, or schools-N.xml)."""
return _serve_sitemap(name)
"""Serve sitemap.xml for search engine indexing."""
global _sitemap_xml
if _sitemap_xml is None:
try:
_sitemap_xml = build_sitemap()
except Exception as e:
raise HTTPException(status_code=503, detail=f"Sitemap unavailable: {e}")
return Response(content=_sitemap_xml, media_type="application/xml")
@app.post("/api/admin/regenerate-sitemap")
@@ -1608,14 +1105,10 @@ async def regenerate_sitemap(
_: bool = Depends(verify_admin_api_key),
):
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
global _sitemaps, _place_registry
# Places and sitemap are rebuilt together — they read the same marts, and
# letting them drift apart would submit URLs for places that no longer
# exist.
_place_registry = None
_sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in _sitemaps.values())
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
global _sitemap_xml
_sitemap_xml = build_sitemap()
n = _sitemap_xml.count("<url>")
return {"status": "ok", "urls": n}
# Mount static files directly (must be after all routes to avoid catching API calls)
-15
View File
@@ -35,11 +35,6 @@ class Settings(BaseSettings):
# Security
admin_api_key: str = Field(default_factory=lambda: secrets.token_urlsafe(32))
rate_limit_per_minute: int = 60 # Requests per minute per IP
# A ceiling on total /api/ traffic, independent of any client identity.
# client_key trusts headers only Cloudflare can vouch for, so a caller
# reaching the origin directly could otherwise mint a fresh bucket per
# request. See GlobalRateLimitMiddleware in backend/app.py.
global_rate_limit_per_minute: int = 3000
rate_limit_burst: int = 10 # Allow burst of requests
max_request_size: int = 1024 * 1024 # 1MB max request size
@@ -47,16 +42,6 @@ class Settings(BaseSettings):
typesense_url: str = "http://localhost:8108"
typesense_api_key: str = ""
# Feature flags (Unleash). An empty unleash_url disables flags entirely and
# every flag evaluates False — the correct behaviour for local development
# and CI, and the reason no test needs a running Unleash.
unleash_url: str = ""
unleash_api_token: str = ""
unleash_app_name: str = "schoolcompare-backend"
# On a named volume, so a restart during an Unleash outage keeps
# last-known state instead of reverting a released feature to dark.
unleash_cache_directory: str = "/app/.unleash"
# Analytics
ga_measurement_id: Optional[str] = "G-J0PCVT14NY" # Google Analytics 4 Measurement ID
+1 -337
View File
@@ -18,9 +18,8 @@ from .config import settings
from .database import SessionLocal, engine
from .models import (
DimSchool, DimLocation, KS2Performance,
FactOfstedInspection, FactAdmissions, FactAdmissionDistance,
FactOfstedInspection, FactAdmissions,
FactDeprivation, FactFinance, FactPupilCharacteristics,
FactKs4Destinations, FactKs5Destinations,
)
from .ofsted_codes import ofsted_page_url, report_card_labels
from .schemas import SCHOOL_TYPE_MAP
@@ -101,58 +100,6 @@ def search_schools_typesense(query: str, limit: int = 250) -> List[int]:
return []
# The most a public endpoint will return in one response.
SUGGEST_MAX_LIMIT = 20
# Fields a suggestion row carries, and the default when the document omits an
# optional one. phase and school_type are optional in the Typesense schema.
_SUGGEST_FIELDS = ("school_name", "local_authority", "postcode",
"phase", "school_type")
def suggest_schools_typesense(query: str, limit: int = 8) -> List[dict]:
"""Autosuggest rows straight from Typesense. Never raises.
Returns documents rather than URNs, unlike search_schools_typesense, so the
caller needs no DataFrame. Every field below is already in the index — see
pipeline/scripts/sync_typesense.py — which is what makes this cheap enough
to run per keystroke.
"""
client = _get_typesense_client()
if client is None:
return []
try:
result = client.collections["schools"].documents.search({
"q": query,
"query_by": "school_name,local_authority",
"per_page": max(1, min(limit, SUGGEST_MAX_LIMIT)),
"typo_tokens_threshold": 1,
})
except Exception:
# A dropdown that quietly stops appearing is the right failure here.
return []
rows = []
for hit in result.get("hits", []) or []:
doc = (hit or {}).get("document") or {}
try:
urn = int(doc["urn"])
except (KeyError, TypeError, ValueError):
# Skip the row, keep the rest. Typesense declares urn as int32 so
# this should be unreachable, but the index is a separate system
# that something other than this code can reindex — and "never
# raises" is a promise the keystroke path actually depends on.
# Dropping one malformed document is right; blanking the whole
# dropdown, or serving a suggestion pointing at /school/0, is not.
logging.getLogger(__name__).warning(
"skipping malformed suggestion document: %r", doc)
continue
row = {"urn": urn}
row.update({f: str(doc.get(f, "") or "") for f in _SUGGEST_FIELDS})
rows.append(row)
return rows
def normalize_school_type(school_type: Optional[str]) -> Optional[str]:
"""Convert cryptic school type codes to user-friendly names."""
if not school_type:
@@ -776,17 +723,6 @@ def _admissions_row_dict(a) -> dict:
}
def _admission_distance_dict(d) -> dict:
"""Serialize one fact_admission_distance row for API responses."""
return {
"year": d.year,
"distance_m": d.distance_m,
"route_count": d.route_count,
"la_name": d.la_name,
"distance_unit_raw": d.distance_unit_raw,
}
def _census_dict(pc) -> dict:
return {
"year": pc.year,
@@ -817,230 +753,16 @@ def _finance_dict(f) -> dict:
}
# Destination measures that are totals DfE published itself, rather than one of
# the categories that partition the cohort.
_AGGREGATE_MEASURES = {"agg_sustained_education", "agg_sustained_all"}
def _format_cohort_year(year) -> str | None:
"""202223 -> '2022/23'.
The section has to date its own cohort. Destination measures run about two
GCSE years behind the results shown above them on the same page, so an
undated figure reads as stale data rather than as a different question.
"""
if not year:
return None
text = str(year)
if len(text) == 6:
return f"{text[:4]}/{text[4:6]}"
if len(text) == 8:
return f"{text[:4]}/{text[6:8]}"
return text
_PUPIL_GROUPS = ("disadvantaged", "other", "all")
def _lone_hidden_groups(groups: dict) -> list:
"""Pupil groups hiding exactly one category — solvable by subtraction."""
return [
key for key, group in groups.items()
if sum(1 for c in group["categories"] if c["status"] == "suppressed") == 1
]
def _lone_hidden_categories(groups: dict) -> list:
"""Categories hidden in exactly one of several pupil groups."""
lone = []
categories = {c["category"] for g in groups.values() for c in g["categories"]}
for category in categories:
found = [
c for g in groups.values() for c in g["categories"]
if c["category"] == category
]
hidden = [c for c in found if c["status"] == "suppressed"]
if len(hidden) == 1 and len(found) > 1:
lone.append(category)
return lone
def disclosure_invariant_holds(groups: dict) -> bool:
"""Every row and every column hides none, or at least two.
Public so the tests can assert it directly rather than re-deriving it.
"""
return not _lone_hidden_groups(groups) and not _lone_hidden_categories(groups)
def _mask_for_disclosure(groups: dict) -> None:
"""Withhold further cells until nothing suppressed can be solved for.
Not rendering a figure is not the same as not publishing it. This endpoint
is public and unauthenticated, so anything left in the payload is
published, whatever the UI chooses to draw — the same reasoning the
admission_distance field carries in app.py.
Two identities let a caller solve for a withheld cell:
* within a pupil group, the categories sum to the cohort, so a group with
exactly ONE suppressed category gives it away as cohort - sum(rest);
* across groups, disadvantaged + other = all for every category, so a
category suppressed in exactly ONE of the three gives itself away.
DfE's own answer is secondary suppression: withhold a second cell so the
residual spans two unknowns and identifies neither.
Where no companion can do that — a sparse cohort whose every other category
is `not_applicable`, which is common in special schools and alternative
provision — there is nothing left to withhold, so the pupil group is
DROPPED entirely. An earlier version simply gave up here and returned with
the violation intact and no signal, which is the one outcome this function
must never produce: a disclosure-control pass that fails silently is worse
than none, because everything downstream trusts it.
Mutates `groups` in place. Guaranteed to return with
disclosure_invariant_holds(groups) true.
"""
def suppress(cell):
if cell["status"] == "published":
cell["status"] = "suppressed"
cell["pupils"] = None
cell["percentage"] = None
return True
return False
def add_companion(candidates) -> bool:
"""Withhold a second cell so the residual spans two unknowns.
The companion must carry pupils. Suppressing a zero looks like
secondary suppression and protects nothing: the residual still equals
the original withheld figure exactly. Returns False when no cell can
do the job, which escalates to dropping the group.
"""
published = [c for c in candidates if c["status"] == "published"]
useful = sorted(
(c for c in published if (c["pupils"] or 0) > 0),
key=lambda c: c["pupils"],
)
if useful:
return suppress(useful[0])
# Every remaining cell is zero or not applicable: withholding any of
# them leaves the residual equal to the original figure.
return False
# Fixpoint: each new suppression can break the other identity. Terminates
# because every pass either adds a suppression, drops a group, or stops.
while not disclosure_invariant_holds(groups):
changed = False
for category in _lone_hidden_categories(groups):
siblings = [
c for g in groups.values() for c in g["categories"]
if c["category"] == category
]
if add_companion(siblings):
changed = True
for key in _lone_hidden_groups(groups):
if add_companion(groups[key]["categories"]):
changed = True
if changed:
continue
# Nothing left to withhold. Drop the groups that are still solvable,
# and any category still solvable across the groups that remain.
for key in _lone_hidden_groups(groups):
del groups[key]
changed = True
for category in _lone_hidden_categories(groups):
for group in groups.values():
for cell in group["categories"]:
if cell["category"] == category and suppress(cell):
changed = True
if not changed:
# Unreachable given the two escalations above, but a masking pass
# must never spin or exit unsafely. Withhold everything.
groups.clear()
return
def _destinations_block(rows: list) -> dict | None:
"""Shape destination rows for one phase into the API's block.
Applies secondary suppression before returning, so no caller of this public
endpoint can solve for a figure DfE withheld. See _mask_for_disclosure.
Aggregate measures are dropped entirely. DfE publishes them, and they would
be useful for a "what is published for this group" fallback, but nothing
renders them today and an aggregate spanning exactly one suppressed
component names that component. An unused field that leaks is not a
trade-off worth carrying — re-add them with their own guard if the fallback
is ever built.
Deliberately computes no residual, no "remaining pupils" figure, and no
total that would close a gap left by a suppressed category.
"""
if not rows:
return None
years = [r["year"] for r in rows if r.get("year") is not None]
if not years:
return None
latest_year = max(years)
rows = [r for r in rows if r.get("year") == latest_year]
groups: dict = {}
for row in rows:
group = groups.setdefault(
row["pupil_group"],
{"cohort": row.get("cohort_pupils"), "categories": []},
)
measure = row["destination_measure"]
published = row.get("status") == "published"
# Belt and braces: percentage is derived from the same source cell as
# pupils, but publishing one without the other would hand back the
# cohort (pupils / percentage) and with it the residual.
cell = {
"category": measure,
"pupils": row.get("pupils") if published else None,
"percentage": row.get("percentage") if published else None,
"status": row.get("status"),
}
if measure in _AGGREGATE_MEASURES:
continue
group["categories"].append(cell)
if not groups:
return None
_mask_for_disclosure(groups)
# Masking can empty the block entirely — a sparse cohort where no group
# could be made safe. Return None so the section is absent rather than
# rendering an empty shell.
if not groups:
return None
return {"cohort_year": _format_cohort_year(latest_year), "groups": groups}
def _empty_supplementary() -> dict:
return {
"ofsted": None,
"census": None,
"admissions": None,
"admissions_history": [],
"admission_distance": None,
"sen_detail": None,
"phonics": None,
"deprivation": None,
"finance": None,
"destinations": None,
}
@@ -1115,32 +837,6 @@ def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict:
result[urn]["admissions"] = rows_for_urn[-1] if rows_for_urn else None
_safe(_admissions)
# Last distance offered — the latest year per URN, and only that.
#
# The mart holds every published year and the DAG keeps loading them; what
# changed is what leaves this process. Earlier years are being held back as
# a paid feature, and this API is public and unauthenticated — serving the
# history here would hand it to anyone who opened the network tab, whatever
# the page chose to render. Withholding it in the client would have been
# decoration, not a decision.
#
# Restoring it for entitled callers is a change to this function, not to
# the pipeline: fact_admission_distance is untouched and complete.
def _admission_distance():
rows = (
db.query(FactAdmissionDistance)
.filter(FactAdmissionDistance.urn.in_(urns))
.order_by(FactAdmissionDistance.urn, FactAdmissionDistance.year.desc())
.all()
)
seen = set()
for d in rows:
if d.urn in seen:
continue
seen.add(d.urn)
result[d.urn]["admission_distance"] = _admission_distance_dict(d)
_safe(_admission_distance)
# Deprivation — one row per URN.
def _deprivation():
rows = (
@@ -1168,38 +864,6 @@ def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict:
result[f.urn]["finance"] = _finance_dict(f)
_safe(_finance)
# Destinations — KS4 and 16-18. Both marts are long-format, so every row
# for a URN is collected and _destinations_block picks the latest year and
# shapes the pupil groups. A phase with no rows serialises as null rather
# than an empty shell, so the frontend renders nothing rather than an empty
# section.
def _destinations():
from collections import defaultdict
def _collect(model):
per_urn = defaultdict(list)
for r in db.query(model).filter(model.urn.in_(urns)).all():
per_urn[r.urn].append({
"year": r.year,
"pupil_group": r.pupil_group,
"destination_measure": r.destination_measure,
"cohort_pupils": r.cohort_pupils,
"pupils": r.pupils,
"percentage": r.percentage,
"status": r.status,
})
return per_urn
ks4_rows = _collect(FactKs4Destinations)
ks5_rows = _collect(FactKs5Destinations)
for urn in urns:
ks4 = _destinations_block(ks4_rows.get(urn, []))
ks5 = _destinations_block(ks5_rows.get(urn, []))
result[urn]["destinations"] = (
{"ks4": ks4, "ks5": ks5} if (ks4 or ks5) else None
)
_safe(_destinations)
return result
-135
View File
@@ -1,135 +0,0 @@
"""Feature flags: what can be switched, and what is switched right now.
Ship-dark, not a kill switch. Flags let work merge and deploy without becoming
visible; they are expected to flip about monthly, by a person, deliberately.
Nothing here does percentage rollouts or user targeting — the site has no user
identity to target.
Unleash holds the state. It does not hold the list. REGISTRY below is that
list, and it exists for three reasons: the SDK evaluates an unknown flag to
False, so without a registry that is an *undeclared* False, indistinguishable
from a typo; /api/flags needs a key set to return when Unleash is unreachable;
and a flag in the UI but not in the registry is orphaned and should be visibly
so rather than quietly authoritative.
Every flag defaults to False. There is no per-flag default, because a flag that
defaults on is a kill switch, and this is not one.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from datetime import date
from .config import settings
logger = logging.getLogger(__name__)
# A flag is temporary scaffolding. See test_a_flag_older_than_the_limit.
MAX_FLAG_AGE_DAYS = 90
@dataclass(frozen=True)
class Flag:
# One string: the registry key, the Unleash flag name, and the JSON key in
# /api/flags. snake_case, matching the API's existing convention. No case
# transformation anywhere, so there is no mapping layer to get wrong.
name: str
description: str # one line: what turning this on reveals
added: date # for the staleness tripwire
REGISTRY: dict[str, Flag] = {
f.name: f for f in (
Flag(
name="admission_distance",
description=(
"The last-distance-offered figure on the Admissions tile and "
"the 'How far away are you?' section on school pages."
),
added=date(2026, 8, 23),
),
Flag(
name="school_autosuggest",
description=(
"School name suggestions as you type in the main search box."
),
added=date(2026, 8, 26),
),
Flag(
name="about_page",
description=(
"The /about page, its footer link, its sitemap entry, and the "
"named-author byline on every blog post."
),
added=date(2026, 9, 8),
),
Flag(
name="blog",
description=(
"The /blog index, post pages, the RSS feed, their footer link "
"and their sitemap entries. Not /admin: posts must be "
"writable before the blog is readable."
),
added=date(2026, 9, 8),
),
)
}
_client = None
def init() -> None:
"""Start the Unleash client, or log why flags are all off.
Called once from the app lifespan. Never raises: a flag system that can
stop the API from booting is worse than one that is switched off.
"""
global _client
if not settings.unleash_url or not settings.unleash_api_token:
logger.warning(
"Unleash is not configured (UNLEASH_URL / UNLEASH_API_TOKEN); "
"every feature flag evaluates to False.")
return
try:
from UnleashClient import UnleashClient
_client = UnleashClient(
url=settings.unleash_url,
app_name=settings.unleash_app_name,
custom_headers={"Authorization": settings.unleash_api_token},
cache_directory=settings.unleash_cache_directory,
refresh_interval=15,
)
_client.initialize_client()
logger.info("Unleash client initialised against %s", settings.unleash_url)
except Exception:
# Fail closed and keep serving. The SDK also evaluates everything False
# until its first successful sync, so this is the same direction.
_client = None
logger.exception("Unleash client failed to start; flags are all False.")
def is_enabled(name: str) -> bool:
"""Whether `name` is on. False for anything unknown, unreachable or broken."""
if name not in REGISTRY:
logger.error(
"undeclared feature flag %r was evaluated; returning False. "
"Add it to backend/flags.py REGISTRY or fix the name.", name)
return False
if _client is None:
return False
try:
return bool(_client.is_enabled(
name, fallback_function=lambda feature_name, context: False))
except Exception:
logger.exception("flag %r failed to evaluate; returning False", name)
return False
def all_flags() -> dict[str, bool]:
"""Every declared flag and its current value. Serves /api/flags."""
return {name: is_enabled(name) for name in REGISTRY}
-54
View File
@@ -1,54 +0,0 @@
"""Curated London localities, defined by the postcode districts they cover.
The GIAS `town` field puts 1,819 London schools under the single value
"London", so it cannot answer "schools in Battersea" — a query that appears in
the Search Console baseline. No single field can: parliamentary constituency
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
and colloquial rather than administrative.
So this is curated. Where a locality ends is a judgement, not a fact, and a
reviewable file is the honest place for a judgement. No new ingestion is
needed — the corpus already carries postcodes.
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
mirrors it for anyone querying the warehouse directly; the backend image does
not contain `pipeline/`, which is why the module rather than the seed is
canonical. Same arrangement as `backend/gias_codes.py`.
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
published, so a typo produces no page rather than an empty one. Places that
fail that check are logged at startup, because a locality you meant to publish
quietly not appearing is the failure worth hearing about.
Two rules for anything added here.
**Sub-borough districts only.** A London borough is a local authority and
already has a page at /schools/authority/[la] covering all of its schools; a
locality defined by two or three outcodes would be a partial, near-duplicate
subset of it. Hackney, Islington, Greenwich and Ealing were all in the first
draft for that reason and have been removed.
**The slug must not match a GIAS town.** "Richmond" did — GIAS has a Richmond
in North Yorkshire with 37 schools — so the London one could never publish.
The registry skips any locality that collides and logs it.
"""
# slug -> (display name, outcodes)
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
"battersea": ("Battersea", ("SW11",)),
"canary-wharf": ("Canary Wharf", ("E14",)),
"clapham": ("Clapham", ("SW4",)),
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
"peckham": ("Peckham", ("SE15",)),
"brixton": ("Brixton", ("SW2", "SW9")),
"camden-town": ("Camden Town", ("NW1",)),
"wimbledon": ("Wimbledon", ("SW19",)),
"putney": ("Putney", ("SW15",)),
"fulham": ("Fulham", ("SW6",)),
"chiswick": ("Chiswick", ("W4",)),
"stratford": ("Stratford", ("E15",)),
"walthamstow": ("Walthamstow", ("E17",)),
"tooting": ("Tooting", ("SW17",)),
"dulwich": ("Dulwich", ("SE21", "SE22")),
}
-76
View File
@@ -188,37 +188,6 @@ class FactAdmissions(Base):
admissions_policy = Column(String(100))
class FactAdmissionDistance(Base):
"""Last distance offered — one row per URN per year.
Separate from FactAdmissions because the source is separate: EES publishes
admissions for the whole country, whereas cut-off distances exist only for
the local authorities that choose to publish them (57 at the time of
writing), and the two refresh on unrelated timetables.
"""
__tablename__ = "fact_admission_distance"
__table_args__ = (
Index("ix_admission_distance_urn_year", "urn", "year"),
MARTS,
)
urn = Column(Integer, primary_key=True)
year = Column(Integer, primary_key=True)
# Straight-line distance in metres from the school to the last home offered
# a place that year.
distance_m = Column(Float)
# How many admission routes (ability bands, separate reception/junior
# intakes) were collapsed into distance_m. >1 means the figure is the
# furthest of several and the page must say so.
route_count = Column(Integer)
la_code = Column(Integer)
la_name = Column(String(100))
# The unit the council published in, so the page can lead with the unit a
# parent was given rather than always converting.
distance_unit_raw = Column(String(20))
source_file = Column(Text)
class FactPupilCharacteristics(Base):
"""School pupil composition from EES census — one row per URN per year."""
__tablename__ = "fact_pupil_characteristics"
@@ -321,48 +290,3 @@ class Ks2NationalAverage(Base):
gps_high_pct = Column(Float)
gps_avg_score = Column(Float)
science_expected_pct = Column(Float)
class FactKs4Destinations(Base):
"""KS4 leavers destinations — one row per URN, year, pupil group, measure.
Long format rather than wide because pupil_group is a real third dimension.
`status` is load-bearing: 'suppressed' means DfE withheld a figure it
considered disclosive and the page must print "withheld"; 'not_applicable'
means the measure does not apply and the page must print nothing. `pupils`
is null for both, so collapsing status to a null check loses the
difference — and the categories sum to the cohort, so a consumer that
treats a withheld cell as zero republishes what DfE hid.
"""
__tablename__ = "fact_ks4_destinations"
__table_args__ = (
Index("ix_ks4_dest_urn_year", "urn", "year"),
MARTS,
)
urn = Column(Integer, primary_key=True)
year = Column(Integer, primary_key=True)
pupil_group = Column(String(20), primary_key=True)
destination_measure = Column(String(40), primary_key=True)
cohort_pupils = Column(Integer)
pupils = Column(Integer)
percentage = Column(Float)
status = Column(String(20))
class FactKs5Destinations(Base):
"""16-18 study leavers destinations — same grain as FactKs4Destinations."""
__tablename__ = "fact_ks5_destinations"
__table_args__ = (
Index("ix_ks5_dest_urn_year", "urn", "year"),
MARTS,
)
urn = Column(Integer, primary_key=True)
year = Column(Integer, primary_key=True)
pupil_group = Column(String(20), primary_key=True)
destination_measure = Column(String(40), primary_key=True)
cohort_pupils = Column(Integer)
pupils = Column(Integer)
percentage = Column(Float)
status = Column(String(20))
-356
View File
@@ -1,356 +0,0 @@
"""The place registry: what places the site publishes, and what is in each.
One module owns this question. The pages, the sitemap and the internal-link
modules all read from here, so the threshold and the collision rules exist in
exactly one place and are testable without a browser or a database.
Two namespaces, never one. 67 viable town names collide with a local
authority name, and the authority is the larger set in only 43 of them —
postal towns cross authority boundaries, so neither can absorb the other.
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
"""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass, field
logger = logging.getLogger(__name__)
# Five schools with publishable data. Below this a place has nothing to say
# that a list of schools does not, and publishing it is index bloat.
MIN_SCHOOLS = 5
@dataclass(frozen=True)
class Place:
kind: str # "town" | "locality" | "authority" | "outcode"
slug: str
name: str
urns: tuple[int, ...]
parent_authority: str | None # authority NAME, for the 301 target
# Every authority the place meaningfully sits in, largest first. A quarter
# of outcodes and a third of towns straddle a boundary — SW19 is mostly
# Merton but partly Wandsworth — so naming only one asserts something
# false. parent_authority stays single because a redirect needs one
# target; this is what the page shows.
authorities: tuple[tuple[str, int], ...] = ()
# URNs per phase, so the per-phase threshold can be applied without
# re-querying. A place with 30 primaries and 2 secondaries publishes a
# primary variant and no secondary one.
phase_urns: dict[str, tuple[int, ...]] = field(default_factory=dict)
def publishes_phase(self, phase: str) -> bool:
return len(self.phase_urns.get(phase, ())) >= MIN_SCHOOLS
@property
def key(self) -> str:
return f"{self.kind}:{self.slug}"
def _publishable_urns(df) -> set[int]:
"""URNs with something a page could state, deduplicated across years."""
from backend.app import _PUBLISHABLE_FIELDS
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
if not cols:
return set()
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
# The measure a phase page is built around. A page with no results in this
# column has nothing a list of school names does not already give.
_PHASE_METRIC = {
"primary": "rwm_expected_pct",
"secondary": "attainment_8_score",
}
def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]:
"""URNs per phase, counting only schools with a result for that phase.
Not merely "publishable". A school with an Ofsted grade and no results is
worth a page of its own and belongs in the place list, but it cannot
populate a phase page's results column — and the threshold is there to ask
whether that column will have anything in it.
Counting publishable schools instead let /schools/kent/primary publish
with none of its five rows carrying a result, and left 44 phase pages
majority-blank. It is the same rule as "no page without a local average",
which was never extended per phase.
All-through schools count toward both phases, matching the PHASE_GROUPS
mapping the search filters already use.
"""
from backend.app import PHASE_GROUPS
if "phase" not in group.columns:
return {}
lowered = group["phase"].fillna("").str.lower()
out: dict[str, tuple[int, ...]] = {}
for phase in ("primary", "secondary"):
wanted = PHASE_GROUPS.get(phase, set())
subset = group[lowered.isin(wanted)]
# The page lists every school of the phase; the threshold counts only
# those carrying a result, so a mostly-empty table never publishes.
metric = _PHASE_METRIC[phase]
with_result = (
{int(u) for u in subset.loc[subset[metric].notna(), "urn"]}
if metric in subset.columns else set()
)
if len(with_result & publishable) < MIN_SCHOOLS:
continue
urns = tuple(sorted({int(u) for u in subset["urn"]} & publishable))
if urns:
out[phase] = urns
return out
# A place is described by an authority when it holds at least a tenth of the
# schools, and at least two. GIAS carries occasional postcode errors — EN6
# lists two Shropshire schools among fourteen in Hertfordshire — and a bare
# "any authority present" rule would print those as though they were real.
# There is deliberately no cap on how many are named. An earlier cut stopped
# at three, which silently dropped the fourth in exactly the case where the
# information matters most — a genuinely fragmented place. The share rule is
# the only limit, and it already bounds the list at ten.
_AUTHORITY_MIN_SHARE = 0.10
_AUTHORITY_MIN_SCHOOLS = 2
def _authorities(group) -> tuple[tuple[str, int], ...]:
"""Authorities this place meaningfully sits in, largest first."""
from backend.app import EXCLUDED_FILTER_VALUES
if "local_authority" not in group.columns:
return ()
counts = group["local_authority"].dropna().value_counts()
total = int(counts.sum())
if not total:
return ()
kept = [
(str(name), int(n)) for name, n in counts.items()
if str(name) not in EXCLUDED_FILTER_VALUES
and n >= _AUTHORITY_MIN_SCHOOLS
and n / total >= _AUTHORITY_MIN_SHARE
]
# A place too small or too fragmented for the share rule still names its
# largest authority, or the page would say nothing about where it is.
if not kept:
for name, n in counts.items():
if str(name) not in EXCLUDED_FILTER_VALUES:
return ((str(name), int(n)),)
return ()
return tuple(kept)
def _parent_authority(authorities: tuple[tuple[str, int], ...]) -> str | None:
"""The 301 target: the largest authority a place sits in.
Derived from `authorities` rather than computed separately. The first cut
used `mode()` here while `authorities` used `value_counts()`, and on an
exact tie pandas does not guarantee the two pick the same name — so the
redirect could have pointed somewhere other than the authority the page
named first. One computation, one answer.
Deriving it also inherits the sentinel filter, so a place can no longer
redirect to /schools/authority/does-not-apply.
"""
return authorities[0][0] if authorities else None
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
"""One Place per distinct SLUG in `column` that clears the threshold.
Grouped by slug, not by raw value, because GIAS spells the same place
several ways and they all resolve to one URL. Five town slugs come from
more than one spelling: "London" (1,819 schools) and "LONDON" (12) both
slugify to `london`; Weston-super-Mare is split 14/19 across two
spellings; Newcastle-under-Lyme across three.
Grouping by raw value meant the later group simply overwrote the earlier
one in this dict — so /schools/london could have shown twelve schools
instead of 1,819, silently and depending on row order.
The display name is the most common spelling, which is the one a reader
expects to see.
"""
from backend.app import _slugify
if column not in df.columns:
return {}
working = df.assign(_slug=df[column].map(
lambda v: _slugify(str(v).strip()) if isinstance(v, str) and v.strip() else None))
working = working[working["_slug"].notna() & (working["_slug"] != "")]
out: dict[str, Place] = {}
for slug, group in working.groupby("_slug"):
slug = str(slug)
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
spellings = group[column].dropna().value_counts()
if spellings.empty:
continue
name = str(spellings.index[0]).strip()
authorities = () if kind == "authority" else _authorities(group)
place = Place(
kind=kind, slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(authorities),
authorities=authorities,
phase_urns=_phase_urns(group, publishable),
)
out[place.key] = place
return out
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
def _outcode(postcode) -> str | None:
if not isinstance(postcode, str):
return None
m = _OUTCODE_RE.match(postcode.upper().strip())
return m.group(1) if m else None
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
"""One Place per postcode district clearing the threshold.
These carry no phase variants: nobody searches "primary schools in SW11",
so the spec gives them no /primary or /secondary route. `phase_urns` is
left empty rather than computed and then filtered downstream — the
registry is the one place that decides which phases a place publishes,
and the page links whatever it reports.
Computing them here put a link to a route that does not exist on every one
of the 1,720 outcode pages.
"""
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
working = working[working["_oc"].notna()]
out: dict[str, Place] = {}
for oc, group in working.groupby("_oc"):
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
continue
authorities = _authorities(group)
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
urns=urns, parent_authority=_parent_authority(authorities),
authorities=authorities)
out[place.key] = place
return out
def _locality_places(df, publishable: set[int],
town_slugs: set[str]) -> dict[str, Place]:
"""One Place per curated locality clearing the threshold."""
from backend.localities import LOCALITY_OUTCODES
if "postcode" not in df.columns:
return {}
working = df.assign(_oc=df["postcode"].map(_outcode))
out: dict[str, Place] = {}
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
if slug in town_slugs:
# Skip, do not raise. The guard exists so a locality never
# silently shadows a town — skipping achieves that, and the error
# log makes it loud.
#
# Raising here took down sitemap generation for all 25,000 school
# pages when "richmond" met the GIAS town Richmond in North
# Yorkshire. Worse, GIAS town names change without any code change,
# so a raise means curated data can break the site spontaneously.
# A curation mistake must cost one page, not the sitemap.
logger.error(
"locality %r collides with the published town of the same "
"slug and has been skipped; rename it or remove it", slug)
continue
group = working[working["_oc"].isin(outcodes)]
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
if len(urns) < MIN_SCHOOLS:
# Not an error — a locality can legitimately be too small. Logged
# because one you meant to publish quietly vanishing is the
# failure worth hearing about.
logger.warning(
"locality %s (%s) has %d publishable schools, below the "
"threshold of %d - not published",
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
continue
authorities = _authorities(group)
place = Place(kind="locality", slug=slug, name=name, urns=urns,
parent_authority=_parent_authority(authorities),
authorities=authorities,
phase_urns=_phase_urns(group, publishable))
out[place.key] = place
return out
# Ordered authority → town/locality → outcode, widest first, because that is
# the order a breadcrumb reads. The link module re-sorts for its own purposes.
_PLACE_ORDER = {"authority": 0, "town": 1, "locality": 2, "outcode": 3}
def build_place_index(registry: dict[str, Place]) -> dict[int, tuple[Place, ...]]:
"""URN → the published places containing it, built once per registry.
The reverse of the registry, and the thing school pages link out through.
Derived from the registry rather than maintained beside it, so the two
cannot disagree about which places exist: a place below the publish
threshold is absent from the registry, so it is absent from here too, and
a link is never offered for a page that does not exist.
Built as an index rather than scanned per call because /api/schools/{urn}
is the site's highest-traffic endpoint. Scanning meant walking every place
and doing a tuple membership test against each — on the order of 10^5
comparisons per request, repeated for every school page view. One pass at
registry-build time replaces all of it with a dict lookup.
"""
grouped: dict[int, list[Place]] = {}
for place in registry.values():
for urn in place.urns:
grouped.setdefault(int(urn), []).append(place)
return {
urn: tuple(sorted(places,
key=lambda p: (_PLACE_ORDER.get(p.kind, 9), p.slug)))
for urn, places in grouped.items()
}
def places_for_urn(index: dict[int, tuple[Place, ...]], urn: int) -> tuple[Place, ...]:
"""The published places containing this school, widest first.
Empty is a real answer, not a failure: a school whose town and authority
both fall below the publish threshold has nowhere to link, and the page
renders without the module.
"""
return index.get(int(urn), ())
def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns:
return {}
publishable = _publishable_urns(df)
registry: dict[str, Place] = {}
registry.update(_group(df, "local_authority", "authority", publishable))
towns = _group(df, "town", "town", publishable)
registry.update(towns)
town_slugs = {p.slug for p in towns.values()}
registry.update(_locality_places(df, publishable, town_slugs))
registry.update(_outcode_places(df, publishable))
return registry
-269
View File
@@ -1,269 +0,0 @@
"""The destinations serialiser's contract.
Not rendering a figure is not the same as not publishing it. This endpoint is
public and unauthenticated, so whatever the payload carries is published,
whatever the UI draws. The categories sum to the cohort and the pupil groups
sum to each other, so a lone suppressed cell is solvable by subtraction — the
serialiser adds secondary suppression to prevent it.
See docs/superpowers/specs/2026-08-28-destination-measures-design.md.
"""
from backend.data_loader import (
_destinations_block, _format_cohort_year, disclosure_invariant_holds,
)
def _row(group, measure, pupils, status, cohort=180, percentage=None, year=202223):
return {
"pupil_group": group,
"destination_measure": measure,
"pupils": pupils,
"percentage": percentage,
"status": status,
"cohort_pupils": cohort,
"year": year,
}
def test_suppressed_category_serialises_as_suppressed_with_null_pupils():
rows = [
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
_row("all", "sixth_form_college", None, "suppressed"),
]
block = _destinations_block(rows)
cats = {c["category"]: c for c in block["groups"]["all"]["categories"]}
assert cats["sixth_form_college"]["status"] == "suppressed"
assert cats["sixth_form_college"]["pupils"] is None
assert cats["sixth_form_college"]["percentage"] is None
def test_published_category_keeps_its_figures():
block = _destinations_block([
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
])
cat = block["groups"]["all"]["categories"][0]
assert cat["pupils"] == 75
assert cat["percentage"] == 41.7
assert cat["status"] == "published"
def test_only_the_latest_year_is_served():
rows = [
_row("all", "school_sixth_form", 60, "published", year=202122),
_row("all", "school_sixth_form", 75, "published", year=202223),
]
block = _destinations_block(rows)
assert block["cohort_year"] == "2022/23"
assert len(block["groups"]["all"]["categories"]) == 1
assert block["groups"]["all"]["categories"][0]["pupils"] == 75
def test_all_three_pupil_groups_are_carried():
rows = [
_row("all", "school_sixth_form", 75, "published"),
_row("disadvantaged", "school_sixth_form", 17, "published", cohort=62),
_row("other", "school_sixth_form", 58, "published", cohort=118),
]
block = _destinations_block(rows)
assert set(block["groups"]) == {"all", "disadvantaged", "other"}
assert block["groups"]["disadvantaged"]["cohort"] == 62
def test_cohort_year_is_reported_so_the_page_can_date_itself():
block = _destinations_block([_row("all", "school_sixth_form", 75, "published")])
assert block["cohort_year"] == "2022/23"
def test_format_cohort_year_handles_the_six_digit_form():
assert _format_cohort_year(202223) == "2022/23"
assert _format_cohort_year(None) is None
def test_empty_rows_yield_none_not_an_empty_shell():
assert _destinations_block([]) is None
# ── Disclosure control ──────────────────────────────────────────────────────
#
# The rendering guards in lib/destinations.ts stop a withheld figure being
# DRAWN. They do nothing about it being COMPUTED: this endpoint is public and
# unauthenticated, so whatever the payload carries is published. These tests
# are the ones that matter.
def _solve_residual(group):
"""What any caller can work out: cohort minus everything published."""
published = [c["pupils"] for c in group["categories"] if c["pupils"] is not None]
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
return group["cohort"] - sum(published), len(hidden)
def test_a_lone_suppressed_category_cannot_be_solved_for():
"""Whitley Bay High School's real 2022/23 disadvantaged group: further
education withheld, everything else published, cohort 41. Before secondary
suppression the payload gave the answer away as 41 - 23 = 18."""
rows = [
_row("disadvantaged", "school_sixth_form", 15, "published", cohort=41),
_row("disadvantaged", "sixth_form_college", 0, "published", cohort=41),
_row("disadvantaged", "further_education", None, "suppressed", cohort=41),
_row("disadvantaged", "apprenticeship", 1, "published", cohort=41),
_row("disadvantaged", "employment", 2, "published", cohort=41),
_row("disadvantaged", "not_sustained", 3, "published", cohort=41),
_row("disadvantaged", "not_captured", 2, "published", cohort=41),
]
group = _destinations_block(rows)["groups"]["disadvantaged"]
residual, hidden = _solve_residual(group)
assert hidden >= 2, "a lone suppressed cell must gain a companion"
assert residual != 18, "the withheld figure is recoverable from the payload"
def test_every_group_hides_none_or_at_least_two_categories():
rows = [
_row("all", "school_sixth_form", 75, "published"),
_row("all", "sixth_form_college", None, "suppressed"),
_row("all", "further_education", 61, "published"),
_row("all", "apprenticeship", 8, "published"),
_row("all", "employment", 6, "published"),
_row("all", "not_sustained", 5, "published"),
_row("all", "not_captured", 4, "published"),
]
group = _destinations_block(rows)["groups"]["all"]
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
assert len(hidden) >= 2
def test_a_category_hidden_in_one_group_is_hidden_in_a_second():
"""disadvantaged + other = all for every category, so a category withheld
in exactly one of the three is recoverable from the other two."""
rows = []
for measure, a, d, o in [
("school_sixth_form", 75, None, 58),
("further_education", 61, 27, 34),
("apprenticeship", 8, 4, 4),
("employment", 6, 1, 5),
("not_sustained", 5, 3, 2),
("not_captured", 4, 2, 2),
]:
rows.append(_row("all", measure, a, "published", cohort=159))
rows.append(_row("disadvantaged", measure, d,
"published" if d is not None else "suppressed", cohort=37))
rows.append(_row("other", measure, o, "published", cohort=122))
groups = _destinations_block(rows)["groups"]
measures = {c["category"] for g in groups.values() for c in g["categories"]}
assert len(measures) == 6, "the fixture's six measures must all be checked"
for measure in sorted(measures):
hidden = sum(
1 for g in groups.values() for c in g["categories"]
if c["category"] == measure and c["status"] == "suppressed"
)
# The invariant is "none, or at least two" — not "at least two".
assert hidden != 1, f"{measure} is solvable across the pupil groups"
def test_a_suppressed_cell_never_keeps_its_percentage():
"""percentage / pupils would hand back the cohort, and with it the residual."""
rows = [
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
_row("all", "sixth_form_college", None, "suppressed", percentage=11.7),
_row("all", "further_education", 61, "published", percentage=33.9),
]
group = _destinations_block(rows)["groups"]["all"]
for cell in group["categories"]:
if cell["status"] != "published":
assert cell["pupils"] is None
assert cell["percentage"] is None
def test_aggregates_are_not_served():
"""An aggregate spanning exactly one suppressed component names it, and
nothing renders them today."""
rows = [
_row("all", "school_sixth_form", 75, "published"),
_row("all", "agg_sustained_all", 171, "published"),
]
group = _destinations_block(rows)["groups"]["all"]
assert [c["category"] for c in group["categories"]] == ["school_sixth_form"]
assert "aggregates" not in group
def test_a_fully_published_group_is_left_alone():
"""Secondary suppression must not cost anything where nothing is withheld —
this is the all-pupils view on every mainstream secondary."""
rows = [
_row("all", m, p, "published")
for m, p in [("school_sixth_form", 75), ("sixth_form_college", 21),
("further_education", 61), ("apprenticeship", 8),
("employment", 6), ("not_sustained", 5), ("not_captured", 4)]
]
group = _destinations_block(rows)["groups"]["all"]
assert all(c["status"] == "published" for c in group["categories"])
assert len(group["categories"]) == 7
def test_the_invariant_is_asserted_directly_not_re_derived():
"""A group with one suppressed category and nothing else to withhold."""
rows = [
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
_row("all", "further_education", None, "not_applicable", cohort=9),
]
block = _destinations_block(rows)
assert block is None or disclosure_invariant_holds(block["groups"])
def test_a_sparse_cohort_with_no_companion_drops_the_group():
"""Special schools and AP routinely have one suppressed category and every
other one not applicable. There is nothing left to withhold, so the group
goes — an earlier version returned here with the violation intact."""
rows = [
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
_row("all", "further_education", None, "not_applicable", cohort=9),
_row("all", "apprenticeship", None, "not_applicable", cohort=9),
_row("all", "employment", None, "not_applicable", cohort=9),
_row("all", "not_sustained", None, "not_applicable", cohort=9),
_row("all", "not_captured", None, "not_applicable", cohort=9),
]
block = _destinations_block(rows)
assert block is None or "all" not in block["groups"], (
"a group that cannot be made safe must not be served"
)
def test_zeros_are_not_treated_as_a_usable_companion():
"""Suppressing a zero protects nothing — the residual is unchanged. With
only zeros available the group must be dropped, not falsely 'fixed'."""
rows = [
_row("all", "school_sixth_form", None, "suppressed", cohort=5),
_row("all", "sixth_form_college", 0, "published", cohort=5),
_row("all", "further_education", 0, "published", cohort=5),
]
block = _destinations_block(rows)
if block and "all" in block["groups"]:
group = block["groups"]["all"]
published = sum(c["pupils"] for c in group["categories"]
if c["pupils"] is not None)
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
assert len(hidden) != 1, "a zero companion leaves the figure solvable"
assert group["cohort"] - published != 5
def test_masking_always_terminates_in_a_safe_state():
"""Exhaustive over every suppression pattern of a four-category group."""
from itertools import product
MEASURES = ["school_sixth_form", "sixth_form_college",
"further_education", "apprenticeship"]
for statuses in product(["published", "suppressed", "not_applicable"],
repeat=len(MEASURES)):
rows = [
_row("all", m, 3 if st == "published" else None, st, cohort=12)
for m, st in zip(MEASURES, statuses)
]
block = _destinations_block(rows)
if block is None:
continue
assert disclosure_invariant_holds(block["groups"]), (
f"invariant broken for {statuses}"
)
-163
View File
@@ -1,163 +0,0 @@
"""Tests for the feature flag layer (spec 2026-08-23).
None of these need a running Unleash. That is the point: an unset UNLEASH_URL
means every flag is False, which is what local development and CI get.
"""
from datetime import date, timedelta
from backend import flags
def test_every_declared_flag_is_keyed_by_its_own_name():
# One string is the registry key, the Unleash flag name and the JSON key.
# A mismatch here would mean the UI toggles a flag the code never reads.
for key, flag in flags.REGISTRY.items():
assert key == flag.name
def test_flag_names_are_snake_case():
# Matches the API's existing convention (admission_distance,
# rwm_expected_pct) so no case transformation exists to get wrong.
for name in flags.REGISTRY:
assert name == name.lower()
assert "-" not in name and " " not in name
def test_an_unconfigured_client_evaluates_every_flag_false(monkeypatch):
monkeypatch.setattr(flags, "_client", None)
for name in flags.REGISTRY:
assert flags.is_enabled(name) is False
def test_an_undeclared_flag_is_false_rather_than_an_error(monkeypatch):
# A typo'd flag name must not raise in a request path. It is logged as an
# error, because an undeclared flag is always a bug.
monkeypatch.setattr(flags, "_client", None)
assert flags.is_enabled("no_such_flag") is False
def test_an_exploding_client_is_false_rather_than_a_500(monkeypatch):
class Boom:
def is_enabled(self, *a, **kw):
raise RuntimeError("unleash is on fire")
monkeypatch.setattr(flags, "_client", Boom())
name = next(iter(flags.REGISTRY))
assert flags.is_enabled(name) is False
def test_all_flags_reports_every_declared_flag(monkeypatch):
monkeypatch.setattr(flags, "_client", None)
assert set(flags.all_flags()) == set(flags.REGISTRY)
assert all(v is False for v in flags.all_flags().values())
def test_a_flag_older_than_the_limit_fails_this_test():
"""A tripwire, not an assertion about correctness.
Flags are temporary scaffolding and the failure mode of every flag system
is accumulation. This fails on the day a flag turns 90, on whatever PR
happens to be open — which is the point: someone has to decide.
To fix: delete the flag and the branches that read it, or, if it genuinely
still needs to exist, move its `added` date and say why in the commit.
"""
stale = [
f.name for f in flags.REGISTRY.values()
if date.today() - f.added > timedelta(days=flags.MAX_FLAG_AGE_DAYS)
]
assert not stale, (
f"Flags older than {flags.MAX_FLAG_AGE_DAYS} days: {stale}. "
"Remove the flag and the code branches it guards, or move its `added` "
"date deliberately."
)
def _client():
from fastapi.testclient import TestClient
from backend import app as app_module
return TestClient(app_module.app, raise_server_exceptions=False)
def test_the_flags_endpoint_lists_every_declared_flag(monkeypatch):
monkeypatch.setattr(flags, "_client", None)
body = _client().get("/api/flags").json()
assert set(body) == set(flags.REGISTRY)
def test_the_flags_endpoint_answers_false_when_unleash_is_unreachable(monkeypatch):
# The endpoint must still answer. A frontend that cannot read flags renders
# everything dark, which is right; one that gets a 500 renders nothing.
monkeypatch.setattr(flags, "_client", None)
res = _client().get("/api/flags")
assert res.status_code == 200
assert all(v is False for v in res.json().values())
def _school_payload(monkeypatch, *, flag_on: bool):
"""Fetch one school's payload with the distance flag forced on or off.
The DataFrame shape is copied from test_school_details.py rather than
minimised: the endpoint reads a wide set of GIAS columns, and a trimmed
frame fails for reasons that have nothing to do with flags.
"""
import numpy as np
import pandas as pd
from fastapi.testclient import TestClient
from backend import app as app_module
df = pd.DataFrame([{
"urn": 150275,
"school_name": "West London Performing Arts Academy",
"phase": "Secondary",
"school_type": "Special post 16 institution",
"trust_name": None,
"religious_denomination": "Does not apply",
"gender": None,
"age_range": "16-25",
"admissions_policy": None,
"capacity": np.nan,
"gias_total_pupils": np.nan,
"headteacher_name": None,
"website": None,
"ofsted_grade": np.nan,
"local_authority": "Ealing",
"address": "268 Northfield Avenue, London, W5 4UB",
"postcode": "W5 4UB",
"latitude": 51.4986,
"longitude": -0.3148,
"year": np.nan,
"total_pupils": np.nan,
"eligible_pupils": np.nan,
"rwm_expected_pct": np.nan,
}])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
# Two arguments: get_supplementary_data(db, urn). See backend/app.py.
monkeypatch.setattr(
app_module, "get_supplementary_data",
lambda db, urn: {"admission_distance": {"distance_m": 772.49,
"year": 2024}})
monkeypatch.setattr(flags, "is_enabled", lambda name: flag_on)
client = TestClient(app_module.app, raise_server_exceptions=False)
res = client.get("/api/schools/150275")
assert res.status_code == 200, res.text
return res.json()
def test_the_distance_field_is_absent_when_the_flag_is_off(monkeypatch):
"""Absent, not null, and withheld at the source.
/api/schools/ is public and unauthenticated. Leaving a withheld field in
the payload while declining to render it hands the record to anyone who
opens the network tab — the reasoning already recorded in c9a1892.
"""
body = _school_payload(monkeypatch, flag_on=False)
assert "admission_distance" not in body
def test_the_distance_field_is_present_when_the_flag_is_on(monkeypatch):
body = _school_payload(monkeypatch, flag_on=True)
assert body["admission_distance"]["distance_m"] == 772.49
-497
View File
@@ -1,497 +0,0 @@
"""Tests for the place registry (spec 2026-08-21).
The registry is built from the in-memory school DataFrame, so these build a
small frame directly rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
from backend.places import (MIN_SCHOOLS, build_place_index,
build_place_registry, places_for_urn)
def _df(rows: list[dict]) -> pd.DataFrame:
base = {
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
"phase": "Primary", "postcode": "AA1 1AA",
}
return pd.DataFrame([{**base, **r} for r in rows])
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
"""`start` offsets the URNs so two calls can describe different schools —
the Bedford case needs two authorities' worth of distinct URNs in one
town."""
return [
{"urn": start + i, "school_name": f"{town} School {i}",
"town": town, "local_authority": la, **kw}
for i in range(n)
]
def test_town_clearing_the_threshold_is_published():
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
assert "town:brentwood" in reg
assert reg["town:brentwood"].name == "Brentwood"
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
def test_town_below_the_threshold_is_not_published():
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
assert "town:crosby" not in reg
def test_a_town_below_threshold_still_names_its_authority():
# The route layer needs somewhere to 301 to.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
assert "authority:sefton" in reg
def test_town_and_authority_of_the_same_name_are_separate_places():
# 67 real collisions. Neither set contains the other: Bedford the town has
# 104 schools, Bedford the authority 86, because postal towns cross
# authority boundaries.
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
reg = build_place_registry(_df(rows))
town, authority = reg["town:bedford"], reg["authority:bedford"]
assert set(town.urns) != set(authority.urns)
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
for r in rows:
r["rwm_expected_pct"] = np.nan
r["ofsted_grade"] = np.nan
reg = build_place_registry(_df(rows))
assert "town:ghosttown" not in reg
def test_blank_town_is_ignored():
rows = _town(MIN_SCHOOLS, "", "Essex")
reg = build_place_registry(_df(rows))
assert not any(k.startswith("town:") for k in reg)
def test_a_school_is_counted_once_even_with_several_years_of_rows():
rows = []
for year in (202324, 202425):
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
reg = build_place_registry(_df(rows))
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
def test_locality_groups_schools_by_outcode(monkeypatch):
# The GIAS town field collapses 1,819 London schools into "London", so a
# locality is defined by its postcode districts instead.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"battersea": ("Battersea", ("SW11",))})
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
for r in rows:
r["postcode"] = "SW11 2AA"
reg = build_place_registry(_df(rows))
assert reg["locality:battersea"].name == "Battersea"
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
def test_locality_below_the_threshold_is_not_published(monkeypatch):
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"nowhere": ("Nowhere", ("ZZ99",))})
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
assert "locality:nowhere" not in reg
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch, caplog):
"""A colliding locality is skipped loudly, and the town survives.
This used to raise, which took down sitemap generation for all 25,000
school pages the first time a curated slug met a real GIAS town. Curated
data must not be able to break the site — and GIAS town names change with
no code change at all, so the raise could fire spontaneously.
"""
import logging
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"brentwood": ("Brentwood", ("CM13",))})
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
with caplog.at_level(logging.ERROR):
reg = build_place_registry(_df(rows))
assert "locality:brentwood" not in reg # skipped
assert "town:brentwood" in reg # the town is untouched
assert "brentwood" in caplog.text # and it was loud about it
def test_a_locality_collision_does_not_break_the_rest_of_the_registry(monkeypatch):
# The whole point of skipping rather than raising.
from backend import localities
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
{"brentwood": ("Brentwood", ("CM13",))})
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
reg = build_place_registry(_df(rows))
assert "authority:essex" in reg
assert "outcode:cm13" in reg
def test_outcode_places_are_built_from_postcodes():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "CM13 1AA"
reg = build_place_registry(_df(rows))
assert reg["outcode:cm13"].name == "CM13"
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
def test_malformed_postcodes_do_not_create_places():
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
for r in rows:
r["postcode"] = "not a postcode"
reg = build_place_registry(_df(rows))
assert not any(k.startswith("outcode:") for k in reg)
def test_every_curated_locality_is_structurally_valid():
# Guards the hand-maintained file: real slug, real name, real outcodes.
import re
from backend.localities import LOCALITY_OUTCODES
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
assert name.strip() == name and name, slug
assert outcodes, f"{slug} has no outcodes"
for oc in outcodes:
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
def test_the_pipeline_seed_mirrors_the_canonical_module():
"""Two copies with no drift guard is worse than one copy.
backend/localities.py is canonical because the backend image does not
contain pipeline/. The seed exists so the warehouse can join on the same
definitions, and this is what stops the two diverging — the same
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
"""
import csv
from pathlib import Path
from backend.localities import LOCALITY_OUTCODES
seed_path = (Path(__file__).resolve().parents[2]
/ "pipeline/transform/seeds/locality_outcodes.csv")
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
seed = {
row["locality_slug"]: (row["locality_name"],
tuple(row["outcodes"].split("|")))
for row in csv.DictReader(seed_path.open())
}
assert seed == LOCALITY_OUTCODES
def test_no_curated_locality_names_a_london_borough():
"""Boroughs are authorities and already have a page.
A locality defined by two or three outcodes inside a borough would be a
partial, near-duplicate subset of that authority page — the exact
thin-content failure the two-namespace design exists to avoid. Hackney,
Islington, Greenwich and Ealing were all in the first draft.
Hardcoded rather than read from the corpus because this must fail in CI,
where there is no database.
"""
from backend.localities import LOCALITY_OUTCODES
boroughs = {
"barking-and-dagenham", "barnet", "bexley", "brent", "bromley",
"camden", "croydon", "ealing", "enfield", "greenwich", "hackney",
"hammersmith-and-fulham", "haringey", "harrow", "havering",
"hillingdon", "hounslow", "islington", "kensington-and-chelsea",
"kingston-upon-thames", "lambeth", "lewisham", "merton", "newham",
"redbridge", "richmond-upon-thames", "southwark", "sutton",
"tower-hamlets", "waltham-forest", "wandsworth", "westminster",
}
named = boroughs & set(LOCALITY_OUTCODES)
assert not named, (
f"these are boroughs, not districts: {sorted(named)} - they already "
"have an authority page covering every school"
)
def test_a_place_names_every_authority_it_straddles():
"""SW19 is mostly Merton but partly Wandsworth.
A quarter of viable outcodes and a third of viable towns cross an
authority boundary, so naming only the largest asserts something false.
"""
rows = (_town(26, "London", "Merton", start=300000)
+ _town(7, "London", "Wandsworth", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
names = [n for n, _ in reg["outcode:sw19"].authorities]
assert names == ["Merton", "Wandsworth"] # largest first
assert dict(reg["outcode:sw19"].authorities)["Wandsworth"] == 7
def test_the_redirect_target_stays_a_single_authority():
# parent_authority and authorities do different jobs: a 301 needs one
# target, the page needs the truth.
rows = (_town(26, "London", "Merton", start=300000)
+ _town(7, "London", "Wandsworth", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
assert reg["outcode:sw19"].parent_authority == "Merton"
def test_a_stray_authority_below_the_share_threshold_is_not_named():
# GIAS carries postcode errors — EN6 lists two Shropshire schools among
# fourteen in Hertfordshire. Printing those as though real would be worse
# than omitting them.
rows = (_town(30, "Barnet", "Hertfordshire", start=300000)
+ _town(1, "Barnet", "Shropshire", start=400000))
for r in rows:
r["postcode"] = "EN6 1AA"
reg = build_place_registry(_df(rows))
assert [n for n, _ in reg["outcode:en6"].authorities] == ["Hertfordshire"]
def test_a_sentinel_authority_is_never_named():
rows = (_town(20, "London", "Merton", start=300000)
+ _town(6, "London", "Does not apply", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
reg = build_place_registry(_df(rows))
assert [n for n, _ in reg["outcode:sw19"].authorities] == ["Merton"]
def test_a_place_always_names_at_least_one_authority():
# Even when every authority is below the share threshold, the page has to
# say where the place is.
rows = []
for i, la in enumerate(["A", "B", "C", "D", "E", "F", "G"]):
rows += _town(1, "Fragmented", la, start=300000 + i * 100)
reg = build_place_registry(_df(rows))
place = reg.get("town:fragmented")
assert place is not None
assert len(place.authorities) == 1
def test_every_qualifying_authority_is_named_with_no_cap():
"""An earlier cut stopped at three, dropping the fourth silently.
That truncation bit exactly where the information matters most — a
genuinely fragmented place — and nothing recorded it.
"""
rows = []
for i, la in enumerate(["Hackney", "Lambeth", "Westminster", "Lewisham"]):
rows += _town(3, "Fourway", la, start=300000 + i * 100)
reg = build_place_registry(_df(rows))
assert len(reg["town:fourway"].authorities) == 4
def test_the_redirect_target_is_the_authority_named_first():
"""They were computed separately — mode() against value_counts() — and on
an exact tie pandas does not guarantee the two agree."""
rows = (_town(26, "London", "Merton", start=300000)
+ _town(7, "London", "Wandsworth", start=400000))
for r in rows:
r["postcode"] = "SW19 1AA"
place = build_place_registry(_df(rows))["outcode:sw19"]
assert place.parent_authority == place.authorities[0][0]
def test_a_place_never_redirects_to_a_sentinel_authority():
# Deriving the parent from `authorities` inherits its sentinel filter.
rows = (_town(6, "Someplace", "Does not apply", start=300000)
+ _town(5, "Someplace", "Essex", start=400000))
reg = build_place_registry(_df(rows))
assert reg["town:someplace"].parent_authority == "Essex"
def test_spellings_of_one_place_are_merged_not_overwritten():
"""GIAS spells the same place several ways, and they share a URL.
"London" (1,819 schools) and "LONDON" (12) both slugify to `london`.
Grouping by raw value let the later group overwrite the earlier one, so
the page could have shown twelve schools instead of 1,819 — silently, and
depending on row order.
"""
rows = (_town(6, "Weston-super-Mare", "North Somerset", start=300000)
+ _town(5, "Weston-Super-Mare", "North Somerset", start=400000))
reg = build_place_registry(_df(rows))
assert len(reg["town:weston-super-mare"].urns) == 11
def test_the_merged_place_takes_its_most_common_spelling():
rows = (_town(9, "Newcastle-under-Lyme", "Staffordshire", start=300000)
+ _town(5, "NEWCASTLE-UNDER-LYME", "Staffordshire", start=400000))
reg = build_place_registry(_df(rows))
assert reg["town:newcastle-under-lyme"].name == "Newcastle-under-Lyme"
def test_a_phase_page_needs_results_not_merely_publishable_schools():
"""/schools/kent/primary published with none of its five rows scored.
The threshold counted schools that were publishable — a result OR an
Ofsted grade — while the page exists for its results column. Forty-four
phase pages were majority-blank; one had no results at all.
"""
rows = _town(MIN_SCHOOLS, "Kent", "Kent")
for r in rows:
r["rwm_expected_pct"] = np.nan # Ofsted only, no results
reg = build_place_registry(_df(rows))
assert "town:kent" in reg # the place still publishes
assert not reg["town:kent"].publishes_phase("primary")
def test_a_phase_page_publishes_once_enough_schools_carry_a_result():
rows = _town(MIN_SCHOOLS, "Beccles", "Suffolk")
reg = build_place_registry(_df(rows))
assert reg["town:beccles"].publishes_phase("primary")
def test_a_publishing_phase_page_still_lists_its_unscored_schools():
"""The threshold gates whether the page exists; it does not filter rows.
A parent looking up a school by name has to find it whether or not it
published results.
"""
scored = _town(MIN_SCHOOLS, "Beccles", "Suffolk", start=300000)
unscored = _town(2, "Beccles", "Suffolk", start=400000)
for r in unscored:
r["rwm_expected_pct"] = np.nan
reg = build_place_registry(_df(scored + unscored))
place = reg["town:beccles"]
assert place.publishes_phase("primary")
assert len(place.phase_urns["primary"]) == MIN_SCHOOLS + 2
def test_the_secondary_threshold_counts_its_own_metric():
# A town full of scored primaries must not thereby publish a secondary page.
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
reg = build_place_registry(_df(rows))
assert not reg["town:brentwood"].publishes_phase("secondary")
def test_an_outcode_publishes_no_phase_variants():
"""There is no /schools/near/[outcode]/[phase] route, by design.
Nobody searches "primary schools in SW11", so the spec gives outcodes no
phase variants. The registry computed them anyway, and the place page —
which links whatever phases the registry reports — put two 404s on every
outcode page in the site.
This is the single rule now: a kind with no phase route reports no phases,
so neither the page nor the sitemap can offer one.
"""
rows = [{"urn": 500000 + i, "school_name": f"SW11 School {i}",
"town": "London", "local_authority": "Wandsworth",
"postcode": "SW11 1AA"} for i in range(MIN_SCHOOLS + 3)]
reg = build_place_registry(_df(rows))
place = reg["outcode:sw11"]
assert place.phase_urns == {}
assert not place.publishes_phase("primary")
assert not place.publishes_phase("secondary")
def test_an_authority_still_publishes_phase_variants():
"""Authorities keep theirs — "primary schools in Kent" is a real query,
and /schools/authority/[la]/[phase] is the route that serves it."""
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Maidstone", "Kent")))
assert reg["authority:kent"].publishes_phase("primary")
# ── The reverse index: which published places contain a school ──────────────
#
# School pages link out to the location layer through this. It is the whole
# point of the index: before it, ~27k school pages linked to nothing on the
# site and stranded whatever authority they held.
def test_a_school_resolves_to_every_published_place_containing_it():
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
places = places_for_urn(build_place_index(reg), 100000)
kinds = {p.kind for p in places}
assert "town" in kinds
assert "authority" in kinds
def test_a_school_in_an_unpublished_town_still_resolves_to_its_authority():
# A town below the threshold has no page, so there is no link to offer —
# but the authority above it clears the threshold on the same schools and
# is where that reader should be sent.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS - 1, "Tinytown", "Essex")
+ _town(MIN_SCHOOLS, "Brentwood", "Essex", start=200000)
))
places = places_for_urn(build_place_index(reg), 100000)
# The town is below the threshold, so it has no page and must not be
# offered as a link. The authority above it does, and is the right target.
assert all(p.slug != "tinytown" for p in places)
assert "authority" in {p.kind for p in places}
def test_an_unknown_urn_resolves_to_nothing_rather_than_raising():
# A school page renders for any URN the API knows; the link module is not
# entitled to take the page down when it has nothing to say.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
assert places_for_urn(build_place_index(reg), 999999) == ()
def test_the_index_is_consistent_with_the_registry_it_was_built_from():
# The invariant that matters: a link module must never offer a place whose
# page does not exist, and never omit one that does.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS, "Bedford", "Bedford", start=300000)
))
index = build_place_index(reg)
for key, place in reg.items():
for urn in place.urns:
assert place in places_for_urn(index, urn), (
f"{urn} is in {key} but the index does not say so")
def test_the_index_holds_no_school_the_registry_does_not():
# The reverse direction of the invariant above. An index entry for a URN
# no published place contains would put a link on a page for a place that
# does not list that school.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS - 1, "Tinytown", "Essex", start=400000)
))
index = build_place_index(reg)
for urn, places in index.items():
for place in places:
assert urn in place.urns
assert place.key in reg
def test_the_index_preserves_the_widest_first_order():
# The breadcrumb reads authority then town, and takes this order as given.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
kinds = [p.kind for p in places_for_urn(build_place_index(reg), 100000)]
assert kinds.index("authority") < kinds.index("town")
-185
View File
@@ -1,185 +0,0 @@
"""Tests for the places API (spec 2026-08-21)."""
import numpy as np
import pandas as pd
import pytest
from fastapi.testclient import TestClient
def _schools_df() -> pd.DataFrame:
base = {
"local_authority": "Essex", "school_type": "Academy",
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
"ofsted_date": None, "attainment_8_score": np.nan,
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
}
return pd.DataFrame([
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
"rwm_expected_pct": 50.0 + i}
for i in range(6)
])
@pytest.fixture()
def client(monkeypatch):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False)
def test_registry_lists_each_published_place(client):
body = client.get("/api/places").json()
slugs = {(p["kind"], p["slug"]) for p in body["places"]}
assert ("town", "brentwood") in slugs
assert ("authority", "essex") in slugs
assert ("outcode", "cm13") in slugs
def test_registry_carries_a_count_per_place(client):
body = client.get("/api/places").json()
town = next(p for p in body["places"] if p["slug"] == "brentwood")
assert town["count"] == 6
def test_place_detail_returns_its_schools_alphabetically(client):
"""A place page is read by someone looking for a school they can name.
Scanning for it is what the order should serve, so the list is A-Z.
/api/rankings is where the league-table ordering lives.
"""
body = client.get("/api/places/town/brentwood").json()
assert body["place"]["name"] == "Brentwood"
names = [s["school_name"] for s in body["schools"]]
assert names == sorted(names, key=str.lower)
def test_place_ordering_ignores_case(client):
body = client.get("/api/places/town/brentwood").json()
names = [s["school_name"] for s in body["schools"]]
# A capitalised name must not sort ahead of every lowercase one.
assert names == sorted(names, key=str.lower)
def test_the_rankings_endpoint_still_ranks_by_metric(client):
# Alphabetical is a place-page decision, not a site-wide one.
body = client.get("/api/rankings?metric=rwm_expected_pct&phase=primary").json()
scores = [r["rwm_expected_pct"] for r in body.get("rankings", [])
if r.get("rwm_expected_pct") is not None]
assert scores == sorted(scores, reverse=True)
def test_place_detail_carries_the_local_average(client):
body = client.get("/api/places/town/brentwood").json()
# 50..55 inclusive
assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5)
def test_phase_filter_narrows_the_school_list(client):
body = client.get("/api/places/town/brentwood?phase=secondary").json()
assert body["schools"] == []
def test_unknown_place_404s(client):
assert client.get("/api/places/town/atlantis").status_code == 404
def test_unknown_kind_404s(client):
assert client.get("/api/places/planet/mars").status_code == 404
def _straddling_df() -> pd.DataFrame:
"""Eight schools in CM13: six in Essex, which has a page, and two in an
authority too small to have one.
Two, not one: the registry ignores an authority holding a single school in
a place, because GIAS carries occasional postcode errors."""
df = _schools_df()
extra = df.iloc[:2].copy()
extra["urn"] = [200000, 200001]
extra["school_name"] = ["Scilly School 0", "Scilly School 1"]
extra["local_authority"] = "Isles Of Scilly"
return pd.concat([df, extra], ignore_index=True)
@pytest.fixture()
def straddling_client(monkeypatch):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _straddling_df)
monkeypatch.setattr(app_module, "load_latest_school_data", _straddling_df)
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False)
def test_an_outcode_reports_no_phases_because_it_has_no_phase_route(client):
body = client.get("/api/places/outcode/cm13").json()
assert body["place"]["phases"] == []
def test_an_authority_reports_the_phases_it_publishes(client):
body = client.get("/api/places/authority/essex").json()
assert body["place"]["phases"] == ["primary"]
def test_an_authority_without_a_page_is_named_but_carries_no_slug(straddling_client):
"""Two English authorities — City of London and the Isles of Scilly — hold
fewer than the five schools a page needs, so they have no page.
Naming them is still right: the page says where the place is. Linking them
would not be. A null slug is what tells the page to print the name plainly
rather than invent a URL that 404s.
"""
body = straddling_client.get("/api/places/outcode/cm13").json()
by_name = {a["name"]: a for a in body["place"]["authorities"]}
assert by_name["Essex"]["slug"] == "essex"
assert by_name["Isles Of Scilly"]["slug"] is None
def _attributed_df() -> pd.DataFrame:
"""The same town, with the four attributes the place table now shows."""
df = _schools_df()
df["age_range"] = "4-11"
df["religious_denomination"] = "Church of England"
df["nursery_provision"] = True
df["parliamentary_constituency"] = "Brentwood and Ongar"
return df
@pytest.fixture()
def attributed_client(monkeypatch):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _attributed_df)
monkeypatch.setattr(app_module, "load_latest_school_data", _attributed_df)
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False)
def test_place_detail_carries_the_attributes_the_table_shows(attributed_client):
"""age_range and religious_denomination ride in on SCHOOL_COLUMNS.
nursery_provision and parliamentary_constituency do not, and the place
table needs all four — a column the response cannot fill is a column of
dashes on ~3,900 pages.
"""
body = attributed_client.get("/api/places/town/brentwood").json()
school = body["schools"][0]
assert school["age_range"] == "4-11"
assert school["religious_denomination"] == "Church of England"
assert school["nursery_provision"] is True
assert school["parliamentary_constituency"] == "Brentwood and Ongar"
def test_place_detail_survives_a_mart_without_the_optional_columns(client):
"""The base fixture has neither column, as an unrebuilt mart does not.
data_loader degrades those to NULL rather than failing the load, so the
endpoint must not assume they are present.
"""
res = client.get("/api/places/town/brentwood")
assert res.status_code == 200
assert "nursery_provision" not in res.json()["schools"][0]
-147
View File
@@ -1,147 +0,0 @@
"""The rate-limit bucket must be the caller, not the proxy in front of them.
`get_remote_address` reads request.client.host. In staging and production the
backend has no published ports and its only caller is the Next proxy, so that
host is the Next container — one bucket for every browser user on the site.
Measured before this fix: 70 concurrent requests, 60 served and 10 refused.
"""
from starlette.datastructures import Headers
from backend.app import client_key
class _Req:
"""Enough of a Request for the key function: headers and a client host."""
def __init__(self, headers: dict, host: str = "10.0.0.9"):
self.headers = Headers(headers)
self.client = type("C", (), {"host": host})()
self.scope = {"type": "http", "client": (host, 0),
"headers": [(k.lower().encode(), v.encode())
for k, v in headers.items()]}
def test_cloudflare_header_wins():
# Cloudflare sets CF-Connecting-IP and overwrites any client-supplied
# value, so it is trustworthy in a way a parsed XFF chain is not.
assert client_key(_Req({"cf-connecting-ip": "203.0.113.7"})) == "203.0.113.7"
def test_forwarded_for_is_the_fallback_and_takes_the_first_entry():
# Left-most is the original client; everything after it is proxies.
assert client_key(
_Req({"x-forwarded-for": "203.0.113.7, 10.0.0.2"})) == "203.0.113.7"
def test_remote_address_is_the_last_resort():
assert client_key(_Req({}, host="10.0.0.9")) == "10.0.0.9"
def test_cloudflare_header_beats_forwarded_for():
key = client_key(_Req({"cf-connecting-ip": "203.0.113.7",
"x-forwarded-for": "198.51.100.1"}))
assert key == "203.0.113.7"
def test_two_callers_get_two_buckets():
# The whole point: one user exhausting their limit must not refuse another.
a = client_key(_Req({"cf-connecting-ip": "203.0.113.7"}))
b = client_key(_Req({"cf-connecting-ip": "203.0.113.8"}))
assert a != b
def test_whitespace_is_stripped():
# "a, b" split on comma leaves a leading space on every entry but the
# first; an unstripped key silently creates a second bucket per client.
assert client_key(_Req({"x-forwarded-for": " 203.0.113.7 ,10.0.0.2"})) \
== "203.0.113.7"
# ---------------------------------------------------------------------------
# The ceiling that header rotation cannot raise.
# ---------------------------------------------------------------------------
import pytest
from fastapi.testclient import TestClient
@pytest.fixture()
def api(monkeypatch):
from backend import app as app_module
from backend.config import settings
monkeypatch.setattr(settings, "global_rate_limit_per_minute", 5)
monkeypatch.setattr(app_module, "_global_window", None)
return TestClient(app_module.app, raise_server_exceptions=False)
def _ceiling_req(path: str, host: str):
"""Enough of a Request for exempt_from_ceiling: a path and a peer host."""
return type("R", (), {
"url": type("U", (), {"path": path})(),
"client": type("C", (), {"host": host})(),
})()
def _get(client, path="/api/flags", cf=None):
headers = {"cf-connecting-ip": cf} if cf else {}
return client.get(path, headers=headers)
def test_rotating_the_cloudflare_header_cannot_buy_unlimited_requests(api):
"""The attack the per-client keying opened up.
client_key trusts CF-Connecting-IP, and nothing in this process can tell an
edge-set header from an attacker-set one — that distinction can only be
made at Cloudflare, with Authenticated Origin Pulls or an origin firewall.
A caller reaching the origin directly can therefore mint a fresh
rate-limit bucket per request and evade per-client limits entirely.
Per-client fairness is still the right default; this is the backstop that
bounds what evading it can achieve. Without it, correct keying would be a
net regression against abuse compared with the shared bucket it replaced.
"""
codes = [_get(api, cf=f"203.0.113.{i}").status_code for i in range(8)]
assert codes.count(200) == 5
assert codes.count(429) == 3
def test_the_ceiling_says_which_limit_was_hit(api):
# Distinguishable from slowapi's per-client 429, or an operator reading
# logs cannot tell "one noisy client" from "the origin is saturated".
for i in range(5):
_get(api, cf=f"203.0.113.{i}")
refused = _get(api, cf="203.0.113.99")
assert refused.status_code == 429
assert "capacity" in refused.json()["detail"].lower()
assert refused.headers.get("retry-after")
def test_traffic_below_the_ceiling_is_untouched(api):
codes = [_get(api, cf=f"203.0.113.{i}").status_code for i in range(5)]
assert codes == [200] * 5
def test_the_container_healthcheck_is_exempt(api):
"""The healthcheck runs `curl http://localhost:80/api/data-info` inside the
container. If the ceiling could starve it, saturation would fail the
healthcheck, restart the container, and turn a load spike into an outage
loop — the ceiling has to protect the origin, not kill it.
"""
from backend.app import exempt_from_ceiling
assert exempt_from_ceiling(_ceiling_req("/api/data-info", "127.0.0.1"))
assert exempt_from_ceiling(_ceiling_req("/api/data-info", "::1"))
# Everyone else is counted.
assert not exempt_from_ceiling(_ceiling_req("/api/data-info", "10.0.0.9"))
def test_the_ceiling_ignores_non_api_paths():
# Sitemaps and robots.txt are served by this app too, and a crawler
# fetching them must not be refused because the API is busy.
from backend.app import exempt_from_ceiling
assert exempt_from_ceiling(_ceiling_req("/sitemap.xml", "10.0.0.9"))
assert exempt_from_ceiling(_ceiling_req("/robots.txt", "10.0.0.9"))
-176
View File
@@ -56,11 +56,6 @@ def client(monkeypatch):
monkeypatch.setattr(
app_module, "get_supplementary_data", lambda db, urn: {}
)
# The place registry is a module-level cache, so without this the endpoint
# answers from whatever registry an earlier test happened to leave behind
# — and a `places == []` assertion is satisfied by a stale registry just
# as well as by this fixture's own data, which makes it prove nothing.
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False)
@@ -74,174 +69,3 @@ def test_nan_gias_fields_serialize_as_null(client):
assert info["capacity"] is None
assert info["total_pupils"] is None
assert info["school_name"] == "West London Performing Arts Academy"
# ── Links out to the location layer ─────────────────────────────────────────
#
# School pages carried no link into the site at all: the only anchor on the
# template pointed at the school's own website, so ~27k pages received
# whatever authority the site had and sent it off-site. `places` is what the
# link module and the breadcrumb are built from.
def test_places_is_present_even_when_the_school_belongs_to_none(client):
# This fixture's single school cannot clear any publish threshold, so the
# honest answer is an empty list. The key must still be there: a missing
# key and "no places" are different things to the page rendering it.
body = client.get("/api/schools/150275").json()
assert body["places"] == []
def test_places_names_only_pages_that_exist(monkeypatch):
from backend import app as app_module
from backend.places import MIN_SCHOOLS
def _df():
return pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood",
"local_authority": "Essex",
"postcode": "CM15 8AA",
"phase": "Primary",
"year": 202425,
"rwm_expected_pct": 60.0,
"attainment_8_score": np.nan,
"ofsted_grade": 2.0,
"ofsted_date": None,
}
for i in range(MIN_SCHOOLS)
])
monkeypatch.setattr(app_module, "load_school_data", _df)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
places = client.get("/api/schools/100000").json()["places"]
assert places, "a school in a published town must offer links"
by_kind = {p["kind"]: p for p in places}
assert by_kind["town"]["url"] == "/schools/brentwood"
assert by_kind["authority"]["url"] == "/schools/authority/essex"
# Every entry carries what the link text needs, and a count, so the anchor
# can say what it leads to rather than "click here".
for place in places:
assert place["name"]
assert place["count"] >= 1
assert place["url"].startswith("/schools/")
def _brentwood_df(phase: str = "Primary", n: int = None):
from backend.places import MIN_SCHOOLS
n = n if n is not None else MIN_SCHOOLS
return lambda: pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood", "local_authority": "Essex",
"postcode": "CM15 8AA", "phase": phase, "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": 50.0,
"ofsted_grade": 2.0, "ofsted_date": None,
}
for i in range(n)
])
def _places_for(monkeypatch, df_factory, urn: int):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", df_factory)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
return client.get(f"/api/schools/{urn}").json()["places"]
def test_a_place_offers_the_phase_page_this_school_appears_on(monkeypatch):
# "primary schools in brentwood" is the query the phase pages exist for,
# and ~950 of them were once reachable by nothing at all.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert town["phases"], "a primary school in a published primary town has a link"
assert town["phases"][0]["url"] == "/schools/brentwood/primary"
assert town["phases"][0]["count"] >= 1
def test_an_all_through_school_offers_both_phase_pages(monkeypatch):
# It genuinely appears on both, so there is no tie to break.
places = _places_for(monkeypatch, _brentwood_df("All-through"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert {p["phase"] for p in town["phases"]} == {"primary", "secondary"}
def test_outcodes_never_offer_a_phase_page(monkeypatch):
# The registry gives outcodes no phase route — nobody searches "primary
# schools in SW11" — and computing them anyway once put a link to a
# nonexistent route on all 1,720 outcode pages.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
outcode = next((p for p in places if p["kind"] == "outcode"), None)
if outcode is not None:
assert outcode["phases"] == []
def test_a_school_absent_from_the_phase_page_is_not_linked_to_it(monkeypatch):
# The check is URN membership in the registry's own phase list, not a
# re-derivation of the phase mapping. A secondary school must not be sent
# to a primary phase page that does not list it.
from backend.places import MIN_SCHOOLS
def df():
rows = [
{"urn": 100000 + i, "school_name": f"P{i}", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Primary", "year": 202425, "rwm_expected_pct": 60.0,
"attainment_8_score": np.nan, "ofsted_grade": 2.0,
"ofsted_date": None}
for i in range(MIN_SCHOOLS)
]
rows.append({
"urn": 900000, "school_name": "Lone Secondary", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Secondary", "year": 202425, "rwm_expected_pct": np.nan,
"attainment_8_score": 50.0, "ofsted_grade": 2.0, "ofsted_date": None,
})
return pd.DataFrame(rows)
places = _places_for(monkeypatch, df, 900000)
town = next(p for p in places if p["kind"] == "town")
# The town publishes a primary page, but this secondary school is not on
# it, and there are too few secondaries for a secondary page.
assert town["phases"] == []
def test_the_place_index_rebuilds_when_the_registry_is_replaced(monkeypatch):
"""The reverse index is cached; a stale one would put another dataset's
places on a school page. Invalidation is an identity check against the
registry rather than a second flag, so this asserts the check works."""
from backend import app as app_module
monkeypatch.setattr(app_module, "_place_registry", None)
monkeypatch.setattr(app_module, "_place_index", None)
monkeypatch.setattr(app_module, "_place_index_source", None)
monkeypatch.setattr(app_module, "load_school_data", _brentwood_df("Primary"))
first = app_module.get_place_index()
assert 100000 in first
# Same registry object, so the index is reused rather than rebuilt.
assert app_module.get_place_index() is first
# Drop the registry the way every test that touches place data does. The
# index must follow it, not survive it.
app_module._place_registry = None
monkeypatch.setattr(app_module, "load_school_data",
_brentwood_df("Primary", n=0))
rebuilt = app_module.get_place_index()
assert rebuilt is not first
assert 100000 not in rebuilt, "the index outlived the registry it came from"
-306
View File
@@ -1,306 +0,0 @@
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
The sitemap is built from the in-memory school DataFrame, so these inject a
small frame via monkeypatch rather than touching a database.
"""
import numpy as np
import pandas as pd
import pytest
def _schools_df() -> pd.DataFrame:
"""Two schools: one with results, one with neither results nor Ofsted."""
base = {
"local_authority": "Testshire",
"school_type": "Academy",
"phase": "Primary",
"year": 202425,
"ofsted_date": None,
}
return pd.DataFrame(
[
{**base, "urn": 100001, "school_name": "Alpha Primary",
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
"ofsted_grade": 2.0},
{**base, "urn": 100002, "school_name": "Ghost Primary",
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
"ofsted_grade": np.nan},
]
)
@pytest.fixture()
def sitemap(monkeypatch) -> str:
"""The sitemap index."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemap()
@pytest.fixture()
def schools_child(monkeypatch) -> str:
"""The first school child sitemap, where school URLs actually live."""
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["schools-1.xml"]
@pytest.fixture()
def static_child(monkeypatch) -> str:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()["static.xml"]
def test_every_loc_uses_the_www_host(sitemaps):
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
# Checked across every file, index included, not just one.
#
# A child can legitimately be empty — this fixture holds two schools and no
# town clearing the threshold — so the presence check applies only to files
# that carry URLs. The absence check applies to all of them.
for name, xml in sitemaps.items():
assert "https://schoolcompare.co.uk" not in xml, name
if "<loc>" in xml:
assert "https://www.schoolcompare.co.uk" in xml, name
def test_school_with_results_is_listed(schools_child):
assert "/school/100001-alpha-primary" in schools_child
def test_school_with_no_results_and_no_ofsted_is_omitted(schools_child):
# Nothing for a search result to say about it. Submitting it spends crawl
# budget and drags the corpus-wide quality signal down.
#
# Asserted against the child, not the index: the index carries no school
# URLs at all, so it would pass this trivially and prove nothing.
assert "/school/100002" not in schools_child
def test_no_invented_priority_or_changefreq(sitemaps):
# Google ignores both. They were noise dressed as signal.
for name, xml in sitemaps.items():
assert "<priority>" not in xml, name
assert "<changefreq>" not in xml, name
def test_ofsted_date_becomes_lastmod(monkeypatch):
from backend import app as app_module
import datetime
def _df():
base = _schools_df()
base.loc[base["urn"] == 100001, "ofsted_date"] = datetime.date(2024, 3, 14)
return base
monkeypatch.setattr(app_module, "load_school_data", _df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>2024-03-14</lastmod>" in xml
def test_no_lastmod_invented_when_date_unknown(monkeypatch):
# An always-now lastmod is a claim Google learns to distrust. Absent
# honestly means unknown.
from backend import app as app_module
def _df():
df = _schools_df()
df["ofsted_date"] = None
return df
monkeypatch.setattr(app_module, "load_school_data", _df)
# The child only. The index legitimately carries a lastmod, because there
# it means "when this sitemap file changed", which we do know.
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "<lastmod>" not in xml
def test_static_routes_are_listed(static_child):
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static_child
@pytest.fixture()
def sitemaps(monkeypatch) -> dict:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
return app_module.build_sitemaps()
def test_index_lists_each_child(sitemaps):
index = sitemaps["sitemap.xml"]
assert "<sitemapindex" in index
assert "https://www.schoolcompare.co.uk/sitemaps/static.xml" in index
assert "https://www.schoolcompare.co.uk/sitemaps/schools-1.xml" in index
def test_index_carries_no_url_elements(sitemaps):
# A sitemap index holds <sitemap> entries only; mixing in <url> is invalid.
assert "<url>" not in sitemaps["sitemap.xml"]
def test_index_does_not_list_itself(sitemaps):
assert "<loc>https://www.schoolcompare.co.uk/sitemap.xml</loc>" not in sitemaps["sitemap.xml"]
def test_static_child_holds_the_static_routes(sitemaps):
static = sitemaps["static.xml"]
for path in ("/", "/rankings", "/compare", "/admissions"):
assert f"<loc>https://www.schoolcompare.co.uk{path}</loc>" in static
def test_school_child_holds_the_schools(sitemaps):
assert "/school/100001-alpha-primary" in sitemaps["schools-1.xml"]
def test_children_are_chunked_under_the_limit(monkeypatch):
# Sitemaps cap at 50,000 URLs per file. Chunk at 10,000 so a child stays
# small enough to eyeball in Search Console.
from backend import app as app_module
import pandas as _pd
rows = [
{"urn": 200000 + i, "school_name": f"School {i}", "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": None,
"ofsted_grade": 2.0, "ofsted_date": None}
for i in range(10_001)
]
monkeypatch.setattr(app_module, "load_school_data", lambda: _pd.DataFrame(rows))
maps = app_module.build_sitemaps()
assert maps["schools-1.xml"].count("<url>") == 10_000
assert maps["schools-2.xml"].count("<url>") == 1
def test_build_sitemap_still_returns_the_index(sitemap):
# lifespan and the admin endpoint call build_sitemap(); keep it working.
assert "<sitemapindex" in sitemap
def test_school_with_results_in_an_earlier_year_is_still_listed(monkeypatch):
"""Regression: The Mallard Academy (150367).
Real KS2 results 2015-16 to 2018-19, then null rows from 2022-23 onward
because the school stopped reporting. The first cut tested the latest
year's row alone and dropped it, along with ~220 others, even though its
detail page shows all four years of results.
"""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "urn": 150367,
"school_name": "Mallard Academy"}
df = _pd.DataFrame([
{**base, "year": 201819, "rwm_expected_pct": 67.0},
{**base, "year": 202324, "rwm_expected_pct": np.nan},
{**base, "year": 202425, "rwm_expected_pct": np.nan},
])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
xml = app_module.build_sitemaps()["schools-1.xml"]
assert "/school/150367-mallard-academy" in xml
def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
"""The fix must not turn into "list everything"."""
from backend import app as app_module
import pandas as _pd
base = {"local_authority": "Testshire", "school_type": "Academy",
"phase": "Primary", "ofsted_date": None, "ofsted_grade": np.nan,
"attainment_8_score": np.nan, "rwm_expected_pct": np.nan,
"urn": 100002, "school_name": "Ghost Primary"}
df = _pd.DataFrame([{**base, "year": y} for y in (202324, 202425)])
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]
def _places_df() -> pd.DataFrame:
base = {
"local_authority": "Essex", "school_type": "Academy",
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
"ofsted_date": None, "attainment_8_score": np.nan,
"town": "Brentwood", "postcode": "CM13 1AA",
}
return pd.DataFrame([
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
"rwm_expected_pct": 60.0}
for i in range(6)
])
@pytest.fixture()
def place_sitemaps(monkeypatch) -> dict:
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", _places_df)
monkeypatch.setattr(app_module, "_place_registry", None)
return app_module.build_sitemaps()
def test_place_children_are_listed_in_the_index(place_sitemaps):
index = place_sitemaps["sitemap.xml"]
assert "/sitemaps/places-1.xml" in index
assert "/sitemaps/outcodes-1.xml" in index
def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps):
xml = place_sitemaps["places-1.xml"]
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood</loc>" in xml
assert "<loc>https://www.schoolcompare.co.uk/schools/authority/essex</loc>" in xml
def test_outcode_urls_live_in_their_own_child(place_sitemaps):
assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"]
assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"]
def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps):
for name in ("places-1.xml", "outcodes-1.xml"):
assert "<priority>" not in place_sitemaps[name]
assert "<changefreq>" not in place_sitemaps[name]
def test_phase_variants_are_submitted_where_the_phase_clears_the_threshold(place_sitemaps):
# "primary schools in beccles" is the query shape the baseline showed, so
# each variant is its own page and has to be submitted. Emitting only the
# bare place URL left ~950 of them reachable by nothing.
xml = place_sitemaps["places-1.xml"]
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood/primary</loc>" in xml
def test_a_phase_below_its_own_threshold_is_not_submitted(place_sitemaps):
# The fixture is six primaries and no secondaries.
xml = place_sitemaps["places-1.xml"]
assert "/schools/brentwood/secondary" not in xml
def test_outcodes_get_no_phase_variants(place_sitemaps):
# Nobody searches "primary schools in CM13"; the routes do not exist.
xml = place_sitemaps["outcodes-1.xml"]
assert "/primary" not in xml and "/secondary" not in xml
def test_authority_phase_variants_are_submitted_in_their_own_namespace(place_sitemaps):
"""302 of these were already in the sitemap, and every one 404'd.
The spec gives authorities a phase route; the plan built the bare
authority route and dropped it. Nothing noticed because the sitemap was
written from the registry, which was right, while the routes were written
by hand. This test fails if the URL ever leaves the sitemap; the e2e
journey fails if the route ever leaves the app.
"""
xml = place_sitemaps["places-1.xml"]
assert ("<loc>https://www.schoolcompare.co.uk"
"/schools/authority/essex/primary</loc>") in xml
# And never in the town namespace, which is a different set of schools.
assert "/schools/essex/primary" not in xml
-157
View File
@@ -1,157 +0,0 @@
"""Tests for school autosuggest (spec 2026-08-26)."""
from backend import data_loader
class _FakeDocs:
def __init__(self, hits, explode=False):
self._hits = hits
self._explode = explode
self.last_params = None
def search(self, params):
self.last_params = params
if self._explode:
raise RuntimeError("typesense is down")
return {"hits": [{"document": d} for d in self._hits]}
class _FakeClient:
def __init__(self, hits, explode=False):
self.docs = _FakeDocs(hits, explode)
self.collections = {"schools": type("C", (), {"documents": self.docs})()}
_HIT = {
"urn": 100010, "school_name": "Brecknock Primary School",
"local_authority": "Camden", "postcode": "NW1 1AA",
"phase": "Primary", "school_type": "Community school",
}
def _use(monkeypatch, client):
monkeypatch.setattr(data_loader, "_get_typesense_client", lambda: client)
def test_returns_the_fields_a_suggestion_needs(monkeypatch):
# Local authority is not decoration: there are many schools called
# "St Mary's", and a list without it cannot be chosen between.
_use(monkeypatch, _FakeClient([_HIT]))
out = data_loader.suggest_schools_typesense("breck")
assert out == [{
"urn": 100010, "school_name": "Brecknock Primary School",
"local_authority": "Camden", "postcode": "NW1 1AA",
"phase": "Primary", "school_type": "Community school",
}]
def test_a_missing_optional_field_becomes_an_empty_string(monkeypatch):
# phase and school_type are optional in the Typesense schema. A missing
# key must not KeyError in the keystroke path.
_use(monkeypatch, _FakeClient([{"urn": 1, "school_name": "X",
"local_authority": "Y", "postcode": "Z"}]))
out = data_loader.suggest_schools_typesense("x")
assert out[0]["phase"] == "" and out[0]["school_type"] == ""
def test_typesense_unavailable_gives_no_suggestions_rather_than_raising(monkeypatch):
_use(monkeypatch, None)
assert data_loader.suggest_schools_typesense("anything") == []
def test_a_typesense_error_gives_no_suggestions_rather_than_raising(monkeypatch):
_use(monkeypatch, _FakeClient([], explode=True))
assert data_loader.suggest_schools_typesense("anything") == []
def test_the_limit_is_passed_through_and_clamped(monkeypatch):
client = _FakeClient([])
_use(monkeypatch, client)
data_loader.suggest_schools_typesense("x", limit=500)
assert client.docs.last_params["per_page"] == 20
def _client(monkeypatch, rows, *, blow_up_dataframe=False):
from fastapi.testclient import TestClient
from backend import app as app_module
monkeypatch.setattr(app_module, "suggest_schools_typesense",
lambda q, limit=8: rows)
if blow_up_dataframe:
def _boom():
raise AssertionError("the suggest path must not load the DataFrame")
monkeypatch.setattr(app_module, "load_school_data", _boom)
monkeypatch.setattr(app_module, "load_latest_school_data", _boom)
return TestClient(app_module.app, raise_server_exceptions=False)
def test_the_endpoint_returns_suggestions(monkeypatch):
body = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck").json()
assert body["suggestions"][0]["school_name"] == "Brecknock Primary School"
def test_the_endpoint_never_touches_the_dataframe(monkeypatch):
"""The whole reason this is not a mode of /api/schools.
That endpoint filters and sorts 25,000 rows of pandas per query, holding
the GIL. Per keystroke, that is the cost this endpoint exists to avoid.
"""
res = _client(monkeypatch, [_HIT], blow_up_dataframe=True).get("/api/suggest?q=breck")
assert res.status_code == 200
assert res.json()["suggestions"]
def test_a_one_character_query_returns_nothing_and_does_not_error(monkeypatch):
# The keystroke path never errors on ordinary input.
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=b")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_a_blank_query_returns_nothing_and_does_not_error(monkeypatch):
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_typesense_down_is_an_empty_list_not_a_500(monkeypatch):
res = _client(monkeypatch, []).get("/api/suggest?q=breck")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_the_response_is_cacheable(monkeypatch):
# Prefix queries repeat enormously across users, and school names change
# once a year. Without this the endpoint pays full price every keystroke.
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck")
assert "s-maxage" in res.headers.get("cache-control", "")
assert res.headers.get("etag")
def test_a_malformed_urn_does_not_raise(monkeypatch):
"""The docstring promises "never raises"; the parsing loop sat outside the
try, so int(None) or int("abc") would have turned a keystroke into a 500.
Typesense declares urn as int32, so this should be unreachable — but the
contract is what the caller relies on, and a search index is a separate
system that can be reindexed by something other than this code.
"""
_use(monkeypatch, _FakeClient([{"urn": None, "school_name": "X",
"local_authority": "Y", "postcode": "Z"}]))
assert data_loader.suggest_schools_typesense("x") == []
def test_a_malformed_row_does_not_discard_the_good_ones(monkeypatch):
# One bad document must not blank the whole dropdown.
_use(monkeypatch, _FakeClient([
{"urn": "not-a-number", "school_name": "Bad", "local_authority": "Y",
"postcode": "Z"},
_HIT,
]))
out = data_loader.suggest_schools_typesense("x")
assert [r["urn"] for r in out] == [100010]
def test_a_hit_with_no_document_does_not_raise(monkeypatch):
_use(monkeypatch, _FakeClient([{}]))
assert data_loader.suggest_schools_typesense("x") == []
+4 -76
View File
@@ -8,31 +8,8 @@ from backend import data_loader
from backend.data_loader import get_supplementary_data_batch
def _sort_key(criterion):
"""(column name, descending) for a SQLAlchemy order_by argument.
A bare column (Model.year) arrives as an InstrumentedAttribute carrying
.key; Model.year.desc() wraps it in a UnaryExpression whose column sits on
.element.
"""
name = getattr(criterion, "key", None)
if name is not None:
return name, False
element = getattr(criterion, "element", None)
name = getattr(element, "key", None)
return name, "DESC" in str(criterion).upper()
class _FakeQuery:
"""Records that a query ran and serves canned rows filtered by an in-list.
order_by is honoured rather than ignored. The batch loader picks a row per
URN by position — first for "latest Ofsted", last for "latest cut-off
distance" — which is only correct because the database returned them
sorted. A double that drops the ORDER BY makes those picks depend on
fixture insertion order instead, so the test would pass with the sort
reversed or removed and prove nothing about the query.
"""
"""Records that a query ran and serves canned rows filtered by an in-list."""
def __init__(self, recorder, model_name, rows):
self._rec = recorder
@@ -42,20 +19,7 @@ class _FakeQuery:
def filter(self, *args, **kwargs):
return self
def order_by(self, *criteria):
for crit in reversed(criteria): # reversed = stable multi-key sort
name, descending = _sort_key(crit)
if not name:
continue
values = [getattr(r, name, None) for r in self._rows]
# Only sort on plainly comparable values. Some fixtures stand dates
# up as namespace objects, which raise on <; leaving those in their
# given order matches what the real query would produce for them.
if not all(isinstance(v, (int, float, str)) for v in values):
continue
self._rows = sorted(
self._rows, key=lambda r: getattr(r, name), reverse=descending
)
def order_by(self, *args, **kwargs):
return self
def all(self):
@@ -105,13 +69,6 @@ def _adm_row(urn, year):
)
def _dist_row(urn, year, distance_m, route_count=1):
return types.SimpleNamespace(
urn=urn, year=year, distance_m=distance_m, route_count=route_count,
la_name="Camden", distance_unit_raw="miles", source_file="camden/guide.pdf",
)
def test_one_query_per_table_and_latest_row_per_urn():
rows = {
# URN 1 has two Ofsted rows; the batch must keep the most recent (2023).
@@ -121,34 +78,19 @@ def test_one_query_per_table_and_latest_row_per_urn():
_ofsted_row(2, "2021-06-01", 1),
],
"FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)],
# URN 1 has three years of cut-offs; only the most recent is served.
# Deliberately not in year order — the ordering is the query's job.
"FactAdmissionDistance": [
_dist_row(1, 2026, 529.47),
_dist_row(1, 2024, 772.49),
_dist_row(1, 2025, 1421.05),
_dist_row(2, 2023, 2029.38, route_count=4),
],
"FactPupilCharacteristics": [],
"FactDeprivation": [],
"FactFinance": [],
"FactKs4Destinations": [],
"FactKs5Destinations": [],
}
session = _FakeSession(rows)
out = get_supplementary_data_batch(session, [1, 2])
# Exactly one query per table — eight total, regardless of two URNs.
# Exactly one query per table — five total, regardless of two URNs.
assert sorted(session.queries) == [
"FactAdmissionDistance", "FactAdmissions", "FactDeprivation",
"FactFinance", "FactKs4Destinations", "FactKs5Destinations",
"FactAdmissions", "FactDeprivation", "FactFinance",
"FactOfstedInspection", "FactPupilCharacteristics",
]
# A school with no destination rows gets null, not an empty shell — the
# frontend renders the section from the block's presence.
assert out[1]["destinations"] is None
# Latest Ofsted kept per URN
assert out[1]["ofsted"]["overall_effectiveness"] == 2
assert out[2]["ofsted"]["overall_effectiveness"] == 1
@@ -158,18 +100,6 @@ def test_one_query_per_table_and_latest_row_per_urn():
assert out[1]["admissions"]["year"] == 202627
assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}]
# Cut-off distance: the latest year only. Earlier years stay in the mart
# but are held back as a paid feature, and this API is public — serving
# them here would hand them to anyone reading the response. The fixture
# rows are deliberately out of order, so "latest" only comes out right if
# the query's ORDER BY is doing the work.
assert out[1]["admission_distance"]["year"] == 2026
assert "admission_distance_history" not in out[1]
assert out[1]["admission_distance"]["distance_m"] == 529.47
# route_count travels with the figure — the page needs it to say the
# distance is the furthest of several bands rather than the only one.
assert out[2]["admission_distance"]["route_count"] == 4
# Empty tables degrade to the null block, not a crash
assert out[1]["census"] is None and out[1]["deprivation"] is None
@@ -179,5 +109,3 @@ def test_single_wrapper_matches_batch(monkeypatch):
single = data_loader.get_supplementary_data(session, 5)
assert single["ofsted"]["overall_effectiveness"] == 2
assert single["admissions_history"] == []
assert single["admission_distance"] is None
assert "admission_distance_history" not in single
-46
View File
@@ -23,52 +23,6 @@ Key files:
- `backend/data_loader.py` - Data queries, geocoding, legacy DataFrame compatibility
- `backend/schemas.py` - Column mappings, metric definitions, LA code mappings
### Content / CMS (Payload)
Payload CMS runs **inside** the Next.js app — one image, one container, no
separate service. It powers `/blog`; `/about` is a plain coded page.
- **Admin panel:** `/admin`. The only authenticated surface on the site.
`noindex` via both `robots.txt` and `X-Robots-Tag`.
- **CMS API:** `/cms-api`, **not** `/api`. `/api/*` is a catch-all proxy to
FastAPI (`app/(frontend)/api/[...path]`) which would silently swallow every
admin call and forward it to the backend. Mount points are defined once in
`lib/payloadRoutes.ts`.
- **Database:** the existing Postgres, in its own `payload` schema, so no
pipeline operation on `public` — including
`scripts/migrate_csv_to_db.py --drop` — can reach blog content.
- **Uploads:** the `payload_media` Docker volume at `/app/media`. Not
reproducible from the pipeline; must be backed up.
- **New env vars:** `DATABASE_URL` and `PAYLOAD_SECRET` on the frontend service.
Staging must use a different `PAYLOAD_SECRET` from production.
- Publishing workflow and house style: `nextjs-app/docs/PUBLISHING.md`.
- **Admin field components resolve through a generated import map**
(`app/(payload)/admin/importMap.js`). Payload hands the client a *path* per
field and looks it up there; a missing entry renders no field and reports no
error, while `required` still blocks the save. After adding or changing any
field, editor or lexical feature, run `npm run generate:importmap` in
`nextjs-app/` and commit the result.
### Two route groups
`nextjs-app/app/` has no root `layout.tsx`. It cannot: Payload's admin panel
ships its own root layout rendering `<html>`/`<body>`, and Next permits
multiple root layouts only when no `app/layout.tsx` exists.
- `app/(frontend)/` — the site. Its `layout.tsx` is the site's root layout.
- `app/(payload)/` — the admin panel and `/cms-api`.
Route groups are invisible to routing, so every public URL is unchanged.
**The metadata file conventions stay at the `app/` root** — `robots.ts`,
`opengraph-image.tsx`, `icon.png`, `apple-icon.png`. Inside a route group Next
treats them as segment-scoped: it renames `/icon.png` to `/icon-<hash>.png` and
drops `/robots.txt` entirely. Route handlers are unaffected.
The build must succeed with `DATABASE_URL` unset, because CI builds it that
way. Never call `getCachedPayload()` at module scope, and never add
`generateStaticParams` to a DB-backed route.
### Frontend (Vanilla JS)
- Single-page application with hash-based routing
- Chart.js for data visualization
+2 -48
View File
@@ -16,16 +16,7 @@
# ADMIN_API_KEY — Backend admin API key
# TYPESENSE_API_KEY — Typesense admin API key
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
# UNLEASH_API_TOKEN — Unleash *client* token, environment: development
# PAYLOAD_SECRET — Payload CMS encryption secret. REQUIRED: long and
# random, and DIFFERENT from production's. Sharing
# it would let a staging session authenticate
# against production.
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
# refuses to start without it, rather than falling
# back to a generated one that changes on restart.
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
# STAGING_DB_IP — macvlan IP for staging Postgres (default 10.0.1.190)
# STAGING_FRONTEND_IP — macvlan IP for staging frontend (default 10.0.1.151)
@@ -64,12 +55,6 @@ services:
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
TYPESENSE_URL: http://typesense:8108
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
# Unset means every feature flag is False — the correct dark state for an
# environment with no Unleash, not a failure.
UNLEASH_URL: ${UNLEASH_URL:-}
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
volumes:
- unleash_cache:/app/.unleash
depends_on:
sc_database:
condition: service_healthy
@@ -93,20 +78,9 @@ services:
- FASTAPI_URL=http://backend:80/api
- TYPESENSE_URL=http://typesense:8108
- TYPESENSE_API_KEY=${TYPESENSE_SEARCH_KEY:-changeme}
# Payload CMS runs inside this container, in the `payload` schema of the
# staging database. Staging has its own stack, its own Postgres and its
# own admin account — never production's.
- DATABASE_URL=postgresql://${DB_USERNAME}:${DB_PASSWORD}@sc_database:5432/${DB_DATABASE_NAME}
- PAYLOAD_SECRET=${PAYLOAD_SECRET:?set PAYLOAD_SECRET in the staging Portainer stack environment}
volumes:
# Portainer prefixes volume names with the stack name, so this is
# automatically isolated from production's media.
- payload_media:/app/media
depends_on:
backend:
condition: service_healthy
sc_database:
condition: service_healthy
networks:
backend: {}
macvlan:
@@ -142,23 +116,7 @@ services:
airflow-api-server:
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:staging
container_name: sc_staging_airflow_api
# The simple auth manager generates a random password on first start and
# writes it to a file, so every container restart invalidates the last one.
# Writing the file ourselves from an environment variable makes the login
# deterministic. Airflow does not generate anything when the file exists.
#
# Built with python rather than echo/printf so a password containing quotes,
# backslashes or spaces is escaped correctly by json.dumps. An unset
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
# back to a generated password would silently undo the point of this.
command:
- bash
- -c
- |
set -euo pipefail
mkdir -p /opt/airflow
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
exec airflow api-server --port 8080
command: airflow api-server --port 8080
ports:
- "8081:8080"
environment:
@@ -170,8 +128,6 @@ services:
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-staging-airflow-jwt-secret-key-long-enough-for-sha512"
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
PG_HOST: sc_database
PG_PORT: "5432"
@@ -256,5 +212,3 @@ volumes:
postgres_data:
typesense_data:
airflow_logs:
unleash_cache:
payload_media:
-73
View File
@@ -1,73 +0,0 @@
# Portainer Stack Definition for School Compare — UNLEASH (feature flags)
#
# Deploy as a *separate* Portainer stack ("schoolcompare-unleash"), alongside
# the production and staging stacks. It deliberately belongs to neither: a
# staging redeploy must not be able to disturb production's flag state, and a
# production redeploy must not disturb staging's.
#
# One instance serves both environments. Open-source Unleash ships with
# `development` and `production` environments and environment-scoped client
# tokens, so the same flag holds independent state in each — which is what
# lets a feature be on in staging, where the E2E journeys exercise it, while
# production stays dark.
#
# Portainer environment variables (set in Portainer UI -> Stack -> Environment):
# UNLEASH_DB_PASSWORD — PostgreSQL password for the Unleash database
# UNLEASH_ADMIN_PASSWORD — initial admin password for the Unleash UI
# UNLEASH_IP — macvlan IP for the Unleash server (default 10.0.1.152)
services:
# ── PostgreSQL (Unleash's own; nothing else uses it) ──────────────────
unleash_db:
container_name: sc_unleash_postgres
image: postgres:16-alpine
environment:
POSTGRES_USER: unleash
POSTGRES_PASSWORD: ${UNLEASH_DB_PASSWORD}
POSTGRES_DB: unleash
volumes:
- unleash_postgres_data:/var/lib/postgresql/data
networks:
- unleash
healthcheck:
test: ["CMD-SHELL", "pg_isready -U unleash"]
interval: 10s
timeout: 5s
retries: 5
start_period: 10s
restart: unless-stopped
# ── Unleash server (UI + client API on 4242) ──────────────────────────
unleash:
container_name: sc_unleash
image: unleashorg/unleash-server:6
environment:
DATABASE_URL: postgres://unleash:${UNLEASH_DB_PASSWORD}@unleash_db:5432/unleash
DATABASE_SSL: "false"
INIT_ADMIN_API_TOKENS: ""
UNLEASH_DEFAULT_ADMIN_PASSWORD: ${UNLEASH_ADMIN_PASSWORD}
depends_on:
unleash_db:
condition: service_healthy
networks:
unleash: {}
macvlan:
ipv4_address: ${UNLEASH_IP:-10.0.1.152}
healthcheck:
test: ["CMD-SHELL", "wget -qO- http://localhost:4242/health || exit 1"]
interval: 30s
timeout: 10s
retries: 3
start_period: 30s
restart: unless-stopped
networks:
unleash:
driver: bridge
macvlan:
external:
name: macvlan
volumes:
unleash_postgres_data:
+2 -48
View File
@@ -7,15 +7,7 @@
# ADMIN_API_KEY — Backend admin API key
# TYPESENSE_API_KEY — Typesense admin API key
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
# UNLEASH_API_TOKEN — Unleash *client* token, environment: production
# PAYLOAD_SECRET — Payload CMS encryption secret. REQUIRED: long and
# random. Changing it invalidates every admin
# session. Staging MUST use a different value.
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
# refuses to start without it, rather than falling
# back to a generated one that changes on restart.
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
services:
@@ -52,12 +44,6 @@ services:
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
TYPESENSE_URL: http://typesense:8108
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
# Unset means every feature flag is False — the correct dark state for an
# environment with no Unleash, not a failure.
UNLEASH_URL: ${UNLEASH_URL:-}
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
volumes:
- unleash_cache:/app/.unleash
depends_on:
sc_database:
condition: service_healthy
@@ -81,21 +67,9 @@ services:
- FASTAPI_URL=http://backend:80/api
- TYPESENSE_URL=http://typesense:8108
- TYPESENSE_API_KEY=${TYPESENSE_SEARCH_KEY:-changeme}
# Payload CMS runs inside this container. It reaches Postgres over the
# `backend` network and keeps its tables in the `payload` schema, so no
# pipeline operation on `public` can touch blog content.
- DATABASE_URL=postgresql://${DB_USERNAME}:${DB_PASSWORD}@sc_database:5432/${DB_DATABASE_NAME}
# Same :? form as AIRFLOW_ADMIN_PASSWORD: refuse to start rather than
# boot with an empty secret and silently accept forged sessions.
- PAYLOAD_SECRET=${PAYLOAD_SECRET:?set PAYLOAD_SECRET in the Portainer stack environment}
volumes:
# Blog images. Not reproducible from the pipeline — must be backed up.
- payload_media:/app/media
depends_on:
backend:
condition: service_healthy
sc_database:
condition: service_healthy
networks:
backend: {}
macvlan:
@@ -131,23 +105,7 @@ services:
airflow-api-server:
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:prod
container_name: schoolcompare_airflow_api
# The simple auth manager generates a random password on first start and
# writes it to a file, so every container restart invalidates the last one.
# Writing the file ourselves from an environment variable makes the login
# deterministic. Airflow does not generate anything when the file exists.
#
# Built with python rather than echo/printf so a password containing quotes,
# backslashes or spaces is escaped correctly by json.dumps. An unset
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
# back to a generated password would silently undo the point of this.
command:
- bash
- -c
- |
set -euo pipefail
mkdir -p /opt/airflow
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
exec airflow api-server --port 8080
command: airflow api-server --port 8080
ports:
- "8080:8080"
environment:
@@ -159,8 +117,6 @@ services:
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
PG_HOST: sc_database
PG_PORT: "5432"
@@ -245,5 +201,3 @@ volumes:
postgres_data:
typesense_data:
airflow_logs:
unleash_cache:
payload_media:
+1 -23
View File
@@ -36,10 +36,6 @@ services:
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
TYPESENSE_URL: http://typesense:8108
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
# Unset means every feature flag is False — the correct dark state for an
# environment with no Unleash, not a failure.
UNLEASH_URL: ${UNLEASH_URL:-}
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
volumes:
- ./data:/app/data:ro
depends_on:
@@ -105,23 +101,7 @@ services:
airflow-api-server:
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:latest
container_name: schoolcompare_airflow_api
# The simple auth manager generates a random password on first start and
# writes it to a file, so every container restart invalidates the last one.
# Writing the file ourselves from an environment variable makes the login
# deterministic. Airflow does not generate anything when the file exists.
#
# Built with python rather than echo/printf so a password containing quotes,
# backslashes or spaces is escaped correctly by json.dumps. An unset
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
# back to a generated password would silently undo the point of this.
command:
- bash
- -c
- |
set -euo pipefail
mkdir -p /opt/airflow
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
exec airflow api-server --port 8080
command: airflow api-server --port 8080
ports:
- "8080:8080"
environment: &airflow-env
@@ -133,8 +113,6 @@ services:
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "admin:admin"
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:-admin}
PG_HOST: db
PG_PORT: "5432"
PG_USER: schoolcompare
-106
View File
@@ -98,12 +98,6 @@ fail the E2E gate. That's the point: staging absorbs the risk.
pr-checks status checks (frontend, backend, builds, ai-review) to pass.
5. **Bootstrap staging data via Airflow** (no prod dump — staging populates
itself from source, exercising the pipeline image end-to-end):
- Set `AIRFLOW_ADMIN_PASSWORD` in the stack environment first. The
api-server refuses to start without it. Airflow's simple auth manager
otherwise generates a password on first start and writes it to a file, so
the login changes every time the container restarts; the stack writes that
file itself from this variable instead. `AIRFLOW_ADMIN_USER` defaults to
`admin`.
- Open the staging Airflow UI (`http://<host>:8081`) and trigger, in order:
`school_data_daily`, `school_data_monthly_ofsted`, then the manual-schedule
`school_data_annual_ees` and `school_data_annual_idaci`.
@@ -156,103 +150,3 @@ token Gitea Actions provides automatically (`secrets.GITEA_TOKEN` — no setup
needed), and fails the check only when a finding is rated
**severe** (would break prod, leak data, or corrupt data). Minor findings are
informational and never block a merge.
## Rate limiting, and the Cloudflare gap
Two independent limits protect the API:
- **Per client**, via slowapi, keyed on `CF-Connecting-IP` (falling back to
`X-Forwarded-For`, then the peer address). 60/minute by default;
`/api/suggest` gets 120/minute because typing is bursty.
- **Globally**, via `GlobalRateLimitMiddleware`: a fixed 60-second window over
all `/api/` traffic, `GLOBAL_RATE_LIMIT_PER_MINUTE` (default 3000),
independent of any client identity. Requests from `127.0.0.1` are exempt so
the container healthcheck cannot be starved into a restart loop.
### Open: the origin must only accept Cloudflare
`CF-Connecting-IP` is only meaningful for requests that actually reached the
origin through Cloudflare, and **the application cannot verify that they did**.
Anything able to reach the origin directly can set that header freely and, by
rotating it, mint a fresh rate-limit bucket per request — defeating per-client
limits on every endpoint.
The global ceiling bounds the damage to total origin capacity. It does not fix
the underlying gap, and nothing in the code can. Closing it needs one of:
- **Authenticated Origin Pulls** — Cloudflare presents a client certificate the
origin requires, so non-Cloudflare traffic is refused at TLS.
- **An origin firewall** restricted to Cloudflare's published IP ranges.
Until one is in place, treat per-client limits as protection against accidents
and ordinary load, not against a determined caller.
## Feature flags (Unleash)
Flag state lives in a self-hosted Unleash instance, deployed as its own
Portainer stack from `docker-compose.portainer.unleash.yml`. It is separate
from the application stacks on purpose — redeploying staging must not be able
to disturb production's flags.
The flags themselves are declared in `backend/flags.py`. Unleash holds the
state; the registry holds the list. A flag in the UI that is not in the
registry is orphaned and nothing reads it.
### First-time setup
1. Deploy the stack in Portainer. Set `UNLEASH_DB_PASSWORD`,
`UNLEASH_ADMIN_PASSWORD` and (optionally) `UNLEASH_IP`.
2. Log in to the UI at `http://<UNLEASH_IP>:4242` as `admin`.
3. Create one **client** API token per environment:
- `schoolcompare-staging`, environment **development**
- `schoolcompare-prod`, environment **production**
Client tokens, not admin tokens — the backend only reads.
4. Put each token in the matching Portainer stack's `UNLEASH_API_TOKEN`
variable, and set `UNLEASH_URL` to `http://<UNLEASH_IP>:4242/api`.
5. Redeploy the application stacks.
### Adding a flag to Unleash
**Unleash does not create flags by itself.** The SDK reads definitions from the
server and never registers anything, and metrics for a flag the server has
never heard of are discarded. So a flag declared in `backend/flags.py` will be
evaluated on every request, stay `False` forever, and never appear in the UI
until someone creates it there by hand.
For each flag in the registry, create one in Unleash with:
- **Name** — character for character what `backend/flags.py` declares.
snake_case, no hyphens or spaces. A typo produces a flag that looks correct
in the UI and is read by nothing.
- **Type** — Release. No strategies, constraints or variants: these are plain
on/off switches, by design.
### Turning a feature on
Toggle the flag in the environment matching the stack you mean: **development**
for staging, **production** for prod. The token in each stack is scoped to one
environment, so toggling the other one has no visible effect.
The SDK refreshes every 15 seconds, so the API reflects the change almost at
once; the pages follow on their own schedule, below.
A flip reaches school pages within about five minutes and place pages within
the hour. Next's ISR does the propagating — it revalidates a route at the
*lowest* `revalidate` among that route's fetches, which is 300s for
`/school/[slug]` and 3600s for the place pages. There is no webhook, and
adding one would only be worth it if flips ever needed to be instant.
### When Unleash is unreachable
Every flag evaluates to `False` and the site serves as though nothing were
switched on. That is deliberate — an unfinished feature staying hidden is the
safe direction — but it means a *released* feature disappears if a backend
container cold-starts with an empty cache while Unleash is down. The SDK's
disk cache is on a named volume so restarts keep last-known state, and flags
are removed from the code within 90 days (enforced by a test), which bounds
how long any feature is exposed to this.
If `UNLEASH_URL` is unset, every flag is `False` and no connection is
attempted. That is the correct behaviour for local development and CI, and it
means the test suites need no flag server.
-85
View File
@@ -1,85 +0,0 @@
# Remote branch cleanup, 2026-08-20
# Restore any branch with: git push origin <sha>:refs/heads/<name>
## Deleted: fully merged into main (content is in main)
75677f4252b759ef895e7d5f7c19f8f1745bdb59 add-contact-form-footer
fa1abff642683dfd26ba88a295a0a6710d77147a chore/byline-removal-and-audit-figure
95081d38bdf87764ef5d298676c25fae4cd3b792 chore/remove-parent-view
6877abedebfc1c7d95f1f6ebe68945c62be528ab chore/staged-prod-promotion
090d5f7bec824e083d3252e2c6e636686016304d ci/frontend-checks-speedup
8c3a5cc4e9f551f0190d85357ce7741ad87f3a4c design/cohort-identity
955659580067ca8b81bd77e31e4e2554103f094d feat/allow-analytics-iframe-embed
6828f6cd4417284ea3eb6f088fa20945b8b40ed3 feat/compare-chips-two-per-row
d5cd0abfee226885119665da5d2aa8288b59217f feat/compare-data-foundation
6dd9b04b50bee146682da87efad8fc8b526251c5 feat/compare-frontend-rebuild
96d5fcf5b07b6f175b48e9b20fcb765a320a907f feat/detail-header-details-reveal
f1388ff5bd0af1409823a1e047b7ba84246a0f70 feat/gias-sixth-form-flag
eddf74745f86c9c6d9eb07d867246ff9cc90dc20 feat/hero-artwork-v2
3015c37bac6dc28db58c80fdb9235942c25b83d7 feat/hero-byline
8e763e39d17a4964cf558e51c03f044186371f6d feat/info-popover-tooltip
88c653215d520ab6e902c9de55bf27d86eeb90c3 feat/last-distance-offered
a72323874f7aebdb2e64b6d64a5febd61152d09d feat/last-distance-offered-full
c9a1892bfb0370e0672e5849cba294ddfabe3c65 feat/latest-cutoff-only
4e8df006d75d8be2a1d8529ddf855c445854cba1 feat/near-me-by-search
45ab479062c6a1639facad636fc0cc0cf0fd9155 feat/proposed-to-close-schools
3bf2e8f262cbe058fda6de6f8ea3e050224a51b9 feat/school-detail-visualisations
1f80571b1ff217dc92b660a936c02b5f49d07f0b feat/umami-heatmap-recorder
609bb923d96aa5730131463ea35d5efdd404bf96 feature/ingest-independent-schools
94151c58ea38a9256d66d15161293486a500c7a2 fix/admissions-section-height
79246edc22961c2beb3520437e9d064d3b809d10 fix/annual-dag-ks4-national-selector
59ac9c10b97e0ef1143f57fea06b324e72ac3d4a fix/chart-marker-contrast
9f8dba227c95706ca3527bd48d381e7622cc0a5e fix/compare-chart-refetch-resilience
e74d3882ce78a141fa1a57daa3102d7a58852dc3 fix/compare-expert-fixes
80176cac4db4820e76ea2a156c7a2974bee2f203 fix/compare-final-review-mustfix
f579630fab6c456e26a6984a3e8eebdbe3184202 fix/compare-mockup-drift
dc85254ad2ddf134b4434065d4762cef374d2220 fix/compare-null-year-blanks-chart
43a2c4a6bc539b621f31655aec05ef319a25f343 fix/compare-refresh-and-fetch
d677b5453365b72c81d6df2de62b1fa0d05d684d fix/daily-dag-cache-invalidation
f6bb037c471553e8195b5a8b147467ce0d07a688 fix/detail-all-through
17bd4d5a5eb0b12ca79b97db587f14f7d671e85b fix/detail-chart-truthfulness
4e6be0ce65647410b4ff74f8b763207c28920c26 fix/detail-inclusion-admissions
fdda52ff0af3fae03a4b059a655973cdbb269f91 fix/detail-ofsted-correctness
e36125b24aba254a8d15c5c33b4d2a296e691995 fix/detail-provenance-anchoring
b31e71ac884df6507f567adc046c9fc52d9d310a fix/detail-report-card-render-date
32f8a02862be6a1d49f4c3928b17fcf15d4c94cc fix/detail-trend-chart-taller
e65688d600a86818fe21ae4c61ba27e5b6ec8d7c fix/e2e-brand-assertions
3adea73ee04cdedfab54b0351878b297f72756ad fix/e2e-compare-chips-phase
06e4898c30feaedc471f97aba28ddb0d61379f4d fix/e2e-compare-samephase
9abd020967670a855e80fe5a908c8048a3aa9f14 fix/e2e-distance-locator
acec8135e1ec7c9c3c255e5b23733a0ef862b590 fix/e2e-rankings-year-pick
5944d88f0b1517ef1ef56af1b62270d2cf28e717 fix/expert-signoff-mustfixes
2433101fa08be5df6f170d41790512ffe823d33e fix/font-cascade-and-map-palette
74ca76d150deec6725259d9637ea86d7bb90c683 fix/gias-legacy-fallback
bdaa05cd542f563ef74c45307cd8f7fc465193c9 fix/hero-fallback-and-sharp
d52d384cf23d282b44e9251176f8f3402d808600 fix/hero-map-ios-fullscreen
4043270a77bbe4edb18207fa5f1d94d4747fe12f fix/hero-mobile-and-wording
22e9eb2d48b0d6623e88fd67cb6ca8e5e4074583 fix/homepage-education-accuracy
8d50afef1e8a2b621b7344eadf475b0d609ad7a2 fix/leaflet-specificity-and-font-assertion
b2b2cad5acf534ae7a667d3fb2be15efff37c4a7 fix/list-map-report-card-signal
dc21e80a5e9eebd13aaab84735642eb77cef35e6 fix/mobile-cell-name-size
e5f7f4c959f024c073472122d858333a0f24866c fix/mobile-compare-polish
a00cbe916182d1e04661c750a1ce2e7ad27a68ae fix/mobile-sort-select-overflow
2fd997bfe640c419a6713e85df008463de7f56e8 fix/modal-keyboard-viewport
3e7705756776a0c44d27966dfc972023f1f38b69 fix/ofsted-link-text
ce422e64363e2b03c186ef316832b6de15502d67 fix/promote-status-token
4522cbf64560db1e1cd519e119aa466b42cd16a1 fix/proposed-to-close-copy
15da060e4af37fbae919e0edf25f108266de2585 fix/rankings-admissions-accuracy
6c872ce726f210433354ca38dc5314bc6a467534 fix/rankings-year-validation
1c1df7796194af3d47f8e5ac0a0fbe6f323700e7 fix/report-card-chip-alignment
b2dc4d0779ced02709429afaa5985acf4abda794 fix/results-map-ios-fullscreen
95a5783da1fc994df76cb97238b55596dce4cd8f fix/runtime-api-proxy
8a9ba30cc24e29653a72c03c0e817684b7db7c07 fix/sats-per-level-national
536832a524fc4f9ed858f047e00b9a4d9d429c46 fix/school-detail-nan-500
fef83b3bf244a9bf3cb4afa75dbc433d8725014f fix/schoolbar-sticky-offset
b0c5b6bb57c879477da12c37b15cff950c29ebd3 fix/secondary-anchors-button-affordance
261403bcd2b01aa4f26ee212e26302fc0f769bf9 fix/special-note-full-width
ea5249a2ea6faf6bfa5a1522644387ba59770ec8 fix/standardise-distance-units
e4565e9f158721d4df6b918f2b065de82851f8d9 fix/trends-chart-height
3aad5101a842539105022f9059e85c224fc5973a fix/welsh-establishment-leak
315f1feede70bdf3101d2fdd305b3d06e037fdac perf/batch-supplementary
d2dc78aeb599e16b7ef5019be2b360b08df463bc perf/compare-loading
e098ad4bd1130152705788d4773837b7d1e7112e perf/server-client-split
## Deleted: superseded by PR #110 (content preserved on feat/seo-crawl-hygiene-main)
786ec80dd4de4a3cb674a89e35b3b5e461639289 feat/seo-crawl-hygiene
a5ac0bcd1b37bc10dcbce88f8601d01bf7b3eaaf feat/england-only-corpus
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -19,7 +19,7 @@
- Screenshot `j1-postcode-results-mobile.png` confirms cards carry only the LA name ("Solihull"), no "X miles away".
**Desktop (1440×900) — Attempt A repeat + keyboard + zoom.**
- Desktop hero additionally shows a trust badge ("● UPDATED WITH 2026/2027 ADMISSIONS RESULTS") and a value-prop subheading ("27,000+ primary and secondary schools with Key Stage 2 SATs, GCSE results, Ofsted grades, progress scores and admissions data — side by side, in one place"). **Both are absent on the mobile hero** (mobile jumps H1 → search box).
- Desktop hero additionally shows a trust badge ("● UPDATED WITH 2026/2027 ADMISSIONS RESULTS") and a value-prop subheading ("24,000+ primary and secondary schools with Key Stage 2 SATs, GCSE results, Ofsted grades, progress scores and admissions data — side by side, in one place"). **Both are absent on the mobile hero** (mobile jumps H1 → search box).
- Name search identical, correct, instant (`j1-search-results-desktop.png`).
- Keyboard: tab order is logical — Skip link → logo → Search/Compare/Rankings/Admissions nav → search input → Search button → Schools near me → content. Skip link, logo, nav links and Search button all get a clear **2px solid orange (#E07256) focus outline**. The search input uses an orange border + very faint ring (`box-shadow rgba(224,114,86,0.12) 0 0 0 3px`, `outline:none`) — visible but weaker than the other controls. Search → results → school link is fully keyboard-operable (standard links/buttons).
- 200% reflow (720×450): **no horizontal scroll** (`scrollWidth == clientWidth == 720`); deadline cards reflow from 1×4 to 2×2, no overlap or clipping. Pass.
@@ -57,7 +57,7 @@
- Severity guess: P2
- **F6. Mobile hero omits the value proposition shown on desktop**
- Evidence: Desktop hero has the "UPDATED WITH 2026/2027…" badge + subheading "27,000+ primary and secondary schools with KS2 SATs, GCSE results, Ofsted grades… side by side, in one place" (`j1-home-desktop-fold.png`). The mobile hero (`j1-home-mobile-fold.png`) drops both — below the poetic-but-vague H1 there is only a search box.
- Evidence: Desktop hero has the "UPDATED WITH 2026/2027…" badge + subheading "24,000+ primary and secondary schools with KS2 SATs, GCSE results, Ofsted grades… side by side, in one place" (`j1-home-desktop-fold.png`). The mobile hero (`j1-home-mobile-fold.png`) drops both — below the poetic-but-vague H1 there is only a search box.
- Criterion violated: Nielsen #1 (Visibility of system status) / recognition-over-recall; mobile content-parity best practice (the primary 63%-of-entries viewport should not lose the core "what is this and why trust it" copy).
- Argument: A first-time parent landing on mobile sees "Every school in England, compared." + a bare box, with no statement of coverage, data sources, or freshness. Weak/absent value proposition above the fold is a classic driver of immediate exits — directly relevant to the 46% home-exit rate.
- Severity guess: P2 (candidate P1 given mobile is the primary, highest-traffic viewport)
@@ -1,309 +0,0 @@
# SEO Programme — Design
Date: 2026-08-20
Status: awaiting review
## Problem
schoolcompare ranks second for "school compare" — an exact match for the
brand and the domain. It ranks poorly for "compare schools", "school
comparison" and "schools near me". The first is a naming artefact and
transfers to nothing; the rest are the queries that actually carry parent
demand.
The cause is structural, not editorial. The site publishes five route
families:
/ /compare /rankings /admissions /school/[slug]
Location intent has no landing page at all. Every competitor outranking us
on those queries wins with programmatic location pages:
| Competitor | URL pattern |
|---------------|--------------------------------------------|
| School Guide | `/best-schools-in/manchester` |
| Locrating | `/the-best-primary-schools-in-Manchester_…`|
| FindMySchool | `/best-primary-schools/manchester` |
| Snobe | `/best-primary-schools/manchester` |
| School Atlas | `/guides/best-primary-schools-manchester` |
"Schools near me" is a local-intent query. Google resolves it against the
user's coordinates and serves pages that are *about a place*. A national
homepage cannot win it. No title or description change fixes this; only
pages Google can localise will.
## Baseline (measured 2026-08-20, production API)
| Measure | Value |
|--------------------------------------------|---------|
| Unique schools | 27,229 |
| URLs in sitemap.xml | 27,232 |
| Schools with 2024/25 performance data | 21,266 |
| Schools with **no** current performance data| ~5,963 (22%) |
| Welsh establishments (all metrics null) | 1,569 |
| Overseas / offshore establishments | 467 |
| Static URLs in sitemap | 3 |
| Routes setting a canonical | 1 of 5 |
Three findings from that table drive the plan.
**We submit ~6,000 thin pages to Google.** `build_sitemap()`
(`backend/app.py:73`) enumerates every URN regardless of whether the school
has any data. Welsh establishments return `school_type: "Welsh
establishment"` with every performance metric, Ofsted grade and phase field
null. Overseas and offshore establishments ("BFPO Overseas Establishments",
"Gibraltar Overseas Establishments", "Jersey Offshore Establishments") are
in the local-authority list too. At 22% of the submitted corpus this is a
site-wide quality signal problem and a crawl-budget waste, not a rounding
error.
W1 item 4 removes 2,036 of those — every non-England establishment — taking
the corpus to 25,193. The 3,927 that remain are English schools with no
current data: mostly newly opened, special, nursery or alternative provision.
Those are a template problem, not a corpus problem, and item 5 handles them
separately.
**The homepage is its own competitor.** `app/page.tsx` accepts eleven search
params (`search`, `local_authority`, `school_type`, `phase`, `page`,
`postcode`, `radius`, `sort`, `gender`, `admissions_policy`,
`has_sixth_form`) and sets no canonical. Every filter combination is a
crawlable near-duplicate of the single page we are asking to rank for
"compare schools".
**School pages are near-orphans.** Reachable from the sitemap and from site
search, but almost nothing links to them contextually, so they accrue no
internal authority.
Also noted: the sitemap emits invented `priority` values and no `lastmod`.
Google ignores `priority` and `changefreq` entirely; `lastmod` is the field
it does read, and we omit it.
## Keyword clusters
Ranked by judgement of UK parent search behaviour and by the competitive
SERP evidence above. Google Search Console is connected, so cluster
priorities are to be re-derived from measured impressions before build
starts (see Workstream 0).
**C1 — Head "compare" terms.** compare schools · school comparison · school
comparison tool · compare school performance · compare primary schools ·
compare secondary schools · compare two schools
**C2 — League tables and rankings.** primary school league tables ·
secondary school league tables · school league tables 2026 · SATs results by
school · GCSE results by school · KS2 league tables · Progress 8 rankings ·
best primary schools in [town] · top 10 primary schools in [LA]
**C3 — Local / near me.** schools near me · primary schools near me ·
secondary schools near me · best schools near me · good schools near me ·
schools in [town] · primary schools in [LA] · schools near [postcode] ·
[postcode] school catchment
**C4 — Individual school long tail.** [school] ofsted · [school] SATs
results · [school] catchment area · [school] reviews · [school] URN
**C5 — Admissions.** primary school admissions 2027 · national offer day
2027 · school application deadline · school admissions appeal ·
oversubscription criteria · distance criteria school admissions · didn't get
first choice school · school admissions [LA]
**C6 — Metric explainers.** what is a good SATs score · what is Progress 8 ·
what is Attainment 8 · expected standard KS2 meaning · scaled score
explained · Ofsted grades explained · Ofsted report cards · pupil premium
explained
**C7 — Head to head.** [school A] vs [school B] · academy vs community
school · grammar school vs comprehensive · faith school vs community school
## Workstreams
### W0 — Measure before touching anything
Export a Google Search Console baseline: impressions, clicks, average
position and CTR by query and by page, for the trailing 16 months. Bucket
queries into C1–C7. This sets the counterfactual — without it, no later
claim about lift is defensible, because school-search traffic is strongly
seasonal (results day in December, offer day in March/April).
Re-rank C1–C7 against measured impressions and adjust the sequence below if
the data disagrees with the judgement calls.
### W1 — Crawl hygiene and index sanity
Cheap, and it unblocks everything after it. Adding 5,000 pages on top of a
corpus that is 22% thin would compound the existing problem.
1. Canonical on every route. `/`, `/rankings`, `/compare` and `/admissions`
currently set none.
2. The homepage canonicalises to `/` regardless of search params.
3. `/compare?urns=…` gets `noindex, follow` — it is an unbounded parameter
space with no standalone value.
4. **England only — DONE.** Wales, the Crown Dependencies, Gibraltar and the
service/overseas schools are removed from the corpus at the mart boundary,
not hidden at the view layer. `dim_school` and `dim_location` both exclude
`TypeOfEstablishment` in {25, 26, 30, 37} — Offshore schools, Service
children's education, Welsh establishment, British schools overseas —
listed once as `vars.non_england_school_type_codes` in `dbt_project.yml`.
That removes 2,036 establishments and 29 local authorities, and because
`build_sitemap()` reads the same marts, it drops them from the sitemap in
the same stroke. `assert_england_only_schools` fails the pipeline if a GIAS
refresh reintroduces them or if the two models drift apart.
5. Prune the remaining thin pages: exclude any school with no performance data
**and** no Ofsted record. Distinct from item 4 — these are English schools
with nothing yet to show, so the fix may be a better template rather than
removal.
6. Rebuild the sitemap as a sitemap **index**: one child per page family,
real `lastmod` from the data-load timestamp, `priority` and `changefreq`
dropped.
### W2 — The location layer
The dominant lever. `dim_location` already carries `town`, `county`,
`local_authority_name`, `parliamentary_constituency`, `latitude`,
`longitude` and `postcode`, so no new ingestion is required.
Routes:
/schools/[la] e.g. /schools/manchester
/schools/[la]/primary
/schools/[la]/secondary
/best-primary-schools/[town]
/best-secondary-schools/[town]
/schools/near/[outcode] e.g. /schools/near/m20
/schools/near-me geolocating hub
**Thin-page threshold: generate a town or outcode page only where at least
five schools have current performance data.** Below that, 301 to the parent
LA page. This is the single most important constraint in the workstream —
it is what separates a location layer from index bloat.
Each page must earn its place with content a parent would actually use, not
a template shell:
- H1 matching the query intent ("Best primary schools in Manchester")
- Counts framed usefully: "137 primary schools, 9 rated Outstanding"
- A ranked table of the top 20 on the headline metric
- Local average against the England average
- Ofsted grade distribution
- Map
- Links to neighbouring towns and to the parent LA
- An FAQ block (feeds `FAQPage` in W4)
- Links to every school page in scope — this is what de-orphans W1's corpus
Sizing estimate: ~150 usable LAs × 3 ≈ 450; towns clearing the threshold
≈ 1,200 × 2 ≈ 2,400; outcodes ≈ 2,300. Roughly **5,000 new pages**,
comfortably inside a sitemap index and well under the per-file 50,000 limit.
### W3 — Make rankings indexable
`/rankings` is driven entirely by query params, so Google indexes
approximately one page where there should be hundreds.
/rankings/[phase]/[metric]
/rankings/[phase]/[metric]/[la]
The interactive filter UI stays; its state moves into real paths. Param
forms canonicalise to the clean path. This is the direct play for C2.
### W4 — Structured data and internal linking
- Replace the bare `EducationalOrganization` on school pages with `School`,
and populate it properly.
- `BreadcrumbList` site-wide.
- `ItemList` on every rankings and location page.
- `FAQPage` on admissions and on location pages.
- New school-page modules: "Other schools in [town]", "Nearby schools",
"Compare with similar schools". Each links out to W2 and W3 pages, which
is what circulates authority instead of stranding it.
Explicitly **not** doing `Dataset` or `AggregateRating` — no review corpus
exists, and fabricating one would be both useless and a policy violation.
### W5 — Admissions expansion
One static page currently carries an entire cluster.
/admissions/[la] per-authority deadlines and offer day
/admissions/appeals
/admissions/national-offer-day
`school admissions [LA]` is high-intent and highly seasonal; per-authority
pages are the natural unit.
### W6 — Explainer content
/guides/progress-8
/guides/attainment-8
/guides/sats-scaled-scores
/guides/ofsted-grades
/guides/expected-standard
Each links into the corresponding W3 rankings page. Cheap to build, and it
is what gives the metric vocabulary enough topical weight to support C1–C4.
### W7 — Head-to-head pages
/compare/[school-a]-vs-[school-b]
C7 is uncontested and native to the product. It is also the easiest way to
destroy everything W1 fixes: 27,229 schools generate 370 million pairs.
**Curated pairs only** — same town, both with current data, both with real
search demand — capped in the low thousands. Gated behind evidence that W2
is indexing cleanly.
### W8 — Metadata rewrite for C1
Current homepage title is `schoolcompare | Compare every school in England`,
which spends the most valuable position on the brand. Rewrite the homepage,
rankings and compare titles and descriptions around C1 phrasing. Small
change, and the cheapest item in the programme.
## Sequencing
W0 → W1 → W2 → W3 → W4 → W5 → W6 → (W7 if W2 indexes cleanly)
W1 before W2 is not negotiable: adding pages to a corpus that is 22% thin
compounds the problem rather than diluting it.
This spec is a programme, not a single implementation plan. Each workstream
gets its own plan and its own PR; W2 will likely need several. Only W0 and W1
are ready to plan against today — the rest should be re-read after W0's
Search Console baseline lands, because that data may reorder them.
## Testing
Per CLAUDE.md, user-facing behaviour changes extend the `e2e/` journeys in
the same PR. Each workstream adds:
- W1: canonical present and correct on every route; `/compare?urns=` carries
`noindex`; sitemap excludes a known dataless URN.
- W2: a known LA, town and outcode page renders with the expected school
count; a below-threshold town redirects to its LA.
- W3: a clean rankings path renders; the param form canonicalises to it.
- W4: JSON-LD parses and validates against the declared types.
## Risks
**Index bloat.** The failure mode of every programmatic SEO programme. The
five-school threshold, the W1 prune and the W7 gate are the three controls.
**Helpful-content exposure.** Google's stance on templated location pages
has hardened. The mitigation is that each page carries genuinely local
computed data — real counts, real distributions, real local-vs-national
comparison — rather than a name substituted into boilerplate.
**Build cost.** School pages already use ISR with a 7-day revalidate and
`PRERENDER_SCHOOLS` gating full prerender. 5,000 more routes need the same
treatment; full static generation of 32,000 pages is likely impractical in
CI.
**Seasonality.** Results day and offer day dominate the traffic curve.
Judging the programme on a mid-summer window would misread it in either
direction. W0's baseline must be year-on-year, not month-on-month.
## Open questions
1. Catchment areas are Locrating's moat and a strong C3 driver
(`[postcode] school catchment`). `fact_admissions` carries admission
distances. Is deriving approximate catchment a later workstream, or out
of scope?
@@ -1,278 +0,0 @@
# W2: The Location Layer — Design
Date: 2026-08-21
Status: awaiting review
Supersedes: workstream W2 in `2026-08-20-seo-programme-design.md`
Scope note: this covers four page families in one spec. Splitting them — towns
and authorities first, outcodes and localities after — was proposed and
declined in favour of building the layer in one pass. The decomposition
argument was that the curated locality seed needs human review and would hold
up 783 pages of measured demand behind it; that risk is accepted here, and the
implementation plan should sequence the seed early enough that review time
does not become the critical path.
## Problem
Location intent is the largest unserved demand the site has. In the 16-month
Search Console baseline it draws **874 impressions, one click, average
position 49.5**. The site does not compete.
Unlike named-school queries — which the same baseline showed to be
navigational and unwinnable, since a parent typing "audley junior school"
wants that school's own website — location queries have no incumbent owner.
Nobody owns "primary schools in Brentwood" the way a school owns its name.
The cause is structural: the site has no page about a place. Every competitor
ranking above it does.
## What the demand actually looks like
Every location query in the baseline is **town or district level**. Not one is
an administrative area:
| Query | Impressions | Position |
|-------|-------------|----------|
| colleges in solihull | 112 | 51.2 |
| schools in ramsey | 64 | 42.5 |
| schools in crosby | 57 | 47.7 |
| primary schools in beccles | 44 | 40.9 |
| private schools in battersea | 41 | 71.9 |
| secondary schools in brentwood | 37 | 56.1 |
| secondary schools in canary wharf | 30 | 35.9 |
Three patterns follow directly, and they drive the whole design.
**Towns, not authorities.** The superseded W2 put `/schools/[la]` first and
towns second. The data inverts that. Brentwood appears four times in different
phrasings; Beccles twice. Both are towns, not authorities.
**Phase is part of the query**, not a filter applied afterwards: "primary
schools in beccles", "secondary schools in brentwood", "colleges in solihull".
**London is searched by district** — Battersea, Canary Wharf — and the GIAS
`town` field cannot serve it at all.
## Measured sizing
Counted against the live corpus of 25,185 schools, not estimated.
| Family | Viable (≥5 schools) | Below threshold |
|--------|--------------------|-----------------|
| Towns | **783** | 907 → redirect to authority |
| Outcodes | **1,760** | 305 |
| Local authorities | 154 | — |
| London localities | ~100–150 (curated) | — |
With phase variants — 783 town pages plus roughly 700 primary and 250
secondary variants, 154 authorities across three variants, 1,760 outcodes and
the curated localities — the total lands near **4,000 pages**. Phase variants
need their own threshold: there are 17,426 primaries but only 4,456 secondaries nationally,
so most towns will support a primary page and not a secondary one.
## Two design problems this spec exists to solve
### 1. Town and authority names collide, and neither contains the other
67 viable towns share a name with a local authority. The obvious fix — let the
authority absorb the town, since it sounds like a superset — **does not work**:
| Place | Schools in the town | Schools in the authority |
|-------|--------------------|-----------------------|
| Bedford | 104 | 86 |
| Birmingham | 520 | 518 |
| Derby | 157 | 119 |
| Doncaster | 152 | 145 |
The authority is the larger set in only 43 of the 67. Postal towns cross
authority boundaries, so these are overlapping sets that happen to share a
name. Publishing both into one namespace produces near-duplicate pages, which
is the specific failure that sinks programmatic SEO.
**Resolution: two namespaces.**
```
/schools/[place] towns and London localities
/schools/[place]/primary
/schools/[place]/secondary
/schools/authority/[la] local authorities
/schools/authority/[la]/primary
/schools/authority/[la]/secondary
/schools/near/[outcode]
```
Outcodes carry no phase variants: nobody searches "primary schools in SW11",
so the variants would be pages without demand.
Every collision disappears by construction. `/schools/[place]` keeps the clean
URL for the pattern that carries the demand; authorities get a namespace whose
purpose is genuinely different — admissions are authority-run, and the
authority page is the one that can speak to catchment policy and LA averages.
A place page and an authority page of the same name must each say plainly
which set of schools they cover, or they read as duplicates to a reader even
when they differ in fact.
### 2. London has no locality field
`town` collapses **1,819 London schools into the single value "London"**. A
page listing all of them is useless, and borough pages do not help because
people search "Battersea", not "Wandsworth".
No single field solves it:
| Search term | `parliamentary_constituency` | postcodes.io `admin_ward` |
|-------------|------------------------------|---------------------------|
| Battersea | **Battersea** ✓ | Northcote / Wandsworth Town ✗ |
| Canary Wharf | Poplar and Limehouse ✗ | **Canary Wharf** ✓ |
| Vauxhall | Vauxhall and Camberwell Green ✗ | **Vauxhall** ✓ |
And neither covers Clapham, Shoreditch or Peckham, which are postal and
colloquial rather than administrative.
**Resolution: a curated seed mapping locality to outcodes.**
```
pipeline/transform/seeds/locality_outcodes.csv
locality_slug,locality_name,outcodes,region
battersea,Battersea,"SW11|SW8",London
canary-wharf,Canary Wharf,"E14",London
clapham,Clapham,"SW4|SW9",London
```
This needs **no new ingestion** — the corpus already has postcodes. It puts
the fuzzy, contested part of the problem in a reviewable file rather than in
derived logic, which suits it: locality boundaries are a judgement, not a
fact. The repo already uses dbt seeds for curated reference data
(`la_code_names.csv`, `gias_code_names.csv`), so this follows an established
pattern.
The seed generalises past London. Any colloquial place — Jesmond, Chorlton,
Clifton — can be defined by its outcodes without a schema change.
**Constraint:** a locality slug may not collide with a viable town slug. The
place registry enforces this and fails the build rather than silently
shadowing a town.
## Architecture
### The place registry
One module owns the question "what places do we publish, and what is in each".
Everything else reads from it: the pages, the sitemap, the internal links.
```
backend/places.py
Place = { kind: "town"|"locality"|"authority"|"outcode",
slug, name, urn_list, parent_authority | None }
build_place_registry(df) -> dict[str, Place]
place_schools(slug, phase=None) -> list[School]
```
Built once at startup from the same DataFrame the sitemap uses, and rebuilt by
the existing `/api/admin/regenerate-sitemap` path after a pipeline run.
Registry construction is where the threshold, the collision rules and the
seed's uniqueness constraint are enforced — in one place, testable without a
browser or a database.
### API
```
GET /api/places the registry: slug, kind, name, count
GET /api/places/{slug}?phase= aggregate + ranked schools for one place
```
`/api/places` is what the sitemap and the internal-link modules enumerate.
### Routes
Next App Router, ISR with the same 7-day revalidate the school pages use.
`generateStaticParams` gated behind an env flag, matching
`PRERENDER_SCHOOLS`, because 3,900 more routes cannot be statically built in
CI on every deploy.
## What each page must contain
A place page that is a name substituted into a template is the thing Google's
helpful-content stance exists to demote. Each page carries computed local
facts that exist nowhere else on the site:
- **H1** matching the query: "Primary schools in Brentwood"
- **Counts framed usefully**: "29 schools, 4 rated Outstanding"
- **A ranked table** of the top 20 on the phase's headline metric —
`rwm_expected_pct` for primary, `attainment_8_score` for secondary, and for
an unphased place page the metric matching whichever phase it holds more of
- **The local average against the England average** — the one number a parent
cannot get from a list
- **Ofsted grade distribution** for the place
- **A map**
- **Links to neighbouring places** and to the parent authority
- **An FAQ block**, feeding `FAQPage` structured data
- **A link to every school page in scope** — this is what finally de-orphans
the 23,000 school pages the original spec identified as near-orphans
## Thin-page controls
Three, and they are the difference between a location layer and index bloat:
1. **Five schools with current data minimum.** Below it, 301 to the parent
authority. This drops 907 towns and 305 outcodes.
2. **Per-phase thresholds.** A town with 30 primaries and 2 secondaries
publishes a primary page and no secondary page.
3. **No page without a local average.** If a place has too few schools with
results to compute one, it has nothing to say that a list does not, and it
falls back to the authority.
## Sitemap
Two new children in the existing index: `/sitemaps/places-{n}.xml` and
`/sitemaps/outcodes-{n}.xml`. Per-family children are why the index was built
in W1 — Search Console reports coverage per submitted sitemap, so indexation
of the location layer is measurable separately from the school pages.
## Testing
Per `CLAUDE.md`, user-facing behaviour extends `e2e/tests/journeys.spec.ts` in
the same PR.
**Unit (registry, no DB):** threshold enforcement; a sub-threshold town
resolves to its authority; a locality slug colliding with a town fails the
build; Bedford's town and authority pages hold different URN sets; per-phase
thresholds.
**Backend:** `/api/places` shape; `/api/places/{slug}` aggregate correctness
against a fixture; unknown slug 404s.
**e2e:** a known town, authority, locality and outcode page each render with
the expected count; a below-threshold town 301s; every place page declares a
canonical and appears in the sitemap; `/schools/bedford` and
`/schools/authority/bedford` both resolve and state which set they cover.
## Risks
**Index bloat** is the failure mode of every programmatic SEO programme. The
three controls above are the answer, and the per-family sitemap is how we find
out early if they were not enough.
**Helpful-content exposure.** Templated location pages are exactly what
Google's stance targets. The mitigation is that every page carries real
computed local data — counts, distributions, local-versus-national comparison
— rather than a name dropped into boilerplate. If indexation of the places
sitemap stalls below roughly half, that is the signal to stop and rethink
rather than to add more pages.
**Build cost.** ~4,000 additional ISR routes on top of 23,000 school pages.
The env-flag gate on `generateStaticParams` keeps CI viable.
**Curation drift.** The locality seed is hand-maintained and will go stale as
places change. It is small and reviewable, and a dbt test asserts every seed
outcode matches at least one school so a typo fails the pipeline rather than
publishing an empty page.
## Out of scope
Catchment-area estimation. It is a strong driver for this cluster and
`fact_admissions` carries the distances, but it is a modelling problem with
real accuracy risk and deserves its own design.
@@ -1,294 +0,0 @@
# Feature Flags — Design
**Date:** 2026-08-23
**Status:** approved for planning
**First consumer:** the last-distance-offered feature (`admission_distance`)
## Goal
Let work merge to `main` and deploy to production without becoming visible,
so that releasing a feature stops being the same event as deploying it.
The site has no way to do this today. A feature is either on `main` and live,
or it is on a branch. That forces long-lived branches for anything not ready,
and it makes every promotion to production an all-or-nothing decision about
everything queued behind it.
This is a **ship-dark** capability, not a kill switch. Flags are expected to
flip on the order of once a month, by a person, deliberately. Nothing here is
designed for flipping something off in seconds under pressure, and nothing
here does percentage rollouts, user targeting or A/B tests — the site has no
user identity to target.
## Decision: Unleash
Flag state is held in a self-hosted [Unleash](https://www.getunleash.io/)
instance (Apache-2.0), not in the repository.
A lighter option was considered and rejected by the project owner: a typed
registry in each runtime with environment-variable overrides set in the
Portainer stack files, which would have needed no new container and kept flag
state in git. The argument for Unleash is that it provides a UI and an audit
log without a deploy, and that flags are expected to become an ongoing
operational tool rather than an occasional one.
Two consequences follow from choosing a service, and this design exists mostly
to handle them:
1. **Flag state lives outside the repository.** `main` is no longer the whole
truth about what is switched on. The registry in §2 exists to bound that.
2. **A flag can change without a deploy**, so nothing else clears the caches
that a deploy would have cleared. §4 establishes how long a flip takes to
become visible, and why that is short enough to need no extra mechanism.
Also considered: Flagsmith (heavier — Django, Postgres and Redis), GrowthBook
(requires MongoDB), and Flipt v2 (the closest conceptual fit, git-native, but
now under the Fair Core Licence — source-available, not OSI open source).
## 1. Topology
A third Portainer stack, `docker-compose.portainer.unleash.yml`, holding
`unleashorg/unleash-server` and its own PostgreSQL 16. It is on the macvlan so
both application stacks can reach it, and it belongs to neither of them — a
staging redeploy must not be able to disturb production's flag state, and vice
versa.
One instance serves both environments. Open-source Unleash ships with
`development` and `production` environments and environment-scoped client
tokens, so the same flag holds independent state in each: staging's FastAPI
carries a `development` token, production's carries a `production` one.
That property is what makes ship-dark testable. A feature can be **on in
staging and off in production** for as long as it takes, which means the `e2e/`
journeys exercise it against staging while production stays unchanged.
## 2. The registry
Unleash supplies flag *state* and the toggle UI. It does not supply the list of
flags. `backend/flags.py` declares every flag the code knows about:
```python
@dataclass(frozen=True)
class Flag:
name: str # identical in the registry, in Unleash, and in JSON
description: str # one line: what turning this on reveals
added: date # for the staleness test in §8
```
**Every flag defaults to `False`.** There is no per-flag default field, because
a flag that defaults on is not a ship-dark flag — it is a kill switch, and this
design does not offer one. A single unconditional default also means the
fallback path has no branching to get wrong.
Three reasons the registry is not optional:
- The Unleash SDK evaluates an unknown flag to `False`. Without a registry that
is an *undeclared* false — indistinguishable from a typo in a flag name.
- `/api/flags` needs a key set to return when Unleash is unreachable. It cannot
enumerate flags it has never heard of.
- A flag present in the Unleash UI but absent from the registry is orphaned,
and should be visibly so rather than quietly authoritative.
**Naming.** One string, used unchanged as the registry key, the Unleash flag
name, and the JSON key in `/api/flags`. It is snake_case, matching the API's
existing convention (`admission_distance`, `rwm_expected_pct`) and the mirrored
types in `nextjs-app/lib/types.ts`. No case transformation anywhere, so there
is no mapping layer to get wrong.
## 3. Read paths
### Backend
`backend/flags.py` wraps `UnleashClient` behind `is_enabled(name: str) -> bool`.
Fail-closed is the default rather than something added: the Python SDK
evaluates every flag to `False` until it has synchronised with the server. An
unfinished feature therefore stays hidden when Unleash is unreachable, which is
the correct direction for ship-dark.
The SDK's fcache directory is mounted on a named volume so a container restart
during an Unleash outage keeps last-known state rather than reverting a
released feature to dark. The registry default remains `False`, so the worst
case is a feature disappearing, never one appearing.
### Frontend
`nextjs-app/lib/flags.ts` exposes `getFlags(): Promise<Flags>`, a single
server-side fetch of `/api/flags` returning a typed record. Server components
only — no flag value reaches the browser bundle, and `package.json` gains no
Unleash dependency. The Unleash client library stays entirely inside the
service that already owns every other piece of data the frontend renders.
The cost, named plainly: a purely front-end flag must still be declared in a
Python file. It is a flat data edit rather than programming, and the return is
one list, so nobody has to ask which service knows about a given flag.
### `/api/flags` must not be publicly reachable
`nextjs-app/app/api/[...path]/route.ts` proxies **everything** under `/api/` to
FastAPI. Left alone, `https://www.schoolcompare.co.uk/api/flags` would return
`{"admission_distance": false, ...}` — publishing the name and state of every
unreleased feature, which defeats the purpose of shipping dark.
The proxy therefore gains a denylist, and `flags` is on it: a request for a
denied path returns 404 rather than being forwarded. Next's own `getFlags()` is
unaffected because it calls `FASTAPI_URL` directly across the Docker network
and never transits the public proxy.
This is a general hole rather than a flags-specific one — the proxy will
forward any future internal endpoint too — so the denylist is written as a
named constant with a comment saying what belongs on it.
## 4. Propagation
**Time-based revalidation is sufficient. There is no webhook.**
An earlier draft of this section specified two Unleash webhooks and a
`revalidateTag('flags')` purge, on the premise that pages cache for seven days.
That premise was wrong, and checking it removed the most complex part of the
design.
Next uses the **lowest** `revalidate` among a route's fetches to set the
revalidation frequency of the whole route — the segment-level
`export const revalidate` does not override a lower value inside it. Measured
against this codebase:
| Page family | Segment | Lowest fetch | Effective |
|---|---|---|---|
| `/school/[slug]` | 604800 | `fetchSchoolDetails` at 300 | **5 minutes** |
| `/schools/*` | 604800 | `fetchNationalAverages` at 3600 | **1 hour** |
The Unleash SDK polls every 15 seconds, so a flip reaches school pages within
about five minutes and place pages within the hour, unaided. Flags flip
monthly, by hand, deliberately. That is fast enough.
What this removes: two webhook integrations, a `/api/revalidate-flags` route, a
shared-secret-in-a-query-string scheme, an idempotency requirement against
duplicate and out-of-order delivery, and a rule that every fetch in
`nextjs-app/lib/` carry a cache tag. None of it has to be built, maintained, or
kept correct as new fetches are added.
**If instant flips are ever wanted**, the webhook is the way to add them, and it
is purely additive — nothing in this design has to change first.
### Two constraints this leaves behind
**Never flag content on a `force-static` page.** `app/admissions/page.tsx`
declares `export const dynamic = 'force-static'`, so it is baked at build time
and never revalidates. A flag gating anything on such a page would not take
effect until the next deploy, silently. If a flag ever needs to reach one, that
page must first move to ISR.
**A route-family flag still needs the sitemap rebuilt.** The sitemap is held in
memory and rebuilt only at startup or via `POST /api/admin/regenerate-sitemap`.
No flag in scope touches the sitemap (§6), so this is deferred with the route
case rather than solved now — but a route flag must not ship without it, or the
sitemap will advertise URLs that `notFound()`.
## 5. What "off" means, per surface
| Surface | Off |
|---|---|
| Route | `notFound()`, **and** absent from the sitemap, **and** absent from nav |
| UI element | Not rendered; surrounding page byte-identical to today |
| API field | Key **absent**, not `null` |
| API endpoint | 404, not 403 |
The three parts of the route rule move together or not at all. Submitting URLs
to Google that return 404 is the bug fixed in PR #124, and a flag is a new way
to reintroduce it.
An API field is withheld **at the source**, never rendered-but-hidden. The
precedent is already set in this codebase by commit `c9a1892`: `/api/schools/`
is public and unauthenticated, so leaving a withheld field in the payload hands
the record to anyone who opens the network tab.
## 6. First consumer: `admission_distance`
The last-distance-offered feature is merged to `main` and live on staging.
Production has never received it: `/api/schools/100010` on production carries
no `admission_distance` key, and no Distance section renders.
It needs **exactly one gate** — `backend/app.py:809`, where the field is
attached to the school payload:
```python
"admission_distance": (
supplementary.get("admission_distance")
if flags.is_enabled("admission_distance") else None
),
```
The frontend follows with no change. `DistanceSection` already returns `null`
when `admission_distance?.distance_m == null`, and `PrimarySchoolSections`
already conditions the admissions block on `(admissions || admissionDistance)`.
The off-state is the commonest state on the site — only 57 local authorities
publish cut-off distances at all — so it is well covered by construction.
The flag does not touch the sitemap: school pages exist either way.
Intended lifecycle: default off, so production receives the code dark on the
next promotion; on in the `development` environment so staging keeps testing
it; flipped on in `production` when the owner chooses.
**This flag exercises two of the three surfaces** in §5 — API field and UI
element. No route case ships with it. The route rule is specified but unproven
until a route-shaped flag exists, and should be treated as such.
## 7. Testing
**Backend unit.** The registry is well-formed; an unknown flag evaluates
`False`; `/api/flags` returns every declared flag with its default when the
SDK is unreachable; `admission_distance` is absent from the school payload when
the flag is off and present when on.
**Frontend unit.** `getFlags()` returns declared defaults when `/api/flags`
fails, rather than throwing and taking the page with it.
**E2E.** Journeys read `/api/flags` and gate flag-dependent assertions on it,
matching the `test.skip` shape the suite already uses.
One trap to avoid, worth stating because the existing distance journeys walk
straight into it: they already skip when no school has a published figure, so
with the flag off they would skip silently and the suite would go green. The
gate must be explicit — **if `/api/flags` reports `admission_distance` on, then
a school with a cut-off must be found**, converting a silent skip into a real
assertion.
## 8. Lifecycle
A flag is temporary scaffolding, and the failure mode of every flag system is
accumulation.
The registry records the date each flag was added, and a backend test fails any
flag older than **90 days**. Removing a flag means deleting the registry entry,
the branches that read it, and the flag in the Unleash UI.
Unleash SDK usage metrics stay enabled, so the UI shows which flags are still
being evaluated — the evidence needed to retire one safely.
## 9. Risks
**Production gains a homelab dependency.** If Unleash is unreachable when a
production container cold-starts with an empty cache, every flag evaluates
`False` and any feature currently switched on disappears. The fcache volume
covers restarts; the 90-day lifecycle rule bounds how long any feature is
exposed to this. It is a real regression risk and the reason flags must be
retired rather than left on indefinitely.
**Flag state is not in git.** `main` no longer tells you what production is
showing. The registry lists what *can* be flagged; only the Unleash UI says
what *is*. This is inherent to the choice of a service.
**A large promotion backlog exists.** Production is running the
pre-SEO-programme build — no place pages, and a sitemap still declaring the
apex host. The first promotion after this work ships that entire backlog. The
flag isolates the distance feature from it and nothing else.
## Out of scope
- Percentage rollouts, user targeting, A/B testing, and Unleash strategies
beyond simple on/off. Flags are booleans.
- Pipeline and dbt flags. Airflow and dbt are not flag consumers.
- Client-side flag evaluation. Flags are server-side only.
- Automatic flag removal. The staleness test reports; a person deletes.
@@ -1,294 +0,0 @@
# School Autosuggest — Design
**Date:** 2026-08-26
**Status:** approved for planning
**Depends on:** the feature-flag layer (PR #125, merged)
## Goal
Suggest schools by name as someone types in the site's main search box, so a
parent who knows the school they want reaches it in one step instead of
searching, scanning a result list, and clicking.
Scope is **schools only**. Places and postcodes were considered and excluded —
see *Out of scope*.
## The finding that shapes everything
The site's rate limiter does not do what it looks like it does.
`limiter = Limiter(key_func=get_remote_address)` with `60/minute` reads
`request.client.host`. In staging and production the backend has no published
ports and sits on the internal `backend` network, so its only caller is the
Next proxy — and `request.client.host` is therefore **the Next container**, for
every browser user on the site.
Measured against staging: 70 concurrent requests to `/api/schools` returned
**60 × 200 and 10 × 429**. One machine consumed the whole site's budget for
that minute.
Autosuggest is the worst possible feature to build on that. One person typing
"st marys primary" produces six to eight debounced requests; **eight concurrent
searchers would 429 the site.** The compare modal's search-as-you-type already
shares this bucket, so the exposure exists today — autosuggest makes it
certain.
Fixing the keying is therefore part of this work, not a follow-up.
## 1. Rate-limit keying
Both environments sit behind Cloudflare (`server: cloudflare`, `cf-ray` present
on staging and production). Cloudflare sets `CF-Connecting-IP` on every request
to the origin and **overwrites any client-supplied value**, which makes it
trustworthy in a way a parsed `X-Forwarded-For` chain is not.
```python
def client_key(request: Request) -> str:
"""Rate-limit bucket: the real caller, not the proxy in front of them."""
cf = request.headers.get("cf-connecting-ip")
if cf:
return cf.strip()
xff = request.headers.get("x-forwarded-for")
if xff:
return xff.split(",")[0].strip()
return get_remote_address(request)
```
`nextjs-app/app/api/[...path]/route.ts` already forwards every inbound header
except `host` and `connection`, so `CF-Connecting-IP` reaches the backend with
no proxy change.
**This header is trustworthy only for traffic that actually passed through
Cloudflare, and nothing in the application can verify that it did.** An earlier
draft of this section claimed Cloudflare "replaces the header, so a browser
cannot forge it", and that only the `X-Forwarded-For` fallback was forgeable.
That was wrong. Cloudflare does overwrite the header *on requests it handles* —
but a caller reaching the origin directly sets whatever it likes, and this
process cannot distinguish an edge-set header from an attacker-set one. Both
headers are equally forgeable in that scenario.
The consequence is sharper than a weakened defence. An attacker rotating
`CF-Connecting-IP` per request mints a fresh rate-limit bucket every time and
evades per-client limits entirely — including on the DataFrame-heavy
`/api/schools`. Against abuse that is *worse* than the shared bucket it
replaced, which at least capped everyone at 60/minute together.
Two mitigations, and they are not interchangeable:
1. **The real fix is at Cloudflare** — Authenticated Origin Pulls, or an origin
firewall that refuses connections not from Cloudflare's ranges. Only the
edge can vouch for its own header. This is infrastructure work and is not
part of this change; it is the thing that makes the header mean anything.
2. **The ceiling in §1.1 bounds what evading the keying can achieve** while
that remains open. It does not make the header trustworthy — it makes
trusting it survivable.
The backend being unreachable from outside the Docker network is a real second
layer, but it depends on the ingress path in front of the frontend, which this
design does not control and should not assume.
### 1.1 The ceiling, which is back
The shared bucket was acting as an accidental global throttle on a
single-process uvicorn backend that filters a 25,000-row DataFrame in-process.
Correct per-user keying removes it: the origin becomes reachable at 60/min *per
user* rather than 60/min in total, and — per above — at an unbounded rate by
anyone willing to rotate a header.
An earlier draft dropped the in-app ceiling, arguing it belonged at Cloudflare.
That argument assumed the keying was sound. It is not, so the ceiling is
load-bearing rather than redundant, and it ships here:
`GlobalRateLimitMiddleware` counts all `/api/` requests in a fixed 60-second
window against `global_rate_limit_per_minute` (3000), independent of any client
identity, and refuses with a 429 that names capacity rather than the client —
an operator has to be able to tell "one noisy client" from "the origin is
saturated". It is registered last so it is outermost: a ceiling that applies
after the expensive work has run is not a ceiling.
slowapi cannot express this. `default_limits` and `application_limits` are both
evaluated with the same `key_func`, making them per-client rather than global,
and `application_limits` only apply with `SlowAPIMiddleware` installed, which
this app does not use. Hence the explicit middleware — about thirty lines, and
obviously correct, which is what a backstop needs to be.
Requests from `127.0.0.1` are exempt. The container healthcheck runs
`curl http://localhost:80/api/data-info` from inside the container, and
starving it would fail the check, restart the container, and turn a load spike
into an outage loop. The exemption keys on the peer address, never the `Host`
header, which the caller sets.
3000/minute is an estimate, not a measurement, and worth revisiting against
real traffic.
### Per-user limits
Per-user fairness and origin protection are different jobs, and this design now
does both separately: the ceiling above for the origin, and per-route limits
for fairness. Conflating them is what produced the original behaviour, where
one bucket served the whole internet.
The existing 60/minute default is unchanged, and `/api/suggest` gets
120/minute. Both are estimates rather than measurements, and are a starting
point to revisit once the keying is correct enough for real per-user traffic to
be visible — which it was not before, because everyone shared one bucket.
## 2. `GET /api/suggest`
A dedicated endpoint, not a mode of `/api/schools`.
The existing search path calls Typesense for URNs and then filters, ranks and
sorts the full in-memory DataFrame — a pandas pass per keystroke, holding the
GIL and blocking other requests in the same worker. Suggestions need none of
it: `urn`, `school_name`, `phase`, `school_type`, `local_authority`,
`postcode` and `ofsted_rating` are all already in the Typesense document
(`pipeline/scripts/sync_typesense.py`).
```
GET /api/suggest?q=<query>&limit=8
→ 200 {"suggestions": [
{"urn": 100010, "school_name": "Brecknock Primary School",
"local_authority": "Camden", "postcode": "NW1 1AA",
"phase": "Primary", "school_type": "Community school"}
]}
```
- **Under two characters** returns `{"suggestions": []}` with 200. The
keystroke path never returns an error for ordinary input.
- **Typesense unavailable** returns `{"suggestions": []}` with 200. There is
deliberately **no DataFrame fallback**: the substring scan `/api/schools`
falls back to is precisely the cost this endpoint exists to avoid, and a
silent 25,000-row scan per keystroke is worse than no suggestions.
- **`limit` is clamped** to 20. It is a public endpoint.
- **Rate limit `120/minute`** per client, not the default 60. A 200 ms
debounce tops out near 5 requests/second while someone is actively typing,
but averages far below that across a real search; 120 leaves headroom for
bursts while still bounding one client.
- **Local authority is part of the payload, not decoration.** There are many
schools called "St Mary's"; a suggestion list without the authority is
unusable for exactly the queries autosuggest is meant to serve.
### Caching
`CACHE_RULES` gains `("/api/suggest", (60, 3600, 86400))`. Prefix queries
repeat enormously across users and school names change once a year.
The client fetch must **not** use `cache: "no-store"`. The compare modal does,
and copying that pattern would throw away both the browser cache and the ETag
304s the existing `CacheAndETagMiddleware` already provides.
Both environments currently report `cf-cache-status: DYNAMIC` — Cloudflare
ignores the `Cache-Control` the API already sends, because it does not cache
dynamic paths by default. **A Cloudflare Cache Rule for `/api/suggest*` would
let the edge absorb most of this traffic and never reach the origin.** That is
a dashboard change, it is optional, and nothing here depends on it.
## 3. The combobox
This is an ARIA combobox, not a text input with a list underneath.
**Files.** `FilterBar.tsx` is already long. The work splits three ways:
`hooks/useSchoolSuggest.ts` owns fetching, debouncing and cancellation;
`components/SuggestList.tsx` owns rendering and ARIA; `FilterBar.tsx` wires
them to the existing input and form.
**Fetching.** 200 ms debounce; minimum two characters; an `AbortController`
cancels the superseded request on every keystroke. Cancellation is not an
optimisation — without it, a slow response for `"st"` can land after the fast
one for `"st marys"` and replace a correct list with a stale one.
**Suppressed during postcode entry.** The box takes a school name *or* a
postcode, and `isValidPostcode` already distinguishes them. Suggestions do not
appear once the value parses as a postcode.
**Keyboard.** `ArrowDown`/`ArrowUp` move the active option, `Escape` closes and
keeps the typed text, `Tab` closes. `Enter` **with an option active** navigates
to that school's page. `Enter` **with none active** submits the free-text
search exactly as it does today — the existing behaviour is preserved, not
replaced.
**ARIA.** `role="combobox"` with `aria-expanded` and `aria-controls` on the
input, `aria-activedescendant` pointing at the active option, `role="listbox"`
on the list and `role="option"` on each row.
**Both instances get it.** `HomeView` renders `FilterBar` twice — hero and
sticky — from one component, so there is one implementation.
## 4. Behind a flag
Flag `school_autosuggest`, declared in `backend/flags.py`, default off.
This is the most-used control on the site and the first change to it in a
while. `app/page.tsx` is an async server component, so it reads the flag and
threads it to `FilterBar` through `HomeView` — two prop hops, explicit, no
client-side flag read.
Off means the input behaves exactly as it does today: no listener, no fetch, no
markup. Not a rendered-then-hidden dropdown.
The rate-limit keying is **not** flagged. It is a correctness fix that should
apply whether or not autosuggest is on, and flagging it would mean shipping a
known-wrong limiter into production deliberately.
## 5. Analytics
`search_submitted` already carries `via: 'input'`. Selecting a suggestion fires
it with `via: 'suggestion'` plus the chosen `urn`, so the obvious question —
does this actually help, or do people ignore it — has an answer in the data
rather than an opinion.
## 6. Testing
**Backend.** `client_key` prefers `CF-Connecting-IP`, falls back through
`X-Forwarded-For` to the remote address, and two different values get two
different buckets. `/api/suggest` returns matches, returns empty below two
characters, returns empty and 200 when Typesense is unavailable, and clamps
`limit`. That it never touches the DataFrame is asserted by making
`load_school_data` raise and requiring the endpoint to answer anyway.
**Frontend.** The hook debounces, aborts superseded requests, and drops a
late-arriving response for a stale query. The list renders the ARIA
attributes. Keyboard navigation moves the active option; `Enter` on an option
navigates; `Enter` on none submits the search.
**E2E.** With the flag on, typing a known school name shows it and selecting it
lands on that school's page. With the flag off, no combobox markup exists.
Gated on the flag the same way the distance journeys are — read the observable
effect, since `/api/flags` is denied to the public.
## 7. Risks
**Removing the accidental throttle.** Covered in §1. Correct per-user keying
means the origin is reachable at 60/minute *per user* where it was 60/minute
in total, and no in-app global cap replaces it — that job goes to Cloudflare,
which is not done as part of this change. Until it is, a determined caller
with many source addresses can put more load on a single-process origin than
they can today. Against this site's traffic that is a theoretical risk rather
than a live one, but it is a real one and it is the price of the fix.
**Cloudflare bypass — the open one.** If the origin is reachable without
passing through Cloudflare, `CF-Connecting-IP` is attacker-controlled, and
rotating it per request defeats per-client limits on every endpoint. The
ceiling in §1.1 bounds the damage to the origin's total capacity; it does not
restore per-client fairness under attack, and it cannot. Closing this properly
means Authenticated Origin Pulls or an origin firewall restricted to
Cloudflare's published ranges — infrastructure work, outside this change, and
the single most valuable follow-up here.
**Typesense becomes user-visible.** Today a Typesense outage degrades search to
a slow substring match. With autosuggest it also means the dropdown silently
stops appearing. That is the correct failure — quiet, not broken — but it makes
Typesense health worth monitoring in a way it was not before.
## Out of scope
- **Place suggestions.** The 2,646 town, authority and outcode pages are a
strong candidate and would route people onto the pages W2 built, but they
live in the place registry rather than Typesense, so it is a second index and
a ranking rule for comparing two kinds of result. Worth its own change.
- **Postcode completion.** Would put postcodes.io in the keystroke path, with
its own latency and rate limits.
- **The compare modal.** It already has search-as-you-type. Converting it to
this component is a reasonable follow-up, not part of this.
- **Recent or popular searches.** No storage for either, and no evidence yet
that they are wanted.
@@ -1,399 +0,0 @@
# Destination Measures — Design
**Date:** 2026-08-28
**Status:** awaiting review
**Scope:** secondary school detail pages only
## Goal
Say what happened to a school's leavers after they left. Two sections on the
secondary template:
- **After Year 11** — every secondary, from the KS4 destination measures
- **After the sixth form** — sixth-form schools only, from the 16-18 measures
This replaces the "Post-16 destination data coming soon" placeholder standing in
`nextjs-app/components/school/SecondaryAdmissionsSection.tsx:117` since the exam
phase taxonomy work, and fills the `ks5_destinations_pct` slot specified but
never built in `2026-07-07-exam-phase-taxonomy-design.md:201`.
Mockup, with all three data states live:
<https://claude.ai/code/artifact/5be149d6-252f-473c-9a4f-4c36b05161b0>
## The finding that shapes everything
**Suppression is per cell, and the cells sum to the cohort.**
DfE withholds a figure it considers disclosive by writing `c`. It does this at
the level of an individual destination category, not the whole school, and it
publishes the cohort total alongside. The categories form a clean partition. So
where exactly one category is suppressed, subtracting the published ones from the
cohort recovers it exactly.
Verified against three real schools in the 2022/23 file:
| School | URN | Withheld | Recovers to |
|---|---|---|---|
| North East Futures UTC | 145900 | School sixth form | **3 pupils** |
| Whitley Bay High School | 108638 | Further education | **18 pupils** |
| St Matthew's RC High School | 148389 | School sixth form | **4 pupils** |
Those are the precise numbers the `c` exists to hide, and in a random 400-school
sample **22% of mainstream secondaries** have exactly one suppressed category in
their disadvantaged group. This is the normal case, not an edge case.
Three rules follow, and everything else in this document is downstream of them.
**R1 — Never *publish* enough to derive a remainder.**
An earlier draft of this rule said "never *render* a derived remainder", and
that was the defect code review caught in PR #137. Not drawing a number does
nothing to stop it being computed: `GET /api/schools/{urn}` is public and
unauthenticated, so anything in the payload is published whatever the UI
chooses to draw. The rendering guards shipped; the payload still carried the
cohort and every published category, and `cohort - sum(published)` returned
Whitley Bay's withheld figure exactly.
The rule is therefore about the serialiser, and the UI guards are a second line
of defence behind it. Two identities have to be closed:
- within a pupil group the categories sum to the cohort, so a group with
exactly **one** suppressed category gives it away;
- across groups, disadvantaged + other = all for every category, so a category
suppressed in exactly **one** of the three gives itself away.
`_mask_for_disclosure` applies DfE's own answer — secondary suppression —
withholding a companion cell until every row and every column hides either none
or at least two. It iterates, because each new suppression can break the other
identity, and terminates because cells are only ever added.
The companion must carry pupils. Suppressing a zero looks like secondary
suppression and protects nothing: the residual still equals the original
withheld figure.
Where no companion can do the job — a sparse cohort whose every other category
is `not_applicable`, routine in special schools and alternative provision — the
pupil group is **dropped from the payload entirely**. A first version simply
returned at that point with the violation intact and no signal, which review
caught: a disclosure-control pass that fails silently is worse than none,
because everything downstream trusts it. The function now cannot terminate
except in a state where `disclosure_invariant_holds()` is true, and an
exhaustive test sweeps all 81 suppression patterns of a four-category group to
prove it.
Measured cost on the 400-school sample: the all-pupils bar survives on **94%**
of mainstream secondaries rather than 100%. That is the price of not
republishing what DfE withheld.
**R2 — Never aggregate across a suppression boundary.** Summing published
components to fill a gap is R1 with extra steps.
DfE's own aggregates (`Sustained education destination`, `Sustained education,
employment & apprenticeships`) are ingested but **not served**. An aggregate
spanning exactly one suppressed component names it, and nothing renders them
today — an unused field that leaks is not a trade-off worth carrying. They can
be re-added with their own guard if the fallback ladder is ever built.
**R3 — The three pupil groups are one disclosure surface, not three.**
Disadvantaged and Not-known-to-be-disadvantaged partition All pupils, so
rendering any *two* of them recovers the third. Where a category is suppressed in
the disadvantaged group, it must therefore also be withheld from **all other
pupils** — the all-pupils view is the primary one and keeps it.
This costs almost nothing, because DfE already applies the same masking: across
the sample, 493 of 498 suppressed disadvantaged cells were suppressed in the
other group too. The mart enforces the remaining 5, which fell on 2 schools of
262. **The all-pupils bar is unaffected** — masking the whole page wherever the
disadvantaged group is thin would remove the bar from 80% of schools, and is not
what this rule says.
R1 and R2 both hold within a group and still leak across the switch, which is why
R3 is stated separately.
### The convention that would break this quietly
`macros/safe_numeric.sql` coerces every EES sentinel — `z`, `c`, `x`, `q`, `u` —
to `NULL`, deliberately and correctly for attainment, where "suppressed" and "no
data" are equally unrenderable. Here they are not the same thing: one must print
*withheld*, the other must print nothing at all, and the difference is what keeps
R1 enforceable.
**`safe_numeric` must not be used on destination counts.** The staging model
keeps the sentinel in a companion status column. This is the single most likely
way for this feature to regress into a disclosure, so it gets its own dbt test.
## What is actually available
Measured against the EES public API (open, no key). Both datasets carry
`geographicLevel: School` with `urn` on every location option, so the join to
`dim_school` is direct.
| | KS4 | 16-18 |
|---|---|---|
| Dataset id | `019d4f41-22d1-71b2-a1a7-f3b91026815b` | `019d4e73-6440-7523-b60c-bfab1ad4a30d` |
| Rows | 1,871,739 | 3,862,658 |
| Institutions | 4,946 | 3,065 |
| Time periods | 2009/10–2022/23 | 2016/17–2022/23 |
**Destination categories (KS4).** School sixth form · Sixth form college ·
Further education · Other education destination · Sustained apprenticeships (with
level breakdown) · Sustained employment destination · Not recorded as a sustained
destination · Activity not captured. Plus the aggregates `Sustained education
destination` and `Sustained education, employment & apprenticeships`.
**16-18 adds** UK higher education institution and FE split by level, which is
what makes the post-16 section worth having.
**Breakdowns.** `Disadvantage Status` gives Disadvantaged / Not known to be
disadvantaged / Total — exactly the three-way switch. Sex, ethnicity, FSM status,
prior attainment and SEN provision also travel in the same table; we ingest none
of them.
**Indicators.** Both counts and percentages, plus the cohort size. Bar widths use
the counts — the published percentages do not sum to 100.
### Coverage, and what degrades
Random 400-school sample, 2022/23, mainstream secondaries (n=262):
| View | As published by DfE | After R1–R3 masking | Consequence |
|---|---|---|---|
| All pupils, all categories | 100% | **94%** | Bar works nearly everywhere |
| Disadvantaged, headline rate | 95% | 95% | Gap panel works |
| Disadvantaged, three grouped cards | 68% | 68% | Degrades card by card |
| Disadvantaged, all six categories | 20% | **20%** | Bar unusable for this group |
The middle column is what the site actually serves. Masking costs the
all-pupils bar on 6% of mainstream secondaries — those are schools where a
category was suppressed in exactly one pupil group and no non-zero companion
existed below the all-pupils row.
Special schools and alternative provision are far worse: 13% and 41% respectively
have the whole cohort suppressed even for all pupils. The empty state is
load-bearing, not defensive.
## The display
Question-led. Three cards over one bar, with the cards acting as a lens on the
bar rather than a summary beside it — hovering a card dims the bar, table and
England reference to the categories that card is built from. The full mockup is
linked above; what matters for implementation:
**The headline is not the sustained rate.** That figure sits between 92% and 97%
for nearly every school in England. The mix is what varies, so the mix leads.
**The grouping is ours, not DfE's.** "Academic route" = school sixth form +
sixth-form college; "College" = FE and other colleges; "Work" = apprenticeship +
employment. This is the most arguable thing on the page, so it lives in one place
in `lib/destinations.ts`, is explained in a tooltip, and is reversible in one
edit.
**The absence is hatched neutral, never a colour.** "Activity not captured" means
no record in the sources DfE holds — it includes independent schools, moving
abroad and private training. Colouring it as a bad outcome would be a factual
error rendered in CSS. The hatch also fixes a real contrast problem: neutral
against the employment blue failed CVD separation at ΔE 7.6, and texture is the
secondary encoding that rescues it. Every other adjacent pair clears ΔE 10.9
under protanopia.
**Colour tokens.** Education is one hue in three steps (school-like to
college-like); apprenticeship and employment are separate hues. Six new tokens in
`globals.css`, defined in both themes, per the existing token discipline.
**The disadvantage split rides the same control.** One visualisation serving
three cohorts, with the England reference repointing to the matching national
group. The gap statement stays visible below the bar whatever is selected,
because a gap nobody clicks on is a gap nobody sees.
## Data model
### Extraction
A new `tap-uk-ees-destinations` extractor, separate from `tap-uk-ees`. The
existing tap downloads a release ZIP and reads a CSV inside it; the destinations
files are far larger than we need and the query API filters server-side, so this
one POSTs to `/v1/data-sets/{id}/query` and pages through results.
With every dimension pinned — destination measures, disadvantage status, sex
Total, characteristic topic Total — one year returns **252,610 rows** across all
geographic levels. Three school-level years is comfortably tractable.
Pinning is mandatory, not an optimisation: leaving the characteristic dimensions
unconstrained returned 45 rows where 9 were wanted, because every breakdown
shares one table.
The tap emits the raw value as text. **It does not coerce `c`.**
### Staging
`stg_ees_ks4_destinations` / `stg_ees_ks5_destinations`. Each raw value becomes
two columns:
```sql
case when raw ~ '^-?[0-9]+(\.[0-9]+)?$' then raw::numeric end as pupils,
case
when raw ~ '^-?[0-9]+(\.[0-9]+)?$' then 'published'
when lower(trim(raw)) = 'c' then 'suppressed'
else 'not_applicable'
end as status
```
### Marts
`fact_ks4_destinations` and `fact_ks5_destinations`, **long format**:
```
urn, year, pupil_group, destination_category, cohort_pupils, pupils, percentage, status
```
This departs from the wide house pattern (`fact_ks4_performance` and friends) on
purpose. `pupil_group` is a genuine third dimension; going wide would need three
sets of every column, and R2 is far easier to test on rows than on columns.
Roughly 8 categories × 3 groups × 4,946 schools × 3 years ≈ 356k rows.
`fact_destination_national` carries the same grain for England, so the page's
England reference repoints with the switch.
### dbt tests
- `assert_destinations_no_derived_remainder` — for every (urn, year,
pupil_group) with exactly one suppressed category, assert no aggregate row
exists that would let the residual be recovered. **This is the R1 guard.**
- `assert_destinations_group_masking` — for every (urn, year, category), if the
disadvantaged group carries `suppressed`, so does the other-pupils group.
**This is the R3 guard**, applied in the mart so no consumer can reach an
unmasked combination.
- `assert_destination_status_null_agreement` — `pupils is null` wherever
`status != 'published'`, and never null where it is.
- `assert_destinations_join_dim_school` — no orphaned URNs, matching the
existing `assert_no_orphaned_facts` pattern.
## API
`GET /api/schools/{urn}` gains a `destinations` block:
```json
{
"ks4": {
"cohort_year": "2022/23",
"published": "2026-04",
"groups": {
"all": { "cohort": 180, "categories": [ … ], "aggregates": { … } },
"disadvantaged": { … },
"other": { … }
}
},
"ks5": { … }
}
```
Each category carries `pupils`, `percentage` and `status`. **The serialiser never
emits a computed remainder**, and a backend test asserts that a group containing a
suppressed category serialises no total that closes the gap.
`null` for the whole block where nothing is published — the frontend renders the
empty state from its absence, not from a sentinel.
## Frontend
| File | Kind | Job |
|---|---|---|
| `lib/destinations.ts` | pure | Category list, the academic/college/work grouping, `canAggregate()` enforcing R2, percentage derivation from counts |
| `components/school/DestinationsSection.tsx` | server | Section shell, renders **all pupils** into the HTML |
| `components/school/DestinationsView.tsx` | client | Cohort switch, card↔bar linkage |
| `components/school/Post16DestinationsSection.tsx` | server | Year 13 section, sixth-form schools only |
| `app/globals.css` | tokens | Six destination colours, both themes |
Server-first matches the directory's existing discipline — every component in
`components/school/` is a server component except `AdmissionsViewToggle`, which
is the precedent this follows. All-pupils figures are in the HTML for crawlers
and for no-JS; only the switch and the hover linkage need the client.
`lib/schoolSections.ts` gains `hasKs4Destinations` / `hasKs5Destinations` flags
and the nav items, following the existing `computeSchoolFlags` pattern.
**Placement** on the secondary template: GCSE results → After Year 11 → After the
sixth form → admissions. Destinations follow attainment because they answer "and
then what happened".
**Dating.** The latest destination year is 2022/23, published April 2026, while
the site's newest KS4 year is 2024/25. The section header states its own cohort
year, or it reads as stale data next to the GCSE section above it.
## Edge states
| State | Frequency | Behaviour |
|---|---|---|
| Whole cohort suppressed | 13% of special, 41% of AP | Section renders the explanation, no chart |
| Some categories withheld | 80% of disadvantaged views | Cards degrade individually; **no bar**; table marks withheld rows |
| Disadvantaged group suppressed entirely | 5% | Switch drops to two options, gap panel not rendered |
| No sixth form | — | Post-16 section not rendered at all — absence is correct, a "no data" placeholder would imply something is missing |
| School too new | — | "First figures expected in 2026", not a bare no |
## Testing
Per CLAUDE.md, user-facing behaviour extends `e2e/` in the same PR.
**Unit** — `lib/destinations.ts` is where R1 and R2 live, so it carries the
heaviest tests: `canAggregate()` refuses a group containing one suppressed cell,
allows one spanning two, and the bar builder refuses to emit segments for any
group with suppression. These are the tests that must fail loudly if someone
later "fixes" a gap in the chart.
**dbt** — the three tests above.
**Backend** — the serialiser emits no closing total for a partially suppressed
group.
**E2E** — a school with full data renders three cards and a bar; a school with a
partially suppressed disadvantaged group renders the withheld state and **no bar
element**; a suppressed school renders the explanation; a school with no sixth
form renders no post-16 section.
Note the staging caveat: mart changes are inert until the Airflow pipeline runs,
and the staging E2E gate runs post-merge.
## Out of scope
- **Compare view and rankings.** The long mart shape supports both; neither is
built here. Flagged because "% to a school sixth form" is a plausible rankings
metric and the mart shape should not have to change to allow it.
- **Ethnicity, sex, SEN and prior-attainment breakdowns.** Available in the same
file, ingested deliberately not at all — each is a separate editorial decision
about what a school page should assert.
- **Longer term destinations** (3 and 5 years out) and **Progression to higher
education** — separate publications, worth a later look for sixth forms.
- **Primary schools.** No KS2 destination measures publication exists; DfE
tracking starts at KS4. Naming the secondaries a primary's leavers go to needs
the National Pupil Database, which is not publishable at that grain.
## Risks
**A later change reintroduces the disclosure.** The likeliest routes are
applying `safe_numeric` to a destination column for consistency, adding a
`coalesce` in a mart, or — as happened in review — enforcing a disclosure rule
at the rendering layer instead of the publishing layer. Mitigation is the dbt
tests plus `backend/tests/test_destinations_api.py`, which reconstructs the
residual the way an attacker would and asserts it no longer resolves.
**The two-year lag reads as staleness.** Mitigated by dating the cohort in the
section header rather than only in a tooltip.
**Sixth-form retention will be misread.** "41% went to a school sixth form" says
nothing about *which* school. The published file reports destination type, never
destination institution. Copy must never imply "stayed on here", and the tooltip
should say so.
**Section length.** The secondary template is already long and this adds two
sections. If it becomes a problem the post-16 section is the one to collapse
behind a disclosure, not the Year 11 one.
## Open questions
1. Is the disadvantage split its own section or a sub-block inside the
destinations section? Modelled as a sub-block; it is the most differentiating
figure on the page and the most easily misread on a small cohort.
2. Do we ingest the apprenticeship level breakdown (intermediate / advanced /
higher) now, or collapse to one apprenticeship figure and revisit? Collapsed
in this design.
@@ -1,368 +0,0 @@
# Giving schoolcompare a human author: an About page and a blog
**Date:** 2026-09-02
**Status:** Design — awaiting review
**Scope:** A named author for the site, an `/about` page, and a Payload-CMS-backed
blog at `/blog`.
## Why
The site reads as synthetic. Not because of its tone, but because of three
specific absences:
1. **Nobody is accountable for the numbers.** There is no author, no statement
of why the site exists, and no one who can be wrong. The only human trace on
the entire site is `contact@schoolcompare.co.uk` in the footer.
2. **No visible judgement.** Every figure is presented as though it fell out of
a machine. Hundreds of editorial decisions went into this codebase — which
metrics to show, when a benchmark is invalid, what to suppress — and not one
of them is visible to a reader. `isSpecialSchool()` silently drops the
England comparison for special schools and PRUs because that comparison is
meaningless; nowhere does the site *say* so.
3. **The voice is institutional third person.** "schoolcompare brings it all
into one place." "Built for parents, governors, journalists." That is
brochure register, and it is precisely the register that machine-generated
content defaults to.
There is a second, independent reason. The SEO programme
(`2026-08-20-seo-programme-design.md`) defines eight workstreams and none of
them address E-E-A-T or authorship. School performance data is YMYL territory;
an anonymous site republishing DfE figures has no authorship signal at all. This
work fills that hole, and the blog gives W6 (explainer content) somewhere to
live.
### The failure mode to avoid
The standard fix — a stock photo and "Hi, I'm Tudor, and I'm passionate about
education!" — reads as *more* synthetic than the current coldness. Manufactured
warmth is a stronger machine-tell than plain institutional voice. Everything
here has to be specific, occasionally awkward, and willing to be unflattering,
or it makes the problem worse.
## Positioning
The author is **Tudor**: first name only, real photograph, no surname, no
employer named.
The credibility claim is deliberately **not** educational expertise. The About
page states plainly: *"I'm not an education expert."* Authority comes from two
things that are actually true:
- **Experience.** A parent going through primary admissions in south-west London
right now. Google's E-E-A-T leads with Experience, and lived experience of the
thing is exactly what the DfE's own service lacks.
- **Method.** Every number's provenance is stated, so a reader can check the
site rather than trust it.
This is more durable than borrowed expertise: it cannot be undermined by someone
noticing the author has no teaching qualification.
**Consequence for the design.** A `Person` entity with no surname is a weak
search signal and cannot be corroborated off-site. The credibility load
therefore shifts onto the methodology being visibly rigorous. That is a design
constraint, not a caveat — it is why the About page carries a substantial
"how this is built and where it can be wrong" section rather than a short bio.
### Voice rules
Applied to About and every post. Recorded here so the voice does not drift.
- First person singular. "I built", not "we provide".
- Concrete over general. "when we were looking at schools in Wandsworth" beats
any amount of stated warmth.
- State limits before someone else finds them. Every post that presents a
metric says what it does not show.
- No mission statements, no "passionate about", no invented team.
- No em dashes. One of the clearest tells of machine-written prose, which is
the exact problem this work exists to fix.
- Short sentences. The existing code comments in this repo are already written
this way; the prose should match.
## Scope
**In:**
- `/about` — a coded page (not CMS-managed).
- `/blog` and `/blog/[slug]` — Payload-backed, with an index and post pages.
- Payload CMS installed into the existing Next application.
- Footer and navigation links to both.
- `Person`, `Organization`, `BlogPosting`, `BreadcrumbList` JSON-LD.
- RSS feed and sitemap integration.
- One first post, so the blog does not launch empty.
**Out (deliberately):**
- Rewriting existing homepage/how-it-works copy into first person. Worth doing,
but it would double the review surface of this PR. Separate change.
- In-product signed notes on school pages (the "distributed humanity" idea).
Revisit once About and the blog exist.
- Comments, newsletter, author accounts beyond one.
- A team page. There is no team.
## Architecture
### Topology
Payload 3 installs **into the existing Next application** and serves `/admin`
from the same container. One image, one deploy, no new service. This is
Payload 3's native model and it makes on-demand revalidation trivial, because
the CMS hooks run in the same process as the Next cache.
Accepted costs: the public site's image now carries Payload, so a CMS security
patch redeploys the whole site; and the image grows substantially.
### Two collisions that must be handled
**1. `/api` is already taken.** `app/api/[...path]/route.ts` is a catch-all that
proxies `/api/*` to FastAPI at runtime. Payload's default API route is also
`/api`. Left alone, these fight, and the failure is not clean — the catch-all
would swallow Payload's admin API calls and forward them to FastAPI.
Payload's API route is therefore remapped:
```ts
routes: { api: '/cms-api', admin: '/admin' }
```
with its route group at `app/(payload)/cms-api/[...slug]/route.ts`. The
`/cms-api` prefix must also be added to the FastAPI proxy's excluded-paths list
as a defensive second line.
**2. `next.config.js` is CommonJS.** Payload's `withPayload()` wrapper is ESM
only. The config must become `next.config.mjs`, converting `module.exports` to
`export default` and wrapping the export. All existing content — the standalone
output, `outputFileTracingIncludes`, the staging `X-Robots-Tag` header block,
the CSP — carries over unchanged. This is mechanical but it touches the file
that controls staging's noindex, so it needs care and an explicit test.
### Database
Payload uses the existing `sc_database` Postgres instance, in its **own
`payload` schema**:
```ts
db: postgresAdapter({
pool: { connectionString: process.env.DATABASE_URL },
schemaName: 'payload',
})
```
The frontend container is already on the `backend` Docker network, so it can
reach `sc_database:5432` with no networking change. It needs a new
`DATABASE_URL` environment variable.
Schema isolation is not cosmetic. `public` currently holds the application
tables and Airflow's metadata, and `scripts/migrate_csv_to_db.py --drop` exists
to drop and reimport. Blog content living in its own schema means no data
pipeline operation can destroy it.
**Verified 2026-09-02** (this was an open question when the spec was written).
`--drop` calls `run_full_migration()` in `backend/migration.py`, which drops
exactly two tables by name:
```python
ks2_tables = ["school_results", "schools"]
for tname in ks2_tables:
if tname in existing:
Base.metadata.tables[tname].drop(bind=engine)
```
There is no `Base.metadata.drop_all()` anywhere in `backend/`, and no
`DROP SCHEMA`. The only other drop is `_apply_schema_drops()`, a single
schema-qualified `DROP TABLE IF EXISTS marts.fact_parent_view CASCADE`.
Nothing sets `search_path`, so the SQLAlchemy metadata resolves to `public`,
and `inspector.get_table_names()` does not even enumerate other schemas.
So the guarantee is stronger than schema isolation alone: `--drop` targets two
named tables that Payload does not have, and would not reach `posts`, `media`
or `users` even if they shared a schema. The `payload` schema remains the right
choice — it protects against a *future* broadening of that script rather than
today's behaviour — but the safety claim rests on verified code, not on
assumption.
Putting CMS tables in this instance is consistent with existing practice —
Airflow already stores its metadata there.
### Migrations
Payload's Postgres adapter auto-pushes schema in development and requires
explicit migrations in production. Use `prodMigrations`, which runs pending
migrations during server initialisation:
```ts
db: postgresAdapter({ /* ... */, prodMigrations: migrations })
```
This is preferred over a one-shot init container (the `airflow-init` pattern)
because the app is a single long-running process and there is no ordering
problem to solve. Migration files are generated with `payload migrate:create`
and committed, so schema changes travel through the same PR and staging gate as
code.
### Media
Uploads go to a Docker named volume, consistent with `postgres_data`,
`typesense_data` and `airflow_logs`.
- `staticDir` must be an **absolute** path in Payload 3: `/app/media`.
- The container runs as `nextjs` (uid 1001). The Dockerfile must
`mkdir -p /app/media && chown nextjs:nodejs /app/media` **before** the volume
is mounted, or Docker will create the mountpoint root-owned and every upload
will fail with EACCES.
- `sharp` moves from `devDependencies` to `dependencies` — Payload needs it at
runtime to generate `imageSizes`.
- The volume must be added to the backup routine alongside Postgres. A blog
post's images are not reproducible from the pipeline.
### Rendering
**Constraint:** CI builds the image with no database reachable. Blog pages
therefore cannot use build-time `generateStaticParams` — that would either fail
the build or bake in an empty post list.
Instead: ISR. Post and index pages declare a `revalidate` window and render on
first request, with Payload `afterChange` / `afterDelete` hooks calling
`revalidatePath('/blog')` and `revalidatePath('/blog/' + slug)` for immediate
publication. Because Payload runs in the same process, the hook calls
`revalidatePath` from `next/cache` directly — no webhook, no shared secret.
The ISR cache lives on container disk and is cleared by a redeploy. For a
single container serving a handful of posts this is fine.
### Collections
- **`posts`** — `title`, `slug`, `publishedAt`, `excerpt`, `heroImage`
(relation to `media`), `content` (Lexical rich text), `seo` group
(`metaTitle`, `metaDescription`), `_status` (drafts enabled).
- **`media`** — upload collection, `alt` required, `imageSizes` for thumbnail
and hero widths, public read access.
- **`users`** — Payload's auth collection. One account. Public creation
disabled.
Drafts are enabled so posts can be written over several sittings and previewed
before publication.
**Payload Blocks** are how posts embed live product components — a real trend
chart or comparison table inside a post, rendered from live data rather than
screenshotted. This is the main thing the CMS has to earn back against
file-based MDX, and it directly serves the goal: showing judgement in context.
Ship with one block (a callout/aside for "what this number doesn't tell you");
add a live-chart block once a post needs it.
### Security
`/admin` is the first authenticated surface on this site. Public, hardened:
- `PAYLOAD_SECRET` — long, random, set in the Portainer stack environment, never
committed. The same variable must exist in staging with a *different* value.
- Strong unique password on the single admin account.
- Login rate limiting via Payload's `maxLoginAttempts` / `lockTime`.
- `X-Robots-Tag: noindex, nofollow` on `/admin/*` and `/cms-api/*`, and a
`robots.ts` disallow. The admin panel must never be indexed.
- Public user creation disabled; no open registration.
- Verify the existing CSP `frame-ancestors` directive does not break the admin
panel.
Residual risk, accepted: a future Payload authentication CVE is live against the
public internet. Mitigation is prompt patching, which the staging→prod pipeline
already supports. If this becomes uncomfortable, restricting `/admin` at the
proxy to LAN/VPN is a one-line change later.
Staging note: staging runs the same image on `stx.`, so it gets its own admin
panel and its own database. It must have its own `PAYLOAD_SECRET` and its own
credentials — never production's.
## Deployment changes
- `nextjs-app/Dockerfile` — create and chown `/app/media`; ensure Payload's
admin bundle and `sharp` survive standalone output file tracing.
- `docker-compose.portainer.yml` and the staging equivalent — add
`DATABASE_URL` and `PAYLOAD_SECRET` to the `frontend` service, add a
`payload_media` volume mounted at `/app/media`, and add
`depends_on: sc_database`.
- Document both new environment variables in the compose header comment block,
which is where this stack records its configuration.
## SEO
- `Person` (Tudor, with photo) and `Organization` JSON-LD on `/about`.
- `BlogPosting` + `BreadcrumbList` on post pages, with `author` referencing the
same `Person`.
- Canonical URLs on `/blog` and every post.
- Posts and `/about` added to the existing sitemap (`app/sitemap.xml/route.ts`
and `app/sitemaps/[...parts]`). Post URLs come from Payload at request time.
- RSS feed at `/blog/rss.xml`.
- Footer links to both pages, under a new "About" column.
**Navigation is deliberately left alone.** `Navigation.tsx` renders a bottom tab
bar on mobile that already carries four items (Search, Compare, Rankings,
Admissions). A fifth tab makes each one cramped at 320px, and About and Blog are
both lower-intent than any of the four. Both live in the footer; About
additionally gets a byline link from every post, which is where a reader who
cares actually asks the question. Revisit only if analytics show people hunting
for it.
## Testing
Unit (Jest):
- Post rendering, including a post with no hero image and one with no excerpt.
- Slug generation and collision handling.
- JSON-LD shape for `BlogPosting` and `Person`.
- The `next.config.mjs` conversion preserves the staging `X-Robots-Tag` rule —
this guards the riskiest mechanical change in the plan.
E2E (Playwright, `e2e/`, required by CLAUDE.md for user-facing change):
- `/about` renders, shows the author name and photo, and is reachable from the
footer and nav.
- `/blog` lists at least one post; clicking through reaches the post.
- A post page renders title, date, body and byline.
- `/admin` responds with `noindex` and does not leak a stack trace when
unauthenticated.
Note the known constraint: new journeys cannot be proven in PR checks, because
the staging E2E gate runs post-merge.
## Risks
| Risk | Mitigation |
|---|---|
| `next.config.mjs` conversion silently drops the staging noindex header, making staging a crawlable duplicate | Unit test asserting the header rule; verify on staging before promotion |
| Payload API route collides with the FastAPI `/api` proxy | Remap to `/cms-api`; add to the proxy's exclusion list |
| Media volume mounts root-owned; all uploads fail with EACCES | `mkdir`+`chown` in the Dockerfile before the mount; test an upload on staging |
| Build fails or bakes empty content because CI has no DB | No build-time DB access; ISR only |
| A pipeline `--drop` destroys blog content | Separate `payload` schema; verify `--drop` blast radius before building |
| Media volume not backed up; images unrecoverable | Add `payload_media` to the backup routine |
| Payload auth CVE exposed publicly | Prompt patching; proxy restriction available as a fallback |
| Blog launches empty or goes stale | Ship with one post; cadence is explicitly "a few times a year", so no cadence is promised anywhere on the page — no dates implying a schedule |
## Sequence
Each step is independently reviewable and mergeable.
1. **Payload foundation** — install, `next.config.mjs` conversion, `payload`
schema, `/cms-api` remap, `users` collection, `/admin` hardening, compose and
Dockerfile changes. No public-facing change yet. Verify on staging that the
site is unchanged and `/admin` works.
2. **`/about`** — coded page, photo, `Person`/`Organization` JSON-LD, footer and
nav links, e2e journey. Independently valuable and does not depend on the
blog.
3. **Blog** — `posts` and `media` collections, `/blog` index and post pages, ISR
plus revalidation hooks, RSS, sitemap, structured data, e2e journeys.
4. **First post** — written in the admin panel, published through the normal
flow, proving the whole path end to end.
Step 1 carries all the infrastructure risk and none of the visible benefit, so
it should be verified on staging carefully before step 2 starts.
## Dependencies on Tudor
- **A photograph.** Blocks step 2. Nothing else in the plan is blocked by it.
- **The first post's subject.** Blocks step 4 only. Suggested: what school
performance data cannot tell you — it demonstrates judgement, is genuinely
useful, and is the kind of thing an anonymous or machine-written site will not
publish.
- ~~Confirmation that `scripts/migrate_csv_to_db.py --drop` is schema-scoped.~~
**Resolved 2026-09-02** — verified in `backend/migration.py`; see the
Database section. No action needed.
File diff suppressed because it is too large. Load diff
-591
View File
@@ -1,591 +0,0 @@
<title>Last distance offered — detail page mockup</title>
<style>
:root{
--bg-primary:#faf7f2; --bg-secondary:#f3ede4; --bg-card:#fff;
--text-primary:#1a1612; --text-secondary:#5c564d; --text-muted:#6d685f;
--accent-coral:#e07256; --accent-coral-dark:#b04a2e;
--accent-teal:#296f6f; --accent-teal-light:#3a9e9e;
--accent-gold:#c9a227; --accent-gold-text:#7a6800;
--coral-bg:rgba(224,114,86,.12); --teal-bg:rgba(45,125,125,.12); --gold-bg:rgba(201,162,39,.12);
--border:#e5dfd5; --shadow:0 2px 8px rgba(26,22,18,.06); --shadow-md:0 4px 20px rgba(26,22,18,.1);
--radius-sm:4px; --radius-md:8px; --radius-lg:16px;
--serif:'Playfair Display',Georgia,'Iowan Old Style',serif;
--sans:'DM Sans',-apple-system,BlinkMacSystemFont,'Segoe UI',sans-serif;
}
/* Mockup is a fixed light artefact — it mirrors the live app, which is light-only. */
*{margin:0;padding:0;box-sizing:border-box}
body{font-family:var(--sans);background:var(--bg-primary);color:var(--text-primary);line-height:1.6;
padding:44px 20px 110px;font-variant-numeric:tabular-nums}
.wrap{max-width:820px;margin:0 auto;display:flex;flex-direction:column;gap:0}
.pagehead h1{font-family:var(--serif);font-weight:600;font-size:clamp(26px,4vw,34px);letter-spacing:-.015em;text-wrap:balance}
.pagehead p{color:var(--text-secondary);margin-top:10px;font-size:15px;max-width:64ch}
.pagehead p + p{margin-top:8px}
.step{margin:60px 0 6px;display:flex;align-items:baseline;gap:10px;flex-wrap:wrap}
.step h2{font-family:var(--serif);font-size:20px;font-weight:600;letter-spacing:-.01em}
.step .where{font-size:12px;font-weight:600;letter-spacing:.05em;text-transform:uppercase;color:var(--accent-teal);
background:var(--teal-bg);padding:3px 9px;border-radius:999px}
.stepdesc{color:var(--text-muted);font-size:14px;margin-bottom:18px;max-width:66ch}
.stepdesc code{font-family:ui-monospace,SFMono-Regular,Menlo,monospace;font-size:12.5px;background:var(--bg-secondary);padding:1px 5px;border-radius:var(--radius-sm)}
/* ---- card shell mirrors SchoolDetailView .card ---- */
.card{background:var(--bg-card);border:1px solid var(--border);border-radius:var(--radius-lg);
box-shadow:var(--shadow);padding:26px 26px 24px}
.cardhead{display:flex;align-items:center;justify-content:space-between;gap:16px;flex-wrap:wrap;margin-bottom:8px}
.sectionTitle{font-family:var(--serif);font-weight:600;font-size:22px;letter-spacing:-.01em}
.sectionSub{color:var(--text-secondary);font-size:14px;margin-bottom:18px}
.seg{display:inline-flex;background:var(--bg-secondary);border-radius:999px;padding:3px;gap:2px;flex:none}
.seg button{appearance:none;border:none;background:none;cursor:pointer;font:inherit;font-size:13px;font-weight:600;
color:var(--text-muted);padding:6px 13px;border-radius:999px;white-space:nowrap;transition:background .15s,color .15s}
.seg button[aria-pressed="true"]{background:var(--bg-card);color:var(--text-primary);box-shadow:var(--shadow)}
.seg button:focus-visible{outline:2px solid var(--accent-teal);outline-offset:2px}
/* ---- tiles ---- */
.tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(146px,1fr));gap:10px;margin-top:4px}
.tile{background:var(--bg-secondary);border-radius:var(--radius-md);padding:14px 15px 13px}
.tile .num{font-family:var(--serif);font-size:27px;font-weight:600;line-height:1.15;letter-spacing:-.01em;display:block}
.tile .num .unit{font-family:var(--sans);font-size:15px;font-weight:600;margin-left:2px}
.tile .num .sub{display:block;font-family:var(--sans);font-size:12.5px;font-weight:400;color:var(--text-muted);margin-top:1px}
.tile .lbl{display:block;font-size:12.5px;color:var(--text-secondary);margin-top:5px;line-height:1.35}
.tile.accent{background:var(--coral-bg)}
.tile.accent .num{color:var(--accent-coral-dark)}
.tile.newtile{background:var(--coral-bg);box-shadow:inset 0 0 0 1.5px var(--accent-coral)}
.tile.newtile .num{color:var(--accent-coral-dark)}
.newflag{display:inline-block;font-size:10px;font-weight:700;letter-spacing:.08em;text-transform:uppercase;
color:var(--accent-coral-dark);background:rgba(224,114,86,.2);padding:1px 6px;border-radius:999px;margin-bottom:6px}
/* ---- verdict banner ---- */
.verdict{display:flex;gap:13px;align-items:flex-start;padding:15px 17px;border-radius:var(--radius-md);margin:2px 0 20px}
.verdict.hard{background:var(--coral-bg)}
.verdict .vic{flex:none;width:22px;height:22px;border-radius:50%;display:grid;place-items:center;margin-top:1px;
background:var(--accent-coral-dark);color:#fff;font-size:12px;font-weight:700}
.verdict .vhead{font-weight:700;font-size:16.5px;letter-spacing:-.01em;color:var(--accent-coral-dark);text-wrap:balance}
.verdict .vsub{color:var(--text-secondary);font-size:13.5px;margin-top:3px}
/* ---- chart ---- */
.chartwrap{overflow-x:auto}
.chart{display:block;width:100%;min-width:460px;height:auto}
.axtxt{font-family:var(--sans);font-size:11px;fill:var(--text-muted)}
.ptlbl{font-family:var(--sans);font-size:12px;font-weight:700;fill:var(--accent-coral-dark)}
.keyrow{display:flex;flex-wrap:wrap;gap:16px;margin-top:12px;font-size:12.5px;color:var(--text-secondary)}
.keyrow span{display:inline-flex;align-items:center;gap:7px}
.kdot{width:11px;height:11px;border-radius:50%;flex:none}
.kdot.line{background:var(--accent-coral)}
.kdot.open{background:#fff;box-shadow:inset 0 0 0 2px var(--accent-teal)}
.kdot.gap{background:repeating-linear-gradient(90deg,var(--text-muted) 0 2px,transparent 2px 4px);border-radius:0;height:2px}
/* ---- table ---- */
.tblwrap{overflow-x:auto;margin-top:4px}
table{width:100%;border-collapse:collapse;font-size:14px;min-width:420px}
thead th{text-align:right;font-size:11px;font-weight:600;letter-spacing:.04em;text-transform:uppercase;
color:var(--text-muted);padding:0 10px 9px;border-bottom:1px solid var(--border)}
thead th:first-child{text-align:left}
tbody td{text-align:right;padding:11px 10px;border-bottom:1px solid var(--border)}
tbody td:first-child{text-align:left;font-weight:600}
tbody tr:last-child td{border-bottom:none}
tbody tr.now{background:var(--bg-secondary)}
td.miss{color:var(--text-muted);font-weight:400}
.pill{display:inline-block;font-size:11px;font-weight:600;padding:2px 9px;border-radius:999px;white-space:nowrap}
.pill.over{background:var(--coral-bg);color:var(--accent-coral-dark)}
.pill.ok{background:var(--teal-bg);color:var(--accent-teal)}
.pill.na{background:var(--bg-secondary);color:var(--text-muted)}
/* ---- map ---- */
.mapfig{border-radius:var(--radius-md);overflow:hidden;border:1px solid var(--border);background:#eef2ec}
.mapsvg{display:block;width:100%;height:auto}
.maplegend{display:flex;flex-wrap:wrap;gap:14px 20px;margin-top:12px;font-size:12.5px;color:var(--text-secondary)}
.maplegend span{display:inline-flex;align-items:center;gap:7px}
.swatch{width:14px;height:14px;border-radius:50%;flex:none}
.swatch.now{background:rgba(224,114,86,.18);box-shadow:inset 0 0 0 2px var(--accent-coral)}
.swatch.past{background:transparent;box-shadow:inset 0 0 0 1.5px rgba(176,74,46,.4)}
.swatch.you{background:var(--accent-teal);border-radius:2px;transform:rotate(45deg);width:11px;height:11px}
/* ---- postcode check ---- */
.checkbox{margin-top:20px;border-top:1px solid var(--border);padding-top:18px}
.checkhead{font-weight:700;font-size:15px;margin-bottom:3px}
.checksub{font-size:13.5px;color:var(--text-muted);margin-bottom:12px}
.checkform{display:flex;gap:8px;flex-wrap:wrap}
.checkform input{font:inherit;font-size:15px;padding:10px 13px;border:1px solid var(--border);border-radius:var(--radius-md);
background:var(--bg-card);color:var(--text-primary);min-width:150px;flex:1 1 150px;text-transform:uppercase}
.checkform input:focus-visible{outline:2px solid var(--accent-teal);outline-offset:1px;border-color:var(--accent-teal)}
.checkform button{font:inherit;font-weight:600;font-size:15px;padding:10px 20px;border:none;border-radius:var(--radius-md);
background:var(--accent-coral-dark);color:#fff;cursor:pointer;transition:background .15s}
.checkform button:hover{background:#9c3f26}
.checkform button:focus-visible{outline:2px solid var(--accent-teal);outline-offset:2px}
.result{margin-top:14px;background:var(--teal-bg);border-radius:var(--radius-md);padding:15px 17px}
.result .rhead{font-weight:700;font-size:16px;color:var(--accent-teal);letter-spacing:-.01em;text-wrap:balance}
.result .rsub{font-size:13.5px;color:var(--text-secondary);margin-top:4px}
.yearstrip{display:flex;gap:4px;margin-top:12px;flex-wrap:wrap}
.yr{font-size:11px;font-weight:600;padding:3px 7px;border-radius:var(--radius-sm);white-space:nowrap}
.yr.in{background:rgba(45,125,125,.22);color:var(--accent-teal)}
.yr.out{background:var(--coral-bg);color:var(--accent-coral-dark)}
.yr.none{background:var(--bg-secondary);color:var(--text-muted)}
/* ---- notes / disclosure ---- */
.note{display:flex;gap:9px;margin-top:16px;font-size:13px;color:var(--text-muted);line-height:1.55}
.note .i{flex:none;width:17px;height:17px;border-radius:50%;background:var(--bg-secondary);color:var(--text-muted);
font-size:11px;font-weight:700;display:grid;place-items:center;margin-top:2px}
.disclosure{margin-top:16px;border-top:1px solid var(--border);padding-top:4px}
.disclosure>summary{list-style:none;cursor:pointer;display:flex;align-items:center;gap:8px;padding:10px 0;
font-weight:600;font-size:14px;color:var(--accent-teal)}
.disclosure>summary::-webkit-details-marker{display:none}
.disclosure>summary:focus-visible{outline:2px solid var(--accent-teal);outline-offset:2px;border-radius:var(--radius-sm)}
.disclosure .chev{transition:transform .2s ease}
.disclosure[open]>summary .chev{transform:rotate(90deg)}
.disclosure .body{padding:2px 0 10px;font-size:13.5px;color:var(--text-secondary);display:flex;flex-direction:column;gap:10px}
.disclosure .body b{color:var(--text-primary)}
/* ---- viewport stack (keeps card height stable across views) ---- */
.viewport{display:grid}
.viewport>.view{grid-area:1/1}
.viewport>.view[hidden]{display:block;visibility:hidden;pointer-events:none}
/* ---- edge cases ---- */
.cases{display:grid;grid-template-columns:repeat(auto-fit,minmax(250px,1fr));gap:14px}
.case{background:var(--bg-card);border:1px solid var(--border);border-radius:var(--radius-md);padding:16px 17px}
.case h3{font-size:12px;font-weight:700;letter-spacing:.05em;text-transform:uppercase;color:var(--text-muted);margin-bottom:10px}
.case .body{font-size:14px;color:var(--text-secondary);line-height:1.5}
.case .body strong{color:var(--text-primary)}
.emptybox{background:var(--bg-secondary);border-radius:var(--radius-md);padding:13px 15px;font-size:13.5px;color:var(--text-secondary)}
/* ---- phone ---- */
.phonerow{display:flex;gap:24px;flex-wrap:wrap;align-items:flex-start}
.phone{width:330px;max-width:100%;border:9px solid #1a1612;border-radius:34px;overflow:hidden;box-shadow:var(--shadow-md);background:var(--bg-primary)}
.phonebody{padding:14px 13px 20px;display:flex;flex-direction:column;gap:12px}
.phone .card{padding:17px 16px 16px;border-radius:var(--radius-md)}
.phone .sectionTitle{font-size:18px}
.phone .tiles{grid-template-columns:1fr 1fr;gap:8px}
.phone .tile{padding:11px 12px}
.phone .tile .num{font-size:22px}
.phone .chart{min-width:0}
.phone .chartwrap{overflow:visible}
.phonenote{font-size:13px;color:var(--text-muted);flex:1 1 240px;min-width:220px}
.phonenote h3{font-family:var(--serif);font-size:17px;color:var(--text-primary);margin-bottom:8px;font-weight:600}
.phonenote ul{padding-left:18px;display:flex;flex-direction:column;gap:7px}
@media (prefers-reduced-motion:reduce){*{transition:none!important;animation:none!important}}
</style>
<div class="wrap">
<div class="pagehead">
<h1>Last distance offered — school detail page</h1>
<p>Adds the final-offer cut-off distance to the existing <b>Admissions</b> card, plus a catchment
view on the map. Data covers one to ten years depending on the school and local authority, so every
screen here is built around partial coverage rather than assuming a full run.</p>
<p>Sample school: <b>Fairlawn Primary School</b>, Lewisham — 8 years of distance data out of 10 years of admissions data.</p>
</div>
<!-- ============ 1. ADMISSIONS CARD ============ -->
<div class="step">
<h2>1. A distance tile joins the admissions tiles</h2>
<span class="where">SchoolDetailView · #admissions</span>
</div>
<p class="stepdesc">No new section and no new nav entry — the number a parent actually asks for
("how close do we need to live?") sits with the rest of the intake story. The segmented control gains a
third view, <code>Distance</code>.</p>
<div class="card">
<div class="cardhead">
<h2 class="sectionTitle">Admissions</h2>
<div class="seg" role="group" aria-label="Admissions view">
<button type="button" aria-pressed="true" data-view="year">This year</button>
<button type="button" aria-pressed="false" data-view="trend">10-year trend</button>
<button type="button" aria-pressed="false" data-view="dist">Distance</button>
</div>
</div>
<p class="sectionSub">Reception entry, September 2025.</p>
<div class="viewport">
<!-- view: this year -->
<div class="view" id="v-year">
<dl class="tiles">
<div class="tile">
<dd class="num">60</dd><dt class="lbl">Places offered</dt>
</div>
<div class="tile">
<dd class="num">142</dd><dt class="lbl">Wanted it first</dt>
</div>
<div class="tile accent">
<dd class="num">54<span class="sub">of 142 · 38%</span></dd>
<dt class="lbl">Got their first choice</dt>
</div>
<div class="tile newtile">
<span class="newflag">New</span>
<dd class="num">0.31<span class="unit">mi</span><span class="sub">≈ 500 m · 6 min walk</span></dd>
<dt class="lbl">Last distance offered</dt>
</div>
</dl>
<div class="note">
<span class="i" aria-hidden="true">i</span>
<span>The furthest home offered a place once siblings, faith and EHCP priority were applied.
It is not a fixed catchment — it moves every year with the number of applications.</span>
</div>
</div>
<!-- view: distance -->
<div class="view" id="v-dist" hidden>
<div class="verdict hard">
<span class="vic" aria-hidden="true">↓</span>
<div>
<div class="vhead">The catchment has halved in nine years</div>
<div class="vsub">0.62 mi in 2016 → 0.31 mi in 2025. Four of the last five years tightened.</div>
</div>
</div>
<div class="chartwrap">
<svg class="chart" viewBox="0 0 700 250" role="img"
aria-label="Last distance offered by year: 0.62 miles in 2016, 0.55 in 2017, not published in 2018, 0.48 in 2019, 0.51 in 2020, 0.44 in 2021, no cut-off needed in 2022, 0.39 in 2023, 0.35 in 2024, 0.31 in 2025.">
<!-- grid -->
<g stroke="#e5dfd5" stroke-width="1">
<line x1="46" y1="30" x2="686" y2="30"/>
<line x1="46" y1="80" x2="686" y2="80"/>
<line x1="46" y1="130" x2="686" y2="130"/>
<line x1="46" y1="180" x2="686" y2="180"/>
</g>
<line x1="46" y1="206" x2="686" y2="206" stroke="#d8d0c3" stroke-width="1.5"/>
<g class="axtxt" text-anchor="end">
<text x="38" y="34">0.8</text><text x="38" y="84">0.6</text>
<text x="38" y="134">0.4</text><text x="38" y="184">0.2</text>
</g>
<text class="axtxt" x="46" y="16" text-anchor="start">miles</text>
<!-- gap segments (dashed = no figure published) -->
<g fill="none" stroke="#6d685f" stroke-width="1.5" stroke-dasharray="4 4" opacity=".55">
<path d="M114 92 L182 110"/>
<path d="M318 122 L386 132.5"/>
</g>
<!-- solid series -->
<polyline fill="none" stroke="#e07256" stroke-width="2.5" stroke-linejoin="round" stroke-linecap="round"
points="46,75 114,92"/>
<polyline fill="none" stroke="#e07256" stroke-width="2.5" stroke-linejoin="round" stroke-linecap="round"
points="182,110 250,102 318,122"/>
<polyline fill="none" stroke="#e07256" stroke-width="2.5" stroke-linejoin="round" stroke-linecap="round"
points="386,132.5 454,142.5 522,152.5"/>
<!-- points -->
<g fill="#e07256" stroke="#fff" stroke-width="2">
<circle cx="46" cy="75" r="5"/><circle cx="114" cy="92" r="5"/>
<circle cx="182" cy="110" r="5"/><circle cx="250" cy="102" r="5"/>
<circle cx="318" cy="122" r="5"/><circle cx="386" cy="132.5" r="5"/>
<circle cx="454" cy="142.5" r="5"/>
</g>
<!-- latest, emphasised -->
<circle cx="522" cy="152.5" r="7.5" fill="#b04a2e" stroke="#fff" stroke-width="2.5"/>
<!-- "no cut-off needed" marker, 2022 -->
<circle cx="352" cy="46" r="6" fill="#fff" stroke="#296f6f" stroke-width="2.5"/>
<text class="axtxt" x="352" y="34" text-anchor="middle" fill="#296f6f" font-weight="600">all offered</text>
<line x1="352" y1="54" x2="352" y2="196" stroke="#296f6f" stroke-width="1.5" stroke-dasharray="3 4" opacity=".45"/>
<!-- endpoint labels -->
<text class="ptlbl" x="46" y="63" text-anchor="start">0.62</text>
<text class="ptlbl" x="530" y="157" text-anchor="start">0.31 mi</text>
<!-- year axis -->
<g class="axtxt" text-anchor="middle">
<text x="46" y="226">2016</text><text x="114" y="226">2017</text>
<text x="182" y="226">2019</text><text x="250" y="226">2020</text>
<text x="318" y="226">2021</text><text x="386" y="226">2023</text>
<text x="454" y="226">2024</text><text x="522" y="226">2025</text>
</g>
<text class="axtxt" x="148" y="243" text-anchor="middle" fill="#6d685f">2018 not published</text>
<text class="axtxt" x="352" y="243" text-anchor="middle" fill="#296f6f">2022 undersubscribed</text>
</svg>
</div>
<div class="keyrow">
<span><i class="kdot line" aria-hidden="true"></i>Distance of the last place offered</span>
<span><i class="kdot open" aria-hidden="true"></i>No cut-off needed — every applicant offered</span>
<span><i class="kdot gap" aria-hidden="true"></i>Not published by the local authority</span>
</div>
<div class="tblwrap" style="margin-top:22px">
<table>
<caption class="sr-only" style="position:absolute;width:1px;height:1px;overflow:hidden;clip:rect(0 0 0 0)">Last distance offered by year</caption>
<thead>
<tr><th scope="col">Year</th><th scope="col">Last distance</th><th scope="col">Places</th><th scope="col">Status</th></tr>
</thead>
<tbody>
<tr class="now"><td>2025</td><td>0.31 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2024</td><td>0.35 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2023</td><td>0.39 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2022</td><td class="miss">No cut-off needed</td><td>60</td><td><span class="pill ok">All offered</span></td></tr>
<tr><td>2021</td><td>0.44 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2020</td><td>0.51 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2019</td><td>0.48 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2018</td><td class="miss">—</td><td>60</td><td><span class="pill na">Not published</span></td></tr>
<tr><td>2017</td><td>0.55 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
<tr><td>2016</td><td>0.62 mi</td><td>60</td><td><span class="pill over">Oversubscribed</span></td></tr>
</tbody>
</table>
</div>
<details class="disclosure">
<summary><span class="chev" aria-hidden="true">›</span>What this number does and doesn't tell you</summary>
<div class="body">
<span><b>It is a result, not a rule.</b> It records how far the last successful applicant lived
in a given year. Move one large sibling cohort and the figure shifts.</span>
<span><b>Distance is the final tiebreak.</b> Children in care, EHCP places, siblings and — at faith
schools — the faith criteria are ranked first. A family inside the distance can still miss out.</span>
<span><b>Lewisham measures straight-line distance</b> from home to the school's main gate. Other
authorities use walking routes, which are always longer for the same home.</span>
<span><b>Gaps are normal.</b> An authority may not publish a figure, or the school may not have
needed a distance cut-off that year. Both are shown here rather than hidden.</span>
</div>
</details>
</div>
</div>
</div>
<!-- ============ 2. MAP ============ -->
<div class="step">
<h2>2. The cut-off drawn on the map</h2>
<span class="where">SchoolHeroMap · catchment layer</span>
</div>
<p class="stepdesc">"0.31 miles" is abstract until you see it over your own streets. The latest year is a
filled ring; earlier years sit behind it as hairlines, so the tightening reads instantly as a set of
shrinking circles. Straight-line rings only — they're an illustration of the number, not a boundary.</p>
<div class="card">
<div class="cardhead" style="margin-bottom:14px">
<h2 class="sectionTitle">Where the last place went</h2>
<div class="seg" role="group" aria-label="Map years">
<button type="button" aria-pressed="true">2025 only</button>
<button type="button" aria-pressed="false">Last 5 years</button>
</div>
</div>
<figure class="mapfig">
<svg class="mapsvg" viewBox="0 0 700 400" role="img"
aria-label="Map showing concentric catchment rings around Fairlawn Primary School: 0.62 miles in 2016 shrinking to 0.31 miles in 2025, with a marker for a sample home 0.24 miles away, inside the current ring.">
<rect width="700" height="400" fill="#eef2ec"/>
<!-- park -->
<path d="M470 20 h230 v150 h-160 q-70 -20 -70 -80 z" fill="#dfe9dc"/>
<!-- water -->
<path d="M0 330 q120 -40 250 -10 t260 -20 l190 -30 v130 H0 z" fill="#dbe6ee"/>
<!-- street grid -->
<g stroke="#fff" stroke-width="7" stroke-linecap="round" opacity=".95">
<path d="M-10 90 H710"/><path d="M-10 200 H710"/><path d="M-10 300 H710"/>
<path d="M120 -10 V410"/><path d="M300 -10 V410"/><path d="M470 -10 V410"/><path d="M620 -10 V410"/>
</g>
<g stroke="#fff" stroke-width="3.5" opacity=".8">
<path d="M-10 145 H710"/><path d="M-10 250 H710"/><path d="M210 -10 V410"/><path d="M385 -10 V410"/><path d="M550 -10 V410"/>
</g>
<!-- building blocks -->
<g fill="#e4e2dc" opacity=".85">
<rect x="132" y="102" width="60" height="30" rx="2"/><rect x="222" y="102" width="64" height="30" rx="2"/>
<rect x="132" y="212" width="60" height="26" rx="2"/><rect x="222" y="212" width="64" height="26" rx="2"/>
<rect x="400" y="102" width="56" height="30" rx="2"/><rect x="400" y="212" width="56" height="26" rx="2"/>
</g>
<!-- historic rings (hairline) -->
<g fill="none" stroke="#b04a2e" opacity=".38" stroke-width="1.5">
<circle cx="300" cy="200" r="150"/>
<circle cx="300" cy="200" r="126"/>
<circle cx="300" cy="200" r="106"/>
<circle cx="300" cy="200" r="90"/>
</g>
<text class="axtxt" x="300" y="44" text-anchor="middle" fill="#b04a2e" font-weight="600">2016 · 0.62 mi</text>
<!-- current ring -->
<circle cx="300" cy="200" r="75" fill="rgba(224,114,86,.18)" stroke="#e07256" stroke-width="3"/>
<text class="axtxt" x="300" y="118" text-anchor="middle" fill="#b04a2e" font-weight="700" font-size="12.5">2025 · 0.31 mi</text>
<!-- school pin -->
<circle cx="300" cy="200" r="11" fill="#b04a2e" stroke="#fff" stroke-width="3"/>
<text class="axtxt" x="300" y="232" text-anchor="middle" fill="#1a1612" font-weight="700" font-size="12">Fairlawn Primary</text>
<!-- your home -->
<g transform="translate(246,158)">
<rect x="-7" y="-7" width="14" height="14" rx="2" fill="#296f6f" stroke="#fff" stroke-width="2.5" transform="rotate(45)"/>
</g>
<text class="axtxt" x="246" y="140" text-anchor="middle" fill="#296f6f" font-weight="700" font-size="12">Your home · 0.24 mi</text>
</svg>
</figure>
<div class="maplegend">
<span><i class="swatch now" aria-hidden="true"></i>2025 cut-off — 0.31 mi</span>
<span><i class="swatch past" aria-hidden="true"></i>Earlier years, 2016–2024</span>
<span><i class="swatch you" aria-hidden="true"></i>Your home</span>
</div>
<div class="checkbox">
<div class="checkhead">Would you have got in?</div>
<p class="checksub">We measure straight-line distance from your postcode, the same way Lewisham does.</p>
<form class="checkform" onsubmit="return false">
<label class="sr-only" for="pc" style="position:absolute;width:1px;height:1px;overflow:hidden;clip:rect(0 0 0 0)">Your postcode</label>
<input id="pc" type="text" value="SE23 3NA" autocomplete="postal-code" spellcheck="false">
<button type="submit">Check</button>
</form>
<div class="result">
<div class="rhead">0.24 miles away — inside the cut-off in all 8 years on record</div>
<div class="rsub">That's 0.07 miles of headroom on 2025, the tightest year so far. 2018 has no published
figure, and in 2022 every applicant was offered a place.</div>
<div class="yearstrip">
<span class="yr in">2016 ✓</span>
<span class="yr in">2017 ✓</span>
<span class="yr none">2018 –</span>
<span class="yr in">2019 ✓</span>
<span class="yr in">2020 ✓</span>
<span class="yr in">2021 ✓</span>
<span class="yr in">2022 ✓</span>
<span class="yr in">2023 ✓</span>
<span class="yr in">2024 ✓</span>
<span class="yr in">2025 ✓</span>
</div>
</div>
<div class="note">
<span class="i" aria-hidden="true">!</span>
<span>An indication only. Distance is applied after siblings, faith and EHCP priority, and next year's
cut-off depends on next year's applicants. Always check the school's own admissions policy.</span>
</div>
</div>
</div>
<!-- ============ 3. COVERAGE STATES ============ -->
<div class="step">
<h2>3. Coverage states</h2>
<span class="where">Partial data is the normal case</span>
</div>
<p class="stepdesc">Coverage runs from ten years to none. Each state says something true rather than
falling back on a generic "no data" — the reason a figure is absent is itself useful to a parent.</p>
<div class="cases">
<div class="case">
<h3>4+ years</h3>
<div class="body">Full treatment: verdict banner, chart, table, map rings. <strong>The verdict line only
appears with 4+ points</strong> — below that a two-year swing isn't a trend.</div>
</div>
<div class="case">
<h3>2–3 years</h3>
<div class="body">Tile and table, no verdict banner, single map ring for the latest year.
<strong>"Only 3 years available"</strong> sits under the table.</div>
</div>
<div class="case">
<h3>1 year</h3>
<div class="body">Tile plus one map ring. No <code>Distance</code> tab — the tile and its footnote are
the whole story.</div>
</div>
<div class="case">
<h3>Never oversubscribed</h3>
<div class="body" style="margin-bottom:10px">Not missing data — good news, and it should read that way.</div>
<div class="emptybox"><b style="color:var(--accent-teal)">Everyone who applied was offered a place</b>
in each of the last 6 years, so no distance cut-off was needed.</div>
</div>
<div class="case">
<h3>No data at all</h3>
<div class="body" style="margin-bottom:10px">Replaces the current placeholder text in
<code>SecondarySchoolDetailView.tsx:780</code>.</div>
<div class="emptybox">Lewisham hasn't published cut-off distances for this school.
<a href="#" style="color:var(--accent-teal);font-weight:600">The council's admissions page</a> may list them.</div>
</div>
<div class="case">
<h3>Not distance-ranked</h3>
<div class="body" style="margin-bottom:10px">Grammar and some faith schools rank on test score or faith
practice, so a distance figure would mislead.</div>
<div class="emptybox">Places here are ranked by the entrance test, not by distance.</div>
</div>
</div>
<!-- ============ 4. MOBILE ============ -->
<div class="step">
<h2>4. Mobile</h2>
<span class="where">≤ 640 px</span>
</div>
<p class="stepdesc">Tiles fall to two columns and the distance tile takes the full width beneath them, so the
headline number survives the reflow. The chart drops its table on mobile behind a "See all years" disclosure.</p>
<div class="phonerow">
<div class="phone">
<div class="phonebody">
<div class="card">
<div class="cardhead" style="margin-bottom:6px">
<h2 class="sectionTitle">Admissions</h2>
</div>
<p class="sectionSub" style="font-size:13px;margin-bottom:12px">Reception, Sept 2025</p>
<div class="seg" style="margin-bottom:14px">
<button type="button" aria-pressed="true">Year</button>
<button type="button" aria-pressed="false">Trend</button>
<button type="button" aria-pressed="false">Distance</button>
</div>
<dl class="tiles">
<div class="tile"><dd class="num">60</dd><dt class="lbl">Places offered</dt></div>
<div class="tile"><dd class="num">142</dd><dt class="lbl">Wanted it first</dt></div>
</dl>
<div class="tile newtile" style="margin-top:8px">
<span class="newflag">New</span>
<dd class="num" style="font-size:26px">0.31<span class="unit">mi</span>
<span class="sub">≈ 500 m · 6 min walk</span></dd>
<dt class="lbl">Last distance offered, 2025</dt>
</div>
</div>
<div class="card">
<h2 class="sectionTitle" style="margin-bottom:12px">Distance</h2>
<div class="verdict hard" style="padding:12px 13px;margin:0 0 14px">
<span class="vic" aria-hidden="true">↓</span>
<div><div class="vhead" style="font-size:15px">Halved in nine years</div>
<div class="vsub" style="font-size:12.5px">0.62 mi → 0.31 mi</div></div>
</div>
<div class="chartwrap">
<svg class="chart" viewBox="0 0 300 150" role="img" aria-label="Last distance offered falling from 0.62 miles in 2016 to 0.31 miles in 2025.">
<g stroke="#e5dfd5" stroke-width="1">
<line x1="26" y1="20" x2="292" y2="20"/><line x1="26" y1="62" x2="292" y2="62"/><line x1="26" y1="104" x2="292" y2="104"/>
</g>
<g class="axtxt" text-anchor="end" font-size="9">
<text x="21" y="23">0.8</text><text x="21" y="65">0.5</text><text x="21" y="107">0.2</text>
</g>
<path d="M26 55 L64 69" fill="none" stroke="#e07256" stroke-width="2.5" stroke-linecap="round"/>
<path d="M64 69 L102 84" fill="none" stroke="#6d685f" stroke-width="1.5" stroke-dasharray="3 3" opacity=".55"/>
<path d="M102 84 L140 78 L178 95" fill="none" stroke="#e07256" stroke-width="2.5" stroke-linejoin="round" stroke-linecap="round"/>
<path d="M178 95 L216 109" fill="none" stroke="#6d685f" stroke-width="1.5" stroke-dasharray="3 3" opacity=".55"/>
<path d="M216 109 L254 116 L280 120" fill="none" stroke="#e07256" stroke-width="2.5" stroke-linejoin="round" stroke-linecap="round"/>
<g fill="#e07256" stroke="#fff" stroke-width="1.8">
<circle cx="26" cy="55" r="4"/><circle cx="64" cy="69" r="4"/><circle cx="102" cy="84" r="4"/>
<circle cx="140" cy="78" r="4"/><circle cx="178" cy="95" r="4"/><circle cx="216" cy="109" r="4"/><circle cx="254" cy="116" r="4"/>
</g>
<circle cx="280" cy="120" r="6" fill="#b04a2e" stroke="#fff" stroke-width="2"/>
<circle cx="197" cy="30" r="4.5" fill="#fff" stroke="#296f6f" stroke-width="2"/>
<g class="axtxt" font-size="9" text-anchor="middle">
<text x="26" y="132">'16</text><text x="140" y="132">'20</text><text x="280" y="132">'25</text>
</g>
<text class="ptlbl" x="280" y="140" text-anchor="middle" font-size="11">0.31 mi</text>
</svg>
</div>
<details class="disclosure" style="margin-top:10px">
<summary style="font-size:13px"><span class="chev" aria-hidden="true">›</span>See all 10 years</summary>
</details>
</div>
</div>
</div>
<div class="phonenote">
<h3>Mobile decisions</h3>
<ul>
<li>Distance tile spans both columns — the number a parent came for shouldn't be a half-width cell.</li>
<li>Chart keeps every year but labels only first, middle and last; the endpoint stays labelled.</li>
<li>Year-by-year table collapses into a disclosure rather than forcing a horizontal scroll.</li>
<li>Map rings reuse the existing full-screen hero map sheet, opened from the tile.</li>
</ul>
</div>
</div>
</div>
<script>
// Segmented control on the admissions card swaps the stacked views.
document.querySelectorAll('.seg button[data-view]').forEach(function (btn) {
btn.addEventListener('click', function () {
var group = btn.closest('.seg');
group.querySelectorAll('button').forEach(function (b) { b.setAttribute('aria-pressed', String(b === btn)); });
var target = btn.dataset.view === 'dist' ? 'v-dist' : 'v-year';
['v-year', 'v-dist'].forEach(function (id) {
document.getElementById(id).hidden = id !== target;
});
});
});
</script>
-1
View File
@@ -39,4 +39,3 @@ yarn-error.log*
# typescript
*.tsbuildinfo
next-env.d.ts
-7
View File
@@ -53,13 +53,6 @@ COPY --from=builder /app/.next/static ./.next/static
# a miss here is a silent 500 on /opengraph-image, not a build failure.
COPY --from=builder /app/assets ./assets
# Payload writes uploads here, and the compose file mounts a named volume over
# it. The directory must exist and be owned by the runtime user BEFORE the
# mount: Docker seeds a fresh named volume from the image path, so a missing or
# root-owned directory here makes every upload fail with EACCES at runtime,
# long after the build passed. The chown below covers it.
RUN mkdir -p /app/media
# Set correct permissions
RUN chown -R nextjs:nodejs /app
@@ -1,31 +0,0 @@
/**
* The /api/* proxy is public. Anything it forwards is on the internet.
*
* @jest-environment node
*/
// The docblock above is load-bearing. jest.config.js sets jsdom globally, and
// NextRequest/NextResponse need the Web Fetch API globals that only the node
// environment provides — under jsdom this suite fails on import, not on an
// assertion.
import { NextRequest } from 'next/server';
import { GET } from '@/app/(frontend)/api/[...path]/route';
function request(path: string) {
return new NextRequest(`http://localhost:3000/api/${path}`);
}
describe('public API proxy', () => {
it('refuses to forward internal-only paths', async () => {
// /api/flags names every unreleased feature and its state. Forwarding it
// publishes the thing shipping dark exists to keep quiet.
const res = await GET(request('flags'), { params: Promise.resolve({ path: ['flags'] }) });
expect(res.status).toBe(404);
});
it('does not deny a path that merely starts with the same letters', async () => {
// A prefix match would take /api/flagship down with /api/flags.
const res = await GET(
request('flagship'), { params: Promise.resolve({ path: ['flagship'] }) });
expect(res.status).not.toBe(404);
});
});
@@ -1,34 +0,0 @@
import { metadata } from '@/app/(frontend)/about/page';
import { personJsonLd, organizationJsonLd } from '@/lib/jsonld';
describe('/about metadata', () => {
it('canonicalises to the bare path', () => {
expect(metadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/about');
});
});
describe('author structured data', () => {
it('describes a Person with a first name and a photo', () => {
const person = personJsonLd();
expect(person['@type']).toBe('Person');
expect(person.name).toBe('Tudor');
expect(person.image).toBe('https://www.schoolcompare.co.uk/brand/tudor.jpg');
expect(person.url).toBe('https://www.schoolcompare.co.uk/about');
});
it('never publishes a surname or an employer', () => {
// Author identity constraint: first name only. A surname here would be
// the one place it leaks, since JSON-LD is machine-read and archived.
const serialised = JSON.stringify(personJsonLd());
expect(serialised).not.toMatch(/familyName|Sitaru/i);
expect(serialised).not.toMatch(/worksFor|affiliation/i);
});
it('describes the site as an Organization the Person authors for', () => {
const org = organizationJsonLd();
expect(org['@type']).toBe('Organization');
expect(org.name).toBe('schoolcompare');
expect(org.url).toBe('https://www.schoolcompare.co.uk');
});
});
@@ -1,66 +0,0 @@
/**
* The blog index imports getCachedPayload, which pulls in Payload — ESM-only,
* and next/jest will not transform node_modules. Mocking that one module keeps
* the page's metadata testable without loading the CMS; the mock is never
* called, because `metadata` is a static export evaluated at import time.
*/
jest.mock('@/lib/payload', () => ({ getCachedPayload: jest.fn() }));
import { metadata } from '@/app/(frontend)/blog/page';
import { blogPostingJsonLd, breadcrumbJsonLd } from '@/lib/jsonld';
const post = {
title: 'What the data cannot tell you',
slug: 'what-the-data-cannot-tell-you',
excerpt: 'Results describe one year group on a handful of days.',
publishedAt: '2026-09-15T00:00:00.000Z',
};
describe('/blog metadata', () => {
it('canonicalises to the bare path', () => {
expect(metadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/blog');
});
});
describe('BlogPosting structured data', () => {
it('names the same Person entity the about page declares', () => {
// By @id, not by repeating the person: search engines must resolve every
// post and the about page to one author entity, or the site has several.
const ld = blogPostingJsonLd(post, { namedAuthor: true });
expect(ld['@type']).toBe('BlogPosting');
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk/about#tudor');
expect(ld.publisher['@id']).toBe('https://www.schoolcompare.co.uk#organization');
});
it('attributes to the organization when the about page is dark', () => {
/*
* The two flags are independent, so blog-on-about-off is a reachable
* state. The Person entity lives at /about#tudor and that URL 404s while
* the flag is dark, so claiming it would declare an author that resolves
* to nothing — worse for the blog's credibility than having no named
* author at all. Attribute to the publisher instead.
*/
const ld = blogPostingJsonLd(post, { namedAuthor: false });
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk#organization');
expect(JSON.stringify(ld)).not.toContain('/about');
});
it('carries a self-referencing canonical url and the publish date', () => {
const ld = blogPostingJsonLd(post, { namedAuthor: true });
expect(ld.url).toBe(
'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you',
);
expect(ld.datePublished).toBe('2026-09-15T00:00:00.000Z');
});
});
describe('breadcrumbs', () => {
it('places the post under the blog index', () => {
const ld = breadcrumbJsonLd(post);
expect(ld.itemListElement[0].item).toBe('https://www.schoolcompare.co.uk/blog');
expect(ld.itemListElement[1].item).toBe(
'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you',
);
});
});
-130
View File
@@ -1,130 +0,0 @@
import { metadata as homeMetadata } from '@/app/(frontend)/page';
import { metadata as rankingsMetadata } from '@/app/(frontend)/rankings/page';
import { metadata as admissionsMetadata } from '@/app/(frontend)/admissions/page';
import { generateMetadata as compareMetadata } from '@/app/(frontend)/compare/page';
describe('canonical URLs', () => {
it('the homepage canonicalises to the bare root', () => {
// page.tsx reads eleven search params. Without this, every filter
// combination is a crawlable near-duplicate of the one page we want to
// rank for "compare schools".
expect(homeMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/');
});
it('rankings canonicalises to the bare path', () => {
expect(rankingsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/rankings');
});
it('admissions canonicalises to the bare path', () => {
expect(admissionsMetadata.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/admissions');
});
});
describe('/compare indexability', () => {
it('the bare compare page is indexable and canonical to itself', async () => {
// This is the landing page for the "compare schools" head term.
const meta = await compareMetadata({ searchParams: Promise.resolve({}) });
expect(meta.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/compare');
expect(meta.robots).toBeUndefined();
});
it('a comparison of specific schools is noindex, follow', async () => {
// ~317 million pairs before triples. Indexing the parameter space would
// swamp everything else in the corpus.
const meta = await compareMetadata({
searchParams: Promise.resolve({ urns: '100001,100002' }),
});
expect(meta.robots).toEqual({ index: false, follow: true });
});
it('a parameterised comparison still canonicalises to the bare path', async () => {
// follow:true plus a canonical means the outbound links to each school
// page still pass value even though this URL is not indexed.
const meta = await compareMetadata({
searchParams: Promise.resolve({ urns: '100001,100002' }),
});
expect(meta.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/compare');
});
});
/*
* W8 — snippet copy for the C1 cluster.
*
* The baseline (GSC, 16 months to 2026-08-20) showed these pages ranking on
* page one and converting at a tenth of the normal rate: "compare school
* performance" at position 6.1 with 0.43% CTR, against 9.16% for the brand
* query from the same neighbourhood. The SERP is dominated by the DfE's own
* "Compare school performance" service, so the job of this copy is to say
* what that service does not offer, without losing intent match on the title.
*
* These tests guard the mechanics that make a snippet work — length, intent
* keyword, differentiator, no brand-first — not the exact wording, which
* should stay free to iterate.
*/
// Google truncates titles near 60 characters and descriptions near 155.
const TITLE_MAX = 60;
const DESC_MIN = 110;
const DESC_MAX = 155;
type Meta = { title?: unknown; description?: unknown };
const titleOf = (m: Meta): string => {
const t = m.title as string | { absolute?: string } | undefined;
return typeof t === 'string' ? t : (t?.absolute ?? '');
};
describe('C1 snippet copy', () => {
const pages: Array<[string, Meta, RegExp]> = [
['home', homeMetadata as Meta, /compare schools/i],
['rankings', rankingsMetadata as Meta, /league table/i],
['admissions', admissionsMetadata as Meta, /admission/i],
];
for (const [name, meta, intent] of pages) {
it(`${name}: title carries the search intent and fits the SERP`, () => {
const t = titleOf(meta);
expect(t).toMatch(intent);
expect(t.length).toBeLessThanOrEqual(TITLE_MAX);
});
it(`${name}: title does not open with the brand`, () => {
// The measured 0.43% CTR came from a brand-first title. The most
// valuable pixels go to the thing the searcher typed.
expect(titleOf(meta).toLowerCase().startsWith('schoolcompare')).toBe(false);
});
it(`${name}: description is long enough to be worth reading, short enough to survive`, () => {
const d = meta.description as string;
expect(d.length).toBeGreaterThanOrEqual(DESC_MIN);
expect(d.length).toBeLessThanOrEqual(DESC_MAX);
});
}
it('the homepage description names what gov.uk does not publish', () => {
// Admissions distance is the one fact the DfE service has no equivalent
// for. If it ever leaves this description, the snippet is competing with
// gov.uk on gov.uk's own ground.
expect(homeMetadata.description).toMatch(/close you had to live|distance/i);
});
it('/compare targets the tool phrasing rather than repeating the homepage', () => {
// Two pages chasing one phrase is how a site competes with itself.
return compareMetadata({ searchParams: Promise.resolve({}) }).then((m) => {
expect(m.title).toMatch(/comparison tool/i);
expect(m.title).not.toBe(titleOf(homeMetadata as Meta));
});
});
it('no C1 page claims a school count that will drift', () => {
// The corpus moves with every data refresh; this repo has already shipped
// one copy bug of that kind ("three schools" against MAX_SCHOOLS = 5).
for (const [, meta] of pages) {
expect(meta.description as string).not.toMatch(/\b\d{2},\d{3}\b|\b\d{2},000\b/);
}
});
});
@@ -1,66 +0,0 @@
/**
* next.config.mjs carries the staging noindex rule. Breaking it turns
* stx.schoolcompare.co.uk into a fully crawlable duplicate of production,
* and nothing else in the suite would notice.
*
* The non-null assertions are deliberate: every key asserted here is optional
* on NextConfig, and a missing one is precisely the regression under test, so
* the assertion below should fail the test rather than the compile.
*/
import nextConfig from '@/next.config.mjs';
async function headerRules() {
return nextConfig.headers!();
}
describe('next.config.mjs', () => {
it('keeps the staging host out of the index', async () => {
const headers = await headerRules();
const stagingRule = headers.find((rule) =>
rule.has?.some(
(cond) => cond.type === 'host' && cond.value === 'stx.schoolcompare.co.uk',
),
);
expect(stagingRule).toBeDefined();
expect(stagingRule!.headers).toContainEqual({
key: 'X-Robots-Tag',
value: 'noindex, nofollow',
});
});
it('still emits standalone output for the Docker runner', () => {
expect(nextConfig.output).toBe('standalone');
});
it('still traces the share-card fonts into the standalone bundle', () => {
expect(nextConfig.outputFileTracingIncludes!['/opengraph-image']).toEqual([
'./assets/**',
]);
});
it('still allows the analytics subdomain to frame the site', async () => {
const headers = await headerRules();
const csp = headers
.flatMap((rule) => rule.headers)
.find((header) => header.key === 'Content-Security-Policy');
expect(csp).toBeDefined();
expect(csp!.value).toContain('https://analytics.schoolcompare.co.uk');
});
});
describe('admin surface', () => {
it('serves noindex on the admin panel and the CMS API', async () => {
// robots.txt disallows these too, but a Disallow only blocks crawling — a
// URL found from an external link can still be indexed without ever being
// fetched. This header is what actually keeps them out.
const headers = await headerRules();
for (const source of ['/admin/:path*', '/cms-api/:path*']) {
const rule = headers.find((entry) => entry.source === source);
expect(rule).toBeDefined();
expect(rule!.headers).toContainEqual({
key: 'X-Robots-Tag',
value: 'noindex, nofollow',
});
}
});
});
@@ -1,42 +0,0 @@
import { generateMetadata as placeMeta } from '@/app/(frontend)/schools/[place]/page';
jest.mock('@/lib/places', () => ({
...jest.requireActual('@/lib/places'),
fetchPlace: jest.fn(async (kind: string, slug: string) =>
slug === 'atlantis' ? null : ({
place: { kind, slug, name: 'Brentwood', count: 29,
parent_authority: 'Essex' },
schools: [], averages: { rwm_expected_pct: 63, attainment_8_score: null },
})),
fetchPlaces: jest.fn(async () => []),
}));
describe('place page metadata', () => {
it('titles the page the way the place is searched', async () => {
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
expect((m.title as { absolute: string }).absolute).toMatch(/schools in brentwood/i);
});
it('canonicalises to its own path on the www host', async () => {
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
expect(m.alternates?.canonical)
.toBe('https://www.schoolcompare.co.uk/schools/brentwood');
});
it('opts out of the layout template, which would double the brand', () => {
// The root layout appends '| schoolcompare' to a plain string title, and
// these titles already carry it — every place page shipped reading
// '... | schoolcompare | schoolcompare' until this was made absolute.
return placeMeta({ params: Promise.resolve({ place: 'brentwood' }) })
.then((m) => {
expect(typeof m.title).toBe('object');
expect((m.title as { absolute: string }).absolute)
.not.toMatch(/schoolcompare.*schoolcompare/);
});
});
it('an unknown place gets a not-found title rather than inventing one', async () => {
const m = await placeMeta({ params: Promise.resolve({ place: 'atlantis' }) });
expect(m.title).toMatch(/not found/i);
});
});
-20
View File
@@ -1,20 +0,0 @@
import robots from '@/app/robots';
describe('robots.txt', () => {
it('disallows the admin panel and the CMS API', () => {
const rules = robots().rules;
const rule = Array.isArray(rules) ? rules[0] : rules;
expect(rule.disallow).toEqual(
expect.arrayContaining(['/api/', '/_next/', '/admin/', '/cms-api/']),
);
});
});
describe('sitemap discovery', () => {
it('lists both the proxied school sitemap and the Next-owned content sitemap', () => {
expect(robots().sitemap).toEqual([
'https://www.schoolcompare.co.uk/sitemap.xml',
'https://www.schoolcompare.co.uk/content-sitemap.xml',
]);
});
});
@@ -1,152 +0,0 @@
/**
* The postcode check.
*
* This is the one place on the site that answers a question about a specific
* family rather than about a school, so the tests here are mostly about what it
* refuses to say — and that matters more now than it did, because there is only
* one year to answer with. A run of years used to soften a single close call;
* nothing does now, so the "too close to call" band is the whole safety margin.
*/
import { render, screen, fireEvent, waitFor } from '@testing-library/react';
import { CutoffMapPanel } from '@/components/school/CutoffMapPanel';
import { CUTOFF_UNCERTAINTY_M } from '@/components/school/lastDistanceOffered';
import type { School, SchoolAdmissionDistance } from '@/lib/types';
// Leaflet needs a real layout box and network tiles; neither exists in jsdom.
jest.mock('@/components/LeafletCutoffMapInner', () => ({
__esModule: true,
default: () => <div data-testid="cutoff-map" />,
}));
const mockGeocode = jest.fn();
jest.mock('@/lib/api', () => ({
...jest.requireActual('@/lib/api'),
geocodePostcode: (pc: string) => mockGeocode(pc),
}));
const SCHOOL = { urn: 100010, school_name: 'Test Primary', latitude: 51.5, longitude: -0.12 } as School;
const cutoff = (distance_m: number | null, year = 2026): SchoolAdmissionDistance => ({
year, distance_m, route_count: 1, la_name: 'Camden', distance_unit_raw: 'miles',
});
/** A point due north of the school, `metres` away. 1° latitude ≈ 111,320 m. */
function northOf(metres: number) {
return { latitude: SCHOOL.latitude! + metres / 111_320, longitude: SCHOOL.longitude! };
}
const renderPanel = (c = cutoff(800)) =>
render(<CutoffMapPanel schoolInfo={SCHOOL} cutoff={c} />);
async function check(postcode: string) {
fireEvent.change(screen.getByLabelText('Your postcode'), { target: { value: postcode } });
fireEvent.click(screen.getByRole('button', { name: 'Check' }));
}
beforeEach(() => mockGeocode.mockReset());
describe('CutoffMapPanel', () => {
it('renders nothing without a figure to compare against', () => {
const { container } = renderPanel(cutoff(null));
expect(container).toBeEmptyDOMElement();
});
it('renders nothing when the school has no coordinates', () => {
const { container } = render(
<CutoffMapPanel
schoolInfo={{ ...SCHOOL, latitude: null, longitude: null } as School}
cutoff={cutoff(800)}
/>,
);
expect(container).toBeEmptyDOMElement();
});
it('rejects a malformed postcode without calling the geocoder', async () => {
renderPanel();
await check('not a postcode');
expect(await screen.findByRole('alert')).toHaveTextContent(/does not look like a UK postcode/);
expect(mockGeocode).not.toHaveBeenCalled();
});
it('names the year in the verdict, so the figure is never free-floating', async () => {
mockGeocode.mockResolvedValue(northOf(200));
renderPanel(cutoff(800, 2026));
await check('SE23 3NA');
const result = await screen.findByRole('status');
expect(result).toHaveTextContent(/inside the/);
expect(result).toHaveTextContent(/September 2026/);
});
it('reports a home clearly beyond the cut-off', async () => {
mockGeocode.mockResolvedValue(northOf(5000));
renderPanel(cutoff(800));
await check('SE23 3NA');
expect(await screen.findByRole('status')).toHaveTextContent(/beyond the/);
});
it('declines to call a result that sits inside the measurement error', async () => {
// Nominally inside the 800 m cut-off, but by half the uncertainty band —
// which a postcode centroid cannot resolve. With only one year published
// there is nothing else to fall back on, so this must not read as a pass.
mockGeocode.mockResolvedValue(northOf(800 - CUTOFF_UNCERTAINTY_M / 2));
renderPanel(cutoff(800));
await check('SE23 3NA');
const result = await screen.findByRole('status');
expect(result).toHaveTextContent(/too close/);
expect(result).toHaveTextContent(/measurement error/);
// Explanation is supporting text, not part of the bold verdict line.
expect(result.querySelector('[class*="cutoffCheckHeadline"]')!.textContent)
.not.toMatch(/measurement error/);
expect(result).not.toHaveTextContent(/^\S+ away — inside/);
});
it('surfaces a postcode the geocoder cannot find', async () => {
mockGeocode.mockResolvedValue(null);
renderPanel();
await check('ZZ99 9ZZ');
expect(await screen.findByRole('alert')).toHaveTextContent(/could not find that postcode/);
});
it('recovers from a geocoder failure instead of leaving a stale verdict', async () => {
mockGeocode.mockResolvedValue(northOf(200));
renderPanel();
await check('SE23 3NA');
await screen.findByRole('status');
mockGeocode.mockRejectedValue(new Error('network'));
await check('SE23 3NB');
await waitFor(() => expect(screen.queryByRole('status')).not.toBeInTheDocument());
expect(screen.getByRole('alert')).toHaveTextContent(/Something went wrong/);
});
it('keeps the map behind a request until there is a reason to show it', async () => {
renderPanel();
expect(screen.queryByTestId('cutoff-map')).not.toBeInTheDocument();
mockGeocode.mockResolvedValue(northOf(200));
await check('SE23 3NA');
await screen.findByRole('status');
expect(screen.getByTestId('cutoff-map')).toBeInTheDocument();
});
it('can also show the map without a postcode, on request', () => {
renderPanel();
fireEvent.click(screen.getByRole('button', { name: /Show this distance on a map/ }));
expect(screen.getByTestId('cutoff-map')).toBeInTheDocument();
});
it('states its limits before it is used, not with the answer', () => {
renderPanel();
const caveat = screen.getByText(/Distance is the last criterion applied/);
expect(caveat).toBeInTheDocument();
expect(caveat).toHaveTextContent(/not a catchment boundary/);
expect(caveat).toHaveTextContent(/walking route/);
});
});
@@ -1,149 +0,0 @@
import { render, screen } from '@testing-library/react';
import { DestinationsSection } from '@/components/school/DestinationsSection';
import type { DestinationPhase } from '@/lib/types';
import type { DestinationCategory, DestinationStatus } from '@/lib/destinations';
const cell = (
category: DestinationCategory,
pupils: number | null,
status: DestinationStatus = 'published',
) => ({
category, pupils,
percentage: pupils === null ? null : (pupils / 180) * 100,
status,
});
const ALL_PUBLISHED = [
cell('school_sixth_form', 75), cell('sixth_form_college', 21),
cell('further_education', 55), cell('other_education', 6),
cell('apprenticeship', 8), cell('employment', 6),
cell('not_sustained', 5), cell('not_captured', 4),
];
const fullPhase: DestinationPhase = {
cohort_year: '2022/23',
groups: { all: { cohort: 180, categories: ALL_PUBLISHED } },
};
const suppressedPhase: DestinationPhase = {
cohort_year: '2022/23',
groups: {
all: {
cohort: 180,
categories: [
cell('school_sixth_form', 75), cell('sixth_form_college', null, 'suppressed'),
cell('further_education', 55), cell('other_education', 6),
cell('apprenticeship', 8), cell('employment', 6),
cell('not_sustained', 5), cell('not_captured', 4),
],
},
},
};
describe('DestinationsSection', () => {
it('dates its own cohort so it is not read as stale next to the GCSE section', () => {
render(<DestinationsSection destinations={fullPhase} />);
expect(screen.getByText(/2022\/23/)).toBeInTheDocument();
});
it('renders one bar segment per published category', () => {
const { container } = render(<DestinationsSection destinations={fullPhase} />);
expect(container.querySelectorAll('[data-destination-segment]')).toHaveLength(8);
});
it('renders NO bar at all when a category is withheld', () => {
const { container } = render(<DestinationsSection destinations={suppressedPhase} />);
// R1: a bar with a gap in it publishes the withheld figure by its width.
expect(container.querySelectorAll('[data-destination-segment]')).toHaveLength(0);
expect(screen.getAllByText(/withheld/i).length).toBeGreaterThan(0);
});
it('never states the remainder for a partially suppressed group', () => {
const { container } = render(<DestinationsSection destinations={suppressedPhase} />);
// 180 cohort - 159 published = 21, the withheld figure. It must appear nowhere.
expect(container.textContent).not.toMatch(/\b21\b/);
});
it('shows a card value for a group whose components are all published', () => {
render(<DestinationsSection destinations={fullPhase} />);
// academic route = 75 + 21 = 96 of 180 = 53%
expect(screen.getByText('53%')).toBeInTheDocument();
});
it('refuses a card value when one of its components is withheld', () => {
render(<DestinationsSection destinations={suppressedPhase} />);
// academic route needs sixth_form_college, which is suppressed.
expect(screen.getByText(/not published/i)).toBeInTheDocument();
expect(screen.queryByText('53%')).not.toBeInTheDocument();
});
it('never claims a pupil stayed at this school', () => {
const { container } = render(<DestinationsSection destinations={fullPhase} />);
// The published file reports destination TYPE, never destination institution.
expect(container.textContent).not.toMatch(/stayed on (here|at this school)/i);
});
it('renders nothing when no group carries categories', () => {
const empty: DestinationPhase = { cohort_year: '2022/23', groups: {} };
const { container } = render(<DestinationsSection destinations={empty} />);
expect(container.firstChild).toBeNull();
});
});
describe('the detail table keeps the three statuses apart', () => {
// 'suppressed' and 'not_applicable' are different claims, and the mart, the
// SQLAlchemy model and the serialiser all preserve the difference. The table
// used to key its Share column off `percentage === null`, which is true for
// both, so a category that simply does not apply was labelled "withheld" —
// while the Pupils column beside it rendered blank.
const mixedPhase: DestinationPhase = {
cohort_year: '2022/23',
groups: {
all: {
cohort: 180,
categories: [
cell('school_sixth_form', 75),
cell('sixth_form_college', null, 'suppressed'),
cell('further_education', null, 'suppressed'),
cell('apprenticeship', null, 'not_applicable'),
],
},
},
};
const rowFor = (container: HTMLElement, category: string) =>
Array.from(container.querySelectorAll('tbody tr'))
.find(tr => tr.textContent?.includes(category));
it('never labels a not-applicable category as withheld', () => {
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
const row = rowFor(container, 'Apprenticeship');
expect(row).toBeTruthy();
expect(row!.textContent).not.toMatch(/withheld/i);
});
it('labels a genuinely suppressed category as withheld in both columns', () => {
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
const row = rowFor(container, 'Sixth-form college');
expect(row).toBeTruthy();
expect(row!.querySelectorAll('td')).toHaveLength(2);
Array.from(row!.querySelectorAll('td')).forEach(td =>
expect(td.textContent).toMatch(/withheld/i));
});
it('the two columns of a row never disagree about what the row is', () => {
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
Array.from(container.querySelectorAll('tbody tr')).forEach(tr => {
const cells = Array.from(tr.querySelectorAll('td'))
.map(td => /withheld/i.test(td.textContent ?? ''));
expect(new Set(cells).size).toBe(1);
});
});
it('shows a published category its real figures', () => {
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
const row = rowFor(container, 'State-funded school sixth form');
expect(row!.textContent).toMatch(/75/);
expect(row!.textContent).toMatch(/42%/);
});
});
@@ -1,110 +0,0 @@
import { render, screen, fireEvent, waitFor } from '@testing-library/react';
import userEvent from '@testing-library/user-event';
import { FilterBar } from '@/components/FilterBar';
const push = jest.fn();
let searchParams = new URLSearchParams();
jest.mock('next/navigation', () => ({
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
usePathname: () => '/',
useSearchParams: () => searchParams,
}));
const FILTERS = {
local_authorities: [], school_types: [], years: [], phases: [],
genders: [], admissions_policies: [],
};
const realFetch = global.fetch;
beforeEach(() => {
global.fetch = jest.fn(async () => ({
ok: true,
json: async () => ({ suggestions: [{
urn: 100010, school_name: 'Brecknock Primary School',
local_authority: 'Camden', postcode: 'NW1 1AA',
phase: 'Primary', school_type: 'Community school' }] }),
})) as unknown as typeof fetch;
push.mockClear();
searchParams = new URLSearchParams();
});
afterEach(() => { global.fetch = realFetch; });
describe('FilterBar autosuggest', () => {
it('is a combobox only when the flag is on', () => {
const { rerender } = render(<FilterBar filters={FILTERS} autosuggest={false} />);
expect(screen.queryByRole('combobox')).not.toBeInTheDocument();
rerender(<FilterBar filters={FILTERS} autosuggest />);
expect(screen.getByRole('combobox')).toBeInTheDocument();
});
it('makes no request while the flag is off', async () => {
// Off means off: no listener, no fetch, no markup.
render(<FilterBar filters={FILTERS} autosuggest={false} />);
await userEvent.type(screen.getByPlaceholderText(/School name or postcode/i),
'brecknock');
expect(global.fetch).not.toHaveBeenCalled();
});
it('shows suggestions and navigates when one is chosen', async () => {
render(<FilterBar filters={FILTERS} autosuggest />);
await userEvent.type(screen.getByRole('combobox'), 'brecknock');
const option = await screen.findByRole('option', { name: /Brecknock/ });
await userEvent.click(option);
expect(push).toHaveBeenCalledWith(
expect.stringContaining('/school/100010'));
});
it('suppresses suggestions once the value is a postcode', async () => {
// The box takes a name OR a postcode; suggestions must get out of the way.
//
// fireEvent.change, not userEvent.type: typing sets "N", "NW", "NW1"... and
// "NW1" is not a postcode, so a request for it is correct behaviour. Only
// the settled value is the assertion, so set it in one go.
render(<FilterBar filters={FILTERS} autosuggest />);
fireEvent.change(screen.getByRole('combobox'), { target: { value: 'NW1 1AA' } });
await new Promise((r) => setTimeout(r, 300)); // past the 200ms debounce
expect(global.fetch).not.toHaveBeenCalled();
});
it('Enter with no active option still submits the free-text search', async () => {
// The existing behaviour is preserved, not replaced.
render(<FilterBar filters={FILTERS} autosuggest />);
const input = screen.getByRole('combobox');
await userEvent.type(input, 'brecknock{Enter}');
// updateURL pushes inside startTransition, so the call is not synchronous.
await waitFor(() => expect(push).toHaveBeenCalledWith(
expect.stringContaining('search=brecknock')));
});
});
describe('FilterBar autosuggest does not reopen over results', () => {
it('stays shut when the input arrives pre-filled from the URL', async () => {
/*
* The results-page bar renders with the search term already in the input.
* Opening on that would drop the dropdown on top of the results the search
* just produced — which is exactly what happened: the first result became
* unclickable, because the list sat over it and swallowed the pointer.
*
* Suggestions answer typing, not the presence of a value.
*/
searchParams = new URLSearchParams('search=brecknock');
render(<FilterBar filters={FILTERS} autosuggest />);
expect(screen.getByRole('combobox')).toHaveValue('brecknock');
await new Promise((r) => setTimeout(r, 300)); // past the 200ms debounce
expect(global.fetch).not.toHaveBeenCalled();
expect(screen.queryByRole('listbox')).not.toBeInTheDocument();
});
it('closes the dropdown when the search is submitted', async () => {
render(<FilterBar filters={FILTERS} autosuggest />);
const input = screen.getByRole('combobox');
await userEvent.type(input, 'brecknock');
expect(await screen.findByRole('listbox')).toBeInTheDocument();
await userEvent.type(input, '{Enter}');
await waitFor(() =>
expect(screen.queryByRole('listbox')).not.toBeInTheDocument());
});
});
@@ -1,47 +0,0 @@
/**
* The footer is the only navigational route to /about and /blog, so it is
* where a dark flag would otherwise leave a link into a 404.
*
* Both props default to false. A caller that forgets to pass them hides the
* links, which is the direction that cannot break a page — the same reasoning
* as backend/flags.py's "every flag defaults to False".
*/
import { render, screen } from '@testing-library/react';
import { Footer } from '@/components/Footer';
describe('footer feature links', () => {
it('links to both when both flags are on', () => {
render(<Footer aboutEnabled blogEnabled />);
expect(screen.getByRole('link', { name: /who's behind this/i }))
.toHaveAttribute('href', '/about');
expect(screen.getByRole('link', { name: /^blog$/i }))
.toHaveAttribute('href', '/blog');
});
it('omits the about link when that flag is dark', () => {
render(<Footer blogEnabled />);
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
expect(screen.getByRole('link', { name: /^blog$/i })).toBeInTheDocument();
});
it('omits the blog link when that flag is dark', () => {
render(<Footer aboutEnabled />);
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
expect(screen.getByRole('link', { name: /who's behind this/i }))
.toBeInTheDocument();
});
it('drops the whole section when both are dark, not an empty heading', () => {
// Shipping dark means the footer renders as it did before the feature
// existed, not as a section with its contents removed.
render(<Footer />);
expect(screen.queryByRole('heading', { name: /^about$/i })).toBeNull();
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
});
it('defaults to dark when a caller passes nothing', () => {
render(<Footer />);
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
});
});
@@ -1,92 +0,0 @@
/**
* The module that ends the stranding: before it, a school page's only anchor
* pointed at the school's own website, so ~27k pages sent authority off-site
* and none of it reached the location layer.
*/
import { render, screen } from '@testing-library/react';
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const cm15 = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('NearbyPlaces', () => {
it('links to every place the school belongs to', () => {
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
expect(screen.getByRole('link', { name: /Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood');
expect(screen.getByRole('link', { name: /Essex/ }))
.toHaveAttribute('href', '/schools/authority/essex');
expect(screen.getByRole('link', { name: /CM15/ }))
.toHaveAttribute('href', '/schools/near/cm15');
});
it('says how many schools each link leads to', () => {
// An anchor that states its destination's size is worth more to a reader
// and to a crawler than "see more".
render(<NearbyPlaces places={[brentwood]} />);
expect(screen.getByRole('link', { name: /37 schools in Brentwood/ }))
.toBeInTheDocument();
});
it('renders nothing at all when the school has no published places', () => {
// Not an empty heading. A school whose town and authority both fall below
// the threshold has nowhere to point, and the page should look as it did
// before the module existed.
const { container } = render(<NearbyPlaces places={[]} />);
expect(container).toBeEmptyDOMElement();
});
it('puts the narrowest place first, which is the most useful link', () => {
// The API orders widest-first for the breadcrumb; a reader on a school
// page wants its town before its county.
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood'))
.toBeLessThan(hrefs.indexOf('/schools/authority/essex'));
});
it('handles a singular count without saying "1 schools"', () => {
render(<NearbyPlaces places={[{ ...brentwood, count: 1 }]} />);
expect(screen.getByRole('link', { name: /1 school in Brentwood/ }))
.toBeInTheDocument();
});
it('links the phase page the school appears on', () => {
// "primary schools in brentwood" is the query these pages exist for.
render(<NearbyPlaces places={[{
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools in Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood/primary');
});
it('links both phase pages for an all-through school', () => {
render(<NearbyPlaces places={[{
...brentwood,
phases: [
{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' },
{ phase: 'secondary', count: 9, url: '/schools/brentwood/secondary' },
],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools/ })).toBeInTheDocument();
expect(screen.getByRole('link', { name: /9 secondary schools/ })).toBeInTheDocument();
});
it('keeps a phase link next to the place it belongs to', () => {
// Grouping matters: "22 primary schools in Brentwood" directly after
// "37 schools in Brentwood" reads as one place, not two unrelated links.
render(<NearbyPlaces places={[essex, {
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood/primary'))
.toBe(hrefs.indexOf('/schools/brentwood') + 1);
});
});
@@ -1,479 +0,0 @@
import { render, screen } from '@testing-library/react';
import { PlaceView } from '@/components/places/PlaceView';
import type { PlaceDetail } from '@/lib/places';
const detail: PlaceDetail = {
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 29,
parent_authority: 'Essex', phases: ['primary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', rwm_expected_pct: 82,
ofsted_grade: 1, phase: 'Primary' } as never,
{ urn: 2, school_name: 'Beta Primary', rwm_expected_pct: 44,
ofsted_grade: 3, phase: 'Primary' } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: null },
};
describe('PlaceView', () => {
it('leads with an H1 that matches how the place is searched', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getByRole('heading', { level: 1 }))
.toHaveTextContent(/primary schools in brentwood/i);
});
it('states the count so the page says something before the table', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getByText(/29 schools/i)).toBeInTheDocument();
});
it('compares the local average against England, which a list cannot', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('63');
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('61');
});
it('links every school in scope, which is what de-orphans them', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getAllByRole('link', { name: /Primary$/ })).toHaveLength(2);
});
it('links to the parent authority so the place sits in a hierarchy', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getByRole('link', { name: /Essex/i }))
.toHaveAttribute('href', '/schools/authority/essex');
});
it('shows the Ofsted distribution, not just a count of Outstanding', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.getByTestId('ofsted-distribution')).toBeInTheDocument();
});
it('links to neighbouring places so the page is not a dead end', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[{ kind: 'town', slug: 'romford', name: 'Romford', count: 40 }]} />);
expect(screen.getByRole('link', { name: /Romford/ }))
.toHaveAttribute('href', '/schools/romford');
});
it('says nothing about an average it does not have', () => {
render(<PlaceView detail={{ ...detail, averages:
{ rwm_expected_pct: null, attainment_8_score: null } }}
phase="primary" englandAverage={61} neighbours={[]} />);
expect(screen.queryByTestId('local-vs-england')).not.toBeInTheDocument();
});
});
describe('PlaceView structured data', () => {
function jsonLd() {
const { container } = render(<PlaceView detail={detail} phase="primary"
englandAverage={61} neighbours={[]} />);
const el = container.querySelector('script[type="application/ld+json"]');
return JSON.parse(el!.textContent!);
}
it('declares the page as a ranked list, not prose', () => {
const types = jsonLd()['@graph'].map((n: { '@type': string }) => n['@type']);
expect(types).toContain('ItemList');
expect(types).toContain('BreadcrumbList');
});
it('gives every listed school an absolute URL on the canonical host', () => {
const list = jsonLd()['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
expect(list.itemListElement).toHaveLength(2);
for (const item of list.itemListElement) {
expect(item.url).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\//);
}
});
});
describe('PlaceView phase variants', () => {
it('links the phase variants that exist', () => {
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: /Primary schools in Brentwood/i }))
.toHaveAttribute('href', '/schools/brentwood/primary');
});
it('links no variant for a phase below its own threshold', () => {
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
expect(screen.queryByRole('link', { name: /Secondary schools in Brentwood/i }))
.not.toBeInTheDocument();
});
it('does not link sideways from a variant page to itself', () => {
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.queryByRole('link', { name: /Primary schools in Brentwood/i }))
.not.toBeInTheDocument();
});
});
describe('PlaceView presentation', () => {
// /schools/brentwood shipped with 8 of 27 rows blank: an unphased page shows
// one primary-only measure for a list that also holds secondaries.
const mixed: PlaceDetail = {
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 4,
parent_authority: 'Essex', phases: ['primary', 'secondary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null } as never,
{ urn: 2, school_name: 'Beta High', phase: 'Secondary',
rwm_expected_pct: null, attainment_8_score: 47 } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: 45 },
};
it('gives each phase its own table rather than one column of blanks', () => {
render(<PlaceView detail={mixed} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('heading', { name: /^Primary schools/ })).toBeInTheDocument();
expect(screen.getByRole('heading', { name: /^Secondary schools/ })).toBeInTheDocument();
expect(screen.getByText('82%')).toBeInTheDocument();
expect(screen.getByText('47')).toBeInTheDocument();
});
it('names the measure in plain words, not jargon', () => {
// The first cut said "RWM expected", which appears nowhere else on the site.
render(<PlaceView detail={mixed} englandAverage={61} neighbours={[]} />);
expect(screen.getByText('Reading, writing & maths')).toBeInTheDocument();
expect(screen.getByText('Attainment 8')).toBeInTheDocument();
expect(screen.queryByText(/RWM expected/i)).not.toBeInTheDocument();
});
it('says a missing result is unpublished rather than showing a bare dash', () => {
const noResult: PlaceDetail = {
...mixed,
schools: [{ urn: 3, school_name: 'New Primary', phase: 'Primary',
rwm_expected_pct: null, attainment_8_score: null } as never],
};
render(<PlaceView detail={noResult} englandAverage={61} neighbours={[]} />);
expect(screen.getByText('Not published')).toBeInTheDocument();
});
it('styles school links to the site convention rather than browser default', () => {
const { container } = render(<PlaceView detail={mixed} englandAverage={61}
neighbours={[]} />);
const link = container.querySelector('a[href^="/school/"]');
expect(link?.className).toBeTruthy();
});
it('a phased page shows one table and no phase headings', () => {
render(<PlaceView detail={mixed} phase="primary" englandAverage={61}
neighbours={[]} />);
expect(screen.queryByRole('heading', { name: /^Secondary schools/ }))
.not.toBeInTheDocument();
});
});
describe('PlaceView table alignment', () => {
const aligned: PlaceDetail = {
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 2,
parent_authority: 'Essex', phases: ['primary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: null },
};
it('aligns the measure heading and its values with the same class', () => {
// They were aligned by two different selectors whose specificity did not
// match: `.table th:last-child` (0,2,1) won and went right, while `.num`
// (0,1,0) lost to `.table td` (0,1,1) and stayed left. Sharing one class
// is what makes them impossible to drift apart.
const { container } = render(<PlaceView detail={aligned} englandAverage={61}
neighbours={[]} />);
const th = container.querySelectorAll('th')[1];
const td = container.querySelectorAll('tbody td')[1];
expect(th.className).toBeTruthy();
expect(td.className).toBe(th.className);
});
it('leaves the school-name column unclassed so it takes the spare width', () => {
const { container } = render(<PlaceView detail={aligned} englandAverage={61}
neighbours={[]} />);
expect(container.querySelectorAll('th')[0].className).toBe('');
});
});
describe('PlaceView authorities', () => {
const straddling: PlaceDetail = {
place: { kind: 'outcode', slug: 'sw19', name: 'SW19', count: 33,
parent_authority: 'Merton', phases: ['primary'],
authorities: [
{ name: 'Merton', slug: 'merton', count: 26 },
{ name: 'Wandsworth', slug: 'wandsworth', count: 7 },
] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: null },
};
it('names every authority the place straddles, not just the largest', () => {
// SW19 is mostly Merton but partly Wandsworth. Naming one asserts
// something false about a quarter of outcodes.
render(<PlaceView detail={straddling} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: 'Merton' }))
.toHaveAttribute('href', '/schools/authority/merton');
expect(screen.getByRole('link', { name: 'Wandsworth' }))
.toHaveAttribute('href', '/schools/authority/wandsworth');
});
it('joins them readably rather than as a bare list', () => {
// Asserted on the summary line's whole text: a loose /and/ matcher also
// hits "Wandsworth".
const { container } = render(<PlaceView detail={straddling}
englandAverage={61} neighbours={[]} />);
const summary = container.querySelector('header p');
expect(summary?.textContent).toContain('Merton and Wandsworth');
});
it('falls back to the single parent when the field is absent', () => {
// A cached API response predating the authorities field must not blank
// the line entirely.
const legacy = { ...straddling,
place: { ...straddling.place, authorities: undefined } };
render(<PlaceView detail={legacy} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: 'Merton' })).toBeInTheDocument();
});
});
describe('PlaceView list ordering', () => {
const detail3: PlaceDetail = {
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 2,
parent_authority: 'Essex', phases: ['primary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 40, attainment_8_score: null } as never,
{ urn: 2, school_name: 'Beta Primary', phase: 'Primary',
rwm_expected_pct: 90, attainment_8_score: null } as never,
],
averages: { rwm_expected_pct: 65, attainment_8_score: null },
};
it('renders schools in the order the API sent them, not by score', () => {
// The API sorts alphabetically now; the component must not re-sort.
render(<PlaceView detail={detail3} englandAverage={61} neighbours={[]} />);
const links = screen.getAllByRole('link', { name: /Primary$/ });
expect(links.map((l) => l.textContent))
.toEqual(['Alpha Primary', 'Beta Primary']);
});
it('declares the list as ascending rather than implying a ranking', () => {
// An ItemList carrying `position` reads as a ranking unless it says
// otherwise, and the table is A-Z.
const { container } = render(<PlaceView detail={detail3} englandAverage={61}
neighbours={[]} />);
const ld = JSON.parse(
container.querySelector('script[type="application/ld+json"]')!.textContent!);
const list = ld['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
expect(list.itemListOrder).toBe('https://schema.org/ItemListOrderAscending');
});
});
describe('PlaceView phase links', () => {
const authority: PlaceDetail = {
place: { kind: 'authority', slug: 'barnet', name: 'Barnet', count: 156,
parent_authority: null, phases: ['primary', 'secondary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: null },
};
it('keeps an authority phase link in the authority namespace', () => {
// The link was built as `/schools/${slug}/${phase}` for every kind, so an
// authority page pointed into the town namespace. For 87 of 151
// authorities that 404'd; for the other 64 it silently landed on the town
// page of the same name — a different set of schools, and exactly the
// duplicate the two namespaces exist to prevent. Barnet is one of the 64.
render(<PlaceView detail={authority} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: /^Primary schools in Barnet$/ }))
.toHaveAttribute('href', '/schools/authority/barnet/primary');
expect(screen.getByRole('link', { name: /^Secondary schools in Barnet$/ }))
.toHaveAttribute('href', '/schools/authority/barnet/secondary');
});
it('still uses the bare namespace for a town', () => {
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: /^Primary schools in Brentwood$/ }))
.toHaveAttribute('href', '/schools/brentwood/primary');
});
it('offers no phase link when the place publishes none', () => {
// Outcodes are the case: no phase route exists for them, so the registry
// reports no phases and the nav does not render.
const outcode = { ...detail,
place: { ...detail.place, kind: 'outcode', slug: 'cm13', name: 'CM13',
phases: [] } };
render(<PlaceView detail={outcode} englandAverage={61} neighbours={[]} />);
expect(screen.queryByRole('navigation', { name: 'By phase' }))
.not.toBeInTheDocument();
});
});
describe('PlaceView unlinkable authorities', () => {
const withUnpublished: PlaceDetail = {
place: { kind: 'outcode', slug: 'tr21', name: 'TR21', count: 8,
parent_authority: 'Cornwall', phases: [],
authorities: [
{ name: 'Cornwall', slug: 'cornwall', count: 6 },
{ name: 'Isles Of Scilly', slug: null, count: 2 },
] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: null },
};
it('names an authority with no page without linking it', () => {
// City of London and the Isles of Scilly hold fewer schools than a page
// needs. Saying where the place is stays right; linking there would 404.
const { container } = render(<PlaceView detail={withUnpublished}
englandAverage={61} neighbours={[]} />);
expect(screen.getByRole('link', { name: 'Cornwall' })).toBeInTheDocument();
expect(screen.queryByRole('link', { name: 'Isles Of Scilly' }))
.not.toBeInTheDocument();
expect(container.querySelector('header p')?.textContent)
.toContain('Isles Of Scilly');
});
});
describe('PlaceView school attributes', () => {
/*
* The table shipped with one column of scores, which answers "how did they
* do" and nothing about whether the school is one a family could use. Age
* range, faith, nursery and constituency are the four facts a parent
* filters on before they look at a number at all.
*/
const withAttributes: PlaceDetail = {
place: { kind: 'town', slug: 'chelmsford', name: 'Chelmsford', count: 3,
parent_authority: 'Essex', phases: ['primary', 'secondary'] },
schools: [
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
rwm_expected_pct: 82, attainment_8_score: null,
age_range: '4-11', religious_denomination: 'Church of England',
nursery_provision: true,
parliamentary_constituency: 'Chelmsford' } as never,
{ urn: 2, school_name: 'Beta High', phase: 'Secondary',
rwm_expected_pct: null, attainment_8_score: 47,
age_range: '11-16', religious_denomination: 'Does not apply',
nursery_provision: false,
parliamentary_constituency: 'Witham' } as never,
],
averages: { rwm_expected_pct: 63, attainment_8_score: 45 },
};
function headings(container: HTMLElement, table = 0): string[] {
return Array.from(container.querySelectorAll('table')[table]
.querySelectorAll('thead th')).map((th) => th.textContent ?? '');
}
it('heads a primary table with all four attributes', () => {
const { container } = render(<PlaceView detail={withAttributes}
englandAverage={61} neighbours={[]} />);
expect(headings(container)).toEqual([
'School', 'Reading, writing & maths',
'Ages', 'Religious character', 'Nursery', 'Constituency',
]);
});
it('omits nursery from a secondary table, where it does not apply', () => {
const { container } = render(<PlaceView detail={withAttributes}
englandAverage={61} neighbours={[]} />);
expect(headings(container, 1)).toEqual([
'School', 'Attainment 8', 'Ages', 'Religious character', 'Constituency',
]);
});
it('keeps the measure beside the school name, where a phone can see it', () => {
// Six columns overflow a phone and .tableWrap turns that into a swipe.
// With the measure last, the one number the page exists for is the one
// scrolled off the screen.
const { container } = render(<PlaceView detail={withAttributes}
phase="primary" englandAverage={61} neighbours={[]} />);
expect(headings(container)[1]).toBe('Reading, writing & maths');
});
it('shows the age range without repeating the column heading', () => {
render(<PlaceView detail={withAttributes} englandAverage={61}
neighbours={[]} />);
expect(screen.getByText('4–11')).toBeInTheDocument();
expect(screen.queryByText('Ages 4–11')).not.toBeInTheDocument();
});
it('names the faith of a faith school', () => {
render(<PlaceView detail={withAttributes} englandAverage={61}
neighbours={[]} />);
expect(screen.getByText('Church of England')).toBeInTheDocument();
});
it('reads "Does not apply" as no religious character, not as a value', () => {
// GIAS spells the absence of a faith as "Does not apply", which is a
// database answer rather than an English one. The school page already
// suppresses it; the two must not disagree about the same school.
const { container } = render(<PlaceView detail={withAttributes}
englandAverage={61} neighbours={[]} />);
const secondary = container.querySelectorAll('table')[1]
.querySelectorAll('tbody td');
expect(secondary[3].textContent).toBe('—');
expect(screen.queryByText(/Does not apply/)).not.toBeInTheDocument();
});
it('marks a nursery as such and a school without one as not', () => {
const { container } = render(<PlaceView detail={withAttributes}
englandAverage={61} neighbours={[]} />);
const cells = container.querySelectorAll('table')[0]
.querySelectorAll('tbody td');
expect(cells[4].textContent).toBe('Yes');
});
it('names the constituency of each school', () => {
render(<PlaceView detail={withAttributes} englandAverage={61}
neighbours={[]} />);
expect(screen.getByText('Chelmsford', { selector: 'td' })).toBeInTheDocument();
expect(screen.getByText('Witham', { selector: 'td' })).toBeInTheDocument();
});
it('dashes an attribute the data does not carry', () => {
// nursery_provision and parliamentary_constituency are absent from marts
// the pipeline has not rebuilt, and the API degrades them to null rather
// than failing. A row must survive that.
const bare: PlaceDetail = {
...withAttributes,
schools: [{ urn: 3, school_name: 'Gamma Primary', phase: 'Primary',
rwm_expected_pct: 70 } as never],
};
const { container } = render(<PlaceView detail={bare} phase="primary"
englandAverage={61} neighbours={[]} />);
const cells = Array.from(container.querySelectorAll('tbody td'))
.map((td) => td.textContent);
expect(cells.slice(2)).toEqual(['—', '—', '—', '—']);
});
it('gives an all-through school its nursery under primary only', () => {
// All-through schools render in both groups. Nursery belongs to the
// primary reading of the same school, not the secondary one.
const allThrough: PlaceDetail = {
...withAttributes,
schools: [{ urn: 4, school_name: 'Delta Academy', phase: 'All-through',
rwm_expected_pct: 66, attainment_8_score: 51,
age_range: '4-18', religious_denomination: 'None',
nursery_provision: true,
parliamentary_constituency: 'Chelmsford' } as never],
};
const { container } = render(<PlaceView detail={allThrough}
englandAverage={61} neighbours={[]} />);
const tables = container.querySelectorAll('table');
expect(tables[0].textContent).toContain('Yes');
expect(tables[1].textContent).not.toContain('Yes');
});
});
@@ -1,44 +0,0 @@
import { render, screen } from '@testing-library/react';
import { Post16DestinationsSection } from '@/components/school/Post16DestinationsSection';
import type { DestinationPhase } from '@/lib/types';
const phase: DestinationPhase = {
cohort_year: '2022/23',
groups: {
all: {
cohort: 96,
categories: [
{ category: 'higher_education', pupils: 56, percentage: 58.3, status: 'published' },
{ category: 'further_education', pupils: 12, percentage: 12.5, status: 'published' },
{ category: 'apprenticeship', pupils: 9, percentage: 9.4, status: 'published' },
{ category: 'employment', pupils: 13, percentage: 13.5, status: 'published' },
{ category: 'not_sustained', pupils: 6, percentage: 6.3, status: 'published' },
],
},
},
};
describe('Post16DestinationsSection', () => {
it('names the Year 13 cohort, not Year 11', () => {
const { container } = render(<Post16DestinationsSection destinations={phase} />);
expect(container.textContent).toMatch(/Year 13/);
expect(container.textContent).not.toMatch(/Year 11/);
});
it('reports higher education destinations', () => {
render(<Post16DestinationsSection destinations={phase} />);
expect(screen.getByText(/UK higher education/i)).toBeInTheDocument();
});
it('uses its own anchor so the nav does not collide with After Year 11', () => {
const { container } = render(<Post16DestinationsSection destinations={phase} />);
expect(container.querySelector('#post16-destinations')).toBeTruthy();
expect(container.querySelector('#destinations')).toBeNull();
});
it('renders nothing when no group carries categories', () => {
const empty: DestinationPhase = { cohort_year: '2022/23', groups: {} };
const { container } = render(<Post16DestinationsSection destinations={empty} />);
expect(container.firstChild).toBeNull();
});
});
@@ -1,38 +0,0 @@
/**
* The trail has to be written by something, and it has to be written on every
* route — not only the ones that happen to track an event.
*/
import { render } from '@testing-library/react';
const recordVisitedPath = jest.fn();
let pathname = '/schools/brentwood';
jest.mock('next/navigation', () => ({ usePathname: () => pathname }));
jest.mock('@/lib/analytics', () => ({
recordVisitedPath: (p: string) => recordVisitedPath(p),
}));
// eslint-disable-next-line @typescript-eslint/no-var-requires
const { RouteTrail } = require('@/components/RouteTrail');
describe('RouteTrail', () => {
beforeEach(() => recordVisitedPath.mockClear());
it('records the page it is mounted on', () => {
render(<RouteTrail />);
expect(recordVisitedPath).toHaveBeenCalledWith('/schools/brentwood');
});
it('records each new route as the user moves through the app', () => {
const { rerender } = render(<RouteTrail />);
pathname = '/school/115429-brentwood-school';
rerender(<RouteTrail />);
expect(recordVisitedPath).toHaveBeenLastCalledWith(
'/school/115429-brentwood-school');
});
it('renders nothing, so it can sit anywhere in the layout', () => {
const { container } = render(<RouteTrail />);
expect(container).toBeEmptyDOMElement();
});
});
@@ -1,50 +0,0 @@
import { render, screen } from '@testing-library/react';
import { SuggestList, suggestOptionId } from '@/components/SuggestList';
const ROWS = [
{ urn: 1, school_name: "St Mary's Primary", local_authority: 'Camden',
postcode: 'NW1 1AA', phase: 'Primary', school_type: 'Voluntary aided school' },
{ urn: 2, school_name: "St Mary's Primary", local_authority: 'Barnet',
postcode: 'EN5 2AA', phase: 'Primary', school_type: 'Community school' },
];
describe('SuggestList', () => {
it('is a listbox of options', () => {
render(<SuggestList id="s" suggestions={ROWS} activeIndex={-1}
onPick={() => {}} onHover={() => {}} />);
expect(screen.getByRole('listbox')).toBeInTheDocument();
expect(screen.getAllByRole('option')).toHaveLength(2);
});
it('shows the local authority, which is what tells two schools apart', () => {
// Both rows are "St Mary's Primary". Without the authority the list is
// unusable for exactly the query autosuggest exists to serve.
render(<SuggestList id="s" suggestions={ROWS} activeIndex={-1}
onPick={() => {}} onHover={() => {}} />);
expect(screen.getByText('Camden')).toBeInTheDocument();
expect(screen.getByText('Barnet')).toBeInTheDocument();
});
it('marks only the active option selected', () => {
render(<SuggestList id="s" suggestions={ROWS} activeIndex={1}
onPick={() => {}} onHover={() => {}} />);
const options = screen.getAllByRole('option');
expect(options[0]).toHaveAttribute('aria-selected', 'false');
expect(options[1]).toHaveAttribute('aria-selected', 'true');
});
it('gives each option the id the input will point at', () => {
// aria-activedescendant on the input has to name a real element id, or
// a screen reader announces nothing as the user arrows through.
render(<SuggestList id="s" suggestions={ROWS} activeIndex={0}
onPick={() => {}} onHover={() => {}} />);
expect(screen.getAllByRole('option')[0]).toHaveAttribute(
'id', suggestOptionId('s', 0));
});
it('renders nothing when there is nothing to suggest', () => {
const { container } = render(<SuggestList id="s" suggestions={[]}
activeIndex={-1} onPick={() => {}} onHover={() => {}} />);
expect(container).toBeEmptyDOMElement();
});
});
@@ -1,45 +0,0 @@
import { render } from '@testing-library/react';
import { TrackPlaceView } from '@/components/places/TrackPlaceView';
const trackMock = jest.fn();
jest.mock('@/lib/analytics', () => ({
track: (...args: unknown[]) => trackMock(...args),
getNavigationSource: () => 'search',
}));
describe('TrackPlaceView', () => {
beforeEach(() => trackMock.mockClear());
it('reports which kind of location page was viewed', () => {
/*
* `kind` is the reason this event exists. Whether to keep investing in the
* location layer turns on which *sort* of page earns engagement — towns,
* authorities or postcode districts — and a bare pageview cannot say,
* because all four families share the /schools/ prefix.
*/
render(<TrackPlaceView kind="authority" slug="kent" count={412} />);
expect(trackMock).toHaveBeenCalledWith('place_viewed', {
kind: 'authority', slug: 'kent', phase: 'all',
school_count: 412, from: 'search',
});
});
it('names the phase when the page is a phase variant', () => {
render(<TrackPlaceView kind="town" slug="brentwood" count={29} phase="primary" />);
expect(trackMock).toHaveBeenCalledWith('place_viewed',
expect.objectContaining({ phase: 'primary' }));
});
it('fires once, not once per render', () => {
const { rerender } = render(
<TrackPlaceView kind="town" slug="brentwood" count={29} />);
rerender(<TrackPlaceView kind="town" slug="brentwood" count={29} />);
expect(trackMock).toHaveBeenCalledTimes(1);
});
it('renders nothing', () => {
const { container } = render(
<TrackPlaceView kind="town" slug="brentwood" count={29} />);
expect(container).toBeEmptyDOMElement();
});
});
@@ -1,192 +0,0 @@
import fs from 'fs';
import path from 'path';
/**
* Guards against light-theme-only CSS.
*
* The site themes entirely through tokens redefined under
* `@media (prefers-color-scheme: dark)`. A hardcoded colour therefore does not
* fail loudly — it renders perfectly in the theme it was written for and
* quietly wrongly in the other, which nobody sees unless they happen to be in
* dark mode when they look.
*
* Both rules below are drawn from real defects in SchoolHeroMap.module.css,
* found by eye rather than by any test:
*
* - the map's fade to the header ramped through hardcoded white and landed on
* `var(--bg-card)`. Invisible in light; a bright band across the full width
* of a near-black card in dark.
* - the controls floating over the map paired a hardcoded white background
* with `color: var(--text-primary)`, which resolves to #E9EEF0 in dark —
* near-white text on a near-white button.
*/
const COMPONENTS = path.join(__dirname, '..', '..', 'components');
function stylesheets(dir: string): string[] {
return fs.readdirSync(dir, { withFileTypes: true }).flatMap((entry) => {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) return stylesheets(full);
return entry.name.endsWith('.module.css') ? [full] : [];
});
}
/** Innermost `selector { body }` pairs. Nested at-rules never match as rules,
* because their body contains braces.
*
* Comments are stripped before matching rather than after, so that the whole
* selector survives. Taking only its last line — which is what stripping a
* leading comment used to require — silently discarded every selector in a
* grouped rule but the final one, and a safety guard that cannot see half its
* input fails open. */
function rules(css: string): Array<{ selector: string; body: string }> {
const bare = css.replace(/\/\*[\s\S]*?\*\//g, '');
return Array.from(bare.matchAll(/([^{}]+)\{([^{}]*)\}/g), (m) => ({
selector: m[1].trim().replace(/\s*\n\s*/g, ' '),
body: m[2],
}));
}
const HARDCODED_WHITE_BG = /background[^;]*(?:255,\s*255,\s*255|#fff\b|#ffffff\b)/i;
const THEMED_COLOR = /(?:^|[^-])color:\s*var\(--/;
const files = stylesheets(COMPONENTS);
/** Component sources, for the third-party-surface rule below. */
function sources(dir: string): string[] {
return fs.readdirSync(dir, { withFileTypes: true }).flatMap((entry) => {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) return sources(full);
return entry.name.endsWith('.tsx') ? [full] : [];
});
}
describe('dark-theme safety', () => {
it('finds stylesheets to check', () => {
expect(files.length).toBeGreaterThan(0);
});
it('never pairs a hardcoded white background with a themed text colour', () => {
const offenders = files.flatMap((file) =>
rules(fs.readFileSync(file, 'utf8'))
.filter((r) => HARDCODED_WHITE_BG.test(r.body) && THEMED_COLOR.test(r.body))
.map((r) => `${path.relative(COMPONENTS, file)} ${r.selector}`));
// Either the surface follows the theme and so should the text, or it does
// not and the text must be literal too. Mixing them is how near-white text
// ends up on a near-white button.
expect(offenders).toEqual([]);
});
it('never fades to a themed colour through a hardcoded one', () => {
const offenders = files.flatMap((file) =>
rules(fs.readFileSync(file, 'utf8'))
.filter((r) => /linear-gradient/.test(r.body)
&& /var\(--bg-(card|primary|secondary)\)/.test(r.body)
&& /255,\s*255,\s*255|#fff\b/i.test(r.body))
.map((r) => `${path.relative(COMPONENTS, file)} ${r.selector}`));
// A gradient that lands on a token has to be made of that token, or the
// ramp and its destination disagree in one theme. Use the matching
// `--*-rgb` token for the transparent stops.
expect(offenders).toEqual([]);
});
});
/**
* The same defect one stylesheet further out.
*
* The rules above scan our own CSS modules. They cannot see a surface painted
* by a third-party sheet: leaflet.css hardcodes `background: white` on
* `.leaflet-popup-content-wrapper` and `.leaflet-popup-tip`, and
* LeafletMapInner builds its popup as an HTML string with inline
* `color: var(--text-primary)`. Neither half lives in a .module.css, so the
* module scan passed while dark mode rendered #E9EEF0 on #FFFFFF — 1.17:1,
* with the school name and the headline figure effectively invisible.
*
* globals.css already pulls the rest of Leaflet's chrome onto the tokens (the
* attribution bar, the zoom controls) for exactly this reason. The popup was
* simply missed.
*/
describe('third-party surfaces under themed text', () => {
const GLOBALS = path.join(__dirname, '..', '..', 'app', '(frontend)', 'globals.css');
/** Leaflet surfaces our own code writes token-coloured text onto. */
const LEAFLET_POPUP_SURFACES = [
'.leaflet-popup-content-wrapper',
'.leaflet-popup-tip',
];
it('still finds a component painting themed text into a Leaflet popup', () => {
// Guards the rule below against passing vacuously if the popups are ever
// rewritten as React components rather than HTML strings.
const themed = sources(COMPONENTS).filter((file) => {
const src = fs.readFileSync(file, 'utf8');
return /bindPopup\(/.test(src) && /color:var\(--|color: var\(--/.test(src);
});
expect(themed.length).toBeGreaterThan(0);
});
it('themes the Leaflet popup surface, because the text on it is themed', () => {
const globals = rules(fs.readFileSync(GLOBALS, 'utf8'));
const unthemed = LEAFLET_POPUP_SURFACES.filter((surface) => {
const rule = globals.find((r) => r.selector.includes(surface));
return !rule || !/background[^;]*var\(--/.test(rule.body);
});
// Leaflet's white is not a colour this site owns. Either the surface
// follows the theme or the text on it must be literal — and the text is
// already themed.
expect(unthemed).toEqual([]);
});
it('never puts a literal white label on a themed fill', () => {
/*
* The mirror image of the module-CSS rule above, and the half of the popup
* that theming the card does not reach. "View Details" is
* `background:var(--status-above);color:white`; --status-above is #36743F
* in light but #7FCB8A in dark, so the label went from 5.63:1 to 1.94:1.
*
* --text-inverse is the token for ink on a saturated fill — #FFFFFF in
* light, #111A20 in dark — and the popup's Ofsted badge already uses it.
*/
const offenders = sources(COMPONENTS).flatMap((file) => {
const src = fs.readFileSync(file, 'utf8');
return Array.from(
src.matchAll(/background:\s*var\(--[^;"']*;[^"']*?color:\s*(white|#fff\b|#ffffff\b)/gi),
() => path.relative(COMPONENTS, file));
});
expect(offenders).toEqual([]);
});
});
/**
* Destination measures add the first new colour family since the palette was
* set. The tokens have to exist in both blocks or the section renders one
* theme's fills on the other theme's ground — the exact failure the suite
* above exists to catch, but for tokens rather than literals.
*/
describe('destination tokens', () => {
const css = fs.readFileSync(
path.join(__dirname, '..', '..', 'app', '(frontend)', 'globals.css'), 'utf8');
const TOKENS = [
'--dest-sixthform', '--dest-sfcollege', '--dest-fecollege',
'--dest-apprentice', '--dest-employment', '--dest-none', '--dest-none-hatch',
];
const DARK_AT = css.indexOf('@media (prefers-color-scheme: dark)');
it('defines every destination token in the light palette', () => {
const light = css.slice(0, DARK_AT);
expect(TOKENS.filter((t) => !light.includes(`${t}:`))).toEqual([]);
});
it('redefines every destination token for dark', () => {
const dark = css.slice(DARK_AT);
expect(TOKENS.filter((t) => !dark.includes(`${t}:`))).toEqual([]);
});
});
@@ -1,108 +0,0 @@
import fs from 'fs';
import path from 'path';
/**
* The hero search and the results filter bar are the same component in two
* costumes. `.filterBar` is the card — background, border, shadow, padding —
* and `.heroMode` strips all of it so the search sits directly on the hero
* panel.
*
* Both selectors have specificity (0,1,0), so **source order decides**, and
* `.heroMode` only wins because it is declared immediately after. Any later
* bare `.filterBar` rule — which in practice means one inside a media query —
* silently wins instead, and the hero grows a card's padding back.
*
* That is exactly what happened: `@media (max-width: 768px) { .filterBar {
* padding: 0.875rem } }` re-added 14px in hero mode, indenting the search box,
* the hint and the location link 14px past the headline above them and costing
* the search field 28px of width on a 390px screen. The two rules directly
* below it in the same block were correctly written as
* `.filterBar:not(.heroMode)`; this one was missed, and nothing caught it
* because the result is a plausible-looking layout rather than a broken one.
*/
const CSS = path.join(__dirname, '..', '..', 'components', 'FilterBar.module.css');
/** Properties `.heroMode` resets. A later bare `.filterBar` rule setting any
* of these puts the card back on the hero. */
const RESET_BY_HERO_MODE = [
'background', 'border', 'border-radius', 'box-shadow', 'padding',
];
/**
* Comments are stripped before anything is parsed.
*
* A `{` or `}` inside a comment would otherwise desynchronise the brace walk
* below and the rule regex alike, and the selector text captured for each rule
* would carry the preceding comment along with it.
*/
function withoutComments(css: string): string {
return css.replace(/\/\*[\s\S]*?\*\//g, '');
}
/**
* The individual selectors in a rule's prelude.
*
* Split on commas, because a selector list is a list: `.filterBar, .other { }`
* applies to `.filterBar` just as surely as `.filterBar { }` does, and an
* earlier version of this guard compared the whole prelude against the literal
* string '.filterBar' — so writing the regression as a comma list, or across
* two lines, would have walked straight past it.
*/
function selectorsOf(prelude: string): string[] {
return prelude.split(',').map((sel) => sel.trim().replace(/\s+/g, ' '))
.filter(Boolean);
}
function mediaQueryBodies(css: string): string[] {
const bodies: string[] = [];
const re = /@media[^{]*\{/g;
let m: RegExpExecArray | null;
while ((m = re.exec(css)) !== null) {
// Walk braces from the opening one to find this at-rule's whole body.
let depth = 1;
let i = m.index + m[0].length;
const start = i;
while (i < css.length && depth > 0) {
if (css[i] === '{') depth++;
else if (css[i] === '}') depth--;
i++;
}
bodies.push(css.slice(start, i - 1));
}
return bodies;
}
describe('FilterBar hero-mode scoping', () => {
const css = withoutComments(fs.readFileSync(CSS, 'utf8'));
it('confirms heroMode still resets the card, which is what makes this matter', () => {
const hero = css.match(/\.heroMode\s*\{([^}]*)\}/);
expect(hero).not.toBeNull();
expect(hero![1]).toMatch(/padding:\s*0/);
});
it('never re-applies card styling to the hero from inside a media query', () => {
const offenders: string[] = [];
for (const body of mediaQueryBodies(css)) {
for (const rule of body.matchAll(/([^{}]+)\{([^{}]*)\}/g)) {
// Only a *bare* .filterBar is dangerous, and it is dangerous wherever
// it appears in a selector list. Scoped variants
// (`.filterBar:not(.heroMode)`) and descendants are fine.
const selectors = selectorsOf(rule[1]);
if (!selectors.includes('.filterBar')) continue;
for (const prop of RESET_BY_HERO_MODE) {
if (new RegExp(`(^|[;\\s])${prop}\\s*:`).test(rule[2])) {
offenders.push(`${rule[1].trim()} sets ${prop}`);
}
}
}
}
// Fix by scoping the rule as `.filterBar:not(.heroMode)`, the way the
// neighbouring rules in the same block already are.
expect(offenders).toEqual([]);
});
});
@@ -1,202 +0,0 @@
/**
* Last distance offered, rendered on both detail templates.
*
* The figure is the one number on these pages that a parent may act on — it is
* easy to read as "we live inside the catchment, we will get a place". These
* tests pin the things that stop it being read that way: the year is always
* present, the caveat is always present, and the blanket "not available"
* sentence appears only when it is actually true.
*/
import { screen, fireEvent } from '@testing-library/react';
import { renderSchoolDetail, renderSecondarySchoolDetail } from '../support/renderSchoolDetail';
import { primaryFixture, secondaryFixture } from '../support/schoolFixtures';
import type { SchoolAdmissionDistance } from '@/lib/types';
const cutoff = (over: Partial<SchoolAdmissionDistance> = {}): SchoolAdmissionDistance => ({
year: 2025,
distance_m: 500,
route_count: 1,
la_name: 'Camden',
distance_unit_raw: 'miles',
...over,
});
describe('primary detail page', () => {
it('shows the figure with the year it belongs to', () => {
renderSchoolDetail({ ...primaryFixture, admissionDistance: cutoff({ distance_m: 772.49, year: 2024 }) });
expect(screen.getByText('0.48 miles')).toBeInTheDocument();
expect(screen.getByText(/Last distance offered/)).toHaveTextContent('September 2024');
});
it('keeps the figure and its metric support readable as two numbers', () => {
// They are flex children with a CSS gap and nothing between them in the
// text layer, which read as "0.48 miles770 m" to a screen reader and to any
// text matcher. Cheap to lose again, so pinned.
renderSchoolDetail({ ...primaryFixture, admissionDistance: cutoff({ distance_m: 772.49 }) });
const tile = document.querySelector('[class*="admissionsTileDistance"]')!;
expect(tile.textContent).toMatch(/0\.48 miles\s+770 m/);
expect(tile.textContent).not.toMatch(/miles\d/);
});
it('never shows the figure without saying it is not a catchment', () => {
renderSchoolDetail({ ...primaryFixture, admissionDistance: cutoff() });
expect(screen.getByText(/not a fixed catchment/)).toBeInTheDocument();
expect(screen.getByText(/moves every year/)).toBeInTheDocument();
});
it('flags that a banded school\'s figure is the widest of several routes', () => {
renderSchoolDetail({ ...primaryFixture, admissionDistance: cutoff({ route_count: 4 }) });
expect(screen.getByText(/4 admission routes/)).toBeInTheDocument();
});
it('renders nothing distance-related when the LA publishes none', () => {
renderSchoolDetail({ ...primaryFixture, admissionDistance: null });
expect(screen.queryByText(/Last distance offered/)).not.toBeInTheDocument();
expect(screen.queryByText(/not a fixed catchment/)).not.toBeInTheDocument();
});
it('carries the figure even with no EES admissions row', () => {
// The two sources are independent; this school has a cut-off and no
// admissions figures. Before this feature the section did not render at all.
renderSchoolDetail({
...primaryFixture,
admissions: null,
admissionsHistory: [],
admissionDistance: cutoff({ distance_m: 1421.05 }),
});
expect(screen.getByText('0.88 miles')).toBeInTheDocument();
});
});
describe('secondary detail page', () => {
it('shows the figure with the year it belongs to', () => {
renderSecondarySchoolDetail({ ...secondaryFixture, admissionDistance: cutoff({ distance_m: 3472.96 }) });
expect(screen.getByText('2.16 miles')).toBeInTheDocument();
expect(screen.getByText(/Last distance offered/)).toHaveTextContent('September 2025');
});
it('drops the blanket "not available" line once a distance exists', () => {
renderSecondarySchoolDetail({ ...secondaryFixture, admissionDistance: cutoff() });
expect(screen.queryByText(/has not published a cut-off distance/)).not.toBeInTheDocument();
expect(screen.getByText(/not a fixed catchment/)).toBeInTheDocument();
});
it('names the authority that would hold the data when there is none', () => {
// The old copy asserted "Historical distance cut-off data is not available
// for this school" on every secondary page, including the ones whose
// council does publish it.
renderSecondarySchoolDetail({ ...secondaryFixture, admissionDistance: null });
expect(screen.getByText(/has not published a cut-off distance/)).toBeInTheDocument();
});
it('makes no claim about publication when the feature is switched off', () => {
// Absent, not null. The API omits the key entirely while the
// admission_distance flag is off, and "Islington has not published a
// cut-off distance" is then a statement about us, not about Islington —
// false wherever the authority does publish one.
renderSecondarySchoolDetail({ ...secondaryFixture, admissionDistance: undefined });
expect(screen.queryByText(/has not published a cut-off distance/)).not.toBeInTheDocument();
expect(screen.queryByText(/Contact the admissions authority/)).not.toBeInTheDocument();
});
});
// ── The Distance section ───────────────────────────────────────────────
describe('Distance section', () => {
it('appears for a school with a figure and coordinates', () => {
const { container } = renderSchoolDetail({
...primaryFixture,
schoolInfo: { ...primaryFixture.schoolInfo, latitude: 51.5, longitude: -0.12 },
admissionDistance: cutoff({ distance_m: 700, year: 2026 }),
});
expect(container.querySelector('#distance')).toBeInTheDocument();
expect(screen.getByText('How far away are you?')).toBeInTheDocument();
});
it('stays away when the school has no coordinates to measure from', () => {
const { container } = renderSchoolDetail({
...primaryFixture,
schoolInfo: { ...primaryFixture.schoolInfo, latitude: null, longitude: null },
admissionDistance: cutoff({ distance_m: 700, year: 2026 }),
});
expect(container.querySelector('#distance')).not.toBeInTheDocument();
});
it('stays away when no figure has been published', () => {
const { container } = renderSchoolDetail({
...primaryFixture,
schoolInfo: { ...primaryFixture.schoolInfo, latitude: 51.5, longitude: -0.12 },
admissionDistance: null,
});
expect(container.querySelector('#distance')).not.toBeInTheDocument();
});
it('shows no year-by-year record — that is held back as a paid feature', () => {
// The page must not leak the history through a table, a chart or a strip of
// per-year verdicts. The API no longer sends it either; this guards the
// render side so a future component cannot quietly put it back.
renderSchoolDetail({
...primaryFixture,
schoolInfo: { ...primaryFixture.schoolInfo, latitude: 51.5, longitude: -0.12 },
admissionDistance: cutoff({ distance_m: 700, year: 2026 }),
});
// Scoped to the section: the page has other tables (the history section's).
const section = document.querySelector('#distance')!;
expect(section.querySelector('table')).toBeNull();
expect(screen.queryByText(/Last distance offered, by year/)).not.toBeInTheDocument();
expect(screen.queryByText(/Not published/)).not.toBeInTheDocument();
expect(screen.queryByText(/too few to read as a trend/)).not.toBeInTheDocument();
});
});
describe('secondary Distance section', () => {
it('renders on the secondary template too', () => {
const { container } = renderSecondarySchoolDetail({
...secondaryFixture,
schoolInfo: { ...secondaryFixture.schoolInfo, latitude: 51.5, longitude: -0.12 },
admissionDistance: cutoff({ distance_m: 3472.96, year: 2026 }),
});
expect(container.querySelector('#distance')).toBeInTheDocument();
});
it('explains a selective school by how it admits rather than as missing data', () => {
renderSecondarySchoolDetail({
...secondaryFixture,
schoolInfo: { ...secondaryFixture.schoolInfo, admissions_policy: 'Selective' },
admissionDistance: null,
});
expect(screen.getByText(/ranked by the entrance test/)).toBeInTheDocument();
expect(screen.queryByText(/has not published a cut-off distance/)).not.toBeInTheDocument();
});
it('reads a consistently undersubscribed school as good news', () => {
renderSecondarySchoolDetail({
...secondaryFixture,
admissionDistance: null,
admissionsHistory: [
{ year: 2022, oversubscribed: false },
{ year: 2023, oversubscribed: false },
{ year: 2024, oversubscribed: false },
],
});
expect(screen.getByText(/has not needed a distance cut-off/)).toBeInTheDocument();
});
});
@@ -1,73 +0,0 @@
import { renderHook, act, waitFor } from '@testing-library/react';
import { useSchoolSuggest } from '@/hooks/useSchoolSuggest';
const realFetch = global.fetch;
function mockFetch(rows: unknown[], delayMs = 0) {
global.fetch = jest.fn(async (_url: unknown, init?: { signal?: AbortSignal }) => {
if (delayMs) {
await new Promise((resolve, reject) => {
const t = setTimeout(resolve, delayMs);
init?.signal?.addEventListener('abort', () => {
clearTimeout(t);
reject(Object.assign(new Error('aborted'), { name: 'AbortError' }));
});
});
}
return { ok: true, json: async () => ({ suggestions: rows }) };
}) as unknown as typeof fetch;
}
const ROW = {
urn: 1, school_name: 'Brecknock Primary School', local_authority: 'Camden',
postcode: 'NW1 1AA', phase: 'Primary', school_type: 'Community school',
};
describe('useSchoolSuggest', () => {
beforeEach(() => { jest.useFakeTimers(); });
afterEach(() => { jest.useRealTimers(); global.fetch = realFetch; });
it('does not fetch below the minimum query length', () => {
mockFetch([ROW]);
renderHook(() => useSchoolSuggest('b', true));
act(() => { jest.advanceTimersByTime(500); });
expect(global.fetch).not.toHaveBeenCalled();
});
it('does not fetch at all when disabled', () => {
// The flag being off must mean no request, not a hidden dropdown.
mockFetch([ROW]);
renderHook(() => useSchoolSuggest('brecknock', false));
act(() => { jest.advanceTimersByTime(500); });
expect(global.fetch).not.toHaveBeenCalled();
});
it('debounces rather than firing per keystroke', () => {
mockFetch([ROW]);
const { rerender } = renderHook(
({ q }) => useSchoolSuggest(q, true), { initialProps: { q: 'br' } });
rerender({ q: 'bre' });
rerender({ q: 'brec' });
act(() => { jest.advanceTimersByTime(199); });
expect(global.fetch).not.toHaveBeenCalled();
act(() => { jest.advanceTimersByTime(2); });
expect(global.fetch).toHaveBeenCalledTimes(1);
});
it('opens with results once they arrive', async () => {
mockFetch([ROW]);
const { result } = renderHook(() => useSchoolSuggest('brecknock', true));
act(() => { jest.advanceTimersByTime(200); });
await waitFor(() => expect(result.current.suggestions).toHaveLength(1));
expect(result.current.open).toBe(true);
});
it('close() hides the list without clearing the query', async () => {
mockFetch([ROW]);
const { result } = renderHook(() => useSchoolSuggest('brecknock', true));
act(() => { jest.advanceTimersByTime(200); });
await waitFor(() => expect(result.current.open).toBe(true));
act(() => { result.current.close(); });
expect(result.current.open).toBe(false);
});
});
-158
View File
@@ -1,158 +0,0 @@
import { getNavigationSource } from '@/lib/analytics';
/** jsdom's document.referrer is read-only; redefining it is the way in. */
function referrer(url: string) {
Object.defineProperty(document, 'referrer', { value: url, configurable: true });
}
const ORIGIN = 'http://localhost';
describe('getNavigationSource', () => {
afterEach(() => referrer(''));
it('attributes a visit from a location page to the place layer', () => {
/*
* The one this was added for.
*
* W2 published ~3,900 location pages whose entire purpose is to funnel
* search traffic onto school pages. Before this case existed they fell
* through to 'direct' — so the location layer's contribution was not
* merely missing from the funnel, it was being counted in the bucket you
* read as "typed the URL". The measurement that decides whether W2 worked
* was confidently reporting the wrong answer.
*/
referrer(`${ORIGIN}/schools/barnet`);
expect(getNavigationSource()).toBe('place');
});
it.each([
['/schools/authority/kent', 'authority'],
['/schools/near/sw11', 'outcode'],
['/schools/brentwood/primary', 'phase variant'],
])('covers %s (%s)', (path) => {
referrer(`${ORIGIN}${path}`);
expect(getNavigationSource()).toBe('place');
});
it('still calls a school page "detail", one character away', () => {
// /school/ and /schools/ differ by one letter and mean different things.
// A prefix test written in the wrong order silently merges them.
referrer(`${ORIGIN}/school/100010-brecknock-primary-school`);
expect(getNavigationSource()).toBe('detail');
});
it.each([
['/', 'search'],
['/rankings', 'rankings'],
['/compare?urns=1,2', 'compare'],
])('leaves %s attributed as %s', (path, expected) => {
referrer(`${ORIGIN}${path}`);
expect(getNavigationSource()).toBe(expected);
});
it('treats an external referrer as direct', () => {
// Umami records the real referrer on the pageview; this field is only
// about internal navigation.
referrer('https://www.google.com/search?q=schools+in+barnet');
expect(getNavigationSource()).toBe('direct');
});
it('treats no referrer as direct', () => {
referrer('');
expect(getNavigationSource()).toBe('direct');
});
});
/*
* The defect the existing suite could not see.
*
* Every test above sets document.referrer, which the browser writes only when
* a *document* loads. Every internal navigation in this app is an App Router
* soft navigation — history.pushState, no new document — so document.referrer
* keeps naming whatever opened the tab for the whole session. Verified on
* staging: /schools/brentwood → click a school → URL changes to /school/…
* and document.referrer is still "".
*
* So `from` reported 'direct' for essentially every in-app journey, and the
* suite passed because it only ever exercised the full-page-load path.
*/
function freshAnalytics() {
let mod!: typeof import('@/lib/analytics');
jest.isolateModules(() => {
mod = require('@/lib/analytics');
});
return mod;
}
function at(path: string) {
window.history.pushState({}, '', path);
}
describe('getNavigationSource across a soft navigation', () => {
afterEach(() => {
referrer('');
at('/');
});
it('attributes a school view to the place page the user actually came from', () => {
const { recordVisitedPath, getNavigationSource: source } = freshAnalytics();
at('/schools/brentwood');
recordVisitedPath('/schools/brentwood');
at('/school/115429-brentwood-school');
recordVisitedPath('/school/115429-brentwood-school');
expect(source()).toBe('place');
});
it('does not depend on whether the new path was recorded first', () => {
// The trail is written by a layout-level effect and read by a page-level
// one. React orders those by tree position, which is not a contract worth
// resting a measurement on, so the answer must be the same either way.
const { recordVisitedPath, getNavigationSource: source } = freshAnalytics();
recordVisitedPath('/rankings');
at('/school/115429-brentwood-school');
expect(source()).toBe('rankings');
});
it('names the previous page, not the current one, when both are schools', () => {
const { recordVisitedPath, getNavigationSource: source } = freshAnalytics();
at('/school/100010-brecknock-primary-school');
recordVisitedPath('/school/100010-brecknock-primary-school');
at('/school/115429-brentwood-school');
recordVisitedPath('/school/115429-brentwood-school');
expect(source()).toBe('detail');
});
it('looks past a return visit to the page the user came back from', () => {
const { recordVisitedPath, getNavigationSource: source } = freshAnalytics();
for (const p of ['/schools/brentwood', '/school/115429-brentwood-school',
'/schools/brentwood']) {
at(p);
recordVisitedPath(p);
}
expect(source()).toBe('detail');
});
it('falls back to the referrer on a real document load, where it is true', () => {
// A fresh module is a fresh document: nothing has been recorded, and
// document.referrer is meaningful again.
const { getNavigationSource: source } = freshAnalytics();
at('/school/115429-brentwood-school');
referrer(`${ORIGIN}/schools/barnet`);
expect(source()).toBe('place');
});
it('still reads an arrival from outside as direct', () => {
const { recordVisitedPath, getNavigationSource: source } = freshAnalytics();
at('/schools/brentwood');
recordVisitedPath('/schools/brentwood');
referrer('https://www.google.com/search?q=schools+in+brentwood');
expect(source()).toBe('direct');
});
});
@@ -1,87 +0,0 @@
import {
canAggregate, aggregateCells,
canRenderBar, toBarSegments, CARD_GROUPS,
type DestinationCell, type DestinationGroup, type DestinationCategory,
} from '@/lib/destinations';
const pub = (category: DestinationCategory, pupils: number, cohort: number): DestinationCell => ({
category, pupils, percentage: (pupils / cohort) * 100, status: 'published',
});
const sup = (category: DestinationCategory): DestinationCell => ({
category, pupils: null, percentage: null, status: 'suppressed',
});
const fullGroup = (): DestinationGroup => ({
cohort: 180,
cells: [
pub('school_sixth_form', 75, 180), pub('sixth_form_college', 21, 180),
pub('further_education', 55, 180), pub('other_education', 6, 180),
pub('apprenticeship', 8, 180), pub('employment', 6, 180),
pub('not_sustained', 5, 180), pub('not_captured', 4, 180),
],
});
describe('canAggregate — R2, computing from components', () => {
it('allows a sum when every component is published', () => {
expect(canAggregate([pub('apprenticeship', 8, 180), pub('employment', 6, 180)])).toBe(true);
});
it('refuses a sum when any component is suppressed', () => {
expect(canAggregate([pub('apprenticeship', 8, 180), sup('employment')])).toBe(false);
});
it('refuses a sum when every component is suppressed', () => {
expect(canAggregate([sup('apprenticeship'), sup('employment')])).toBe(false);
});
});
describe('aggregateCells', () => {
it('sums published cells and derives a percentage from the cohort', () => {
expect(aggregateCells([pub('apprenticeship', 8, 180), pub('employment', 6, 180)], 180))
.toEqual({ pupils: 14, percentage: (14 / 180) * 100 });
});
it('returns null rather than a partial sum when a component is suppressed', () => {
expect(aggregateCells([pub('apprenticeship', 8, 180), sup('employment')], 180)).toBeNull();
});
});
describe('canRenderBar — R1', () => {
it('allows a bar when the whole group is published', () => {
expect(canRenderBar(fullGroup())).toBe(true);
});
it('refuses a bar when a single category is suppressed', () => {
const g = fullGroup();
g.cells[1] = sup('sixth_form_college');
expect(canRenderBar(g)).toBe(false);
});
});
describe('toBarSegments', () => {
it('derives widths from counts, not from rounded percentages', () => {
const segs = toBarSegments(fullGroup());
expect(segs).toHaveLength(8);
expect(segs[0].widthPct).toBeCloseTo((75 / 180) * 100, 10);
expect(segs.reduce((a, s) => a + s.widthPct, 0)).toBeCloseTo(100, 6);
});
it('throws rather than silently leaving a gap when the group is suppressed', () => {
const g = fullGroup();
g.cells[1] = sup('sixth_form_college');
expect(() => toBarSegments(g)).toThrow(/suppressed/i);
});
});
describe('CARD_GROUPS', () => {
it('partitions every destination category exactly once, plus the absence', () => {
const grouped = Object.values(CARD_GROUPS).flat();
expect(new Set(grouped).size).toBe(grouped.length);
expect(grouped).toEqual(expect.arrayContaining([
'school_sixth_form', 'sixth_form_college', 'further_education',
'other_education', 'apprenticeship', 'employment',
]));
expect(grouped).not.toContain('not_sustained');
expect(grouped).not.toContain('not_captured');
});
});
-58
View File
@@ -1,58 +0,0 @@
import { getFlags, FLAGS_REVALIDATE } from '@/lib/flags';
// jsdom provides no global fetch, so there is nothing for jest.spyOn to attach
// to — assign it and restore the original afterwards. This is the first test
// here to mock fetch; later ones should follow this shape.
const realFetch = global.fetch;
function mockFetch(impl: () => Promise<unknown>) {
global.fetch = jest.fn(impl) as unknown as typeof fetch;
}
describe('getFlags', () => {
afterEach(() => { global.fetch = realFetch; });
it('returns the flags the API reports', async () => {
mockFetch(async () => ({
ok: true,
json: async () => ({ admission_distance: true }),
}));
await expect(getFlags()).resolves.toEqual({ admission_distance: true });
});
it('returns no flags rather than throwing when the API is down', async () => {
// A page that cannot read flags must render everything dark, not 500.
// Fail-closed is the same direction as the backend's default.
mockFetch(async () => { throw new Error('ECONNREFUSED'); });
await expect(getFlags()).resolves.toEqual({});
});
it('returns no flags rather than throwing on a non-200', async () => {
mockFetch(async () => ({ ok: false, status: 503 }));
await expect(getFlags()).resolves.toEqual({});
});
/*
* Reading a flag pins the calling route's ISR floor: Next uses the LOWEST
* revalidate among a route's fetches for the whole route. That is why the
* revalidate is an argument rather than the constant.
*
* Every SEO route here declares `revalidate = 604800`. A gate that read
* flags at the 300s default would drop the whole school and place corpus
* from a weekly cache to a 5-minute one, which is a large origin-load
* regression to pay for a feature flag.
*/
it('reads at the 300s floor by default', async () => {
mockFetch(async () => ({ ok: true, json: async () => ({}) }));
await getFlags();
expect((global.fetch as jest.Mock).mock.calls[0][1])
.toEqual({ next: { revalidate: FLAGS_REVALIDATE } });
});
it('lets a caller pass its own route floor instead', async () => {
mockFetch(async () => ({ ok: true, json: async () => ({}) }));
await getFlags(604800);
expect((global.fetch as jest.Mock).mock.calls[0][1])
.toEqual({ next: { revalidate: 604800 } });
});
});
@@ -1,182 +0,0 @@
/**
* Last distance offered — formatting and the caveats attached to the figure.
*
* The assertions about the route note and the year are not cosmetic. A cut-off
* shown without its year, or a banded school's widest cut-off shown as if it
* were the only one, tells a parent something false about their chances of a
* place — so both are pinned here rather than left to the component.
*/
import { formatCutoffDistance, formatMiles, formatEntryYear } from '@/lib/utils';
import {
describeCutoff, describeCutoffAbsence, compareToCutoff, CUTOFF_UNCERTAINTY_M,
} from '@/components/school/lastDistanceOffered';
import type { SchoolAdmissionDistance } from '@/lib/types';
const distance = (over: Partial<SchoolAdmissionDistance> = {}): SchoolAdmissionDistance => ({
year: 2025,
distance_m: 500,
route_count: 1,
la_name: 'Camden',
distance_unit_raw: 'miles',
...over,
});
describe('formatCutoffDistance', () => {
it('leads with miles, the unit councils publish in', () => {
expect(formatCutoffDistance(500)).toEqual({ primary: '0.31 miles', secondary: '500 m' });
expect(formatCutoffDistance(1609.344)).toEqual({ primary: '1.00 miles', secondary: '1.6 km' });
});
it('switches to kilometres for the support figure above a kilometre', () => {
expect(formatCutoffDistance(3472.96)!.secondary).toBe('3.5 km');
});
it('stays in miles at short range, where it used to swap to metres', () => {
// The swap made a single number easier to read and a comparison harder:
// "69 m away — inside the cut-off of 0.17 miles" asked the reader to
// convert between units to check a claim we had already made for them.
expect(formatCutoffDistance(27)).toEqual({ primary: '0.02 miles', secondary: '30 m' });
expect(formatCutoffDistance(69)!.primary).toMatch(/miles$/);
});
it('describes a distance too short for two decimal places', () => {
// Rather than a flat "0.00 miles", which reads as no distance at all.
expect(formatMiles(5)).toBe('under 0.01 miles');
expect(formatMiles(0)).toBe('under 0.01 miles');
expect(formatMiles(20)).toBe('0.01 miles');
});
it('returns null rather than a zero cut-off', () => {
// 0.0 miles appears in the source where a school filled on a higher
// criterion. Rendered as "0.00 miles" it would read as the opposite.
expect(formatCutoffDistance(0)).toBeNull();
expect(formatCutoffDistance(null)).toBeNull();
expect(formatCutoffDistance(undefined)).toBeNull();
expect(formatCutoffDistance(Number.NaN)).toBeNull();
});
});
describe('formatEntryYear', () => {
it('names the intake, not the academic year', () => {
// formatAcademicYear would render 2025 as "2025/26", which reads as a
// school year rather than the September a child started.
expect(formatEntryYear(2025)).toBe('September 2025');
expect(formatEntryYear(null)).toBe('');
});
});
describe('describeCutoff', () => {
it('always carries the entry year alongside the figure', () => {
const d = describeCutoff(distance({ distance_m: 772.49, year: 2024 }));
expect(d).not.toBeNull();
expect(d!.primary).toBe('0.48 miles');
expect(d!.entryYear).toBe('September 2024');
});
it('says nothing about routes for a school with one', () => {
expect(describeCutoff(distance({ route_count: 1 }))!.routeNote).toBeNull();
expect(describeCutoff(distance({ route_count: null }))!.routeNote).toBeNull();
});
it('warns that a banded school\'s figure is the widest of several', () => {
const note = describeCutoff(distance({ route_count: 4 }))!.routeNote;
expect(note).toContain('4 admission routes');
expect(note).toContain('shorter cut-off');
});
it('is null when there is nothing publishable', () => {
expect(describeCutoff(null)).toBeNull();
expect(describeCutoff(undefined)).toBeNull();
expect(describeCutoff(distance({ distance_m: null }))).toBeNull();
});
});
// ── "Would we have got in?" ────────────────────────────────────────────
describe('compareToCutoff', () => {
it('never states the two figures in different units', () => {
/*
* The reported defect: "69 m away — inside the September 2026 cut-off of
* 0.17 miles". Both numbers are correct and the sentence is still useless,
* because checking it means converting one of them.
*
* Swept across the range where the old formatter switched units, so a
* future readability tweak to one figure cannot reintroduce the mismatch
* in the other.
*/
const mixed: string[] = [];
for (const homeM of [0, 5, 27, 69, 99, 100, 260, 800, 1609, 5000]) {
for (const cutoffM of [30, 69, 100, 270, 1000, 3500]) {
const { headline } = compareToCutoff(homeM, cutoffM, 2026);
const hasMetres = /\d\s?m\b/.test(headline);
const milesCount = (headline.match(/miles/g) ?? []).length;
// Two figures, both in miles, and no metric reading anywhere near them.
if (hasMetres || milesCount !== 2) {
mixed.push(`home=${homeM}m cutoff=${cutoffM}m -> ${headline}`);
}
}
}
expect(mixed).toEqual([]);
});
it('reads back the reported case in one unit', () => {
expect(compareToCutoff(69, 270, 2026).headline)
.toBe('0.04 miles away — inside the September 2026 cut-off of 0.17 miles.');
});
it('calls a clearly nearer home inside, and names the year', () => {
const r = compareToCutoff(300, 800, 2026);
expect(r.verdict).toBe('inside');
expect(r.headline).toContain('September 2026');
});
it('calls a clearly further home beyond', () => {
expect(compareToCutoff(4000, 800, 2026).verdict).toBe('outside');
});
it('refuses to call a result inside the measurement error, either way', () => {
// A postcode centroid covers several addresses, so a margin this fine is
// noise. With one published year there is no other year to fall back on,
// which makes this band the only thing standing between a parent and a
// place they do not have.
expect(compareToCutoff(800 - CUTOFF_UNCERTAINTY_M / 2, 800, 2026).verdict).toBe('too-close');
expect(compareToCutoff(800 + CUTOFF_UNCERTAINTY_M / 2, 800, 2026).verdict).toBe('too-close');
expect(compareToCutoff(800, 800, 2026).detail).toContain('measurement error');
// The explanation is not welded to the headline, so it does not run at
// headline weight in the result block.
expect(compareToCutoff(800, 800, 2026).headline).not.toContain('measurement error');
expect(compareToCutoff(300, 800, 2026).detail).toBeNull();
});
it('treats the band as exclusive at its edge', () => {
// Exactly on the boundary is still too close; one metre past it is not.
expect(compareToCutoff(800 - CUTOFF_UNCERTAINTY_M, 800, 2026).verdict).toBe('too-close');
expect(compareToCutoff(800 - CUTOFF_UNCERTAINTY_M - 1, 800, 2026).verdict).toBe('inside');
});
});
describe('describeCutoffAbsence', () => {
it('explains a selective school by how it admits, not as missing data', () => {
const s = describeCutoffAbsence({ localAuthority: 'Kent', admissionsPolicy: 'Selective' });
expect(s).toContain('entrance test');
expect(s).not.toContain('has not published');
});
it('reads a consistently undersubscribed school as good news', () => {
const s = describeCutoffAbsence({
localAuthority: 'Camden',
admissionsHistory: [
{ year: 2022, oversubscribed: false },
{ year: 2023, oversubscribed: false },
{ year: 2024, oversubscribed: false },
],
});
expect(s).toContain('has not needed a distance cut-off');
});
it('otherwise names the authority that would hold the figure', () => {
expect(describeCutoffAbsence({ localAuthority: 'Camden' }))
.toContain('Camden has not published');
});
});
@@ -1,68 +0,0 @@
/**
* School pages had no BreadcrumbList and no links into the location layer.
* Both are fixed by the same data — the `places` array the API now returns —
* so they are tested together.
*/
import { schoolBreadcrumbJsonLd } from '@/lib/jsonld';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const outcode = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('school breadcrumbs', () => {
it('reads home to authority to town to school', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
expect(ld['@type']).toBe('BreadcrumbList');
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Brentwood', 'Brentwood School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3, 4]);
});
it('skips a level the school has no published place for', () => {
// A school whose town falls below the publish threshold has no town page.
// The trail closes over the gap rather than linking to a 404.
const ld = schoolBreadcrumbJsonLd({
name: 'Lone School', url: '/school/1-lone-school', places: [essex],
});
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Lone School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3]);
});
it('omits outcodes, which are not a place a breadcrumb reads through', () => {
// CM15 is a useful link in the module but nonsense in a trail: nobody
// navigates Essex → CM15 → school.
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood, outcode],
});
expect(JSON.stringify(ld)).not.toContain('cm15');
});
it('still produces a valid trail when the school has no places at all', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Orphan School', url: '/school/2-orphan-school', places: [],
});
expect(ld.itemListElement.map((i) => i.name)).toEqual(['schoolcompare', 'Orphan School']);
});
it('uses absolute urls, as every other entity on the site does', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
for (const item of ld.itemListElement) {
expect(item.item).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\//);
}
// The root is the homepage: there is no /schools index page to link to.
expect(ld.itemListElement[0].item).toBe('https://www.schoolcompare.co.uk/');
});
});
@@ -1,83 +0,0 @@
import { computeSecondaryFlags, buildSecondaryNavItems } from '@/lib/schoolSections';
import type { School, SchoolDestinations } from '@/lib/types';
const schoolInfo = {
urn: 137083, school_name: 'Northbrook Academy', phase: 'Secondary',
has_sixth_form: true,
} as unknown as School;
const base = { schoolInfo, yearlyData: [], deprivation: null, finance: null };
const phase = (categories = 1) => ({
cohort_year: '2022/23',
groups: {
all: {
cohort: 180,
categories: Array.from({ length: categories }, () => ({
category: 'school_sixth_form' as const,
pupils: 75, percentage: 41.7, status: 'published' as const,
})),
},
},
});
const ks4Only: SchoolDestinations = { ks4: phase(), ks5: null };
const both: SchoolDestinations = { ks4: phase(), ks5: phase() };
describe('computeSecondaryFlags — destinations', () => {
it('flags KS4 destinations when the block carries categories', () => {
const flags = computeSecondaryFlags({ ...base, destinations: ks4Only });
expect(flags.hasKs4Destinations).toBe(true);
expect(flags.hasKs5Destinations).toBe(false);
});
it('flags both phases when both are present', () => {
const flags = computeSecondaryFlags({ ...base, destinations: both });
expect(flags.hasKs4Destinations).toBe(true);
expect(flags.hasKs5Destinations).toBe(true);
});
it('flags neither when the block is absent', () => {
const flags = computeSecondaryFlags({ ...base, destinations: null });
expect(flags.hasKs4Destinations).toBe(false);
expect(flags.hasKs5Destinations).toBe(false);
});
it('does not flag a phase whose groups carry no categories', () => {
const empty: SchoolDestinations = {
ks4: { cohort_year: '2022/23', groups: {} }, ks5: null,
};
expect(computeSecondaryFlags({ ...base, destinations: empty }).hasKs4Destinations)
.toBe(false);
});
it('does not flag a phase whose only group has an empty category list', () => {
const empty: SchoolDestinations = { ks4: phase(0), ks5: null };
expect(computeSecondaryFlags({ ...base, destinations: empty }).hasKs4Destinations)
.toBe(false);
});
});
describe('buildSecondaryNavItems — destinations', () => {
const navInput = {
ofsted: null, admissions: null, admissionDistance: null,
hasLocation: false, yearlyDataLength: 0,
};
it('adds both entries, after GCSEs', () => {
const flags = computeSecondaryFlags({ ...base, destinations: both });
const ids = buildSecondaryNavItems({ ...flags, hasResults: true }, navInput)
.map(i => i.id);
expect(ids).toContain('destinations');
expect(ids).toContain('post16-destinations');
expect(ids.indexOf('destinations')).toBeGreaterThan(ids.indexOf('gcse'));
expect(ids.indexOf('post16-destinations')).toBe(ids.indexOf('destinations') + 1);
});
it('adds no entry for a phase that will not render — the nav must not link to a missing anchor', () => {
const flags = computeSecondaryFlags({ ...base, destinations: null });
const ids = buildSecondaryNavItems(flags, navInput).map(i => i.id);
expect(ids).not.toContain('destinations');
expect(ids).not.toContain('post16-destinations');
});
});
@@ -60,7 +60,7 @@ describe('buildNavItems', () => {
it('omits sections with no data', () => {
const flags = computeSchoolFlags(specialFixture);
const ids = buildNavItems(flags, {
ofsted: null, admissions: null, admissionDistance: null, yearlyDataLength: 1,
ofsted: null, admissions: null, yearlyDataLength: 1,
}).map((n) => n.id);
expect(ids).not.toContain('ofsted');
@@ -74,7 +74,6 @@ describe('buildNavItems', () => {
return buildNavItems(flags, {
ofsted: fixture.ofsted,
admissions: fixture.admissions,
admissionDistance: null,
yearlyDataLength: fixture.yearlyData.length,
}).find((n) => n.id === 'results')?.label;
};
@@ -84,27 +83,11 @@ describe('buildNavItems', () => {
expect(label(allThroughFixture)).toBe('Results');
});
it('opens the admissions entry for a cut-off distance with no EES admissions', () => {
// 3% of the schools that render have one source and not the other. The nav
// condition and the section's render condition have to agree, or the sticky
// nav links to an anchor that was never rendered.
const flags = computeSchoolFlags(specialFixture);
const ids = buildNavItems(flags, {
ofsted: null,
admissions: null,
admissionDistance: { year: 2025, distance_m: 500, route_count: 1, la_name: 'Camden', distance_unit_raw: 'miles' },
yearlyDataLength: 1,
}).map((n) => n.id);
expect(ids).toContain('admissions');
});
it('keeps the engagement-led ordering', () => {
const flags = computeSchoolFlags(primaryFixture);
const ids = buildNavItems(flags, {
ofsted: primaryFixture.ofsted,
admissions: primaryFixture.admissions,
admissionDistance: null,
yearlyDataLength: primaryFixture.yearlyData.length,
}).map((n) => n.id);
@@ -144,29 +127,16 @@ describe('buildSecondaryNavItems', () => {
const ids = buildSecondaryNavItems(flags, {
ofsted: secondaryFixture.ofsted,
admissions: secondaryFixture.admissions,
admissionDistance: null,
yearlyDataLength: secondaryFixture.yearlyData.length,
}).map((n) => n.id);
expect(ids).toEqual(['ofsted', 'gcse', 'admissions', 'history', 'wellbeing', 'finances']);
});
it('opens the admissions entry for a cut-off distance alone', () => {
const flags = computeSecondaryFlags(secondaryFixture);
const ids = buildSecondaryNavItems(flags, {
ofsted: null,
admissions: null,
admissionDistance: { year: 2025, distance_m: 2400, route_count: 4, la_name: 'Islington', distance_unit_raw: 'miles' },
yearlyDataLength: 1,
}).map((n) => n.id);
expect(ids).toContain('admissions');
});
it('gates History on more than one year, unlike the primary page', () => {
const flags = computeSecondaryFlags(secondaryFixture);
const ids = buildSecondaryNavItems(flags, {
ofsted: null, admissions: null, admissionDistance: null, yearlyDataLength: 1,
ofsted: null, admissions: null, yearlyDataLength: 1,
}).map((n) => n.id);
expect(ids).not.toContain('history');
-27
View File
@@ -1,27 +0,0 @@
import { SITE_URL, absoluteUrl } from '@/lib/site';
describe('SITE_URL', () => {
it('is the www host, which is the one that serves a 200', () => {
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
// is a wasted signal, so every absolute URL we emit must already be www.
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
});
it('has no trailing slash, so joins never double up', () => {
expect(SITE_URL.endsWith('/')).toBe(false);
});
});
describe('absoluteUrl', () => {
it('joins a rooted path', () => {
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('joins a path missing its leading slash', () => {
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
});
it('maps the site root to a bare trailing slash', () => {
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
});
});
-26
View File
@@ -13,8 +13,6 @@ import {
metricKind,
shortName,
computeYBounds,
formatAgeRange,
formatAgeSpan,
} from '@/lib/utils';
describe('formatPercentage', () => {
@@ -322,27 +320,3 @@ describe('shortName', () => {
expect(shortName('A'.repeat(30), 10)).toBe('AAAAAAAAA…');
});
});
describe('formatAgeSpan', () => {
it('normalises a hyphenated range to an en dash, without a label', () => {
// The place table carries "Ages" in the column heading, so repeating it
// in every cell is noise. formatAgeRange keeps the label for the contexts
// that have no heading to hang it on.
expect(formatAgeSpan('4-11')).toBe('4–11');
});
it('leaves a range it does not recognise alone rather than mangling it', () => {
expect(formatAgeSpan('3-19 (SEN)')).toBe('3-19 (SEN)');
});
it('returns an empty string for a missing range', () => {
expect(formatAgeSpan(null)).toBe('');
expect(formatAgeSpan(undefined)).toBe('');
});
});
describe('formatAgeRange', () => {
it('keeps its label, so the two helpers stay distinguishable', () => {
expect(formatAgeRange('4-11')).toBe('Ages 4–11');
});
});
@@ -1,70 +0,0 @@
/**
* Payload is ESM-only and next/jest will not transform it, so the collections
* cannot be imported and their sanitised config inspected here (see
* lib/payloadRoutes.ts for the full reasoning). These assert the source of the
* collection definitions instead — enough to catch the settings whose loss is
* silent, and cheap. Behaviour is proved by the e2e journeys against staging.
*/
import fs from 'fs';
import path from 'path';
const read = (file: string) =>
fs.readFileSync(path.join(__dirname, '..', '..', 'collections', file), 'utf8');
const POSTS = read('Posts.ts');
const MEDIA = read('Media.ts');
const CONFIG = fs.readFileSync(
path.join(__dirname, '..', '..', 'payload.config.ts'),
'utf8',
);
describe('posts collection', () => {
it('supports drafts, so saving is not publishing', () => {
expect(POSTS).toMatch(/drafts:\s*true/);
});
it('has a unique, indexed slug for stable URLs', () => {
const slugField = POSTS.slice(POSTS.indexOf("name: 'slug'"));
expect(slugField).toMatch(/unique:\s*true/);
expect(slugField).toMatch(/index:\s*true/);
});
it('hides drafts from anonymous readers at the access layer', () => {
// Payload's docs are explicit: "The `draft` argument alone does not
// restrict documents with _status: 'draft' from being returned by the
// API." The blog pages' where-clause is not enforcement — a direct GET
// /cms-api/posts would return unpublished drafts to anyone. Access
// control returning a query constraint is the only thing that stops it.
expect(POSTS).toMatch(/_status:\s*\{\s*equals:\s*'published'\s*\}/);
expect(POSTS).toMatch(/if\s*\(req\.user\)\s*return true/);
});
it('revalidates the post page when a post changes or is deleted', () => {
// /blog/[slug] is ISR — generated on first request and cached — so an edit
// to an already-published post would otherwise not appear until the
// revalidate window expired, up to an hour of a writer concluding that
// saving is broken. The index and feeds are force-dynamic and need no hook.
expect(POSTS).toContain('afterChange');
expect(POSTS).toContain('afterDelete');
expect(POSTS).toMatch(/revalidatePath\(`\/blog\/\$\{[^}]+\}`\)/);
});
});
describe('media collection', () => {
it('writes uploads to the mounted volume, by absolute path', () => {
// Must match the payload_media mount in docker-compose.portainer.yml.
// Payload 3 requires staticDir to be absolute.
expect(MEDIA).toMatch(/staticDir:\s*'\/app\/media'/);
});
it('requires alt text on every upload', () => {
const altField = MEDIA.slice(MEDIA.indexOf("name: 'alt'"));
expect(altField).toMatch(/required:\s*true/);
});
});
describe('payload config', () => {
it('registers every collection', () => {
expect(CONFIG).toMatch(/collections:\s*\[Users,\s*Posts,\s*Media\]/);
});
});
@@ -1,67 +0,0 @@
/**
* The admin panel does not import field components directly. Payload sends the
* client a *path* for each one — a richText field's is
* `@payloadcms/richtext-lexical/rsc#RscEntryLexicalField` — and resolves it
* through this generated map. An entry that is missing from the map is not an
* error the panel reports: the field simply does not render.
*
* That failure is quietly awful, because `required: true` is enforced on the
* server regardless. A writer gets a new-post form with no Content editor and
* a save that refuses on a field they were never shown.
*
* The map is generated by `npx payload generate:importmap`, so it drifts every
* time a field or a lexical feature is added and nobody re-runs it. These
* assert the entries the current config needs.
*/
import fs from 'fs';
import path from 'path';
const MAP = fs.readFileSync(
path.join(__dirname, '..', '..', 'app', '(payload)', 'admin', 'importMap.js'),
'utf8',
);
const POSTS = fs.readFileSync(
path.join(__dirname, '..', '..', 'collections', 'Posts.ts'),
'utf8',
);
describe('admin import map', () => {
it('resolves the richText field, so Content renders in the editor', () => {
// Guarded because Posts.content is required: without this entry the field
// is invisible and the post is unsaveable.
expect(POSTS).toMatch(/type:\s*'richText'/);
expect(MAP).toContain('@payloadcms/richtext-lexical/rsc#RscEntryLexicalField');
});
it('resolves the richText cell, so the list view can render the column', () => {
expect(MAP).toContain('@payloadcms/richtext-lexical/rsc#RscEntryLexicalCell');
});
it('resolves the diff component, which the drafts UI needs', () => {
// versions.drafts is on, so the panel offers version comparison.
expect(POSTS).toMatch(/drafts:\s*true/);
expect(MAP).toContain('@payloadcms/richtext-lexical/rsc#LexicalDiffComponent');
});
it('resolves BlocksFeature, so the Callout block is insertable', () => {
expect(POSTS).toContain('BlocksFeature');
expect(MAP).toContain('@payloadcms/richtext-lexical/client#BlocksFeatureClient');
});
it('resolves the default toolbar features the editor is built with', () => {
// defaultFeatures is spread into the editor config; each one contributes a
// client component the toolbar cannot render without.
for (const feature of [
'BoldFeatureClient',
'ItalicFeatureClient',
'HeadingFeatureClient',
'LinkFeatureClient',
'UploadFeatureClient',
'UnorderedListFeatureClient',
'OrderedListFeatureClient',
'InlineToolbarFeatureClient',
]) {
expect(MAP).toContain(`@payloadcms/richtext-lexical/client#${feature}`);
}
});
});
@@ -1,52 +0,0 @@
/**
* The generated migration is schema-qualified to "payload" throughout but does
* not create that schema — `schemaName` says where tables go, it does not
* create anything. On staging and production, which have never run it, the
* whole migration fails with `schema "payload" does not exist`.
*
* The CREATE SCHEMA is therefore hand-added, which makes it exactly the kind
* of edit a regeneration silently discards. This is the guard.
*/
import fs from 'fs';
import path from 'path';
const DIR = path.join(__dirname, '..', '..', 'migrations');
function migrationFiles() {
return fs
.readdirSync(DIR)
.filter((f) => f.endsWith('.ts') && f !== 'index.ts');
}
describe('payload migrations', () => {
it('ships at least one migration, so a container has tables to find', () => {
expect(migrationFiles().length).toBeGreaterThan(0);
});
it('creates the payload schema before creating anything in it', () => {
const initial = migrationFiles().find((f) => f.includes('initial'))!;
const sql = fs.readFileSync(path.join(DIR, initial), 'utf8');
expect(sql).toMatch(/CREATE SCHEMA IF NOT EXISTS "payload"/);
// Ordering matters: the schema must be created before the first object
// that lives in it, or the migration fails on its first statement.
expect(sql.indexOf('CREATE SCHEMA IF NOT EXISTS "payload"'))
.toBeLessThan(sql.indexOf('CREATE TABLE "payload"'));
});
it('creates the tables the app queries on boot', () => {
const initial = migrationFiles().find((f) => f.includes('initial'))!;
const sql = fs.readFileSync(path.join(DIR, initial), 'utf8');
for (const table of ['users', 'posts', '_posts_v', 'media', 'payload_migrations']) {
expect(sql).toContain(`CREATE TABLE "payload"."${table}"`);
}
});
it('is wired into the adapter, so it runs on server init', () => {
const config = fs.readFileSync(
path.join(__dirname, '..', '..', 'payload.config.ts'), 'utf8',
);
expect(config).toMatch(/prodMigrations:\s*migrations/);
});
});
@@ -1,45 +0,0 @@
/**
* Guards the one thing about Payload's mounting that fails silently.
*
* payload.config.ts itself cannot be imported here — Payload is ESM-only and
* next/jest will not transform it — so this asserts the shared constants and
* that the config actually wires them in, by reading its source. The live
* proof that /api still reaches FastAPI is the e2e journeys, which call
* /api/schools against the running app.
*/
import fs from 'fs';
import path from 'path';
import { PAYLOAD_API_ROUTE, PAYLOAD_ADMIN_ROUTE } from '@/lib/payloadRoutes';
const CONFIG = fs.readFileSync(
path.join(__dirname, '..', '..', 'payload.config.ts'),
'utf8',
);
describe('payload mount points', () => {
it('serves the CMS API from /cms-api, never /api', () => {
// /api is the FastAPI proxy's catch-all. Payload's default would be
// swallowed by it and forwarded to the backend, silently.
expect(PAYLOAD_API_ROUTE).toBe('/cms-api');
expect(PAYLOAD_API_ROUTE).not.toBe('/api');
});
it('serves the admin panel from /admin', () => {
expect(PAYLOAD_ADMIN_ROUTE).toBe('/admin');
});
it('wires both constants into the Payload config', () => {
expect(CONFIG).toContain('PAYLOAD_API_ROUTE');
expect(CONFIG).toContain('PAYLOAD_ADMIN_ROUTE');
});
it('never hardcodes a routes block that could drift from the constants', () => {
expect(CONFIG).not.toMatch(/routes:\s*\{[^}]*api:\s*['"]/);
});
it('isolates CMS tables in their own postgres schema', () => {
// Blog content must sit outside `public`, where the app tables, Airflow's
// metadata and scripts/migrate_csv_to_db.py --drop all live.
expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/);
});
});
@@ -21,7 +21,7 @@ import {
import { nationalAveragesFixture } from './schoolFixtures';
// The shell calls useComparison(), which throws outside the provider. In the
// app this wrapper comes from app/(frontend)/layout.tsx.
// app this wrapper comes from app/layout.tsx.
function withProviders(ui: ReactNode) {
return <ComparisonProvider>{ui}</ComparisonProvider>;
}
@@ -31,7 +31,6 @@ export function renderSchoolDetail(fixture: any) {
const navItems = buildNavItems(flags, {
ofsted: fixture.ofsted,
admissions: fixture.admissions,
admissionDistance: fixture.admissionDistance ?? null,
yearlyDataLength: fixture.yearlyData.length,
});
@@ -45,7 +44,6 @@ export function renderSchoolDetail(fixture: any) {
>
<PrimarySchoolSections
{...fixture}
admissionDistanceHistory={fixture.admissionDistanceHistory ?? []}
nationalAvg={nationalAveragesFixture}
flags={flags}
/>
@@ -59,7 +57,6 @@ export function renderSecondarySchoolDetail(fixture: any) {
const navItems = buildSecondaryNavItems(flags, {
ofsted: fixture.ofsted,
admissions: fixture.admissions,
admissionDistance: fixture.admissionDistance ?? null,
yearlyDataLength: fixture.yearlyData.length,
});
@@ -73,8 +70,6 @@ export function renderSecondarySchoolDetail(fixture: any) {
>
<SecondarySchoolSections
{...fixture}
admissionsHistory={fixture.admissionsHistory ?? []}
admissionDistanceHistory={fixture.admissionDistanceHistory ?? []}
nationalAvg={nationalAveragesFixture}
flags={flags}
/>
@@ -1,82 +0,0 @@
.page {
max-width: 42rem;
margin: 0 auto;
padding: 2.5rem 1.25rem 4rem;
}
.header {
display: flex;
align-items: center;
gap: 1.25rem;
margin-bottom: 2rem;
}
.portrait {
border-radius: 50%;
border: 2px solid var(--border);
object-fit: cover;
flex-shrink: 0;
}
.kicker {
font-family: var(--font-ui);
font-size: 0.75rem;
font-weight: 600;
text-transform: uppercase;
letter-spacing: 0.06em;
color: var(--brand);
margin: 0 0 0.35rem;
}
.heading {
font-family: var(--font-display);
font-size: clamp(1.5rem, 4vw, 2rem);
font-weight: 700;
line-height: 1.2;
color: var(--text-primary);
margin: 0;
}
.subheading {
font-family: var(--font-display);
font-size: 1.15rem;
font-weight: 600;
color: var(--text-primary);
margin: 2.25rem 0 0.75rem;
}
.prose p {
font-family: var(--font-ui);
font-size: 1rem;
line-height: 1.7;
color: var(--text-secondary);
margin: 0 0 1.1rem;
}
/* The opening paragraph carries the page. Larger, and in the primary ink
rather than the secondary, so it reads as a voice rather than as body copy.
Must stay in the descendant form: `.prose p` scores (0,1,1) and would beat a
bare `.lede` at (0,1,0), so simplifying this selector silently reverts the
lede to ordinary body copy. */
.prose .lede {
font-size: 1.125rem;
color: var(--text-primary);
}
.link {
color: var(--brand);
font-weight: 600;
}
.link:hover {
color: var(--brand-strong);
}
@media (max-width: 480px) {
.header {
flex-direction: column;
align-items: flex-start;
gap: 1rem;
}
}
-143
View File
@@ -1,143 +0,0 @@
import type { Metadata } from 'next';
import Image from 'next/image';
import { notFound } from 'next/navigation';
import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import { personJsonLd, organizationJsonLd } from '@/lib/jsonld';
import styles from './About.module.css';
export const metadata: Metadata = {
title: 'About',
description:
'Who builds schoolcompare, why it exists, and where its numbers come from.',
alternates: { canonical: absoluteUrl('/about') },
};
/*
* Gated on about_page. The default 300s read is the right floor here: this
* page declares no revalidate of its own, so nothing is lost by it, and a flip
* lands within five minutes.
*
* notFound(), not a redirect: while the flag is dark this URL does not exist,
* and a 404 is what tells a crawler not to keep it.
*/
export default async function AboutPage() {
const flags = await getFlags();
if (flags.about_page !== true) notFound();
const jsonLd = {
'@context': 'https://schema.org',
'@graph': [personJsonLd(), organizationJsonLd()],
};
return (
<div className={styles.page}>
<script
type="application/ld+json"
dangerouslySetInnerHTML={{ __html: JSON.stringify(jsonLd) }}
/>
<header className={styles.header}>
<Image
src="/brand/tudor.jpg"
alt="Tudor, who builds schoolcompare"
width={96}
height={96}
className={styles.portrait}
priority
/>
<div>
<p className={styles.kicker}>Who&apos;s behind this</p>
<h1 className={styles.heading}>I&apos;m Tudor. I built this site.</h1>
</div>
</header>
<div className={styles.prose}>
<p className={styles.lede}>
I&apos;m a parent in south-west London. When we started looking at
primary schools, I found the information I needed was all published,
and almost impossible to hold in one place.
</p>
<p>
SATs results were in one government table. Ofsted judgements were in a
separate service, in a format that had just changed. Admissions
distances were buried in council PDFs, a different one per borough,
each with its own layout. I ended up building a spreadsheet, and then
I got tired of the spreadsheet.
</p>
<p>
So I built this instead. It pulls the official figures into one place
and puts them side by side, which is what I wanted and could not find.
</p>
<h2 className={styles.subheading}>I&apos;m not an education expert</h2>
<p>
I want to be straightforward about that. I&apos;m not a teacher, a
governor, or an education researcher. I have no qualification that
makes my opinion about a school worth more than yours.
</p>
<p>
What I do have is the problem itself. I&apos;m going through primary
admissions right now, and I work with data for a living. That
combination is enough to take published figures and present them
honestly. It is not enough to tell you which school is right for your
child, and this site never tries to.
</p>
<h2 className={styles.subheading}>Where the numbers come from</h2>
<p>
Everything here is official published data: Key Stage 2 and Key Stage
4 results and school characteristics from the Department for
Education, inspection outcomes from Ofsted, and admissions data from
local authorities. Nothing is estimated, modelled or filled in. Where
a figure is missing, the page says so rather than showing a guess.
</p>
<p>
This is an independent site. It is not affiliated with the Department
for Education or with Ofsted, and nobody pays to appear on it or to
rank higher.
</p>
<h2 className={styles.subheading}>What the data can&apos;t tell you</h2>
<p>
A school is not its results. The figures here describe one year group,
on a handful of days, measured in a way that suits national statistics
rather than your child. A small cohort makes percentages swing wildly.
In a class of thirty, one pupil is worth more than three points.
Results say nothing at all about whether a child will be happy
somewhere.
</p>
<p>
I try to build that honesty into the site rather than just say it
here. Special schools and pupil referral units are never compared
against a mainstream national average, because that comparison is
meaningless and makes good schools look like failing ones. Where a
number is unreliable, the aim is for the page to tell you before you
draw a conclusion from it.
</p>
<h2 className={styles.subheading}>If something&apos;s wrong</h2>
<p>
Tell me and I&apos;ll fix it. If a figure looks wrong, or a page gives
a misleading impression of a school, I genuinely want to know.
It&apos;s the fastest way this gets better.
</p>
<p>
<a href="mailto:contact@schoolcompare.co.uk" className={styles.link}>
contact@schoolcompare.co.uk
</a>
</p>
</div>
</div>
);
}
@@ -1,18 +0,0 @@
import { absoluteUrl } from '@/lib/site';
import type { Metadata } from 'next';
import { AdmissionsView } from '@/components/AdmissionsView';
export const dynamic = 'force-static';
export const metadata: Metadata = {
// Deadlines and offer days are what gets searched, and what this page is
// genuinely best at — the countdowns are live.
title: { absolute: 'School Admissions Deadlines & Offer Days | schoolcompare' },
description:
'Every key date for primary and secondary school admissions in England, with live countdowns to the application deadline and National Offer Day.',
alternates: { canonical: absoluteUrl('/admissions') },
};
export default function AdmissionsPage() {
return <AdmissionsView />;
}
@@ -1,76 +0,0 @@
.page {
max-width: 42rem;
margin: 0 auto;
padding: 2.5rem 1.25rem 4rem;
}
.header { margin-bottom: 2.5rem; }
.kicker {
font-family: var(--font-ui);
font-size: 0.75rem;
font-weight: 600;
text-transform: uppercase;
letter-spacing: 0.06em;
color: var(--brand);
margin: 0 0 0.35rem;
}
.heading {
font-family: var(--font-display);
font-size: clamp(1.5rem, 4vw, 2rem);
font-weight: 700;
line-height: 1.2;
color: var(--text-primary);
margin: 0 0 0.75rem;
}
.standfirst {
font-family: var(--font-ui);
font-size: 1.05rem;
line-height: 1.65;
color: var(--text-secondary);
margin: 0;
}
.list { list-style: none; padding: 0; margin: 0; }
.item {
padding: 1.5rem 0;
border-top: 1px solid var(--border);
}
.date {
font-family: var(--font-ui);
font-size: 0.8rem;
color: var(--text-muted);
/* Inter's tabular numerals keep a column of dates aligned. */
font-variant-numeric: tabular-nums;
}
.itemTitle {
font-family: var(--font-display);
font-size: 1.25rem;
font-weight: 600;
line-height: 1.3;
margin: 0.35rem 0 0.5rem;
}
.itemLink { color: var(--text-primary); text-decoration: none; }
.itemLink:hover { color: var(--brand); }
.excerpt {
font-family: var(--font-ui);
font-size: 0.95rem;
line-height: 1.65;
color: var(--text-secondary);
margin: 0;
}
.empty {
font-family: var(--font-ui);
color: var(--text-muted);
}
.link { color: var(--brand); font-weight: 600; }
.link:hover { color: var(--brand-strong); }
@@ -1,87 +0,0 @@
.page {
max-width: 42rem;
margin: 0 auto;
padding: 2.5rem 1.25rem 4rem;
}
.crumb {
font-family: var(--font-ui);
font-size: 0.85rem;
margin-bottom: 1.25rem;
}
.heading {
font-family: var(--font-display);
font-size: clamp(1.6rem, 5vw, 2.25rem);
font-weight: 700;
line-height: 1.2;
color: var(--text-primary);
margin: 0 0 0.75rem;
}
.byline {
font-family: var(--font-ui);
font-size: 0.9rem;
color: var(--text-muted);
margin: 0 0 2rem;
}
.hero {
width: 100%;
height: auto;
border-radius: 10px;
border: 1px solid var(--border);
margin-bottom: 2rem;
}
/* Rich-text output: the editor emits plain elements, so these are styled by
descendant selector rather than by class. */
.prose p {
font-family: var(--font-ui);
font-size: 1rem;
line-height: 1.7;
color: var(--text-secondary);
margin: 0 0 1.1rem;
}
.prose h2 {
font-family: var(--font-display);
font-size: 1.25rem;
font-weight: 600;
color: var(--text-primary);
margin: 2.25rem 0 0.75rem;
}
.prose h3 {
font-family: var(--font-display);
font-size: 1.05rem;
font-weight: 600;
color: var(--text-primary);
margin: 1.75rem 0 0.6rem;
}
.prose ul,
.prose ol {
font-family: var(--font-ui);
font-size: 1rem;
line-height: 1.7;
color: var(--text-secondary);
padding-left: 1.35rem;
margin: 0 0 1.1rem;
}
.prose li { margin-bottom: 0.4rem; }
.prose a { color: var(--brand); font-weight: 500; }
.prose a:hover { color: var(--brand-strong); }
.prose blockquote {
border-left: 3px solid var(--border-strong);
padding-left: 1rem;
margin: 1.5rem 0;
color: var(--text-muted);
font-style: italic;
}
.link { color: var(--brand); font-weight: 600; }
.link:hover { color: var(--brand-strong); }
@@ -1,194 +0,0 @@
import { cache } from 'react';
import type { Metadata } from 'next';
import Link from 'next/link';
import { notFound } from 'next/navigation';
import { RichText } from '@payloadcms/richtext-lexical/react';
import type { JSXConvertersFunction } from '@payloadcms/richtext-lexical/react';
import { getCachedPayload } from '@/lib/payload';
import type { Post, Media } from '@/payload-types';
import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import {
blogPostingJsonLd,
breadcrumbJsonLd,
personJsonLd,
organizationJsonLd,
} from '@/lib/jsonld';
import { CalloutBlock } from '@/components/blog/CalloutBlock';
import styles from './Post.module.css';
/*
* ISR. Unlike the index, this route has a dynamic param and no
* generateStaticParams, so there is nothing for the build to prerender: each
* post is generated on first request and cached until the collection's
* afterChange hook revalidates it. That hook is what makes an edit to an
* already-published post appear immediately.
*/
export const revalidate = 3600;
/**
* heroImage is `number | Media | null`: an id when the query is shallow, the
* populated document at depth 1. Both pages query at depth 1, but narrowing
* rather than asserting keeps it correct if that ever changes.
*/
function heroOf(post: Post): Media | null {
return typeof post.heroImage === 'object' && post.heroImage !== null
? post.heroImage
: null;
}
/**
* Spreads the default converters and adds the one custom block.
*
* Without the spread, every default node type — paragraphs, headings, links —
* loses its renderer and the post body comes out empty.
*/
const calloutConverters: JSXConvertersFunction = ({ defaultConverters }) => ({
...defaultConverters,
blocks: {
// Annotated because the generic block converter cannot infer a custom
// block's field shape; String() guards the values regardless.
callout: ({ node }: { node: { fields: Record<string, unknown> } }) => (
<CalloutBlock
tone={String(node.fields.tone ?? 'caveat')}
body={String(node.fields.body ?? '')}
/>
),
},
});
/**
* Wrapped in React's cache() because Next calls generateMetadata and the page
* component separately for the same request — without it, every post view runs
* this query against Postgres twice. cache() dedupes within a single request
* only, so it never serves one visitor's request from another's.
*/
const findPost = cache(async (slug: string) => {
const payload = await getCachedPayload();
const { docs } = await payload.find({
collection: 'posts',
where: { slug: { equals: slug }, _status: { equals: 'published' } },
limit: 1,
depth: 1,
});
return docs[0] ?? null;
});
function summarise(post: Post) {
return {
title: post.title,
slug: post.slug,
excerpt: post.excerpt,
publishedAt: post.publishedAt,
};
}
export async function generateMetadata(
{ params }: { params: Promise<{ slug: string }> },
): Promise<Metadata> {
const { slug } = await params;
const post = await findPost(slug);
if (!post) return { title: 'Not found' };
const hero = heroOf(post);
return {
title: post.title,
description: post.excerpt,
alternates: { canonical: absoluteUrl(`/blog/${post.slug}`) },
openGraph: {
type: 'article',
title: post.title,
description: post.excerpt,
url: absoluteUrl(`/blog/${post.slug}`),
publishedTime: post.publishedAt,
// A post with a hero image shares that; one without falls through to the
// generated share card at app/opengraph-image.tsx.
...(hero?.url ? { images: [{ url: hero.url }] } : {}),
},
};
}
export default async function PostPage(
{ params }: { params: Promise<{ slug: string }> },
) {
const { slug } = await params;
/*
* Flags read at this route's own declared floor, so gating costs it nothing.
* Checked before the post is fetched: a dark blog should not query Payload.
*/
const flags = await getFlags(3600);
if (flags.blog !== true) notFound();
const namedAuthor = flags.about_page === true;
const post = await findPost(slug);
if (!post) notFound();
const summary = summarise(post);
const hero = heroOf(post);
const jsonLd = {
'@context': 'https://schema.org',
/*
* The Person entity is anchored at /about#tudor, so it is declared only
* when that page exists. Claiming an author whose URL 404s is a worse
* signal than attributing the post to the publisher.
*/
'@graph': [
blogPostingJsonLd(summary, { namedAuthor }),
breadcrumbJsonLd(summary),
...(namedAuthor ? [personJsonLd()] : []),
organizationJsonLd(),
],
};
return (
<article className={styles.page}>
<script
type="application/ld+json"
dangerouslySetInnerHTML={{ __html: JSON.stringify(jsonLd) }}
/>
<nav className={styles.crumb}>
<Link href="/blog" className={styles.link}>Blog</Link>
</nav>
<h1 className={styles.heading}>{summary.title}</h1>
<p className={styles.byline}>
{/* Unlinked while about_page is dark; the flags are independent. */}
By {namedAuthor
? <Link href="/about" className={styles.link}>Tudor</Link>
: 'Tudor'}
{' · '}
<time dateTime={summary.publishedAt}>
{new Date(summary.publishedAt).toLocaleDateString('en-GB', {
day: 'numeric',
month: 'long',
year: 'numeric',
})}
</time>
</p>
{/*
A plain <img>, not next/image: Payload already generated the sized
derivatives on upload (Media's imageSizes), so routing it through the
optimizer would resize an image that is already the right size.
*/}
{hero?.url && (
<img
className={styles.hero}
src={hero.url}
alt={hero.alt ?? ''}
width={hero.width ?? undefined}
height={hero.height ?? undefined}
/>
)}
<div className={styles.prose}>
<RichText data={post.content} converters={calloutConverters} />
</div>
</article>
);
}
-85
View File
@@ -1,85 +0,0 @@
import type { Metadata } from 'next';
import Link from 'next/link';
import { notFound } from 'next/navigation';
import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import styles from './Blog.module.css';
/*
* Dynamic, not ISR.
*
* This route has no dynamic params, so Next prerenders it at build time — and
* CI builds the image with no database reachable, which fails the build. It is
* a single indexed query against Postgres on the same Docker network, so
* rendering per request is cheap, and it means a newly published post appears
* here immediately rather than waiting on a revalidation.
*/
export const dynamic = 'force-dynamic';
export const metadata: Metadata = {
title: 'Blog',
description:
'Notes on what school performance data shows, and what it does not.',
alternates: { canonical: absoluteUrl('/blog') },
};
function formatDate(value: string) {
return new Date(value).toLocaleDateString('en-GB', {
day: 'numeric',
month: 'long',
year: 'numeric',
});
}
export default async function BlogIndexPage() {
const flags = await getFlags();
if (flags.blog !== true) notFound();
const payload = await getCachedPayload();
const { docs } = await payload.find({
collection: 'posts',
where: { _status: { equals: 'published' } },
sort: '-publishedAt',
limit: 50,
depth: 0,
});
return (
<div className={styles.page}>
<header className={styles.header}>
<p className={styles.kicker}>Blog</p>
<h1 className={styles.heading}>Notes on the numbers</h1>
<p className={styles.standfirst}>
What school performance data shows, what it doesn&apos;t, and how to
read it without being misled. Written by{' '}
{/* Plain text when about_page is dark: the two flags are
independent, so this link would otherwise point at a 404. */}
{flags.about_page === true
? <Link href="/about" className={styles.link}>Tudor</Link>
: 'Tudor'}.
</p>
</header>
{docs.length === 0 ? (
<p className={styles.empty}>No posts yet.</p>
) : (
<ul className={styles.list}>
{docs.map((post) => (
<li key={post.id} className={styles.item}>
<time className={styles.date} dateTime={String(post.publishedAt)}>
{formatDate(String(post.publishedAt))}
</time>
<h2 className={styles.itemTitle}>
<Link href={`/blog/${post.slug}`} className={styles.itemLink}>
{post.title}
</Link>
</h2>
<p className={styles.excerpt}>{post.excerpt}</p>
</li>
))}
</ul>
)}
</div>
);
}
@@ -1,58 +0,0 @@
import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
/*
* Dynamic, not ISR.
*
* This route has no dynamic params, so Next prerenders it at build time — and
* CI builds the image with no database reachable, which fails the build. It is
* a single indexed query against Postgres on the same Docker network, so
* rendering per request is cheap, and it means a newly published post appears
* here immediately rather than waiting on a revalidation.
*/
export const dynamic = 'force-dynamic';
function escapeXml(value: string): string {
return value.replace(/[<>&'"]/g, (char) =>
({ '<': '&lt;', '>': '&gt;', '&': '&amp;', "'": '&apos;', '"': '&quot;' }[char]!));
}
export async function GET() {
// A dark blog has no feed. 404 rather than an empty channel: an empty feed
// is a live feed with nothing in it, which a reader would keep polling.
const flags = await getFlags();
if (flags.blog !== true) return new Response('Not found', { status: 404 });
const payload = await getCachedPayload();
const { docs } = await payload.find({
collection: 'posts',
where: { _status: { equals: 'published' } },
sort: '-publishedAt',
limit: 50,
depth: 0,
});
const items = docs.map((post) => `
<item>
<title>${escapeXml(String(post.title))}</title>
<link>${absoluteUrl(`/blog/${post.slug}`)}</link>
<guid isPermaLink="true">${absoluteUrl(`/blog/${post.slug}`)}</guid>
<description>${escapeXml(String(post.excerpt))}</description>
<pubDate>${new Date(String(post.publishedAt)).toUTCString()}</pubDate>
</item>`).join('');
const xml = `<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0">
<channel>
<title>schoolcompare blog</title>
<link>${absoluteUrl('/blog')}</link>
<description>What school performance data shows, and what it does not.</description>
<language>en-GB</language>${items}
</channel>
</rss>`;
return new Response(xml, {
headers: { 'Content-Type': 'application/rss+xml; charset=utf-8' },
});
}
@@ -1,68 +0,0 @@
/*
* A second sitemap for the URLs Next owns.
*
* /sitemap.xml is proxied from FastAPI (app/(frontend)/sitemap.xml), which
* knows nothing about Payload — the backend and frontend ship as separate
* images. Rather than teach it, the Next-owned URLs get their own sitemap and
* robots.txt lists both.
*/
import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
/*
* Dynamic, not ISR.
*
* This route has no dynamic params, so Next prerenders it at build time — and
* CI builds the image with no database reachable, which fails the build. It is
* a single indexed query against Postgres on the same Docker network, so
* rendering per request is cheap, and it means a newly published post appears
* here immediately rather than waiting on a revalidation.
*/
export const dynamic = 'force-dynamic';
export async function GET() {
/*
* A dark page must not be advertised. Submitting a URL that 404s is the one
* thing a sitemap is not allowed to do, so each entry is gated on the same
* flag that gates the page itself.
*
* With both flags dark this emits a valid, empty <urlset> rather than a 404:
* robots.txt names this sitemap unconditionally, and an empty sitemap is a
* well-formed statement that there is nothing here yet.
*/
const flags = await getFlags();
const aboutEnabled = flags.about_page === true;
const blogEnabled = flags.blog === true;
// Only query Payload when the blog is actually being advertised.
const docs = blogEnabled
? (await (await getCachedPayload()).find({
collection: 'posts',
where: { _status: { equals: 'published' } },
sort: '-publishedAt',
limit: 500,
depth: 0,
})).docs
: [];
const urls: Array<{ loc: string; lastmod: string | null }> = [
...(aboutEnabled ? [{ loc: absoluteUrl('/about'), lastmod: null }] : []),
...(blogEnabled ? [{ loc: absoluteUrl('/blog'), lastmod: null }] : []),
...docs.map((post) => ({
loc: absoluteUrl(`/blog/${post.slug}`),
lastmod: new Date(String(post.updatedAt ?? post.publishedAt)).toISOString(),
})),
];
const xml = `<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
${urls.map(({ loc, lastmod }) =>
` <url><loc>${loc}</loc>${lastmod ? `<lastmod>${lastmod}</lastmod>` : ''}</url>`,
).join('\n')}
</urlset>`;
return new Response(xml, {
headers: { 'Content-Type': 'application/xml; charset=utf-8' },
});
}
@@ -1,63 +0,0 @@
/**
* Phase variants of a place page.
*
* Phase is part of the query — "primary schools in beccles", "secondary
* schools in brentwood" — not a filter applied afterwards, so each gets its
* own indexable path. A place with no schools of the phase has no page: the
* per-phase threshold, not an error.
*/
import { notFound } from 'next/navigation';
import type { Metadata } from 'next';
import { fetchPlace } from '@/lib/places';
import { fetchNationalAverages } from '@/lib/api';
import { PlaceView } from '@/components/places/PlaceView';
import { absoluteUrl } from '@/lib/site';
interface Props { params: Promise<{ place: string; phase: string }> }
export const revalidate = 604800;
export const dynamicParams = true;
const PHASES = ['primary', 'secondary'] as const;
type Phase = (typeof PHASES)[number];
const isPhase = (v: string): v is Phase => (PHASES as readonly string[]).includes(v);
async function resolve(slug: string, phase: Phase) {
return (await fetchPlace('town', slug, phase))
?? (await fetchPlace('locality', slug, phase));
}
export async function generateMetadata({ params }: Props): Promise<Metadata> {
const { place: slug, phase } = await params;
if (!isPhase(phase)) return { title: 'Place Not Found' };
const detail = await resolve(slug, phase);
if (!detail || detail.schools.length === 0) return { title: 'Place Not Found' };
const word = phase === 'secondary' ? 'Secondary' : 'Primary';
const { name } = detail.place;
return {
// Not "Ranked": the table is alphabetical, so the word would be a claim
// the page does not keep.
title: { absolute: `${word} Schools in ${name} | schoolcompare` },
description:
`Every ${phase} school in ${name}, with results, Ofsted grades and the local `
+ `average against England.`,
alternates: { canonical: absoluteUrl(`/schools/${slug}/${phase}`) },
};
}
export default async function PlacePhasePage({ params }: Props) {
const { place: slug, phase } = await params;
if (!isPhase(phase)) notFound();
const detail = await resolve(slug, phase);
if (!detail || detail.schools.length === 0) notFound();
const national = await fetchNationalAverages().catch(() => null);
const englandAverage = phase === 'secondary'
? national?.secondary?.attainment_8_score ?? null
: national?.primary?.rwm_expected_pct ?? null;
return <PlaceView detail={detail} phase={phase}
englandAverage={englandAverage} neighbours={[]} />;
}
@@ -1,99 +0,0 @@
/**
* Town and locality pages.
*
* A place below the five-school threshold is not in the registry, so
* fetchPlace returns null and the request 404s rather than rendering a page
* with nothing to say.
*/
import { notFound, redirect } from 'next/navigation';
import type { Metadata } from 'next';
import { fetchPlace, fetchPlaces, authoritySlug } from '@/lib/places';
import { fetchNationalAverages } from '@/lib/api';
import { PlaceView } from '@/components/places/PlaceView';
import { absoluteUrl } from '@/lib/site';
interface Props { params: Promise<{ place: string }> }
// ISR: place aggregates change only when the pipeline runs.
export const revalidate = 604800;
export const dynamicParams = true;
export async function generateStaticParams(): Promise<Array<{ place: string }>> {
// Off by default: ~2,000 place routes cannot be built in CI on every deploy.
// Matches the PRERENDER_SCHOOLS gate on the school route.
if (process.env.PRERENDER_PLACES !== '1') return [];
try {
return (await fetchPlaces())
.filter((p) => p.kind === 'town' || p.kind === 'locality')
.map((p) => ({ place: p.slug }));
} catch (error) {
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
return [];
}
}
async function resolve(slug: string) {
return (await fetchPlace('town', slug)) ?? (await fetchPlace('locality', slug));
}
/** Other towns in the same authority — the cheapest honest definition of
* "nearby", and enough to stop each place page being a dead end. */
async function neighboursOf(detail: { place: { slug: string; parent_authority: string | null } }) {
if (!detail.place.parent_authority) return [];
const all = await fetchPlaces();
return all
.filter((p) => p.kind === 'town' && p.slug !== detail.place.slug)
.slice(0, 12);
}
export async function generateMetadata({ params }: Props): Promise<Metadata> {
const { place: slug } = await params;
const detail = await resolve(slug);
if (!detail) return { title: 'Place Not Found' };
const { name, count } = detail.place;
return {
// absolute: the root layout's template appends '| schoolcompare' to a
// plain string, and this title already carries it. Without this every
// place title read '... | schoolcompare | schoolcompare'.
title: { absolute: `Schools in ${name} — Compare ${count} Schools | schoolcompare` },
description:
`Every school in ${name}, with SATs and GCSE results, Ofsted grades, the local `
+ `average against England, and how close you had to live to get a place.`,
alternates: { canonical: absoluteUrl(`/schools/${slug}`) },
};
}
export default async function PlacePage({ params }: Props) {
const { place: slug } = await params;
const detail = await resolve(slug);
if (!detail) notFound();
// Global constraint: no page without a local average. A place with too few
// schools carrying results has nothing to say that a list does not, so it
// defers to its authority rather than publishing a thin page.
if (detail.averages.rwm_expected_pct == null
&& detail.averages.attainment_8_score == null) {
// The API's own slug, which is null when that authority is itself under
// the threshold and has no page. Re-slugifying the name here would send
// the reader to a 404 instead of telling them the place has no page.
const target = detail.place.authorities?.[0]?.slug
?? (detail.place.parent_authority
? authoritySlug(detail.place.parent_authority)
: null);
if (target) redirect(`/schools/authority/${target}`);
notFound();
}
const national = await fetchNationalAverages().catch(() => null);
// NationalAverages is nested by phase — { primary: {...}, secondary: {...} }
// — not flat. Reading it flat silently yields undefined and the page renders
// with no comparison, which is the one thing that makes it not a list.
return (
<PlaceView
detail={detail}
englandAverage={national?.primary?.rwm_expected_pct ?? null}
neighbours={await neighboursOf(detail)}
/>
);
}
@@ -1,65 +0,0 @@
/**
* Phase variants of an authority page.
*
* The spec called for these; the plan built the bare authority route and
* dropped them. Nothing caught it, because the sitemap is written from the
* place registry — which was right about them all along — while the routes
* were written by hand. 302 authority phase URLs were submitted to Google and
* every one 404'd, and every authority page linked to a phase page in the
* *town* namespace, which is a different set of schools entirely.
*
* "Primary schools in Kent" is the query these serve, and it is a real one:
* admissions are authority-run, so the authority is the unit a parent thinks
* in when they have not settled on a town.
*/
import { notFound } from 'next/navigation';
import type { Metadata } from 'next';
import { fetchPlace } from '@/lib/places';
import { fetchNationalAverages } from '@/lib/api';
import { PlaceView } from '@/components/places/PlaceView';
import { absoluteUrl } from '@/lib/site';
interface Props { params: Promise<{ la: string; phase: string }> }
export const revalidate = 604800;
export const dynamicParams = true;
const PHASES = ['primary', 'secondary'] as const;
type Phase = (typeof PHASES)[number];
const isPhase = (v: string): v is Phase => (PHASES as readonly string[]).includes(v);
export async function generateMetadata({ params }: Props): Promise<Metadata> {
const { la, phase } = await params;
if (!isPhase(phase)) return { title: 'Place Not Found' };
const detail = await fetchPlace('authority', la, phase);
if (!detail || detail.schools.length === 0) return { title: 'Place Not Found' };
const word = phase === 'secondary' ? 'Secondary' : 'Primary';
const { name } = detail.place;
return {
// "Local Authority" stays in the title for the same reason it is on the
// bare authority page: 67 town names collide with an authority name, and
// a reader landing on both needs to know which set each covers.
title: { absolute: `${word} Schools in ${name} — Local Authority | schoolcompare` },
description:
`Every ${phase} school in the ${name} local authority, with results, Ofsted `
+ `grades and the authority average against England.`,
alternates: { canonical: absoluteUrl(`/schools/authority/${la}/${phase}`) },
};
}
export default async function AuthorityPhasePage({ params }: Props) {
const { la, phase } = await params;
if (!isPhase(phase)) notFound();
const detail = await fetchPlace('authority', la, phase);
if (!detail || detail.schools.length === 0) notFound();
const national = await fetchNationalAverages().catch(() => null);
const englandAverage = phase === 'secondary'
? national?.secondary?.attainment_8_score ?? null
: national?.primary?.rwm_expected_pct ?? null;
return <PlaceView detail={detail} phase={phase}
englandAverage={englandAverage} neighbours={[]} />;
}
@@ -1,67 +0,0 @@
/**
* Local authority pages.
*
* A separate namespace from /schools/[place] because 67 town names collide
* with an authority name and neither set contains the other — Bedford the
* town holds 104 schools, Bedford the authority 86, because postal towns
* cross authority boundaries. The title says "Local Authority" so a reader
* landing on both knows which set each covers.
*/
import { notFound } from 'next/navigation';
import type { Metadata } from 'next';
import { fetchPlace, fetchPlaces } from '@/lib/places';
import { fetchNationalAverages } from '@/lib/api';
import { PlaceView } from '@/components/places/PlaceView';
import { absoluteUrl } from '@/lib/site';
interface Props { params: Promise<{ la: string }> }
export const revalidate = 604800;
export const dynamicParams = true;
export async function generateStaticParams(): Promise<Array<{ la: string }>> {
// Gated like every other prerender in this app. There are only ~154
// authorities, but "few enough to always build" still means the API must be
// reachable at build time, and in CI it is not — the build fails with
// ECONNREFUSED rather than degrading. The catch is the same fallback the
// school route uses.
if (process.env.PRERENDER_PLACES !== '1') return [];
try {
return (await fetchPlaces())
.filter((p) => p.kind === 'authority')
.map((p) => ({ la: p.slug }));
} catch (error) {
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
return [];
}
}
export async function generateMetadata({ params }: Props): Promise<Metadata> {
const { la } = await params;
const detail = await fetchPlace('authority', la);
if (!detail) return { title: 'Place Not Found' };
const { name, count } = detail.place;
return {
title: { absolute: `Schools in ${name} — Local Authority | schoolcompare` },
description:
`All ${count} schools in the ${name} local authority, with SATs and GCSE results, `
+ `Ofsted grades and the authority average against England.`,
alternates: { canonical: absoluteUrl(`/schools/authority/${la}`) },
};
}
export default async function AuthorityPage({ params }: Props) {
const { la } = await params;
const detail = await fetchPlace('authority', la);
if (!detail) notFound();
const national = await fetchNationalAverages().catch(() => null);
return (
<PlaceView
detail={detail}
englandAverage={national?.primary?.rwm_expected_pct ?? null}
neighbours={[]}
/>
);
}
@@ -1,62 +0,0 @@
/**
* Postcode district pages.
*
* No phase variants: nobody searches "primary schools in SW11", so the
* variants would be pages without demand. These exist to catch
* "schools near <postcode>" and to give London districts a geographic page
* where the GIAS town field cannot.
*/
import { notFound } from 'next/navigation';
import type { Metadata } from 'next';
import { fetchPlace, fetchPlaces } from '@/lib/places';
import { fetchNationalAverages } from '@/lib/api';
import { PlaceView } from '@/components/places/PlaceView';
import { absoluteUrl } from '@/lib/site';
interface Props { params: Promise<{ outcode: string }> }
export const revalidate = 604800;
export const dynamicParams = true;
export async function generateStaticParams(): Promise<Array<{ outcode: string }>> {
// 1,760 of these; same CI budget argument as the town routes.
if (process.env.PRERENDER_PLACES !== '1') return [];
try {
return (await fetchPlaces())
.filter((p) => p.kind === 'outcode')
.map((p) => ({ outcode: p.slug }));
} catch (error) {
console.warn('generateStaticParams: API unreachable, falling back to on-demand ISR.', error);
return [];
}
}
export async function generateMetadata({ params }: Props): Promise<Metadata> {
const { outcode } = await params;
const detail = await fetchPlace('outcode', outcode);
if (!detail) return { title: 'Place Not Found' };
const { name, count } = detail.place;
return {
title: { absolute: `Schools near ${name} | schoolcompare` },
description:
`${count} schools in the ${name} postcode district, with results, Ofsted grades `
+ `and how close you had to live to get a place.`,
alternates: { canonical: absoluteUrl(`/schools/near/${outcode}`) },
};
}
export default async function OutcodePage({ params }: Props) {
const { outcode } = await params;
const detail = await fetchPlace('outcode', outcode);
if (!detail) notFound();
const national = await fetchNationalAverages().catch(() => null);
return (
<PlaceView
detail={detail}
englandAverage={national?.primary?.rwm_expected_pct ?? null}
neighbours={[]}
/>
);
}
@@ -1,8 +0,0 @@
import { proxySitemap } from '@/lib/sitemapProxy';
export const dynamic = 'force-dynamic';
export const runtime = 'nodejs';
export async function GET() {
return proxySitemap('/sitemap.xml');
}
@@ -1,24 +0,0 @@
import { NextResponse } from 'next/server';
import { proxySitemap } from '@/lib/sitemapProxy';
export const dynamic = 'force-dynamic';
export const runtime = 'nodejs';
/**
* Children are /sitemaps/static.xml and /sitemaps/schools-{n}.xml. The name is
* validated here rather than passed through, so this route cannot be used to
* reach arbitrary backend paths.
*/
const CHILD = /^(static|schools-\d+|places-\d+|outcodes-\d+)\.xml$/;
export async function GET(
_request: Request,
{ params }: { params: Promise<{ parts: string[] }> },
) {
const { parts } = await params;
const name = parts.join('/');
if (!CHILD.test(name)) {
return new NextResponse('Not found', { status: 404 });
}
return proxySitemap(`/sitemaps/${name}`);
}
@@ -1,16 +0,0 @@
import type { Metadata } from 'next';
import config from '@payload-config';
import { NotFoundPage, generatePageMetadata } from '@payloadcms/next/views';
import { importMap } from '../importMap.js';
type Args = {
params: Promise<{ segments: string[] }>;
searchParams: Promise<{ [key: string]: string | string[] }>;
};
export const generateMetadata = ({ params, searchParams }: Args): Promise<Metadata> =>
generatePageMetadata({ config, params, searchParams });
export default function NotFound({ params, searchParams }: Args) {
return NotFoundPage({ config, importMap, params, searchParams });
}
@@ -1,16 +0,0 @@
import type { Metadata } from 'next';
import config from '@payload-config';
import { RootPage, generatePageMetadata } from '@payloadcms/next/views';
import { importMap } from '../importMap.js';
type Args = {
params: Promise<{ segments: string[] }>;
searchParams: Promise<{ [key: string]: string | string[] }>;
};
export const generateMetadata = ({ params, searchParams }: Args): Promise<Metadata> =>
generatePageMetadata({ config, params, searchParams });
export default function Page({ params, searchParams }: Args) {
return RootPage({ config, importMap, params, searchParams });
}
@@ -1,54 +0,0 @@
import { RscEntryLexicalCell as RscEntryLexicalCell_44fe37237e0ebf4470c9990d8cb7b07e } from '@payloadcms/richtext-lexical/rsc'
import { RscEntryLexicalField as RscEntryLexicalField_44fe37237e0ebf4470c9990d8cb7b07e } from '@payloadcms/richtext-lexical/rsc'
import { LexicalDiffComponent as LexicalDiffComponent_44fe37237e0ebf4470c9990d8cb7b07e } from '@payloadcms/richtext-lexical/rsc'
import { BlocksFeatureClient as BlocksFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { BoldFeatureClient as BoldFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { ItalicFeatureClient as ItalicFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { UnderlineFeatureClient as UnderlineFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { StrikethroughFeatureClient as StrikethroughFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { SubscriptFeatureClient as SubscriptFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { SuperscriptFeatureClient as SuperscriptFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { InlineCodeFeatureClient as InlineCodeFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { ParagraphFeatureClient as ParagraphFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { HeadingFeatureClient as HeadingFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { AlignFeatureClient as AlignFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { IndentFeatureClient as IndentFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { UnorderedListFeatureClient as UnorderedListFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { OrderedListFeatureClient as OrderedListFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { ChecklistFeatureClient as ChecklistFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { LinkFeatureClient as LinkFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { RelationshipFeatureClient as RelationshipFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { BlockquoteFeatureClient as BlockquoteFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { UploadFeatureClient as UploadFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { HorizontalRuleFeatureClient as HorizontalRuleFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { InlineToolbarFeatureClient as InlineToolbarFeatureClient_e70f5e05f09f93e00b997edb1ef0c864 } from '@payloadcms/richtext-lexical/client'
import { CollectionCards as CollectionCards_f9c02e79a4aed9a3924487c0cd4cafb1 } from '@payloadcms/next/rsc'
/** @type import('payload').ImportMap */
export const importMap = {
"@payloadcms/richtext-lexical/rsc#RscEntryLexicalCell": RscEntryLexicalCell_44fe37237e0ebf4470c9990d8cb7b07e,
"@payloadcms/richtext-lexical/rsc#RscEntryLexicalField": RscEntryLexicalField_44fe37237e0ebf4470c9990d8cb7b07e,
"@payloadcms/richtext-lexical/rsc#LexicalDiffComponent": LexicalDiffComponent_44fe37237e0ebf4470c9990d8cb7b07e,
"@payloadcms/richtext-lexical/client#BlocksFeatureClient": BlocksFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#BoldFeatureClient": BoldFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#ItalicFeatureClient": ItalicFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#UnderlineFeatureClient": UnderlineFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#StrikethroughFeatureClient": StrikethroughFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#SubscriptFeatureClient": SubscriptFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#SuperscriptFeatureClient": SuperscriptFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#InlineCodeFeatureClient": InlineCodeFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#ParagraphFeatureClient": ParagraphFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#HeadingFeatureClient": HeadingFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#AlignFeatureClient": AlignFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#IndentFeatureClient": IndentFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#UnorderedListFeatureClient": UnorderedListFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#OrderedListFeatureClient": OrderedListFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#ChecklistFeatureClient": ChecklistFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#LinkFeatureClient": LinkFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#RelationshipFeatureClient": RelationshipFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#BlockquoteFeatureClient": BlockquoteFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#UploadFeatureClient": UploadFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#HorizontalRuleFeatureClient": HorizontalRuleFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/richtext-lexical/client#InlineToolbarFeatureClient": InlineToolbarFeatureClient_e70f5e05f09f93e00b997edb1ef0c864,
"@payloadcms/next/rsc#CollectionCards": CollectionCards_f9c02e79a4aed9a3924487c0cd4cafb1
}
@@ -1,20 +0,0 @@
/*
* Payload's REST API, mounted at /cms-api rather than /api.
* See lib/payloadRoutes.ts — /api is the FastAPI proxy's catch-all.
*/
import config from '@payload-config';
import {
REST_DELETE,
REST_GET,
REST_OPTIONS,
REST_PATCH,
REST_POST,
REST_PUT,
} from '@payloadcms/next/routes';
export const GET = REST_GET(config);
export const POST = REST_POST(config);
export const DELETE = REST_DELETE(config);
export const PATCH = REST_PATCH(config);
export const PUT = REST_PUT(config);
export const OPTIONS = REST_OPTIONS(config);
@@ -1,4 +0,0 @@
import config from '@payload-config';
import { GRAPHQL_PLAYGROUND_GET } from '@payloadcms/next/routes';
export const GET = GRAPHQL_PLAYGROUND_GET(config);
@@ -1,5 +0,0 @@
import config from '@payload-config';
import { GRAPHQL_POST, REST_OPTIONS } from '@payloadcms/next/routes';
export const POST = GRAPHQL_POST(config);
export const OPTIONS = REST_OPTIONS(config);
-27
View File
@@ -1,27 +0,0 @@
/**
* Root layout for the Payload admin panel.
*
* This is a SECOND root layout: it renders its own <html>/<body>, as does
* app/(frontend)/layout.tsx. Next permits that only while no app/layout.tsx
* exists — which is why the site's routes were moved into (frontend). Adding
* an app/layout.tsx would nest the admin panel inside the site's nav, footer
* and providers and emit nested <html>.
*/
import type { ServerFunctionClient } from 'payload';
import config from '@payload-config';
import { RootLayout, handleServerFunctions } from '@payloadcms/next/layouts';
import { importMap } from './admin/importMap.js';
import '@payloadcms/next/css';
const serverFunction: ServerFunctionClient = async function (args) {
'use server';
return handleServerFunctions({ ...args, config, importMap });
};
export default function PayloadLayout({ children }: { children: React.ReactNode }) {
return (
<RootLayout config={config} importMap={importMap} serverFunction={serverFunction}>
{children}
</RootLayout>
);
}
+14
View File
@@ -0,0 +1,14 @@
import type { Metadata } from 'next';
import { AdmissionsView } from '@/components/AdmissionsView';
export const dynamic = 'force-static';
export const metadata: Metadata = {
title: 'School Admissions Guide',
description:
'Understand the Primary and Secondary school admissions process in England, with live countdowns to every key deadline and National Offer Day.',
};
export default function AdmissionsPage() {
return <AdmissionsView />;
}
Loaded 100 of 242 files, more files were not shown because too many files have changed in this diff. Show more