Merge pull request 'feat(seo): link school pages into the location layer' (#145) from feat/school-page-place-links into main
Stage (build -> staging -> E2E gate) / Build Backend (FastAPI) (push) Successful in 21s
Stage (build -> staging -> E2E gate) / Build Frontend (Next.js) (push) Successful in 1m26s
Stage (build -> staging -> E2E gate) / Build Pipeline (Meltano + dbt + Airflow) (push) Successful in 13s
Stage (build -> staging -> E2E gate) / Deploy to Staging (push) Successful in 1s
Stage (build -> staging -> E2E gate) / E2E Journeys against Staging (push) Failing after 2m29s

Reviewed-on: #145
This commit was merged in pull request #145.
This commit is contained in:
tudor committed 2026-09-14 20:29:48 +00:00
commit dc79d653e5
12 files changed
+809 -4

No files matched your search

+69 -1
View File
@@ -38,7 +38,7 @@ from .data_loader import (
)
from .data_loader import get_data_info as get_db_info
from . import flags
from .places import build_place_registry
from .places import build_place_index, build_place_registry, places_for_urn
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
from .utils import clean_for_json, convert_to_native
@@ -65,6 +65,10 @@ _sitemaps: dict[str, str] | None = None
# Built from the same DataFrame the sitemap uses, so places and sitemap can
# never describe different corpora. Reset by the same admin endpoint.
_place_registry: dict | None = None
# Cached beside the registry, and invalidated by identity against it — see
# get_place_index. Never cleared independently.
_place_index: dict | None = None
_place_index_source: dict | None = None
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
@@ -188,6 +192,24 @@ def get_place_registry() -> dict:
return _place_registry
def get_place_index() -> dict:
"""URN → its published places, cached against the registry it came from.
Invalidation is an identity check rather than a second flag to remember to
clear. Anything that drops `_place_registry` — the tests all do — gets a
fresh registry object here, which no longer matches the one the index was
built from, so the index rebuilds with it. A separate `_place_index = None`
would be one more thing to forget, and a stale reverse index is exactly the
bug that would put links to another dataset's places on a school page.
"""
global _place_index, _place_index_source
registry = get_place_registry()
if _place_index is None or _place_index_source is not registry:
_place_index = build_place_index(registry)
_place_index_source = registry
return _place_index
def _urlset(rows: list[str]) -> str:
return "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>',
@@ -211,6 +233,45 @@ def _place_url(place) -> str:
return f"/schools/{place.slug}"
def _places_payload(urn: int) -> list[dict]:
"""The published places containing this school, as the school page needs
them: a name to write in the link, a count so the anchor can say what it
leads to, and the canonical path.
`phases` carries the phase variants this school actually appears on, which
is usually one and is two for an all-through school — it is listed on both
pages, so there is no tie to break.
Membership is read straight from the registry's own `phase_urns` rather
than re-derived from the school's phase string. The registry is the one
place that decides which phases a place publishes and who is on them;
computing it a second time here is how a page comes to link a school to a
phase page that does not list it, or to a route that does not exist. That
is also why outcodes need no special case: they carry empty `phase_urns`,
so they report no phase links on their own.
"""
payload = []
for place in places_for_urn(get_place_index(), int(urn)):
phases = [
{
"phase": phase,
"count": len(phase_urns),
"url": f"{_place_url(place)}/{phase}",
}
for phase, phase_urns in sorted(place.phase_urns.items())
if int(urn) in phase_urns
]
payload.append({
"kind": place.kind,
"slug": place.slug,
"name": place.name,
"count": len(place.urns),
"url": _place_url(place),
"phases": phases,
})
return payload
def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
"""A <url> per place, plus a phase variant wherever that phase clears the
threshold on its own.
@@ -902,6 +963,13 @@ async def get_school_details(request: Request, urn: int):
return {
"school_info": school_info,
# Where this school sits in the location layer, for the page's link
# module and breadcrumb. Derived from the same registry the place
# pages and the sitemap use, so a link is never offered for a page
# that does not exist. Empty is a valid answer: a school whose town
# and authority both fall below the publish threshold has nowhere to
# point, and the page renders without the module.
"places": _places_payload(urn),
"yearly_data": clean_for_json(school_data),
# Supplementary data (null if not yet populated by Kestra)
"ofsted": supplementary.get("ofsted"),
+42
View File
@@ -296,6 +296,48 @@ def _locality_places(df, publishable: set[int],
return out
# Ordered authority → town/locality → outcode, widest first, because that is
# the order a breadcrumb reads. The link module re-sorts for its own purposes.
_PLACE_ORDER = {"authority": 0, "town": 1, "locality": 2, "outcode": 3}
def build_place_index(registry: dict[str, Place]) -> dict[int, tuple[Place, ...]]:
"""URN → the published places containing it, built once per registry.
The reverse of the registry, and the thing school pages link out through.
Derived from the registry rather than maintained beside it, so the two
cannot disagree about which places exist: a place below the publish
threshold is absent from the registry, so it is absent from here too, and
a link is never offered for a page that does not exist.
Built as an index rather than scanned per call because /api/schools/{urn}
is the site's highest-traffic endpoint. Scanning meant walking every place
and doing a tuple membership test against each — on the order of 10^5
comparisons per request, repeated for every school page view. One pass at
registry-build time replaces all of it with a dict lookup.
"""
grouped: dict[int, list[Place]] = {}
for place in registry.values():
for urn in place.urns:
grouped.setdefault(int(urn), []).append(place)
return {
urn: tuple(sorted(places,
key=lambda p: (_PLACE_ORDER.get(p.kind, 9), p.slug)))
for urn, places in grouped.items()
}
def places_for_urn(index: dict[int, tuple[Place, ...]], urn: int) -> tuple[Place, ...]:
"""The published places containing this school, widest first.
Empty is a real answer, not a failure: a school whose town and authority
both fall below the publish threshold has nowhere to link, and the page
renders without the module.
"""
return index.get(int(urn), ())
def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns:
+78 -1
View File
@@ -8,7 +8,8 @@ import numpy as np
import pandas as pd
import pytest
from backend.places import MIN_SCHOOLS, build_place_registry
from backend.places import (MIN_SCHOOLS, build_place_index,
build_place_registry, places_for_urn)
def _df(rows: list[dict]) -> pd.DataFrame:
@@ -418,3 +419,79 @@ def test_an_authority_still_publishes_phase_variants():
and /schools/authority/[la]/[phase] is the route that serves it."""
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Maidstone", "Kent")))
assert reg["authority:kent"].publishes_phase("primary")
# ── The reverse index: which published places contain a school ──────────────
#
# School pages link out to the location layer through this. It is the whole
# point of the index: before it, ~27k school pages linked to nothing on the
# site and stranded whatever authority they held.
def test_a_school_resolves_to_every_published_place_containing_it():
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
places = places_for_urn(build_place_index(reg), 100000)
kinds = {p.kind for p in places}
assert "town" in kinds
assert "authority" in kinds
def test_a_school_in_an_unpublished_town_still_resolves_to_its_authority():
# A town below the threshold has no page, so there is no link to offer —
# but the authority above it clears the threshold on the same schools and
# is where that reader should be sent.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS - 1, "Tinytown", "Essex")
+ _town(MIN_SCHOOLS, "Brentwood", "Essex", start=200000)
))
places = places_for_urn(build_place_index(reg), 100000)
# The town is below the threshold, so it has no page and must not be
# offered as a link. The authority above it does, and is the right target.
assert all(p.slug != "tinytown" for p in places)
assert "authority" in {p.kind for p in places}
def test_an_unknown_urn_resolves_to_nothing_rather_than_raising():
# A school page renders for any URN the API knows; the link module is not
# entitled to take the page down when it has nothing to say.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
assert places_for_urn(build_place_index(reg), 999999) == ()
def test_the_index_is_consistent_with_the_registry_it_was_built_from():
# The invariant that matters: a link module must never offer a place whose
# page does not exist, and never omit one that does.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS, "Bedford", "Bedford", start=300000)
))
index = build_place_index(reg)
for key, place in reg.items():
for urn in place.urns:
assert place in places_for_urn(index, urn), (
f"{urn} is in {key} but the index does not say so")
def test_the_index_holds_no_school_the_registry_does_not():
# The reverse direction of the invariant above. An index entry for a URN
# no published place contains would put a link on a page for a place that
# does not list that school.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS - 1, "Tinytown", "Essex", start=400000)
))
index = build_place_index(reg)
for urn, places in index.items():
for place in places:
assert urn in place.urns
assert place.key in reg
def test_the_index_preserves_the_widest_first_order():
# The breadcrumb reads authority then town, and takes this order as given.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
kinds = [p.kind for p in places_for_urn(build_place_index(reg), 100000)]
assert kinds.index("authority") < kinds.index("town")
+176
View File
@@ -56,6 +56,11 @@ def client(monkeypatch):
monkeypatch.setattr(
app_module, "get_supplementary_data", lambda db, urn: {}
)
# The place registry is a module-level cache, so without this the endpoint
# answers from whatever registry an earlier test happened to leave behind
# — and a `places == []` assertion is satisfied by a stale registry just
# as well as by this fixture's own data, which makes it prove nothing.
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False)
@@ -69,3 +74,174 @@ def test_nan_gias_fields_serialize_as_null(client):
assert info["capacity"] is None
assert info["total_pupils"] is None
assert info["school_name"] == "West London Performing Arts Academy"
# ── Links out to the location layer ─────────────────────────────────────────
#
# School pages carried no link into the site at all: the only anchor on the
# template pointed at the school's own website, so ~27k pages received
# whatever authority the site had and sent it off-site. `places` is what the
# link module and the breadcrumb are built from.
def test_places_is_present_even_when_the_school_belongs_to_none(client):
# This fixture's single school cannot clear any publish threshold, so the
# honest answer is an empty list. The key must still be there: a missing
# key and "no places" are different things to the page rendering it.
body = client.get("/api/schools/150275").json()
assert body["places"] == []
def test_places_names_only_pages_that_exist(monkeypatch):
from backend import app as app_module
from backend.places import MIN_SCHOOLS
def _df():
return pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood",
"local_authority": "Essex",
"postcode": "CM15 8AA",
"phase": "Primary",
"year": 202425,
"rwm_expected_pct": 60.0,
"attainment_8_score": np.nan,
"ofsted_grade": 2.0,
"ofsted_date": None,
}
for i in range(MIN_SCHOOLS)
])
monkeypatch.setattr(app_module, "load_school_data", _df)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
places = client.get("/api/schools/100000").json()["places"]
assert places, "a school in a published town must offer links"
by_kind = {p["kind"]: p for p in places}
assert by_kind["town"]["url"] == "/schools/brentwood"
assert by_kind["authority"]["url"] == "/schools/authority/essex"
# Every entry carries what the link text needs, and a count, so the anchor
# can say what it leads to rather than "click here".
for place in places:
assert place["name"]
assert place["count"] >= 1
assert place["url"].startswith("/schools/")
def _brentwood_df(phase: str = "Primary", n: int = None):
from backend.places import MIN_SCHOOLS
n = n if n is not None else MIN_SCHOOLS
return lambda: pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood", "local_authority": "Essex",
"postcode": "CM15 8AA", "phase": phase, "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": 50.0,
"ofsted_grade": 2.0, "ofsted_date": None,
}
for i in range(n)
])
def _places_for(monkeypatch, df_factory, urn: int):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", df_factory)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
return client.get(f"/api/schools/{urn}").json()["places"]
def test_a_place_offers_the_phase_page_this_school_appears_on(monkeypatch):
# "primary schools in brentwood" is the query the phase pages exist for,
# and ~950 of them were once reachable by nothing at all.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert town["phases"], "a primary school in a published primary town has a link"
assert town["phases"][0]["url"] == "/schools/brentwood/primary"
assert town["phases"][0]["count"] >= 1
def test_an_all_through_school_offers_both_phase_pages(monkeypatch):
# It genuinely appears on both, so there is no tie to break.
places = _places_for(monkeypatch, _brentwood_df("All-through"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert {p["phase"] for p in town["phases"]} == {"primary", "secondary"}
def test_outcodes_never_offer_a_phase_page(monkeypatch):
# The registry gives outcodes no phase route — nobody searches "primary
# schools in SW11" — and computing them anyway once put a link to a
# nonexistent route on all 1,720 outcode pages.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
outcode = next((p for p in places if p["kind"] == "outcode"), None)
if outcode is not None:
assert outcode["phases"] == []
def test_a_school_absent_from_the_phase_page_is_not_linked_to_it(monkeypatch):
# The check is URN membership in the registry's own phase list, not a
# re-derivation of the phase mapping. A secondary school must not be sent
# to a primary phase page that does not list it.
from backend.places import MIN_SCHOOLS
def df():
rows = [
{"urn": 100000 + i, "school_name": f"P{i}", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Primary", "year": 202425, "rwm_expected_pct": 60.0,
"attainment_8_score": np.nan, "ofsted_grade": 2.0,
"ofsted_date": None}
for i in range(MIN_SCHOOLS)
]
rows.append({
"urn": 900000, "school_name": "Lone Secondary", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Secondary", "year": 202425, "rwm_expected_pct": np.nan,
"attainment_8_score": 50.0, "ofsted_grade": 2.0, "ofsted_date": None,
})
return pd.DataFrame(rows)
places = _places_for(monkeypatch, df, 900000)
town = next(p for p in places if p["kind"] == "town")
# The town publishes a primary page, but this secondary school is not on
# it, and there are too few secondaries for a secondary page.
assert town["phases"] == []
def test_the_place_index_rebuilds_when_the_registry_is_replaced(monkeypatch):
"""The reverse index is cached; a stale one would put another dataset's
places on a school page. Invalidation is an identity check against the
registry rather than a second flag, so this asserts the check works."""
from backend import app as app_module
monkeypatch.setattr(app_module, "_place_registry", None)
monkeypatch.setattr(app_module, "_place_index", None)
monkeypatch.setattr(app_module, "_place_index_source", None)
monkeypatch.setattr(app_module, "load_school_data", _brentwood_df("Primary"))
first = app_module.get_place_index()
assert 100000 in first
# Same registry object, so the index is reused rather than rebuilt.
assert app_module.get_place_index() is first
# Drop the registry the way every test that touches place data does. The
# index must follow it, not survive it.
app_module._place_registry = None
monkeypatch.setattr(app_module, "load_school_data",
_brentwood_df("Primary", n=0))
rebuilt = app_module.get_place_index()
assert rebuilt is not first
assert 100000 not in rebuilt, "the index outlived the registry it came from"
+57
View File
@@ -1935,6 +1935,63 @@ async function firstPlaceOfKind(page: Page, kind: string) {
return hit as { kind: string; slug: string; name: string; count: number };
}
/**
* The round trip. Place pages always linked down to school pages; school
* pages linked nowhere on the site, so the ~27k of them that carry most of
* the inbound authority stranded it — their only anchor pointed at the
* school's own website.
*
* Asserting both directions is the point. A one-way link is what already
* existed and is not what this journey is for.
*/
test('a school page links back into the location layer, and the place page links down', async ({ page }) => {
const town = await firstPlaceOfKind(page, 'town');
// Start from the place page and take its first school, so the pair is
// guaranteed to be genuinely related rather than a hardcoded guess.
await page.goto(`/schools/${town.slug}`);
const schoolHref = await page.locator('a[href^="/school/"]').first()
.getAttribute('href');
expect(schoolHref, 'the town page listed no school to follow').toBeTruthy();
await page.goto(schoolHref!);
// Down: the school page must offer a link back to the town it sits in.
const backToTown = page.locator(`a[href="/schools/${town.slug}"]`);
await expect(backToTown).toHaveCount(1);
await expect(backToTown).toBeVisible();
// The anchor says what it leads to, which is worth more than "see more".
await expect(backToTown).toContainText(town.name, { ignoreCase: true });
await expect(backToTown).toContainText(/\d+ schools?/);
// And the breadcrumb resolves the school into a real hierarchy.
const blocks = await page.locator('script[type="application/ld+json"]')
.allTextContents();
const graph = blocks.join(' ');
expect(graph).toContain('"BreadcrumbList"');
// The narrower type, not the EducationalOrganization parent it used to be.
expect(graph).toContain('"School"');
/*
* The phase variants are the pages this most needs to reach: ~950 of them
* were once reachable by nothing at all, absent from every sitemap and
* unlinked from the place page. Conditional because not every school sits
* in a town that publishes one.
*/
const phaseLink = page.locator(`a[href^="/schools/${town.slug}/"]`).first();
if (await phaseLink.count()) {
const phaseHref = await phaseLink.getAttribute('href');
expect((await page.request.get(phaseHref!)).status()).toBe(200);
await expect(phaseLink).toContainText(/primary|secondary/);
}
// Following it lands on a real page, not a 404.
await backToTown.click();
await page.waitForURL(new RegExp(`/schools/${town.slug}$`));
await expect(page.locator('h1')).toContainText(town.name, { ignoreCase: true });
});
for (const [kind, prefix, article] of [
['town', '/schools/', 'a'],
['authority', '/schools/authority/', 'an'],
@@ -0,0 +1,92 @@
/**
* The module that ends the stranding: before it, a school page's only anchor
* pointed at the school's own website, so ~27k pages sent authority off-site
* and none of it reached the location layer.
*/
import { render, screen } from '@testing-library/react';
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const cm15 = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('NearbyPlaces', () => {
it('links to every place the school belongs to', () => {
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
expect(screen.getByRole('link', { name: /Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood');
expect(screen.getByRole('link', { name: /Essex/ }))
.toHaveAttribute('href', '/schools/authority/essex');
expect(screen.getByRole('link', { name: /CM15/ }))
.toHaveAttribute('href', '/schools/near/cm15');
});
it('says how many schools each link leads to', () => {
// An anchor that states its destination's size is worth more to a reader
// and to a crawler than "see more".
render(<NearbyPlaces places={[brentwood]} />);
expect(screen.getByRole('link', { name: /37 schools in Brentwood/ }))
.toBeInTheDocument();
});
it('renders nothing at all when the school has no published places', () => {
// Not an empty heading. A school whose town and authority both fall below
// the threshold has nowhere to point, and the page should look as it did
// before the module existed.
const { container } = render(<NearbyPlaces places={[]} />);
expect(container).toBeEmptyDOMElement();
});
it('puts the narrowest place first, which is the most useful link', () => {
// The API orders widest-first for the breadcrumb; a reader on a school
// page wants its town before its county.
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood'))
.toBeLessThan(hrefs.indexOf('/schools/authority/essex'));
});
it('handles a singular count without saying "1 schools"', () => {
render(<NearbyPlaces places={[{ ...brentwood, count: 1 }]} />);
expect(screen.getByRole('link', { name: /1 school in Brentwood/ }))
.toBeInTheDocument();
});
it('links the phase page the school appears on', () => {
// "primary schools in brentwood" is the query these pages exist for.
render(<NearbyPlaces places={[{
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools in Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood/primary');
});
it('links both phase pages for an all-through school', () => {
render(<NearbyPlaces places={[{
...brentwood,
phases: [
{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' },
{ phase: 'secondary', count: 9, url: '/schools/brentwood/secondary' },
],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools/ })).toBeInTheDocument();
expect(screen.getByRole('link', { name: /9 secondary schools/ })).toBeInTheDocument();
});
it('keeps a phase link next to the place it belongs to', () => {
// Grouping matters: "22 primary schools in Brentwood" directly after
// "37 schools in Brentwood" reads as one place, not two unrelated links.
render(<NearbyPlaces places={[essex, {
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood/primary'))
.toBe(hrefs.indexOf('/schools/brentwood') + 1);
});
});
@@ -0,0 +1,68 @@
/**
* School pages had no BreadcrumbList and no links into the location layer.
* Both are fixed by the same data — the `places` array the API now returns —
* so they are tested together.
*/
import { schoolBreadcrumbJsonLd } from '@/lib/jsonld';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const outcode = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('school breadcrumbs', () => {
it('reads home to authority to town to school', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
expect(ld['@type']).toBe('BreadcrumbList');
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Brentwood', 'Brentwood School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3, 4]);
});
it('skips a level the school has no published place for', () => {
// A school whose town falls below the publish threshold has no town page.
// The trail closes over the gap rather than linking to a 404.
const ld = schoolBreadcrumbJsonLd({
name: 'Lone School', url: '/school/1-lone-school', places: [essex],
});
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Lone School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3]);
});
it('omits outcodes, which are not a place a breadcrumb reads through', () => {
// CM15 is a useful link in the module but nonsense in a trail: nobody
// navigates Essex → CM15 → school.
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood, outcode],
});
expect(JSON.stringify(ld)).not.toContain('cm15');
});
it('still produces a valid trail when the school has no places at all', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Orphan School', url: '/school/2-orphan-school', places: [],
});
expect(ld.itemListElement.map((i) => i.name)).toEqual(['schoolcompare', 'Orphan School']);
});
it('uses absolute urls, as every other entity on the site does', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
for (const item of ld.itemListElement) {
expect(item.item).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\//);
}
// The root is the homepage: there is no /schools index page to link to.
expect(ld.itemListElement[0].item).toBe('https://www.schoolcompare.co.uk/');
});
});
@@ -7,6 +7,8 @@
import { fetchSchoolDetails, fetchSchools, fetchNationalAverages } from '@/lib/api';
import { notFound, redirect } from 'next/navigation';
import { SchoolDetailShell } from '@/components/school/SchoolDetailShell';
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
import { schoolBreadcrumbJsonLd, type SchoolPlace } from '@/lib/jsonld';
import { PrimarySchoolSections } from '@/components/school/PrimarySchoolSections';
import { SecondarySchoolSections } from '@/components/school/SecondarySchoolSections';
import {
@@ -149,6 +151,10 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
}
const { school_info, yearly_data, absence_data, ofsted, census, admissions, admissions_history, admission_distance, deprivation, finance, destinations } = data;
// Absent on an older API build; the module and the trail both degrade to
// nothing rather than throwing, which is how this shipped without a
// lockstep deploy of the two images.
const places: SchoolPlace[] = data.places ?? [];
// Redirect bare URN to canonical slug URL
const canonicalSlug = schoolUrl(urn, school_info.school_name).replace('/school/', '');
@@ -185,10 +191,19 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
const primaryNavItems = buildNavItems(primaryFlags, navInput);
const secondaryNavItems = buildSecondaryNavItems(secondaryFlags, navInput);
// Generate JSON-LD structured data for SEO
/*
* `School`, not `EducationalOrganization`.
*
* Both are valid, but EducationalOrganization is the parent type covering
* universities, training providers and nurseries alike. School is the
* specific one, and a type that says what the page is about is the whole
* point of declaring it. Google's own guidance treats the narrower type as
* the correct choice where it applies.
*/
const structuredData = {
'@context': 'https://schema.org',
'@type': 'EducationalOrganization',
'@graph': [{
'@type': 'School',
name: school_info.school_name,
identifier: school_info.urn.toString(),
...(school_info.address && {
@@ -210,6 +225,15 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
...(school_info.school_type && {
additionalType: school_info.school_type,
}),
},
// The trail the page sits at the end of. School pages carried no
// breadcrumb at all, while every place page already emitted one.
schoolBreadcrumbJsonLd({
name: school_info.school_name,
url: `/school/${slug}`,
places,
}),
],
};
return (
@@ -264,6 +288,7 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
/>
</SchoolDetailShell>
)}
<NearbyPlaces places={places} />
</>
);
}
@@ -0,0 +1,53 @@
/* Tokens only — the same vocabulary schoolSections.module.css uses, so the
module follows both themes without a rule of its own. No hardcoded colour
appears here; darkThemeSafety asserts that across the codebase. */
.section {
margin-top: 2rem;
}
/* Matches .sectionTitle in schoolSections.module.css, including the brand
rule before the text, so this reads as one more section of the page
rather than a footer bolted underneath it. */
.heading {
font-size: 1.125rem;
font-weight: 600;
color: var(--text-primary);
margin-bottom: 0.875rem;
padding-bottom: 0.5rem;
border-bottom: 2px solid var(--border);
font-family: var(--font-display);
display: flex;
align-items: center;
gap: 0.375rem;
}
.heading::before {
content: "";
display: inline-block;
width: 3px;
height: 1em;
background: var(--brand);
border-radius: 2px;
flex-shrink: 0;
}
.list {
display: flex;
flex-wrap: wrap;
gap: 0.5rem 1.25rem;
list-style: none;
margin: 0;
padding: 0;
}
.link {
color: var(--brand-strong);
font-weight: 500;
text-decoration: underline;
text-underline-offset: 2px;
}
.link:hover {
text-decoration-thickness: 2px;
}
@@ -0,0 +1,67 @@
import Link from 'next/link';
import type { SchoolPlace, SchoolPhasePage } from '@/lib/jsonld';
import styles from './NearbyPlaces.module.css';
/**
* Links from a school page into the location layer.
*
* This exists for a structural reason rather than a decorative one. Before
* it, the only anchor on a school page pointed at the school's own website,
* so the ~27k pages that carry most of the site's inbound authority passed it
* straight off-site and none of it reached the place pages. These links are
* what circulate it instead.
*
* Every entry comes from the place registry via the API, so a link is only
* ever offered for a page that exists: a place below the publish threshold is
* absent from the registry and therefore absent here.
*/
/** Narrowest first: a reader on a school page wants its town before its
* county. The API orders widest-first because that is what the breadcrumb
* reads, so the two orders are deliberately different. */
const ORDER: Record<string, number> = {
town: 0, locality: 0, outcode: 1, authority: 2,
};
function label(place: SchoolPlace): string {
const noun = place.count === 1 ? 'school' : 'schools';
const preposition = place.kind === 'outcode' ? 'near' : 'in';
return `${place.count} ${noun} ${preposition} ${place.name}`;
}
/** "22 primary schools in Brentwood" — the phrasing the query itself uses. */
function phaseLabel(place: SchoolPlace, page: SchoolPhasePage): string {
const noun = page.count === 1 ? 'school' : 'schools';
return `${page.count} ${page.phase} ${noun} in ${place.name}`;
}
export function NearbyPlaces({ places }: { places: SchoolPlace[] }) {
if (places.length === 0) return null;
const sorted = [...places].sort(
(a, b) => (ORDER[a.kind] ?? 9) - (ORDER[b.kind] ?? 9),
);
return (
<section className={styles.section} aria-labelledby="nearby-places">
<h2 id="nearby-places" className={styles.heading}>More schools near here</h2>
<ul className={styles.list}>
{sorted.flatMap((place) => [
<li key={`${place.kind}:${place.slug}`}>
<Link href={place.url} className={styles.link}>{label(place)}</Link>
</li>,
/* Immediately after its own place, so "22 primary schools in
Brentwood" reads as part of Brentwood rather than as an
unrelated link further down the row. */
...place.phases.map((page) => (
<li key={`${place.kind}:${place.slug}:${page.phase}`}>
<Link href={page.url} className={styles.link}>
{phaseLabel(place, page)}
</Link>
</li>
)),
])}
</ul>
</section>
);
}
+69
View File
@@ -69,6 +69,75 @@ export function blogPostingJsonLd(
} as const;
}
/**
* A place the location layer publishes a page for, as the school API reports
* it. `count` is what lets a link say "All 37 schools in Brentwood" rather
* than "click here".
*/
export interface SchoolPhasePage {
phase: string;
count: number;
url: string;
}
export interface SchoolPlace {
kind: string;
slug: string;
name: string;
count: number;
url: string;
/**
* The phase variants this school is actually listed on: usually one, two
* for an all-through school, none for an outcode, which publishes no phase
* route. Decided by the place registry, never re-derived here.
*/
phases: SchoolPhasePage[];
}
/**
* The trail a school page sits at the end of: Schools → authority → town.
*
* Only authority and town/locality appear. An outcode is a useful link in the
* module beside this — a parent does search "schools near CM15" — but it is
* not a step anyone navigates through, and a breadcrumb that claims otherwise
* describes a hierarchy the site does not have.
*
* Levels are skipped rather than faked. A school whose town falls below the
* publish threshold has no town page, so the trail closes over the gap; the
* alternative is a breadcrumb linking to a 404.
*/
export function schoolBreadcrumbJsonLd(
school: { name: string; url: string; places: SchoolPlace[] },
) {
/*
* Rooted at the homepage, not at /schools. There is no /schools index page
* — the location layer is /schools/[place], /schools/authority/[la] and
* /schools/near/[outcode], with nothing at the bare path — so a trail
* starting there would open with a link to a 404.
*/
const trail: Array<{ name: string; url: string }> = [
{ name: 'schoolcompare', url: '/' },
];
const authority = school.places.find((p) => p.kind === 'authority');
if (authority) trail.push({ name: authority.name, url: authority.url });
const town = school.places.find((p) => p.kind === 'town' || p.kind === 'locality');
if (town) trail.push({ name: town.name, url: town.url });
trail.push({ name: school.name, url: school.url });
return {
'@type': 'BreadcrumbList',
itemListElement: trail.map((step, index) => ({
'@type': 'ListItem',
position: index + 1,
name: step.name,
item: absoluteUrl(step.url),
})),
} as const;
}
export function breadcrumbJsonLd(post: PostSummary) {
return {
'@type': 'BreadcrumbList',
+11
View File
@@ -1,3 +1,5 @@
import type { SchoolPlace } from '@/lib/jsonld';
/**
* TypeScript type definitions for SchoolCompare API
* Generated from backend/models.py and backend/schemas.py
@@ -346,6 +348,15 @@ export interface SchoolsResponse {
export interface SchoolDetailsResponse {
school_info: School;
/**
* The published location-layer pages containing this school, widest first.
*
* Optional because the frontend and backend ship as separate images: a
* frontend deployed ahead of the API that serves this must render without
* it, not throw. Empty is also a real answer — a school whose town and
* authority both fall below the publish threshold has nowhere to link.
*/
places?: SchoolPlace[];
yearly_data: SchoolResult[];
absence_data: AbsenceData | null;
// Supplementary data (null until Kestra populates)