feat(api): decide which nearby schools a page may offer

Hard filters encode claims the section may not make — a selective school is
not an alternative to a non-selective one, a special school is not comparable
to a mainstream one, a Girls school is not an option for a Boys school's
reader — so they never relax. Soft preferences describe closeness of fit, so
they relax across three tiers, and only far enough to reach three; the
remaining slots up to six fill from the tiers already opened.

PHASE_GROUPS moves to schemas.py so this module can share it without
importing app, which would be a cycle.

_mask() exists because Series.apply on an empty Series returns a DataFrame,
and using that as a mask drops every column — so the next lookup raises
KeyError instead of yielding no rows. A special school with no special school
near it empties the frame at the provision filter, which is the ordinary case
for most special schools, so this was a crash on a common path rather than an
edge case.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5 committed 2026-09-21 22:40:10 +01:00
1 parent 8a23e3657d
commit b571d9c549
4 files changed
+508 -9

No files matched your search

+1 -9
View File
@@ -40,20 +40,12 @@ from .data_loader import (
from .data_loader import get_data_info as get_db_info from .data_loader import get_data_info as get_db_info
from . import flags from . import flags
from .places import build_place_index, build_place_registry, places_for_urn from .places import build_place_index, build_place_registry, places_for_urn
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS from .schemas import METRIC_DEFINITIONS, PHASE_GROUPS, RANKING_COLUMNS, SCHOOL_COLUMNS
from .utils import clean_for_json, convert_to_native from .utils import clean_for_json, convert_to_native
# Values to exclude from filter dropdowns (empty strings, non-applicable labels) # Values to exclude from filter dropdowns (empty strings, non-applicable labels)
EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"} EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"}
# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they include.
# All-through schools appear in both primary and secondary results.
PHASE_GROUPS: dict[str, set[str]] = {
"primary": {"primary", "middle deemed primary", "all-through"},
"secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"},
"all-through": {"all-through"},
}
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a # Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
# sitemap <loc> that redirects wastes a crawl on every URL it lists. # sitemap <loc> that redirects wastes a crawl on every URL it lists.
BASE_URL = "https://www.schoolcompare.co.uk" BASE_URL = "https://www.schoolcompare.co.uk"
+12
View File
@@ -532,6 +532,18 @@ RANKING_COLUMNS = [
"gcse_grade_91_pct", "gcse_grade_91_pct",
] ]
# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they
# include. All-through schools appear in both primary and secondary results,
# which is why this is a set per phase rather than a single string comparison.
#
# Lives here rather than in app.py because similar_schools.py needs it too, and
# importing app from there would be a cycle.
PHASE_GROUPS: dict[str, set[str]] = {
"primary": {"primary", "middle deemed primary", "all-through"},
"secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"},
"all-through": {"all-through"},
}
# School listing columns # School listing columns
SCHOOL_COLUMNS = [ SCHOOL_COLUMNS = [
"urn", "urn",
+243
View File
@@ -0,0 +1,243 @@
"""Which nearby schools a detail page may offer as alternatives.
Two kinds of rule, and they are not interchangeable.
HARD FILTERS encode claims the section is not allowed to make. A selective
school is not an alternative to a non-selective one, a special school is not
comparable to a mainstream one, and a Girls school is not an option for a Boys
school's reader. These never relax, at any distance, even where that means the
section does not render at all.
SOFT PREFERENCES describe how closely an intake resembles this school's. They
relax in tiers, and every card reports the tier that actually took it so the
page can say what is shared rather than implying more. They relax only far
enough to reach a usable set, never far enough to fill the last of the slots.
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
"""
from __future__ import annotations
import re
import numpy as np
import pandas as pd
from .schemas import PHASE_GROUPS
# A cap, not a quota: the section shows everything that qualified at the tiers
# it used, up to this many. Three fit the row; the rest are behind the arrows.
MAX_SCHOOLS = 6
# Tiers stop relaxing once this many have been found. Without it, a cap of six
# would reliably drag in tier-3 schools ten miles away to fill a row that three
# good matches had already earned.
ENOUGH = 3
MINIMUM = 2
# (tier, radius in miles). Faith relaxes before gender: a faith mismatch
# changes the character of a school, while a gender mismatch can mean the
# school is not available to this reader's child at all.
TIERS: tuple[tuple[int, float], ...] = ((1, 3.0), (2, 5.0), (3, 10.0))
EARTH_RADIUS_MILES = 3958.8
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
# Values that mean "this school has no religious character".
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
def is_special_provision(school_type: str | None) -> bool:
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
Special schools carry a mainstream phase, so phase alone cannot identify
them. The two implementations must agree: a school the frontend treats as
special for benchmarking but this treats as mainstream would be dropped
from its own England comparison and then offered as a peer to a mainstream
school on the next page along.
"""
return bool(_SPECIAL.search(school_type or ""))
def is_selective(admissions_policy: str | None) -> bool:
"""Strictly selective. Unknown counts as non-selective, which is the safe
direction: it can only ever exclude a pairing, never invent one."""
return (admissions_policy or "").strip().lower() == "selective"
def faith_key(denomination: str | None) -> str:
value = (denomination or "").strip().lower()
return "" if value in _NO_FAITH else value
def faith_label(denomination: str | None) -> str:
return denomination.strip() if faith_key(denomination) else "No religious character"
def genders_compatible(a: str | None, b: str | None) -> bool:
single = {"boys", "girls"}
left, right = (a or "").strip().lower(), (b or "").strip().lower()
return not (left in single and right in single and left != right)
def phase_label(phase: str | None) -> str:
text = (phase or "").strip()
if not text:
return "School"
if text.lower() == "all-through":
return "All-through school"
return f"{text.capitalize()} school"
def _phase_group(is_secondary: bool) -> set[str]:
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
"""Vectorised, matching the postcode search in app.py."""
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
def _native(value):
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
the caller never has to remember to."""
if value is None:
return None
if isinstance(value, np.generic):
value = value.item()
if isinstance(value, float) and np.isnan(value):
return None
return value
def _mask(series: pd.Series, predicate) -> pd.Series:
"""A boolean mask that survives an empty frame.
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
that as a mask silently drops every column — so the next column lookup
raises KeyError rather than yielding no rows. This is not hypothetical: a
special school with no special school near it empties the frame at the
provision filter, which is the ordinary case for most special schools.
"""
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
def _chips(subject: pd.Series, candidate: pd.Series, tier: int, is_secondary: bool) -> list[str]:
if tier >= 3:
return [phase_label(candidate.get("phase"))]
chips = [str(subject.get("gender") or "").strip()]
if is_secondary:
policy = (candidate.get("admissions_policy") or "").strip()
if policy and policy.lower() not in {"not applicable", "unknown"}:
chips.append(policy)
if tier == 1:
chips.append(faith_label(candidate.get("religious_denomination")))
return [chip for chip in chips if chip]
def select_similar(frame: pd.DataFrame, urn: int, is_secondary: bool) -> list[dict]:
"""Up to MAX_SCHOOLS nearby schools this page may offer, or [] below MINIMUM.
Selected by tier, displayed by distance: the tier decides which schools
earn a slot, and the render order is then closest-first, because "nearby"
is the promise in the heading.
"""
subject_rows = frame[frame["urn"] == urn]
if subject_rows.empty:
return []
subject = subject_rows.iloc[0]
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
if lat is None or lon is None:
return []
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
candidates = frame[frame["urn"] != urn].copy()
for column in ("latitude", "longitude"):
candidates = candidates[candidates[column].notna()]
if candidates.empty:
return []
# ── Hard filters ────────────────────────────────────────────────────
allowed_phases = _phase_group(is_secondary)
candidates = candidates[
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
]
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
subject_special = is_special_provision(subject.get("school_type"))
special = _mask(candidates["school_type"], is_special_provision)
candidates = candidates[special if subject_special else ~special]
subject_selective = is_selective(subject.get("admissions_policy"))
selective = _mask(candidates["admissions_policy"], is_selective)
candidates = candidates[selective if subject_selective else ~selective]
subject_gender = subject.get("gender")
candidates = candidates[
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
]
if candidates.empty:
return []
candidates["distance_miles"] = _haversine_miles(
lat, lon, candidates["latitude"].values, candidates["longitude"].values
).round(1)
# ── Soft preferences, in tiers ──────────────────────────────────────
subject_faith = faith_key(subject.get("religious_denomination"))
subject_gender_key = (subject_gender or "").strip().lower()
same_gender = candidates["gender"].fillna("").str.strip().str.lower() == subject_gender_key
same_faith = _mask(
candidates["religious_denomination"], lambda d: faith_key(d) == subject_faith
)
tier_masks = {
1: same_gender & same_faith,
2: same_gender,
3: pd.Series(True, index=candidates.index),
}
# Descend the tiers only until the set reaches ENOUGH. The tier that gets
# there is the last one opened, and the remaining slots up to MAX_SCHOOLS
# are filled from the tiers already used — never by widening again.
picked: dict[int, tuple[int, pd.Series]] = {}
for tier, radius in TIERS:
within = candidates[tier_masks[tier] & (candidates["distance_miles"] <= radius)]
for _, row in within.sort_values("distance_miles").iterrows():
candidate_urn = int(row["urn"])
if candidate_urn in picked:
continue
picked[candidate_urn] = (tier, row)
if len(picked) >= MAX_SCHOOLS:
break
if len(picked) >= ENOUGH:
break
if len(picked) < MINIMUM:
return []
selected = sorted(
picked.values(), key=lambda pair: float(pair[1]["distance_miles"])
)[:MAX_SCHOOLS]
return [
{
"urn": int(row["urn"]),
"school_name": str(row.get("school_name") or ""),
"distance_miles": float(row["distance_miles"]),
"school_type": _native(row.get("school_type")),
"age_range": _native(row.get("age_range")),
"shared": _chips(subject, row, tier, is_secondary),
"tier": tier,
"metric_value": _native(row.get(metric_key)),
"metric_key": metric_key,
"metric_year": _native(row.get("year")),
}
for tier, row in selected
]
+252
View File
@@ -0,0 +1,252 @@
"""Selection rules for the "similar schools nearby" section.
The hard filters encode claims the section is not allowed to make — that a
selective school is an alternative to a non-selective one, that a special
school is comparable to a mainstream one, or that a Girls school is an option
for a Boys school's reader. They never relax. The soft preferences describe
how close the intake is, and they do — but only far enough to reach a usable
set, never far enough to fill the last of the six slots.
"""
import numpy as np
import pandas as pd
from backend.similar_schools import select_similar
# Roughly 0.7 miles apart in latitude at this longitude.
BASE_LAT, BASE_LON = 51.5000, -0.1000
def _row(urn, name, **overrides):
base = {
"urn": urn,
"school_name": name,
"local_authority": "Testshire",
"school_type": "Community school",
"phase": "Primary",
"age_range": "4-11",
"status": "Open",
"gender": "Mixed",
"religious_denomination": "None",
"admissions_policy": "Not applicable",
"latitude": BASE_LAT,
"longitude": BASE_LON,
"year": 202425,
"rwm_expected_pct": 70.0,
"attainment_8_score": np.nan,
}
base.update(overrides)
return base
def _frame(*rows):
return pd.DataFrame(list(rows))
def _at(miles):
"""A latitude `miles` north of BASE_LAT."""
return BASE_LAT + miles / 69.0
def test_returns_nearest_same_phase_schools():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Near", latitude=_at(0.5)),
_row(100003, "Mid", latitude=_at(1.0)),
_row(100004, "Far", latitude=_at(2.0)),
)
result = select_similar(frame, 100001, is_secondary=False)
assert [s["urn"] for s in result] == [100002, 100003, 100004]
assert result[0]["distance_miles"] == 0.5
def test_excludes_the_subject_school():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
assert 100001 not in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)}
def test_selective_never_meets_non_selective():
frame = _frame(
_row(100001, "Grammar", phase="Secondary", admissions_policy="Selective"),
_row(100002, "Comp A", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.5)),
_row(100003, "Comp B", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.6)),
)
assert select_similar(frame, 100001, is_secondary=True) == []
reverse = select_similar(frame, 100002, is_secondary=True)
assert 100001 not in {s["urn"] for s in reverse}
def test_special_schools_match_only_each_other():
frame = _frame(
_row(100001, "Special", school_type="Community special school"),
_row(100002, "Mainstream A", latitude=_at(0.5)),
_row(100003, "Mainstream B", latitude=_at(0.6)),
)
assert select_similar(frame, 100001, is_secondary=False) == []
assert select_similar(frame, 100002, is_secondary=False) == []
def test_boys_never_meets_girls():
frame = _frame(
_row(100001, "Boys School", gender="Boys"),
_row(100002, "Girls School", gender="Girls", latitude=_at(0.5)),
_row(100003, "Mixed School", gender="Mixed", latitude=_at(0.6)),
_row(100004, "Another Mixed", gender="Mixed", latitude=_at(0.7)),
)
urns = {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)}
assert 100002 not in urns
assert urns == {100003, 100004}
def test_closed_schools_and_missing_coordinates_are_dropped():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Closed", status="Closed", latitude=_at(0.5)),
_row(100003, "No coords", latitude=np.nan, longitude=np.nan),
_row(100004, "Good A", latitude=_at(0.6)),
_row(100005, "Good B", latitude=_at(0.7)),
)
assert {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} == {100004, 100005}
def test_tiers_relax_faith_before_gender():
frame = _frame(
_row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"),
# Tier 1: same gender and same faith.
_row(100002, "Tier one", gender="Boys", religious_denomination="Roman Catholic", latitude=_at(2.0)),
# Tier 2: same gender, different faith — closer, but a weaker match.
_row(100003, "Tier two", gender="Boys", religious_denomination="None", latitude=_at(0.5)),
# Tier 3: mixed gender, different faith.
_row(100004, "Tier three", gender="Mixed", religious_denomination="None", latitude=_at(0.6)),
)
result = select_similar(frame, 100001, is_secondary=False)
tier_by_urn = {s["urn"]: s["tier"] for s in result}
assert tier_by_urn == {100002: 1, 100003: 2, 100004: 3}
# Selected by tier, displayed by distance.
assert [s["urn"] for s in result] == [100003, 100004, 100002]
def test_caps_at_six_taking_the_nearest():
frame = _frame(
_row(100001, "Subject"),
*[_row(100010 + n, f"Peer {n}", latitude=_at(0.1 * (n + 1))) for n in range(7)],
)
result = select_similar(frame, 100001, is_secondary=False)
assert len(result) == 6
# The seventh-nearest is the one dropped, not an arbitrary one.
assert 100016 not in {s["urn"] for s in result}
def test_tiers_stop_once_enough_are_found():
"""Four tier-1 matches are a usable set, so tier 2 is never opened — even
though it holds a school that is closer than any of them."""
frame = _frame(
_row(100001, "Subject", religious_denomination="Roman Catholic"),
_row(100002, "RC one", religious_denomination="Roman Catholic", latitude=_at(0.5)),
_row(100003, "RC two", religious_denomination="Roman Catholic", latitude=_at(0.6)),
_row(100004, "RC three", religious_denomination="Roman Catholic", latitude=_at(0.7)),
_row(100005, "RC four", religious_denomination="Roman Catholic", latitude=_at(0.8)),
# Closer than every one of them, but only a tier-2 match.
_row(100006, "Secular and nearer", religious_denomination="None", latitude=_at(0.2)),
)
result = select_similar(frame, 100001, is_secondary=False)
assert 100006 not in {s["urn"] for s in result}
assert len(result) == 4
assert all(s["tier"] == 1 for s in result)
def test_a_school_is_never_taken_twice():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
result = select_similar(frame, 100001, is_secondary=False)
assert len(result) == len({s["urn"] for s in result})
def test_fewer_than_two_matches_returns_empty():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Only neighbour", latitude=_at(0.5)),
)
assert select_similar(frame, 100001, is_secondary=False) == []
def test_beyond_the_widest_radius_is_not_offered():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(11.0)),
_row(100003, "B", latitude=_at(12.0)),
)
assert select_similar(frame, 100001, is_secondary=False) == []
def test_all_through_is_offered_on_both_phase_sides():
frame = _frame(
_row(100001, "Primary subject", phase="Primary"),
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
_row(100003, "Primary peer", phase="Primary", latitude=_at(0.6)),
)
assert 100002 in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)}
secondary = _frame(
_row(100010, "Secondary subject", phase="Secondary"),
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
_row(100011, "Secondary peer", phase="Secondary", latitude=_at(0.6)),
)
assert 100002 in {s["urn"] for s in select_similar(secondary, 100010, is_secondary=True)}
def test_chips_state_only_what_the_tier_earned():
frame = _frame(
_row(100001, "Subject", phase="Secondary", gender="Mixed",
religious_denomination="None", admissions_policy="Non-selective"),
_row(100002, "Full match", phase="Secondary", gender="Mixed",
religious_denomination="None", admissions_policy="Non-selective", latitude=_at(0.5)),
_row(100003, "Faith differs", phase="Secondary", gender="Mixed",
religious_denomination="Church of England", admissions_policy="Non-selective", latitude=_at(0.6)),
)
by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)}
assert by_urn[100002]["shared"] == ["Mixed", "Non-selective", "No religious character"]
assert by_urn[100003]["shared"] == ["Mixed", "Non-selective"]
def test_tier_three_chip_is_the_plain_phase():
frame = _frame(
_row(100001, "Subject", gender="Boys"),
_row(100002, "A", gender="Mixed", latitude=_at(0.5)),
_row(100003, "B", gender="Mixed", latitude=_at(0.6)),
)
result = select_similar(frame, 100001, is_secondary=False)
assert all(s["shared"] == ["Primary school"] for s in result)
def test_metric_follows_the_template_not_the_neighbour():
frame = _frame(
_row(100001, "Subject", phase="Secondary", attainment_8_score=50.0),
_row(100002, "A", phase="Secondary", attainment_8_score=52.8, latitude=_at(0.5)),
_row(100003, "B", phase="Secondary", attainment_8_score=np.nan, latitude=_at(0.6)),
)
by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)}
assert by_urn[100002]["metric_key"] == "attainment_8_score"
assert by_urn[100002]["metric_value"] == 52.8
assert by_urn[100002]["metric_year"] == 202425
assert by_urn[100003]["metric_value"] is None
def test_values_are_json_safe_native_types():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
for school in select_similar(frame, 100001, is_secondary=False):
assert isinstance(school["urn"], int)
assert isinstance(school["distance_miles"], float)
assert not isinstance(school["metric_value"], np.generic)