Files
school_compare/backend/similar_schools.py
T
TudorandClaude Opus 5 5c0ccc693d fix: order nearby schools by distance, not by how alike they are
Reported from staging: a Catholic primary showed six Catholic primaries, none
of them close enough to be a real option, and omitted the community school
down the road.

Three causes, compounding. Ranking put tier before distance, so a faith match
at 2.9 miles outranked a community school at 0.3. The ENOUGH=3 stopping rule —
added so a cap of six would not drag in weak distant matches — filled the row
from the best tier before it ever widened, which is what made every card
Catholic. And a 3-mile tier-1 radius is sane for a secondary and most of a city
for a primary, whose catchments are routinely under a mile.

The premise was backwards. For a parent, distance is a constraint and intake is
a preference; a school beyond a primary catchment is not a weaker option, it is
not an option. So distance now decides the order and nothing else does. The
hard filters are untouched — they were always where the defensibility lived.
Similarity survives as chips on the card: reported, so a reader applies their
own weighting, rather than ranked, so we apply ours for them.

Reach is capped per phase (primary 2, secondary 6, post-16 10) as a sanity
bound, not a target: ordering already handles density, so the cap only decides
what happens where an area is sparse. A primary with nothing inside two miles
now renders no section, which is the honest answer.

Deleted: the tier system, the stopping rule, the tier-dependent lede, the
`tier` field, the tier-3 fallback chip and its style. select_similar also stops
taking is_secondary — it reads the phase from the subject's own row, so no
caller can hand it one that disagrees with the data.

The heading is now "Other schools nearby". The hard filters still guarantee a
comparable set, but nothing ranks on likeness, so the heading no longer says it
does.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-22 13:06:32 +01:00

260 lines
10 KiB
Python

"""Which nearby schools a detail page may offer as alternatives.
HARD FILTERS decide eligibility, and encode claims the section is not allowed
to make. A selective school is not an alternative to a non-selective one, a
special school is not comparable to a mainstream one, and a Girls school is not
an option for a Boys school's reader. They never relax, at any distance, even
where that means the section does not render at all.
DISTANCE decides the order, and nothing else does.
An earlier version ranked by intake similarity first and used distance only as
a tiebreak. That put a Catholic school 2.9 miles away above the community
school 0.3 miles down the road, and — because the row filled from the best tier
before widening — filled all six slots with faith matches while omitting every
school a parent could actually walk to. For a primary, a school that far is not
a weaker option; it is not an option. Distance is a constraint and intake is a
preference, and the ranking now says so.
Similarity survives as `shared`: what a candidate genuinely has in common with
this school, reported on its card, so a reader applies their own weighting
instead of having ours applied for them.
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
"""
from __future__ import annotations
import re
import numpy as np
import pandas as pd
from .schemas import PHASE_GROUPS
# Three fit the row; the rest are behind the carousel arrows.
MAX_SCHOOLS = 6
MINIMUM = 2
# How far the section will reach, in miles, when nothing closer exists.
#
# A sanity bound rather than a target: ordering by distance already handles
# density, so a school in a dense area fills all six slots inside a mile and
# never sees this. It decides one thing — what happens where the area is
# sparse — and the answer differs by phase because catchments do. Primary
# catchments are routinely under a mile; beyond two, a primary is not a weaker
# option but not an option, and no section is the honest answer.
PRIMARY_RADIUS_MILES = 2.0
SECONDARY_RADIUS_MILES = 6.0
POST16_RADIUS_MILES = 10.0
EARTH_RADIUS_MILES = 3958.8
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
# Values that mean "this school has no religious character".
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
def is_special_provision(school_type: str | None) -> bool:
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
Special schools carry a mainstream phase, so phase alone cannot identify
them. The two implementations must agree: a school the frontend treats as
special for benchmarking but this treats as mainstream would be dropped
from its own England comparison and then offered as a peer to a mainstream
school on the next page along.
"""
return bool(_SPECIAL.search(school_type or ""))
def is_selective(admissions_policy: str | None) -> bool:
"""Strictly selective. Unknown counts as non-selective, which is the safe
direction: it can only ever exclude a pairing, never invent one."""
return (admissions_policy or "").strip().lower() == "selective"
def faith_key(denomination: str | None) -> str:
value = (denomination or "").strip().lower()
return "" if value in _NO_FAITH else value
def faith_label(denomination: str | None) -> str:
return denomination.strip() if faith_key(denomination) else "No religious character"
def genders_compatible(a: str | None, b: str | None) -> bool:
single = {"boys", "girls"}
left, right = (a or "").strip().lower(), (b or "").strip().lower()
return not (left in single and right in single and left != right)
def is_secondary_phase(phase: str | None) -> bool:
"""Whether this phase takes the secondary side: secondary group membership,
minus all-through.
Membership is read from PHASE_GROUPS rather than tested with `"secondary" in
phase`, because that substring misses "16 plus" — GIAS phase 6, which
PHASE_GROUPS deliberately files as secondary. The substring version fails
silently rather than loudly: a sixth-form college is simply handed the
primary bucket and offered infant schools as peers.
All-through is the exception. PHASE_GROUPS lists it on both sides because it
belongs on both phases' place pages, but the detail page renders it with the
primary template, and the metric follows the template.
"""
text = (phase or "").strip().lower()
return text != "all-through" and text in PHASE_GROUPS["secondary"]
def radius_miles(phase: str | None) -> float:
"""How far this phase's section will reach when nothing closer exists."""
if (phase or "").strip().lower() == "16 plus":
return POST16_RADIUS_MILES
return SECONDARY_RADIUS_MILES if is_secondary_phase(phase) else PRIMARY_RADIUS_MILES
def _phase_group(is_secondary: bool) -> set[str]:
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
"""Vectorised, matching the postcode search in app.py."""
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
def _native(value):
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
the caller never has to remember to."""
if value is None:
return None
if isinstance(value, np.generic):
value = value.item()
if isinstance(value, float) and np.isnan(value):
return None
return value
def _mask(series: pd.Series, predicate) -> pd.Series:
"""A boolean mask that survives an empty frame.
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
that as a mask silently drops every column — so the next column lookup
raises KeyError rather than yielding no rows. This is not hypothetical: a
special school with no special school near it empties the frame at the
provision filter, which is the ordinary case for most special schools.
"""
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
def _shared(subject: pd.Series, candidate: pd.Series, is_secondary: bool) -> list[str]:
"""What this candidate genuinely has in common with the subject.
Empty is a real answer, and renders no chips at all. A card claiming a
shared characteristic it does not have would be worse than a bare one —
and since these no longer affect the order, an empty list costs the school
nothing but its place in the row, which distance already decided.
"""
shared: list[str] = []
gender = str(subject.get("gender") or "").strip()
if gender and str(candidate.get("gender") or "").strip().lower() == gender.lower():
shared.append(gender)
if is_secondary:
policy = str(candidate.get("admissions_policy") or "").strip()
subject_policy = str(subject.get("admissions_policy") or "").strip()
if (
policy
and policy.lower() == subject_policy.lower()
and policy.lower() not in {"not applicable", "unknown"}
):
shared.append(policy)
if faith_key(candidate.get("religious_denomination")) == faith_key(
subject.get("religious_denomination")
):
shared.append(faith_label(candidate.get("religious_denomination")))
return shared
def select_similar(frame: pd.DataFrame, urn: int) -> list[dict]:
"""The nearest eligible schools, closest first — at most MAX_SCHOOLS, and
none at all below MINIMUM.
The phase is read from the subject's own row rather than passed in, so a
caller cannot hand this a phase that disagrees with the data it selects
from.
"""
subject_rows = frame[frame["urn"] == urn]
if subject_rows.empty:
return []
subject = subject_rows.iloc[0]
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
if lat is None or lon is None:
return []
phase = subject.get("phase")
is_secondary = is_secondary_phase(phase)
reach = radius_miles(phase)
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
candidates = frame[frame["urn"] != urn].copy()
for column in ("latitude", "longitude"):
candidates = candidates[candidates[column].notna()]
if candidates.empty:
return []
# ── Hard filters ────────────────────────────────────────────────────
allowed_phases = _phase_group(is_secondary)
candidates = candidates[
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
]
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
subject_special = is_special_provision(subject.get("school_type"))
special = _mask(candidates["school_type"], is_special_provision)
candidates = candidates[special if subject_special else ~special]
subject_selective = is_selective(subject.get("admissions_policy"))
selective = _mask(candidates["admissions_policy"], is_selective)
candidates = candidates[selective if subject_selective else ~selective]
subject_gender = subject.get("gender")
candidates = candidates[
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
]
if candidates.empty:
return []
candidates["distance_miles"] = _haversine_miles(
lat, lon, candidates["latitude"].values, candidates["longitude"].values
).round(1)
# ── Nearest first, and nothing else has a say ───────────────────────
within = candidates[candidates["distance_miles"] <= reach]
if len(within) < MINIMUM:
return []
selected = within.sort_values(["distance_miles", "urn"]).head(MAX_SCHOOLS)
return [
{
"urn": int(row["urn"]),
"school_name": str(row.get("school_name") or ""),
"distance_miles": float(row["distance_miles"]),
"school_type": _native(row.get("school_type")),
"age_range": _native(row.get("age_range")),
"shared": _shared(subject, row, is_secondary),
"metric_value": _native(row.get(metric_key)),
"metric_key": metric_key,
"metric_year": _native(row.get("year")),
}
for _, row in selected.iterrows()
]