Files
school_compare/backend/similar_schools.py
T
TudorandClaude Opus 5 b571d9c549 feat(api): decide which nearby schools a page may offer
Hard filters encode claims the section may not make — a selective school is
not an alternative to a non-selective one, a special school is not comparable
to a mainstream one, a Girls school is not an option for a Boys school's
reader — so they never relax. Soft preferences describe closeness of fit, so
they relax across three tiers, and only far enough to reach three; the
remaining slots up to six fill from the tiers already opened.

PHASE_GROUPS moves to schemas.py so this module can share it without
importing app, which would be a cycle.

_mask() exists because Series.apply on an empty Series returns a DataFrame,
and using that as a mask drops every column — so the next lookup raises
KeyError instead of yielding no rows. A special school with no special school
near it empties the frame at the provision filter, which is the ordinary case
for most special schools, so this was a crash on a common path rather than an
edge case.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-21 22:40:10 +01:00

244 lines
9.5 KiB
Python

"""Which nearby schools a detail page may offer as alternatives.
Two kinds of rule, and they are not interchangeable.
HARD FILTERS encode claims the section is not allowed to make. A selective
school is not an alternative to a non-selective one, a special school is not
comparable to a mainstream one, and a Girls school is not an option for a Boys
school's reader. These never relax, at any distance, even where that means the
section does not render at all.
SOFT PREFERENCES describe how closely an intake resembles this school's. They
relax in tiers, and every card reports the tier that actually took it so the
page can say what is shared rather than implying more. They relax only far
enough to reach a usable set, never far enough to fill the last of the slots.
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
"""
from __future__ import annotations
import re
import numpy as np
import pandas as pd
from .schemas import PHASE_GROUPS
# A cap, not a quota: the section shows everything that qualified at the tiers
# it used, up to this many. Three fit the row; the rest are behind the arrows.
MAX_SCHOOLS = 6
# Tiers stop relaxing once this many have been found. Without it, a cap of six
# would reliably drag in tier-3 schools ten miles away to fill a row that three
# good matches had already earned.
ENOUGH = 3
MINIMUM = 2
# (tier, radius in miles). Faith relaxes before gender: a faith mismatch
# changes the character of a school, while a gender mismatch can mean the
# school is not available to this reader's child at all.
TIERS: tuple[tuple[int, float], ...] = ((1, 3.0), (2, 5.0), (3, 10.0))
EARTH_RADIUS_MILES = 3958.8
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
# Values that mean "this school has no religious character".
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
def is_special_provision(school_type: str | None) -> bool:
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
Special schools carry a mainstream phase, so phase alone cannot identify
them. The two implementations must agree: a school the frontend treats as
special for benchmarking but this treats as mainstream would be dropped
from its own England comparison and then offered as a peer to a mainstream
school on the next page along.
"""
return bool(_SPECIAL.search(school_type or ""))
def is_selective(admissions_policy: str | None) -> bool:
"""Strictly selective. Unknown counts as non-selective, which is the safe
direction: it can only ever exclude a pairing, never invent one."""
return (admissions_policy or "").strip().lower() == "selective"
def faith_key(denomination: str | None) -> str:
value = (denomination or "").strip().lower()
return "" if value in _NO_FAITH else value
def faith_label(denomination: str | None) -> str:
return denomination.strip() if faith_key(denomination) else "No religious character"
def genders_compatible(a: str | None, b: str | None) -> bool:
single = {"boys", "girls"}
left, right = (a or "").strip().lower(), (b or "").strip().lower()
return not (left in single and right in single and left != right)
def phase_label(phase: str | None) -> str:
text = (phase or "").strip()
if not text:
return "School"
if text.lower() == "all-through":
return "All-through school"
return f"{text.capitalize()} school"
def _phase_group(is_secondary: bool) -> set[str]:
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
"""Vectorised, matching the postcode search in app.py."""
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
def _native(value):
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
the caller never has to remember to."""
if value is None:
return None
if isinstance(value, np.generic):
value = value.item()
if isinstance(value, float) and np.isnan(value):
return None
return value
def _mask(series: pd.Series, predicate) -> pd.Series:
"""A boolean mask that survives an empty frame.
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
that as a mask silently drops every column — so the next column lookup
raises KeyError rather than yielding no rows. This is not hypothetical: a
special school with no special school near it empties the frame at the
provision filter, which is the ordinary case for most special schools.
"""
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
def _chips(subject: pd.Series, candidate: pd.Series, tier: int, is_secondary: bool) -> list[str]:
if tier >= 3:
return [phase_label(candidate.get("phase"))]
chips = [str(subject.get("gender") or "").strip()]
if is_secondary:
policy = (candidate.get("admissions_policy") or "").strip()
if policy and policy.lower() not in {"not applicable", "unknown"}:
chips.append(policy)
if tier == 1:
chips.append(faith_label(candidate.get("religious_denomination")))
return [chip for chip in chips if chip]
def select_similar(frame: pd.DataFrame, urn: int, is_secondary: bool) -> list[dict]:
"""Up to MAX_SCHOOLS nearby schools this page may offer, or [] below MINIMUM.
Selected by tier, displayed by distance: the tier decides which schools
earn a slot, and the render order is then closest-first, because "nearby"
is the promise in the heading.
"""
subject_rows = frame[frame["urn"] == urn]
if subject_rows.empty:
return []
subject = subject_rows.iloc[0]
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
if lat is None or lon is None:
return []
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
candidates = frame[frame["urn"] != urn].copy()
for column in ("latitude", "longitude"):
candidates = candidates[candidates[column].notna()]
if candidates.empty:
return []
# ── Hard filters ────────────────────────────────────────────────────
allowed_phases = _phase_group(is_secondary)
candidates = candidates[
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
]
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
subject_special = is_special_provision(subject.get("school_type"))
special = _mask(candidates["school_type"], is_special_provision)
candidates = candidates[special if subject_special else ~special]
subject_selective = is_selective(subject.get("admissions_policy"))
selective = _mask(candidates["admissions_policy"], is_selective)
candidates = candidates[selective if subject_selective else ~selective]
subject_gender = subject.get("gender")
candidates = candidates[
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
]
if candidates.empty:
return []
candidates["distance_miles"] = _haversine_miles(
lat, lon, candidates["latitude"].values, candidates["longitude"].values
).round(1)
# ── Soft preferences, in tiers ──────────────────────────────────────
subject_faith = faith_key(subject.get("religious_denomination"))
subject_gender_key = (subject_gender or "").strip().lower()
same_gender = candidates["gender"].fillna("").str.strip().str.lower() == subject_gender_key
same_faith = _mask(
candidates["religious_denomination"], lambda d: faith_key(d) == subject_faith
)
tier_masks = {
1: same_gender & same_faith,
2: same_gender,
3: pd.Series(True, index=candidates.index),
}
# Descend the tiers only until the set reaches ENOUGH. The tier that gets
# there is the last one opened, and the remaining slots up to MAX_SCHOOLS
# are filled from the tiers already used — never by widening again.
picked: dict[int, tuple[int, pd.Series]] = {}
for tier, radius in TIERS:
within = candidates[tier_masks[tier] & (candidates["distance_miles"] <= radius)]
for _, row in within.sort_values("distance_miles").iterrows():
candidate_urn = int(row["urn"])
if candidate_urn in picked:
continue
picked[candidate_urn] = (tier, row)
if len(picked) >= MAX_SCHOOLS:
break
if len(picked) >= ENOUGH:
break
if len(picked) < MINIMUM:
return []
selected = sorted(
picked.values(), key=lambda pair: float(pair[1]["distance_miles"])
)[:MAX_SCHOOLS]
return [
{
"urn": int(row["urn"]),
"school_name": str(row.get("school_name") or ""),
"distance_miles": float(row["distance_miles"]),
"school_type": _native(row.get("school_type")),
"age_range": _native(row.get("age_range")),
"shared": _chips(subject, row, tier, is_secondary),
"tier": tier,
"metric_value": _native(row.get(metric_key)),
"metric_key": metric_key,
"metric_year": _native(row.get("year")),
}
for tier, row in selected
]