2026-09-21 22:40:10 +01:00
|
|
|
"""Which nearby schools a detail page may offer as alternatives.
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
HARD FILTERS decide eligibility, and encode claims the section is not allowed
|
|
|
|
|
to make. A selective school is not an alternative to a non-selective one, a
|
|
|
|
|
special school is not comparable to a mainstream one, and a Girls school is not
|
|
|
|
|
an option for a Boys school's reader. They never relax, at any distance, even
|
|
|
|
|
where that means the section does not render at all.
|
2026-09-21 22:40:10 +01:00
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
DISTANCE decides the order, and nothing else does.
|
2026-09-21 22:40:10 +01:00
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
An earlier version ranked by intake similarity first and used distance only as
|
|
|
|
|
a tiebreak. That put a Catholic school 2.9 miles away above the community
|
|
|
|
|
school 0.3 miles down the road, and — because the row filled from the best tier
|
|
|
|
|
before widening — filled all six slots with faith matches while omitting every
|
|
|
|
|
school a parent could actually walk to. For a primary, a school that far is not
|
|
|
|
|
a weaker option; it is not an option. Distance is a constraint and intake is a
|
|
|
|
|
preference, and the ranking now says so.
|
|
|
|
|
|
|
|
|
|
Similarity survives as `shared`: what a candidate genuinely has in common with
|
|
|
|
|
this school, reported on its card, so a reader applies their own weighting
|
|
|
|
|
instead of having ours applied for them.
|
2026-09-21 22:40:10 +01:00
|
|
|
|
|
|
|
|
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
import numpy as np
|
|
|
|
|
import pandas as pd
|
|
|
|
|
|
|
|
|
|
from .schemas import PHASE_GROUPS
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
# Three fit the row; the rest are behind the carousel arrows.
|
2026-09-21 22:40:10 +01:00
|
|
|
MAX_SCHOOLS = 6
|
|
|
|
|
MINIMUM = 2
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
# How far the section will reach, in miles, when nothing closer exists.
|
|
|
|
|
#
|
|
|
|
|
# A sanity bound rather than a target: ordering by distance already handles
|
|
|
|
|
# density, so a school in a dense area fills all six slots inside a mile and
|
|
|
|
|
# never sees this. It decides one thing — what happens where the area is
|
|
|
|
|
# sparse — and the answer differs by phase because catchments do. Primary
|
|
|
|
|
# catchments are routinely under a mile; beyond two, a primary is not a weaker
|
|
|
|
|
# option but not an option, and no section is the honest answer.
|
|
|
|
|
PRIMARY_RADIUS_MILES = 2.0
|
|
|
|
|
SECONDARY_RADIUS_MILES = 6.0
|
|
|
|
|
POST16_RADIUS_MILES = 10.0
|
2026-09-21 22:40:10 +01:00
|
|
|
|
|
|
|
|
EARTH_RADIUS_MILES = 3958.8
|
|
|
|
|
|
|
|
|
|
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
|
|
|
|
|
|
|
|
|
|
# Values that mean "this school has no religious character".
|
|
|
|
|
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def is_special_provision(school_type: str | None) -> bool:
|
|
|
|
|
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
|
|
|
|
|
|
|
|
|
|
Special schools carry a mainstream phase, so phase alone cannot identify
|
|
|
|
|
them. The two implementations must agree: a school the frontend treats as
|
|
|
|
|
special for benchmarking but this treats as mainstream would be dropped
|
|
|
|
|
from its own England comparison and then offered as a peer to a mainstream
|
|
|
|
|
school on the next page along.
|
|
|
|
|
"""
|
|
|
|
|
return bool(_SPECIAL.search(school_type or ""))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def is_selective(admissions_policy: str | None) -> bool:
|
|
|
|
|
"""Strictly selective. Unknown counts as non-selective, which is the safe
|
|
|
|
|
direction: it can only ever exclude a pairing, never invent one."""
|
|
|
|
|
return (admissions_policy or "").strip().lower() == "selective"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def faith_key(denomination: str | None) -> str:
|
|
|
|
|
value = (denomination or "").strip().lower()
|
|
|
|
|
return "" if value in _NO_FAITH else value
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def faith_label(denomination: str | None) -> str:
|
|
|
|
|
return denomination.strip() if faith_key(denomination) else "No religious character"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def genders_compatible(a: str | None, b: str | None) -> bool:
|
|
|
|
|
single = {"boys", "girls"}
|
|
|
|
|
left, right = (a or "").strip().lower(), (b or "").strip().lower()
|
|
|
|
|
return not (left in single and right in single and left != right)
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 06:41:49 +01:00
|
|
|
def is_secondary_phase(phase: str | None) -> bool:
|
|
|
|
|
"""Whether this phase takes the secondary side: secondary group membership,
|
|
|
|
|
minus all-through.
|
|
|
|
|
|
|
|
|
|
Membership is read from PHASE_GROUPS rather than tested with `"secondary" in
|
|
|
|
|
phase`, because that substring misses "16 plus" — GIAS phase 6, which
|
|
|
|
|
PHASE_GROUPS deliberately files as secondary. The substring version fails
|
|
|
|
|
silently rather than loudly: a sixth-form college is simply handed the
|
|
|
|
|
primary bucket and offered infant schools as peers.
|
|
|
|
|
|
|
|
|
|
All-through is the exception. PHASE_GROUPS lists it on both sides because it
|
|
|
|
|
belongs on both phases' place pages, but the detail page renders it with the
|
|
|
|
|
primary template, and the metric follows the template.
|
|
|
|
|
"""
|
|
|
|
|
text = (phase or "").strip().lower()
|
|
|
|
|
return text != "all-through" and text in PHASE_GROUPS["secondary"]
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
def radius_miles(phase: str | None) -> float:
|
|
|
|
|
"""How far this phase's section will reach when nothing closer exists."""
|
|
|
|
|
if (phase or "").strip().lower() == "16 plus":
|
|
|
|
|
return POST16_RADIUS_MILES
|
|
|
|
|
return SECONDARY_RADIUS_MILES if is_secondary_phase(phase) else PRIMARY_RADIUS_MILES
|
|
|
|
|
|
|
|
|
|
|
2026-09-21 22:40:10 +01:00
|
|
|
def _phase_group(is_secondary: bool) -> set[str]:
|
|
|
|
|
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
|
|
|
|
|
"""Vectorised, matching the postcode search in app.py."""
|
|
|
|
|
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
|
|
|
|
|
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
|
|
|
|
|
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
|
|
|
|
|
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
|
|
|
|
|
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _native(value):
|
|
|
|
|
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
|
|
|
|
|
the caller never has to remember to."""
|
|
|
|
|
if value is None:
|
|
|
|
|
return None
|
|
|
|
|
if isinstance(value, np.generic):
|
|
|
|
|
value = value.item()
|
|
|
|
|
if isinstance(value, float) and np.isnan(value):
|
|
|
|
|
return None
|
|
|
|
|
return value
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _mask(series: pd.Series, predicate) -> pd.Series:
|
|
|
|
|
"""A boolean mask that survives an empty frame.
|
|
|
|
|
|
|
|
|
|
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
|
|
|
|
|
that as a mask silently drops every column — so the next column lookup
|
|
|
|
|
raises KeyError rather than yielding no rows. This is not hypothetical: a
|
|
|
|
|
special school with no special school near it empties the frame at the
|
|
|
|
|
provision filter, which is the ordinary case for most special schools.
|
|
|
|
|
"""
|
|
|
|
|
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
|
|
|
|
|
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
def _shared(subject: pd.Series, candidate: pd.Series, is_secondary: bool) -> list[str]:
|
|
|
|
|
"""What this candidate genuinely has in common with the subject.
|
|
|
|
|
|
|
|
|
|
Empty is a real answer, and renders no chips at all. A card claiming a
|
|
|
|
|
shared characteristic it does not have would be worse than a bare one —
|
|
|
|
|
and since these no longer affect the order, an empty list costs the school
|
|
|
|
|
nothing but its place in the row, which distance already decided.
|
|
|
|
|
"""
|
|
|
|
|
shared: list[str] = []
|
|
|
|
|
|
|
|
|
|
gender = str(subject.get("gender") or "").strip()
|
|
|
|
|
if gender and str(candidate.get("gender") or "").strip().lower() == gender.lower():
|
|
|
|
|
shared.append(gender)
|
2026-09-21 22:40:10 +01:00
|
|
|
|
|
|
|
|
if is_secondary:
|
2026-09-22 11:58:36 +01:00
|
|
|
policy = str(candidate.get("admissions_policy") or "").strip()
|
|
|
|
|
subject_policy = str(subject.get("admissions_policy") or "").strip()
|
|
|
|
|
if (
|
|
|
|
|
policy
|
|
|
|
|
and policy.lower() == subject_policy.lower()
|
|
|
|
|
and policy.lower() not in {"not applicable", "unknown"}
|
|
|
|
|
):
|
|
|
|
|
shared.append(policy)
|
|
|
|
|
|
|
|
|
|
if faith_key(candidate.get("religious_denomination")) == faith_key(
|
|
|
|
|
subject.get("religious_denomination")
|
|
|
|
|
):
|
|
|
|
|
shared.append(faith_label(candidate.get("religious_denomination")))
|
|
|
|
|
|
|
|
|
|
return shared
|
2026-09-21 22:40:10 +01:00
|
|
|
|
|
|
|
|
|
2026-09-22 11:59:27 +01:00
|
|
|
def select_nearby(frame: pd.DataFrame, urn: int) -> list[dict]:
|
2026-09-22 11:58:36 +01:00
|
|
|
"""The nearest eligible schools, closest first — at most MAX_SCHOOLS, and
|
|
|
|
|
none at all below MINIMUM.
|
2026-09-21 22:40:10 +01:00
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
The phase is read from the subject's own row rather than passed in, so a
|
|
|
|
|
caller cannot hand this a phase that disagrees with the data it selects
|
|
|
|
|
from.
|
2026-09-21 22:40:10 +01:00
|
|
|
"""
|
|
|
|
|
subject_rows = frame[frame["urn"] == urn]
|
|
|
|
|
if subject_rows.empty:
|
|
|
|
|
return []
|
|
|
|
|
subject = subject_rows.iloc[0]
|
|
|
|
|
|
|
|
|
|
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
|
|
|
|
|
if lat is None or lon is None:
|
|
|
|
|
return []
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
phase = subject.get("phase")
|
|
|
|
|
is_secondary = is_secondary_phase(phase)
|
|
|
|
|
reach = radius_miles(phase)
|
2026-09-21 22:40:10 +01:00
|
|
|
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
|
|
|
|
|
|
|
|
|
|
candidates = frame[frame["urn"] != urn].copy()
|
|
|
|
|
for column in ("latitude", "longitude"):
|
|
|
|
|
candidates = candidates[candidates[column].notna()]
|
|
|
|
|
if candidates.empty:
|
|
|
|
|
return []
|
|
|
|
|
|
|
|
|
|
# ── Hard filters ────────────────────────────────────────────────────
|
|
|
|
|
allowed_phases = _phase_group(is_secondary)
|
|
|
|
|
candidates = candidates[
|
|
|
|
|
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
|
|
|
|
|
]
|
|
|
|
|
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
|
|
|
|
|
|
|
|
|
|
subject_special = is_special_provision(subject.get("school_type"))
|
|
|
|
|
special = _mask(candidates["school_type"], is_special_provision)
|
|
|
|
|
candidates = candidates[special if subject_special else ~special]
|
|
|
|
|
|
|
|
|
|
subject_selective = is_selective(subject.get("admissions_policy"))
|
|
|
|
|
selective = _mask(candidates["admissions_policy"], is_selective)
|
|
|
|
|
candidates = candidates[selective if subject_selective else ~selective]
|
|
|
|
|
|
|
|
|
|
subject_gender = subject.get("gender")
|
|
|
|
|
candidates = candidates[
|
|
|
|
|
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
|
|
|
|
|
]
|
|
|
|
|
if candidates.empty:
|
|
|
|
|
return []
|
|
|
|
|
|
|
|
|
|
candidates["distance_miles"] = _haversine_miles(
|
|
|
|
|
lat, lon, candidates["latitude"].values, candidates["longitude"].values
|
|
|
|
|
).round(1)
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
# ── Nearest first, and nothing else has a say ───────────────────────
|
|
|
|
|
within = candidates[candidates["distance_miles"] <= reach]
|
|
|
|
|
if len(within) < MINIMUM:
|
2026-09-21 22:40:10 +01:00
|
|
|
return []
|
|
|
|
|
|
2026-09-22 11:58:36 +01:00
|
|
|
selected = within.sort_values(["distance_miles", "urn"]).head(MAX_SCHOOLS)
|
2026-09-21 22:40:10 +01:00
|
|
|
return [
|
|
|
|
|
{
|
|
|
|
|
"urn": int(row["urn"]),
|
|
|
|
|
"school_name": str(row.get("school_name") or ""),
|
|
|
|
|
"distance_miles": float(row["distance_miles"]),
|
|
|
|
|
"school_type": _native(row.get("school_type")),
|
|
|
|
|
"age_range": _native(row.get("age_range")),
|
2026-09-30 12:04:39 +01:00
|
|
|
# Each peer's own phase, not the subject's: the pool is a phase
|
|
|
|
|
# group, so an all-through school can sit beside a primary. The
|
|
|
|
|
# compare basket counts it against both of its tabs.
|
|
|
|
|
"phase": _native(row.get("phase")),
|
2026-09-22 11:58:36 +01:00
|
|
|
"shared": _shared(subject, row, is_secondary),
|
2026-09-21 22:40:10 +01:00
|
|
|
"metric_value": _native(row.get(metric_key)),
|
|
|
|
|
"metric_key": metric_key,
|
|
|
|
|
"metric_year": _native(row.get("year")),
|
|
|
|
|
}
|
2026-09-22 11:58:36 +01:00
|
|
|
for _, row in selected.iterrows()
|
2026-09-21 22:40:10 +01:00
|
|
|
]
|