"""Which nearby schools a detail page may offer as alternatives. HARD FILTERS decide eligibility, and encode claims the section is not allowed to make. A selective school is not an alternative to a non-selective one, a special school is not comparable to a mainstream one, and a Girls school is not an option for a Boys school's reader. They never relax, at any distance, even where that means the section does not render at all. DISTANCE decides the order, and nothing else does. An earlier version ranked by intake similarity first and used distance only as a tiebreak. That put a Catholic school 2.9 miles away above the community school 0.3 miles down the road, and — because the row filled from the best tier before widening — filled all six slots with faith matches while omitting every school a parent could actually walk to. For a primary, a school that far is not a weaker option; it is not an option. Distance is a constraint and intake is a preference, and the ranking now says so. Similarity survives as `shared`: what a candidate genuinely has in common with this school, reported on its card, so a reader applies their own weighting instead of having ours applied for them. Pure functions over a DataFrame: no I/O, no FastAPI, no database. """ from __future__ import annotations import re import numpy as np import pandas as pd from .schemas import PHASE_GROUPS # Three fit the row; the rest are behind the carousel arrows. MAX_SCHOOLS = 6 MINIMUM = 2 # How far the section will reach, in miles, when nothing closer exists. # # A sanity bound rather than a target: ordering by distance already handles # density, so a school in a dense area fills all six slots inside a mile and # never sees this. It decides one thing — what happens where the area is # sparse — and the answer differs by phase because catchments do. Primary # catchments are routinely under a mile; beyond two, a primary is not a weaker # option but not an option, and no section is the honest answer. PRIMARY_RADIUS_MILES = 2.0 SECONDARY_RADIUS_MILES = 6.0 POST16_RADIUS_MILES = 10.0 EARTH_RADIUS_MILES = 3958.8 _SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I) # Values that mean "this school has no religious character". _NO_FAITH = {"", "none", "does not apply", "not applicable"} def is_special_provision(school_type: str | None) -> bool: """Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts. Special schools carry a mainstream phase, so phase alone cannot identify them. The two implementations must agree: a school the frontend treats as special for benchmarking but this treats as mainstream would be dropped from its own England comparison and then offered as a peer to a mainstream school on the next page along. """ return bool(_SPECIAL.search(school_type or "")) def is_selective(admissions_policy: str | None) -> bool: """Strictly selective. Unknown counts as non-selective, which is the safe direction: it can only ever exclude a pairing, never invent one.""" return (admissions_policy or "").strip().lower() == "selective" def faith_key(denomination: str | None) -> str: value = (denomination or "").strip().lower() return "" if value in _NO_FAITH else value def faith_label(denomination: str | None) -> str: return denomination.strip() if faith_key(denomination) else "No religious character" def genders_compatible(a: str | None, b: str | None) -> bool: single = {"boys", "girls"} left, right = (a or "").strip().lower(), (b or "").strip().lower() return not (left in single and right in single and left != right) def is_secondary_phase(phase: str | None) -> bool: """Whether this phase takes the secondary side: secondary group membership, minus all-through. Membership is read from PHASE_GROUPS rather than tested with `"secondary" in phase`, because that substring misses "16 plus" — GIAS phase 6, which PHASE_GROUPS deliberately files as secondary. The substring version fails silently rather than loudly: a sixth-form college is simply handed the primary bucket and offered infant schools as peers. All-through is the exception. PHASE_GROUPS lists it on both sides because it belongs on both phases' place pages, but the detail page renders it with the primary template, and the metric follows the template. """ text = (phase or "").strip().lower() return text != "all-through" and text in PHASE_GROUPS["secondary"] def radius_miles(phase: str | None) -> float: """How far this phase's section will reach when nothing closer exists.""" if (phase or "").strip().lower() == "16 plus": return POST16_RADIUS_MILES return SECONDARY_RADIUS_MILES if is_secondary_phase(phase) else PRIMARY_RADIUS_MILES def _phase_group(is_secondary: bool) -> set[str]: return PHASE_GROUPS["secondary" if is_secondary else "primary"] def _haversine_miles(lat1: float, lon1: float, lat2, lon2): """Vectorised, matching the postcode search in app.py.""" lat1_r, lon1_r = np.radians(lat1), np.radians(lon1) lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float)) dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2 return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a)) def _native(value): """NaN and numpy scalars both reach JSONResponse badly; normalise here so the caller never has to remember to.""" if value is None: return None if isinstance(value, np.generic): value = value.item() if isinstance(value, float) and np.isnan(value): return None return value def _mask(series: pd.Series, predicate) -> pd.Series: """A boolean mask that survives an empty frame. `Series.apply` on an empty Series returns an empty *DataFrame*, and using that as a mask silently drops every column — so the next column lookup raises KeyError rather than yielding no rows. This is not hypothetical: a special school with no special school near it empties the frame at the provision filter, which is the ordinary case for most special schools. """ return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool) def _shared(subject: pd.Series, candidate: pd.Series, is_secondary: bool) -> list[str]: """What this candidate genuinely has in common with the subject. Empty is a real answer, and renders no chips at all. A card claiming a shared characteristic it does not have would be worse than a bare one — and since these no longer affect the order, an empty list costs the school nothing but its place in the row, which distance already decided. """ shared: list[str] = [] gender = str(subject.get("gender") or "").strip() if gender and str(candidate.get("gender") or "").strip().lower() == gender.lower(): shared.append(gender) if is_secondary: policy = str(candidate.get("admissions_policy") or "").strip() subject_policy = str(subject.get("admissions_policy") or "").strip() if ( policy and policy.lower() == subject_policy.lower() and policy.lower() not in {"not applicable", "unknown"} ): shared.append(policy) if faith_key(candidate.get("religious_denomination")) == faith_key( subject.get("religious_denomination") ): shared.append(faith_label(candidate.get("religious_denomination"))) return shared def select_nearby(frame: pd.DataFrame, urn: int) -> list[dict]: """The nearest eligible schools, closest first — at most MAX_SCHOOLS, and none at all below MINIMUM. The phase is read from the subject's own row rather than passed in, so a caller cannot hand this a phase that disagrees with the data it selects from. """ subject_rows = frame[frame["urn"] == urn] if subject_rows.empty: return [] subject = subject_rows.iloc[0] lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude")) if lat is None or lon is None: return [] phase = subject.get("phase") is_secondary = is_secondary_phase(phase) reach = radius_miles(phase) metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct" candidates = frame[frame["urn"] != urn].copy() for column in ("latitude", "longitude"): candidates = candidates[candidates[column].notna()] if candidates.empty: return [] # ── Hard filters ──────────────────────────────────────────────────── allowed_phases = _phase_group(is_secondary) candidates = candidates[ candidates["phase"].fillna("").str.lower().isin(allowed_phases) ] candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")] subject_special = is_special_provision(subject.get("school_type")) special = _mask(candidates["school_type"], is_special_provision) candidates = candidates[special if subject_special else ~special] subject_selective = is_selective(subject.get("admissions_policy")) selective = _mask(candidates["admissions_policy"], is_selective) candidates = candidates[selective if subject_selective else ~selective] subject_gender = subject.get("gender") candidates = candidates[ _mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g)) ] if candidates.empty: return [] candidates["distance_miles"] = _haversine_miles( lat, lon, candidates["latitude"].values, candidates["longitude"].values ).round(1) # ── Nearest first, and nothing else has a say ─────────────────────── within = candidates[candidates["distance_miles"] <= reach] if len(within) < MINIMUM: return [] selected = within.sort_values(["distance_miles", "urn"]).head(MAX_SCHOOLS) return [ { "urn": int(row["urn"]), "school_name": str(row.get("school_name") or ""), "distance_miles": float(row["distance_miles"]), "school_type": _native(row.get("school_type")), "age_range": _native(row.get("age_range")), # Each peer's own phase, not the subject's: the pool is a phase # group, so an all-through school can sit beside a primary. The # compare basket counts it against both of its tabs. "phase": _native(row.get("phase")), "shared": _shared(subject, row, is_secondary), "metric_value": _native(row.get(metric_key)), "metric_key": metric_key, "metric_year": _native(row.get("year")), } for _, row in selected.iterrows() ]