"""Which nearby schools a detail page may offer as alternatives. Two kinds of rule, and they are not interchangeable. HARD FILTERS encode claims the section is not allowed to make. A selective school is not an alternative to a non-selective one, a special school is not comparable to a mainstream one, and a Girls school is not an option for a Boys school's reader. These never relax, at any distance, even where that means the section does not render at all. SOFT PREFERENCES describe how closely an intake resembles this school's. They relax in tiers, and every card reports the tier that actually took it so the page can say what is shared rather than implying more. They relax only far enough to reach a usable set, never far enough to fill the last of the slots. Pure functions over a DataFrame: no I/O, no FastAPI, no database. """ from __future__ import annotations import re import numpy as np import pandas as pd from .schemas import PHASE_GROUPS # A cap, not a quota: the section shows everything that qualified at the tiers # it used, up to this many. Three fit the row; the rest are behind the arrows. MAX_SCHOOLS = 6 # Tiers stop relaxing once this many have been found. Without it, a cap of six # would reliably drag in tier-3 schools ten miles away to fill a row that three # good matches had already earned. ENOUGH = 3 MINIMUM = 2 # (tier, radius in miles). Faith relaxes before gender: a faith mismatch # changes the character of a school, while a gender mismatch can mean the # school is not available to this reader's child at all. TIERS: tuple[tuple[int, float], ...] = ((1, 3.0), (2, 5.0), (3, 10.0)) EARTH_RADIUS_MILES = 3958.8 _SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I) # Values that mean "this school has no religious character". _NO_FAITH = {"", "none", "does not apply", "not applicable"} def is_special_provision(school_type: str | None) -> bool: """Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts. Special schools carry a mainstream phase, so phase alone cannot identify them. The two implementations must agree: a school the frontend treats as special for benchmarking but this treats as mainstream would be dropped from its own England comparison and then offered as a peer to a mainstream school on the next page along. """ return bool(_SPECIAL.search(school_type or "")) def is_selective(admissions_policy: str | None) -> bool: """Strictly selective. Unknown counts as non-selective, which is the safe direction: it can only ever exclude a pairing, never invent one.""" return (admissions_policy or "").strip().lower() == "selective" def faith_key(denomination: str | None) -> str: value = (denomination or "").strip().lower() return "" if value in _NO_FAITH else value def faith_label(denomination: str | None) -> str: return denomination.strip() if faith_key(denomination) else "No religious character" def genders_compatible(a: str | None, b: str | None) -> bool: single = {"boys", "girls"} left, right = (a or "").strip().lower(), (b or "").strip().lower() return not (left in single and right in single and left != right) def phase_label(phase: str | None) -> str: text = (phase or "").strip() if not text: return "School" if text.lower() == "all-through": return "All-through school" return f"{text.capitalize()} school" def is_secondary_phase(phase: str | None) -> bool: """Whether this phase takes the secondary side: secondary group membership, minus all-through. Membership is read from PHASE_GROUPS rather than tested with `"secondary" in phase`, because that substring misses "16 plus" — GIAS phase 6, which PHASE_GROUPS deliberately files as secondary. The substring version fails silently rather than loudly: a sixth-form college is simply handed the primary bucket and offered infant schools as peers. All-through is the exception. PHASE_GROUPS lists it on both sides because it belongs on both phases' place pages, but the detail page renders it with the primary template, and the metric follows the template. """ text = (phase or "").strip().lower() return text != "all-through" and text in PHASE_GROUPS["secondary"] def _phase_group(is_secondary: bool) -> set[str]: return PHASE_GROUPS["secondary" if is_secondary else "primary"] def _haversine_miles(lat1: float, lon1: float, lat2, lon2): """Vectorised, matching the postcode search in app.py.""" lat1_r, lon1_r = np.radians(lat1), np.radians(lon1) lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float)) dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2 return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a)) def _native(value): """NaN and numpy scalars both reach JSONResponse badly; normalise here so the caller never has to remember to.""" if value is None: return None if isinstance(value, np.generic): value = value.item() if isinstance(value, float) and np.isnan(value): return None return value def _mask(series: pd.Series, predicate) -> pd.Series: """A boolean mask that survives an empty frame. `Series.apply` on an empty Series returns an empty *DataFrame*, and using that as a mask silently drops every column — so the next column lookup raises KeyError rather than yielding no rows. This is not hypothetical: a special school with no special school near it empties the frame at the provision filter, which is the ordinary case for most special schools. """ return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool) def _chips(subject: pd.Series, candidate: pd.Series, tier: int, is_secondary: bool) -> list[str]: if tier >= 3: return [phase_label(candidate.get("phase"))] chips = [str(subject.get("gender") or "").strip()] if is_secondary: policy = (candidate.get("admissions_policy") or "").strip() if policy and policy.lower() not in {"not applicable", "unknown"}: chips.append(policy) if tier == 1: chips.append(faith_label(candidate.get("religious_denomination"))) return [chip for chip in chips if chip] def select_similar(frame: pd.DataFrame, urn: int, is_secondary: bool) -> list[dict]: """Up to MAX_SCHOOLS nearby schools this page may offer, or [] below MINIMUM. Selected by tier, displayed by distance: the tier decides which schools earn a slot, and the render order is then closest-first, because "nearby" is the promise in the heading. """ subject_rows = frame[frame["urn"] == urn] if subject_rows.empty: return [] subject = subject_rows.iloc[0] lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude")) if lat is None or lon is None: return [] metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct" candidates = frame[frame["urn"] != urn].copy() for column in ("latitude", "longitude"): candidates = candidates[candidates[column].notna()] if candidates.empty: return [] # ── Hard filters ──────────────────────────────────────────────────── allowed_phases = _phase_group(is_secondary) candidates = candidates[ candidates["phase"].fillna("").str.lower().isin(allowed_phases) ] candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")] subject_special = is_special_provision(subject.get("school_type")) special = _mask(candidates["school_type"], is_special_provision) candidates = candidates[special if subject_special else ~special] subject_selective = is_selective(subject.get("admissions_policy")) selective = _mask(candidates["admissions_policy"], is_selective) candidates = candidates[selective if subject_selective else ~selective] subject_gender = subject.get("gender") candidates = candidates[ _mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g)) ] if candidates.empty: return [] candidates["distance_miles"] = _haversine_miles( lat, lon, candidates["latitude"].values, candidates["longitude"].values ).round(1) # ── Soft preferences, in tiers ────────────────────────────────────── subject_faith = faith_key(subject.get("religious_denomination")) subject_gender_key = (subject_gender or "").strip().lower() same_gender = candidates["gender"].fillna("").str.strip().str.lower() == subject_gender_key same_faith = _mask( candidates["religious_denomination"], lambda d: faith_key(d) == subject_faith ) tier_masks = { 1: same_gender & same_faith, 2: same_gender, 3: pd.Series(True, index=candidates.index), } # Descend the tiers only until the set reaches ENOUGH. The tier that gets # there is the last one opened, and the remaining slots up to MAX_SCHOOLS # are filled from the tiers already used — never by widening again. picked: dict[int, tuple[int, pd.Series]] = {} for tier, radius in TIERS: within = candidates[tier_masks[tier] & (candidates["distance_miles"] <= radius)] for _, row in within.sort_values("distance_miles").iterrows(): candidate_urn = int(row["urn"]) if candidate_urn in picked: continue picked[candidate_urn] = (tier, row) if len(picked) >= MAX_SCHOOLS: break if len(picked) >= ENOUGH: break if len(picked) < MINIMUM: return [] selected = sorted( picked.values(), key=lambda pair: float(pair[1]["distance_miles"]) )[:MAX_SCHOOLS] return [ { "urn": int(row["urn"]), "school_name": str(row.get("school_name") or ""), "distance_miles": float(row["distance_miles"]), "school_type": _native(row.get("school_type")), "age_range": _native(row.get("age_range")), "shared": _chips(subject, row, tier, is_secondary), "tier": tier, "metric_value": _native(row.get(metric_key)), "metric_key": metric_key, "metric_year": _native(row.get("year")), } for tier, row in selected ]