refactor: rename similar → nearby, so the code says what the section does
The section ranks on distance and is headed "Other schools nearby", but every identifier still called it "similar" — the exact drift that leaves a later reader trusting a name over the behaviour. Mechanical: files, the module, the payload key, the type, the components, the prop. No behaviour change; the suites are unchanged in count and still green. Free to do now because #150 has not merged, so the payload key rename needs no lockstep deploy. Uses of "similar" that are ordinary English — progress measures compared to similar pupils, and unrelated comments — are untouched. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
5c0ccc693d
commit
cd6a45bf7d
15 files changed
+102
-102
No files matched your search
@@ -0,0 +1,259 @@
|
||||
"""Which nearby schools a detail page may offer as alternatives.
|
||||
|
||||
HARD FILTERS decide eligibility, and encode claims the section is not allowed
|
||||
to make. A selective school is not an alternative to a non-selective one, a
|
||||
special school is not comparable to a mainstream one, and a Girls school is not
|
||||
an option for a Boys school's reader. They never relax, at any distance, even
|
||||
where that means the section does not render at all.
|
||||
|
||||
DISTANCE decides the order, and nothing else does.
|
||||
|
||||
An earlier version ranked by intake similarity first and used distance only as
|
||||
a tiebreak. That put a Catholic school 2.9 miles away above the community
|
||||
school 0.3 miles down the road, and — because the row filled from the best tier
|
||||
before widening — filled all six slots with faith matches while omitting every
|
||||
school a parent could actually walk to. For a primary, a school that far is not
|
||||
a weaker option; it is not an option. Distance is a constraint and intake is a
|
||||
preference, and the ranking now says so.
|
||||
|
||||
Similarity survives as `shared`: what a candidate genuinely has in common with
|
||||
this school, reported on its card, so a reader applies their own weighting
|
||||
instead of having ours applied for them.
|
||||
|
||||
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .schemas import PHASE_GROUPS
|
||||
|
||||
# Three fit the row; the rest are behind the carousel arrows.
|
||||
MAX_SCHOOLS = 6
|
||||
MINIMUM = 2
|
||||
|
||||
# How far the section will reach, in miles, when nothing closer exists.
|
||||
#
|
||||
# A sanity bound rather than a target: ordering by distance already handles
|
||||
# density, so a school in a dense area fills all six slots inside a mile and
|
||||
# never sees this. It decides one thing — what happens where the area is
|
||||
# sparse — and the answer differs by phase because catchments do. Primary
|
||||
# catchments are routinely under a mile; beyond two, a primary is not a weaker
|
||||
# option but not an option, and no section is the honest answer.
|
||||
PRIMARY_RADIUS_MILES = 2.0
|
||||
SECONDARY_RADIUS_MILES = 6.0
|
||||
POST16_RADIUS_MILES = 10.0
|
||||
|
||||
EARTH_RADIUS_MILES = 3958.8
|
||||
|
||||
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
|
||||
|
||||
# Values that mean "this school has no religious character".
|
||||
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
|
||||
|
||||
|
||||
def is_special_provision(school_type: str | None) -> bool:
|
||||
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
|
||||
|
||||
Special schools carry a mainstream phase, so phase alone cannot identify
|
||||
them. The two implementations must agree: a school the frontend treats as
|
||||
special for benchmarking but this treats as mainstream would be dropped
|
||||
from its own England comparison and then offered as a peer to a mainstream
|
||||
school on the next page along.
|
||||
"""
|
||||
return bool(_SPECIAL.search(school_type or ""))
|
||||
|
||||
|
||||
def is_selective(admissions_policy: str | None) -> bool:
|
||||
"""Strictly selective. Unknown counts as non-selective, which is the safe
|
||||
direction: it can only ever exclude a pairing, never invent one."""
|
||||
return (admissions_policy or "").strip().lower() == "selective"
|
||||
|
||||
|
||||
def faith_key(denomination: str | None) -> str:
|
||||
value = (denomination or "").strip().lower()
|
||||
return "" if value in _NO_FAITH else value
|
||||
|
||||
|
||||
def faith_label(denomination: str | None) -> str:
|
||||
return denomination.strip() if faith_key(denomination) else "No religious character"
|
||||
|
||||
|
||||
def genders_compatible(a: str | None, b: str | None) -> bool:
|
||||
single = {"boys", "girls"}
|
||||
left, right = (a or "").strip().lower(), (b or "").strip().lower()
|
||||
return not (left in single and right in single and left != right)
|
||||
|
||||
|
||||
def is_secondary_phase(phase: str | None) -> bool:
|
||||
"""Whether this phase takes the secondary side: secondary group membership,
|
||||
minus all-through.
|
||||
|
||||
Membership is read from PHASE_GROUPS rather than tested with `"secondary" in
|
||||
phase`, because that substring misses "16 plus" — GIAS phase 6, which
|
||||
PHASE_GROUPS deliberately files as secondary. The substring version fails
|
||||
silently rather than loudly: a sixth-form college is simply handed the
|
||||
primary bucket and offered infant schools as peers.
|
||||
|
||||
All-through is the exception. PHASE_GROUPS lists it on both sides because it
|
||||
belongs on both phases' place pages, but the detail page renders it with the
|
||||
primary template, and the metric follows the template.
|
||||
"""
|
||||
text = (phase or "").strip().lower()
|
||||
return text != "all-through" and text in PHASE_GROUPS["secondary"]
|
||||
|
||||
|
||||
def radius_miles(phase: str | None) -> float:
|
||||
"""How far this phase's section will reach when nothing closer exists."""
|
||||
if (phase or "").strip().lower() == "16 plus":
|
||||
return POST16_RADIUS_MILES
|
||||
return SECONDARY_RADIUS_MILES if is_secondary_phase(phase) else PRIMARY_RADIUS_MILES
|
||||
|
||||
|
||||
def _phase_group(is_secondary: bool) -> set[str]:
|
||||
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
|
||||
|
||||
|
||||
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
|
||||
"""Vectorised, matching the postcode search in app.py."""
|
||||
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
|
||||
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
|
||||
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
|
||||
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
|
||||
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
|
||||
|
||||
|
||||
def _native(value):
|
||||
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
|
||||
the caller never has to remember to."""
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, np.generic):
|
||||
value = value.item()
|
||||
if isinstance(value, float) and np.isnan(value):
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def _mask(series: pd.Series, predicate) -> pd.Series:
|
||||
"""A boolean mask that survives an empty frame.
|
||||
|
||||
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
|
||||
that as a mask silently drops every column — so the next column lookup
|
||||
raises KeyError rather than yielding no rows. This is not hypothetical: a
|
||||
special school with no special school near it empties the frame at the
|
||||
provision filter, which is the ordinary case for most special schools.
|
||||
"""
|
||||
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
|
||||
|
||||
|
||||
def _shared(subject: pd.Series, candidate: pd.Series, is_secondary: bool) -> list[str]:
|
||||
"""What this candidate genuinely has in common with the subject.
|
||||
|
||||
Empty is a real answer, and renders no chips at all. A card claiming a
|
||||
shared characteristic it does not have would be worse than a bare one —
|
||||
and since these no longer affect the order, an empty list costs the school
|
||||
nothing but its place in the row, which distance already decided.
|
||||
"""
|
||||
shared: list[str] = []
|
||||
|
||||
gender = str(subject.get("gender") or "").strip()
|
||||
if gender and str(candidate.get("gender") or "").strip().lower() == gender.lower():
|
||||
shared.append(gender)
|
||||
|
||||
if is_secondary:
|
||||
policy = str(candidate.get("admissions_policy") or "").strip()
|
||||
subject_policy = str(subject.get("admissions_policy") or "").strip()
|
||||
if (
|
||||
policy
|
||||
and policy.lower() == subject_policy.lower()
|
||||
and policy.lower() not in {"not applicable", "unknown"}
|
||||
):
|
||||
shared.append(policy)
|
||||
|
||||
if faith_key(candidate.get("religious_denomination")) == faith_key(
|
||||
subject.get("religious_denomination")
|
||||
):
|
||||
shared.append(faith_label(candidate.get("religious_denomination")))
|
||||
|
||||
return shared
|
||||
|
||||
|
||||
def select_nearby(frame: pd.DataFrame, urn: int) -> list[dict]:
|
||||
"""The nearest eligible schools, closest first — at most MAX_SCHOOLS, and
|
||||
none at all below MINIMUM.
|
||||
|
||||
The phase is read from the subject's own row rather than passed in, so a
|
||||
caller cannot hand this a phase that disagrees with the data it selects
|
||||
from.
|
||||
"""
|
||||
subject_rows = frame[frame["urn"] == urn]
|
||||
if subject_rows.empty:
|
||||
return []
|
||||
subject = subject_rows.iloc[0]
|
||||
|
||||
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
|
||||
if lat is None or lon is None:
|
||||
return []
|
||||
|
||||
phase = subject.get("phase")
|
||||
is_secondary = is_secondary_phase(phase)
|
||||
reach = radius_miles(phase)
|
||||
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
|
||||
|
||||
candidates = frame[frame["urn"] != urn].copy()
|
||||
for column in ("latitude", "longitude"):
|
||||
candidates = candidates[candidates[column].notna()]
|
||||
if candidates.empty:
|
||||
return []
|
||||
|
||||
# ── Hard filters ────────────────────────────────────────────────────
|
||||
allowed_phases = _phase_group(is_secondary)
|
||||
candidates = candidates[
|
||||
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
|
||||
]
|
||||
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
|
||||
|
||||
subject_special = is_special_provision(subject.get("school_type"))
|
||||
special = _mask(candidates["school_type"], is_special_provision)
|
||||
candidates = candidates[special if subject_special else ~special]
|
||||
|
||||
subject_selective = is_selective(subject.get("admissions_policy"))
|
||||
selective = _mask(candidates["admissions_policy"], is_selective)
|
||||
candidates = candidates[selective if subject_selective else ~selective]
|
||||
|
||||
subject_gender = subject.get("gender")
|
||||
candidates = candidates[
|
||||
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
|
||||
]
|
||||
if candidates.empty:
|
||||
return []
|
||||
|
||||
candidates["distance_miles"] = _haversine_miles(
|
||||
lat, lon, candidates["latitude"].values, candidates["longitude"].values
|
||||
).round(1)
|
||||
|
||||
# ── Nearest first, and nothing else has a say ───────────────────────
|
||||
within = candidates[candidates["distance_miles"] <= reach]
|
||||
if len(within) < MINIMUM:
|
||||
return []
|
||||
|
||||
selected = within.sort_values(["distance_miles", "urn"]).head(MAX_SCHOOLS)
|
||||
return [
|
||||
{
|
||||
"urn": int(row["urn"]),
|
||||
"school_name": str(row.get("school_name") or ""),
|
||||
"distance_miles": float(row["distance_miles"]),
|
||||
"school_type": _native(row.get("school_type")),
|
||||
"age_range": _native(row.get("age_range")),
|
||||
"shared": _shared(subject, row, is_secondary),
|
||||
"metric_value": _native(row.get(metric_key)),
|
||||
"metric_key": metric_key,
|
||||
"metric_year": _native(row.get("year")),
|
||||
}
|
||||
for _, row in selected.iterrows()
|
||||
]
|
||||
Reference in new issue
Block a user