From b571d9c549d413d9071c01c97822446be5ee29ea Mon Sep 17 00:00:00 2001 From: Tudor Date: Mon, 21 Sep 2026 22:40:10 +0100 Subject: [PATCH] feat(api): decide which nearby schools a page may offer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hard filters encode claims the section may not make — a selective school is not an alternative to a non-selective one, a special school is not comparable to a mainstream one, a Girls school is not an option for a Boys school's reader — so they never relax. Soft preferences describe closeness of fit, so they relax across three tiers, and only far enough to reach three; the remaining slots up to six fill from the tiers already opened. PHASE_GROUPS moves to schemas.py so this module can share it without importing app, which would be a cycle. _mask() exists because Series.apply on an empty Series returns a DataFrame, and using that as a mask drops every column — so the next lookup raises KeyError instead of yielding no rows. A special school with no special school near it empties the frame at the provision filter, which is the ordinary case for most special schools, so this was a crash on a common path rather than an edge case. Co-Authored-By: Claude Opus 5 --- backend/app.py | 10 +- backend/schemas.py | 12 ++ backend/similar_schools.py | 243 +++++++++++++++++++++++++ backend/tests/test_similar_schools.py | 252 ++++++++++++++++++++++++++ 4 files changed, 508 insertions(+), 9 deletions(-) create mode 100644 backend/similar_schools.py create mode 100644 backend/tests/test_similar_schools.py diff --git a/backend/app.py b/backend/app.py index e5d5b62..7997cd5 100644 --- a/backend/app.py +++ b/backend/app.py @@ -40,20 +40,12 @@ from .data_loader import ( from .data_loader import get_data_info as get_db_info from . import flags from .places import build_place_index, build_place_registry, places_for_urn -from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS +from .schemas import METRIC_DEFINITIONS, PHASE_GROUPS, RANKING_COLUMNS, SCHOOL_COLUMNS from .utils import clean_for_json, convert_to_native # Values to exclude from filter dropdowns (empty strings, non-applicable labels) EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"} -# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they include. -# All-through schools appear in both primary and secondary results. -PHASE_GROUPS: dict[str, set[str]] = { - "primary": {"primary", "middle deemed primary", "all-through"}, - "secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"}, - "all-through": {"all-through"}, -} - # Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a # sitemap that redirects wastes a crawl on every URL it lists. BASE_URL = "https://www.schoolcompare.co.uk" diff --git a/backend/schemas.py b/backend/schemas.py index 14853f0..8d49c3c 100644 --- a/backend/schemas.py +++ b/backend/schemas.py @@ -532,6 +532,18 @@ RANKING_COLUMNS = [ "gcse_grade_91_pct", ] +# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they +# include. All-through schools appear in both primary and secondary results, +# which is why this is a set per phase rather than a single string comparison. +# +# Lives here rather than in app.py because similar_schools.py needs it too, and +# importing app from there would be a cycle. +PHASE_GROUPS: dict[str, set[str]] = { + "primary": {"primary", "middle deemed primary", "all-through"}, + "secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"}, + "all-through": {"all-through"}, +} + # School listing columns SCHOOL_COLUMNS = [ "urn", diff --git a/backend/similar_schools.py b/backend/similar_schools.py new file mode 100644 index 0000000..72b3318 --- /dev/null +++ b/backend/similar_schools.py @@ -0,0 +1,243 @@ +"""Which nearby schools a detail page may offer as alternatives. + +Two kinds of rule, and they are not interchangeable. + +HARD FILTERS encode claims the section is not allowed to make. A selective +school is not an alternative to a non-selective one, a special school is not +comparable to a mainstream one, and a Girls school is not an option for a Boys +school's reader. These never relax, at any distance, even where that means the +section does not render at all. + +SOFT PREFERENCES describe how closely an intake resembles this school's. They +relax in tiers, and every card reports the tier that actually took it so the +page can say what is shared rather than implying more. They relax only far +enough to reach a usable set, never far enough to fill the last of the slots. + +Pure functions over a DataFrame: no I/O, no FastAPI, no database. +""" + +from __future__ import annotations + +import re + +import numpy as np +import pandas as pd + +from .schemas import PHASE_GROUPS + +# A cap, not a quota: the section shows everything that qualified at the tiers +# it used, up to this many. Three fit the row; the rest are behind the arrows. +MAX_SCHOOLS = 6 +# Tiers stop relaxing once this many have been found. Without it, a cap of six +# would reliably drag in tier-3 schools ten miles away to fill a row that three +# good matches had already earned. +ENOUGH = 3 +MINIMUM = 2 + +# (tier, radius in miles). Faith relaxes before gender: a faith mismatch +# changes the character of a school, while a gender mismatch can mean the +# school is not available to this reader's child at all. +TIERS: tuple[tuple[int, float], ...] = ((1, 3.0), (2, 5.0), (3, 10.0)) + +EARTH_RADIUS_MILES = 3958.8 + +_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I) + +# Values that mean "this school has no religious character". +_NO_FAITH = {"", "none", "does not apply", "not applicable"} + + +def is_special_provision(school_type: str | None) -> bool: + """Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts. + + Special schools carry a mainstream phase, so phase alone cannot identify + them. The two implementations must agree: a school the frontend treats as + special for benchmarking but this treats as mainstream would be dropped + from its own England comparison and then offered as a peer to a mainstream + school on the next page along. + """ + return bool(_SPECIAL.search(school_type or "")) + + +def is_selective(admissions_policy: str | None) -> bool: + """Strictly selective. Unknown counts as non-selective, which is the safe + direction: it can only ever exclude a pairing, never invent one.""" + return (admissions_policy or "").strip().lower() == "selective" + + +def faith_key(denomination: str | None) -> str: + value = (denomination or "").strip().lower() + return "" if value in _NO_FAITH else value + + +def faith_label(denomination: str | None) -> str: + return denomination.strip() if faith_key(denomination) else "No religious character" + + +def genders_compatible(a: str | None, b: str | None) -> bool: + single = {"boys", "girls"} + left, right = (a or "").strip().lower(), (b or "").strip().lower() + return not (left in single and right in single and left != right) + + +def phase_label(phase: str | None) -> str: + text = (phase or "").strip() + if not text: + return "School" + if text.lower() == "all-through": + return "All-through school" + return f"{text.capitalize()} school" + + +def _phase_group(is_secondary: bool) -> set[str]: + return PHASE_GROUPS["secondary" if is_secondary else "primary"] + + +def _haversine_miles(lat1: float, lon1: float, lat2, lon2): + """Vectorised, matching the postcode search in app.py.""" + lat1_r, lon1_r = np.radians(lat1), np.radians(lon1) + lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float)) + dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r + a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2 + return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a)) + + +def _native(value): + """NaN and numpy scalars both reach JSONResponse badly; normalise here so + the caller never has to remember to.""" + if value is None: + return None + if isinstance(value, np.generic): + value = value.item() + if isinstance(value, float) and np.isnan(value): + return None + return value + + +def _mask(series: pd.Series, predicate) -> pd.Series: + """A boolean mask that survives an empty frame. + + `Series.apply` on an empty Series returns an empty *DataFrame*, and using + that as a mask silently drops every column — so the next column lookup + raises KeyError rather than yielding no rows. This is not hypothetical: a + special school with no special school near it empties the frame at the + provision filter, which is the ordinary case for most special schools. + """ + return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool) + + +def _chips(subject: pd.Series, candidate: pd.Series, tier: int, is_secondary: bool) -> list[str]: + if tier >= 3: + return [phase_label(candidate.get("phase"))] + + chips = [str(subject.get("gender") or "").strip()] + if is_secondary: + policy = (candidate.get("admissions_policy") or "").strip() + if policy and policy.lower() not in {"not applicable", "unknown"}: + chips.append(policy) + if tier == 1: + chips.append(faith_label(candidate.get("religious_denomination"))) + return [chip for chip in chips if chip] + + +def select_similar(frame: pd.DataFrame, urn: int, is_secondary: bool) -> list[dict]: + """Up to MAX_SCHOOLS nearby schools this page may offer, or [] below MINIMUM. + + Selected by tier, displayed by distance: the tier decides which schools + earn a slot, and the render order is then closest-first, because "nearby" + is the promise in the heading. + """ + subject_rows = frame[frame["urn"] == urn] + if subject_rows.empty: + return [] + subject = subject_rows.iloc[0] + + lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude")) + if lat is None or lon is None: + return [] + + metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct" + + candidates = frame[frame["urn"] != urn].copy() + for column in ("latitude", "longitude"): + candidates = candidates[candidates[column].notna()] + if candidates.empty: + return [] + + # ── Hard filters ──────────────────────────────────────────────────── + allowed_phases = _phase_group(is_secondary) + candidates = candidates[ + candidates["phase"].fillna("").str.lower().isin(allowed_phases) + ] + candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")] + + subject_special = is_special_provision(subject.get("school_type")) + special = _mask(candidates["school_type"], is_special_provision) + candidates = candidates[special if subject_special else ~special] + + subject_selective = is_selective(subject.get("admissions_policy")) + selective = _mask(candidates["admissions_policy"], is_selective) + candidates = candidates[selective if subject_selective else ~selective] + + subject_gender = subject.get("gender") + candidates = candidates[ + _mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g)) + ] + if candidates.empty: + return [] + + candidates["distance_miles"] = _haversine_miles( + lat, lon, candidates["latitude"].values, candidates["longitude"].values + ).round(1) + + # ── Soft preferences, in tiers ────────────────────────────────────── + subject_faith = faith_key(subject.get("religious_denomination")) + subject_gender_key = (subject_gender or "").strip().lower() + same_gender = candidates["gender"].fillna("").str.strip().str.lower() == subject_gender_key + same_faith = _mask( + candidates["religious_denomination"], lambda d: faith_key(d) == subject_faith + ) + + tier_masks = { + 1: same_gender & same_faith, + 2: same_gender, + 3: pd.Series(True, index=candidates.index), + } + + # Descend the tiers only until the set reaches ENOUGH. The tier that gets + # there is the last one opened, and the remaining slots up to MAX_SCHOOLS + # are filled from the tiers already used — never by widening again. + picked: dict[int, tuple[int, pd.Series]] = {} + for tier, radius in TIERS: + within = candidates[tier_masks[tier] & (candidates["distance_miles"] <= radius)] + for _, row in within.sort_values("distance_miles").iterrows(): + candidate_urn = int(row["urn"]) + if candidate_urn in picked: + continue + picked[candidate_urn] = (tier, row) + if len(picked) >= MAX_SCHOOLS: + break + if len(picked) >= ENOUGH: + break + + if len(picked) < MINIMUM: + return [] + + selected = sorted( + picked.values(), key=lambda pair: float(pair[1]["distance_miles"]) + )[:MAX_SCHOOLS] + return [ + { + "urn": int(row["urn"]), + "school_name": str(row.get("school_name") or ""), + "distance_miles": float(row["distance_miles"]), + "school_type": _native(row.get("school_type")), + "age_range": _native(row.get("age_range")), + "shared": _chips(subject, row, tier, is_secondary), + "tier": tier, + "metric_value": _native(row.get(metric_key)), + "metric_key": metric_key, + "metric_year": _native(row.get("year")), + } + for tier, row in selected + ] diff --git a/backend/tests/test_similar_schools.py b/backend/tests/test_similar_schools.py new file mode 100644 index 0000000..aab8de7 --- /dev/null +++ b/backend/tests/test_similar_schools.py @@ -0,0 +1,252 @@ +"""Selection rules for the "similar schools nearby" section. + +The hard filters encode claims the section is not allowed to make — that a +selective school is an alternative to a non-selective one, that a special +school is comparable to a mainstream one, or that a Girls school is an option +for a Boys school's reader. They never relax. The soft preferences describe +how close the intake is, and they do — but only far enough to reach a usable +set, never far enough to fill the last of the six slots. +""" + +import numpy as np +import pandas as pd + +from backend.similar_schools import select_similar + +# Roughly 0.7 miles apart in latitude at this longitude. +BASE_LAT, BASE_LON = 51.5000, -0.1000 + + +def _row(urn, name, **overrides): + base = { + "urn": urn, + "school_name": name, + "local_authority": "Testshire", + "school_type": "Community school", + "phase": "Primary", + "age_range": "4-11", + "status": "Open", + "gender": "Mixed", + "religious_denomination": "None", + "admissions_policy": "Not applicable", + "latitude": BASE_LAT, + "longitude": BASE_LON, + "year": 202425, + "rwm_expected_pct": 70.0, + "attainment_8_score": np.nan, + } + base.update(overrides) + return base + + +def _frame(*rows): + return pd.DataFrame(list(rows)) + + +def _at(miles): + """A latitude `miles` north of BASE_LAT.""" + return BASE_LAT + miles / 69.0 + + +def test_returns_nearest_same_phase_schools(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "Near", latitude=_at(0.5)), + _row(100003, "Mid", latitude=_at(1.0)), + _row(100004, "Far", latitude=_at(2.0)), + ) + result = select_similar(frame, 100001, is_secondary=False) + assert [s["urn"] for s in result] == [100002, 100003, 100004] + assert result[0]["distance_miles"] == 0.5 + + +def test_excludes_the_subject_school(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "A", latitude=_at(0.5)), + _row(100003, "B", latitude=_at(0.6)), + ) + assert 100001 not in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} + + +def test_selective_never_meets_non_selective(): + frame = _frame( + _row(100001, "Grammar", phase="Secondary", admissions_policy="Selective"), + _row(100002, "Comp A", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.5)), + _row(100003, "Comp B", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.6)), + ) + assert select_similar(frame, 100001, is_secondary=True) == [] + + reverse = select_similar(frame, 100002, is_secondary=True) + assert 100001 not in {s["urn"] for s in reverse} + + +def test_special_schools_match_only_each_other(): + frame = _frame( + _row(100001, "Special", school_type="Community special school"), + _row(100002, "Mainstream A", latitude=_at(0.5)), + _row(100003, "Mainstream B", latitude=_at(0.6)), + ) + assert select_similar(frame, 100001, is_secondary=False) == [] + assert select_similar(frame, 100002, is_secondary=False) == [] + + +def test_boys_never_meets_girls(): + frame = _frame( + _row(100001, "Boys School", gender="Boys"), + _row(100002, "Girls School", gender="Girls", latitude=_at(0.5)), + _row(100003, "Mixed School", gender="Mixed", latitude=_at(0.6)), + _row(100004, "Another Mixed", gender="Mixed", latitude=_at(0.7)), + ) + urns = {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} + assert 100002 not in urns + assert urns == {100003, 100004} + + +def test_closed_schools_and_missing_coordinates_are_dropped(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "Closed", status="Closed", latitude=_at(0.5)), + _row(100003, "No coords", latitude=np.nan, longitude=np.nan), + _row(100004, "Good A", latitude=_at(0.6)), + _row(100005, "Good B", latitude=_at(0.7)), + ) + assert {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} == {100004, 100005} + + +def test_tiers_relax_faith_before_gender(): + frame = _frame( + _row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"), + # Tier 1: same gender and same faith. + _row(100002, "Tier one", gender="Boys", religious_denomination="Roman Catholic", latitude=_at(2.0)), + # Tier 2: same gender, different faith — closer, but a weaker match. + _row(100003, "Tier two", gender="Boys", religious_denomination="None", latitude=_at(0.5)), + # Tier 3: mixed gender, different faith. + _row(100004, "Tier three", gender="Mixed", religious_denomination="None", latitude=_at(0.6)), + ) + result = select_similar(frame, 100001, is_secondary=False) + tier_by_urn = {s["urn"]: s["tier"] for s in result} + assert tier_by_urn == {100002: 1, 100003: 2, 100004: 3} + # Selected by tier, displayed by distance. + assert [s["urn"] for s in result] == [100003, 100004, 100002] + + +def test_caps_at_six_taking_the_nearest(): + frame = _frame( + _row(100001, "Subject"), + *[_row(100010 + n, f"Peer {n}", latitude=_at(0.1 * (n + 1))) for n in range(7)], + ) + result = select_similar(frame, 100001, is_secondary=False) + assert len(result) == 6 + # The seventh-nearest is the one dropped, not an arbitrary one. + assert 100016 not in {s["urn"] for s in result} + + +def test_tiers_stop_once_enough_are_found(): + """Four tier-1 matches are a usable set, so tier 2 is never opened — even + though it holds a school that is closer than any of them.""" + frame = _frame( + _row(100001, "Subject", religious_denomination="Roman Catholic"), + _row(100002, "RC one", religious_denomination="Roman Catholic", latitude=_at(0.5)), + _row(100003, "RC two", religious_denomination="Roman Catholic", latitude=_at(0.6)), + _row(100004, "RC three", religious_denomination="Roman Catholic", latitude=_at(0.7)), + _row(100005, "RC four", religious_denomination="Roman Catholic", latitude=_at(0.8)), + # Closer than every one of them, but only a tier-2 match. + _row(100006, "Secular and nearer", religious_denomination="None", latitude=_at(0.2)), + ) + result = select_similar(frame, 100001, is_secondary=False) + assert 100006 not in {s["urn"] for s in result} + assert len(result) == 4 + assert all(s["tier"] == 1 for s in result) + + +def test_a_school_is_never_taken_twice(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "A", latitude=_at(0.5)), + _row(100003, "B", latitude=_at(0.6)), + ) + result = select_similar(frame, 100001, is_secondary=False) + assert len(result) == len({s["urn"] for s in result}) + + +def test_fewer_than_two_matches_returns_empty(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "Only neighbour", latitude=_at(0.5)), + ) + assert select_similar(frame, 100001, is_secondary=False) == [] + + +def test_beyond_the_widest_radius_is_not_offered(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "A", latitude=_at(11.0)), + _row(100003, "B", latitude=_at(12.0)), + ) + assert select_similar(frame, 100001, is_secondary=False) == [] + + +def test_all_through_is_offered_on_both_phase_sides(): + frame = _frame( + _row(100001, "Primary subject", phase="Primary"), + _row(100002, "All through", phase="All-through", latitude=_at(0.5)), + _row(100003, "Primary peer", phase="Primary", latitude=_at(0.6)), + ) + assert 100002 in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} + + secondary = _frame( + _row(100010, "Secondary subject", phase="Secondary"), + _row(100002, "All through", phase="All-through", latitude=_at(0.5)), + _row(100011, "Secondary peer", phase="Secondary", latitude=_at(0.6)), + ) + assert 100002 in {s["urn"] for s in select_similar(secondary, 100010, is_secondary=True)} + + +def test_chips_state_only_what_the_tier_earned(): + frame = _frame( + _row(100001, "Subject", phase="Secondary", gender="Mixed", + religious_denomination="None", admissions_policy="Non-selective"), + _row(100002, "Full match", phase="Secondary", gender="Mixed", + religious_denomination="None", admissions_policy="Non-selective", latitude=_at(0.5)), + _row(100003, "Faith differs", phase="Secondary", gender="Mixed", + religious_denomination="Church of England", admissions_policy="Non-selective", latitude=_at(0.6)), + ) + by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)} + assert by_urn[100002]["shared"] == ["Mixed", "Non-selective", "No religious character"] + assert by_urn[100003]["shared"] == ["Mixed", "Non-selective"] + + +def test_tier_three_chip_is_the_plain_phase(): + frame = _frame( + _row(100001, "Subject", gender="Boys"), + _row(100002, "A", gender="Mixed", latitude=_at(0.5)), + _row(100003, "B", gender="Mixed", latitude=_at(0.6)), + ) + result = select_similar(frame, 100001, is_secondary=False) + assert all(s["shared"] == ["Primary school"] for s in result) + + +def test_metric_follows_the_template_not_the_neighbour(): + frame = _frame( + _row(100001, "Subject", phase="Secondary", attainment_8_score=50.0), + _row(100002, "A", phase="Secondary", attainment_8_score=52.8, latitude=_at(0.5)), + _row(100003, "B", phase="Secondary", attainment_8_score=np.nan, latitude=_at(0.6)), + ) + by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)} + assert by_urn[100002]["metric_key"] == "attainment_8_score" + assert by_urn[100002]["metric_value"] == 52.8 + assert by_urn[100002]["metric_year"] == 202425 + assert by_urn[100003]["metric_value"] is None + + +def test_values_are_json_safe_native_types(): + frame = _frame( + _row(100001, "Subject"), + _row(100002, "A", latitude=_at(0.5)), + _row(100003, "B", latitude=_at(0.6)), + ) + for school in select_similar(frame, 100001, is_secondary=False): + assert isinstance(school["urn"], int) + assert isinstance(school["distance_miles"], float) + assert not isinstance(school["metric_value"], np.generic)