Files
school_compare/backend/tests/test_similar_schools.py
T
TudorandClaude Opus 5 5c0ccc693d fix: order nearby schools by distance, not by how alike they are
Reported from staging: a Catholic primary showed six Catholic primaries, none
of them close enough to be a real option, and omitted the community school
down the road.

Three causes, compounding. Ranking put tier before distance, so a faith match
at 2.9 miles outranked a community school at 0.3. The ENOUGH=3 stopping rule —
added so a cap of six would not drag in weak distant matches — filled the row
from the best tier before it ever widened, which is what made every card
Catholic. And a 3-mile tier-1 radius is sane for a secondary and most of a city
for a primary, whose catchments are routinely under a mile.

The premise was backwards. For a parent, distance is a constraint and intake is
a preference; a school beyond a primary catchment is not a weaker option, it is
not an option. So distance now decides the order and nothing else does. The
hard filters are untouched — they were always where the defensibility lived.
Similarity survives as chips on the card: reported, so a reader applies their
own weighting, rather than ranked, so we apply ours for them.

Reach is capped per phase (primary 2, secondary 6, post-16 10) as a sanity
bound, not a target: ordering already handles density, so the cap only decides
what happens where an area is sparse. A primary with nothing inside two miles
now renders no section, which is the honest answer.

Deleted: the tier system, the stopping rule, the tier-dependent lede, the
`tier` field, the tier-3 fallback chip and its style. select_similar also stops
taking is_secondary — it reads the phase from the subject's own row, so no
caller can hand it one that disagrees with the data.

The heading is now "Other schools nearby". The hard filters still guarantee a
comparable set, but nothing ranks on likeness, so the heading no longer says it
does.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-22 13:06:32 +01:00

384 lines
15 KiB
Python

"""Selection rules for the nearby-schools section.
Hard filters encode claims the section is not allowed to make — that a
selective school is an alternative to a non-selective one, that a special
school is comparable to a mainstream one, or that a Girls school is an option
for a Boys school's reader. They decide who is eligible.
Distance decides the order, and nothing else does. An earlier version ranked by
intake similarity first, which put a Catholic school 2.9 miles away above the
community school 0.3 miles down the road — for a primary, a school that far is
not a weaker option, it is not an option. Similarity is now reported on the
card and never reorders the row.
"""
import numpy as np
import pandas as pd
from backend.similar_schools import (
is_secondary_phase,
radius_miles,
select_similar,
)
BASE_LAT, BASE_LON = 51.5000, -0.1000
def _row(urn, name, **overrides):
base = {
"urn": urn,
"school_name": name,
"local_authority": "Testshire",
"school_type": "Community school",
"phase": "Primary",
"age_range": "4-11",
"status": "Open",
"gender": "Mixed",
"religious_denomination": "None",
"admissions_policy": "Not applicable",
"latitude": BASE_LAT,
"longitude": BASE_LON,
"year": 202425,
"rwm_expected_pct": 70.0,
"attainment_8_score": np.nan,
}
base.update(overrides)
return base
def _frame(*rows):
return pd.DataFrame(list(rows))
def _at(miles):
"""A latitude `miles` north of BASE_LAT."""
return BASE_LAT + miles / 69.0
# ---------------------------------------------------------------------------
# Order: distance, and only distance
# ---------------------------------------------------------------------------
def test_returns_nearest_first():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Mid", latitude=_at(1.0)),
_row(100003, "Near", latitude=_at(0.4)),
_row(100004, "Far", latitude=_at(1.8)),
)
result = select_similar(frame, 100001)
assert [s["urn"] for s in result] == [100003, 100002, 100004]
assert result[0]["distance_miles"] == 0.4
def test_a_faith_match_never_outranks_a_closer_school():
"""The reported defect. A Catholic primary surrounded by Catholic primaries
showed six of them and omitted the community school down the road."""
frame = _frame(
_row(100001, "St Jude's RC Primary", religious_denomination="Roman Catholic"),
_row(100002, "Elm Grove Primary", religious_denomination="None", latitude=_at(0.3)),
_row(100003, "Holy Cross RC", religious_denomination="Roman Catholic", latitude=_at(0.8)),
_row(100004, "Sacred Heart RC", religious_denomination="Roman Catholic", latitude=_at(1.2)),
_row(100005, "St Peter's RC", religious_denomination="Roman Catholic", latitude=_at(1.6)),
)
result = select_similar(frame, 100001)
assert result[0]["urn"] == 100002, "the nearest school leads, whatever its intake"
assert [s["distance_miles"] for s in result] == sorted(s["distance_miles"] for s in result)
def test_the_nearest_eligible_school_is_always_shown():
"""Whatever else changes, a section titled "nearby" cannot omit the nearest
school while listing one four times further away."""
frame = _frame(
_row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"),
_row(100002, "Nearest", gender="Mixed", religious_denomination="None", latitude=_at(0.2)),
*[
_row(100010 + n, f"Match {n}", gender="Boys",
religious_denomination="Roman Catholic", latitude=_at(0.9 + n * 0.1))
for n in range(6)
],
)
assert select_similar(frame, 100001)[0]["urn"] == 100002
def test_caps_at_six_taking_the_nearest():
frame = _frame(
_row(100001, "Subject"),
*[_row(100010 + n, f"Peer {n}", latitude=_at(0.1 * (n + 1))) for n in range(7)],
)
result = select_similar(frame, 100001)
assert len(result) == 6
assert 100016 not in {s["urn"] for s in result}, "the seventh-nearest is the one dropped"
def test_fewer_than_two_matches_returns_empty():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Only neighbour", latitude=_at(0.5)),
)
assert select_similar(frame, 100001) == []
def test_excludes_the_subject_school():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
assert 100001 not in {s["urn"] for s in select_similar(frame, 100001)}
def test_a_school_is_never_listed_twice():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
result = select_similar(frame, 100001)
assert len(result) == len({s["urn"] for s in result})
# ---------------------------------------------------------------------------
# Reach: a sanity bound, not a target
# ---------------------------------------------------------------------------
def test_primary_does_not_reach_past_two_miles():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Just inside", latitude=_at(1.9)),
_row(100003, "Just outside", latitude=_at(2.4)),
_row(100004, "Miles away", latitude=_at(4.0)),
)
# One inside the cap is below the minimum, so nothing renders at all —
# a primary with nothing within two miles has no nearby schools.
assert select_similar(frame, 100001) == []
def test_secondary_reaches_further_than_primary():
frame = _frame(
_row(100001, "Subject", phase="Secondary"),
_row(100002, "A", phase="Secondary", latitude=_at(3.0)),
_row(100003, "B", phase="Secondary", latitude=_at(5.5)),
)
assert {s["urn"] for s in select_similar(frame, 100001)} == {100002, 100003}
def test_the_cap_follows_the_phase():
assert radius_miles("Primary") == 2.0
assert radius_miles("Middle deemed primary") == 2.0
assert radius_miles("All-through") == 2.0
assert radius_miles("Secondary") == 6.0
assert radius_miles("Middle deemed secondary") == 6.0
# Post-16 is the phase people travel furthest for.
assert radius_miles("16 plus") == 10.0
# ---------------------------------------------------------------------------
# Hard filters: eligibility, never order
# ---------------------------------------------------------------------------
def test_selective_never_meets_non_selective():
frame = _frame(
_row(100001, "Grammar", phase="Secondary", admissions_policy="Selective"),
_row(100002, "Comp A", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.5)),
_row(100003, "Comp B", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.6)),
)
assert select_similar(frame, 100001) == []
assert 100001 not in {s["urn"] for s in select_similar(frame, 100002)}
def test_special_schools_match_only_each_other():
frame = _frame(
_row(100001, "Special", school_type="Community special school"),
_row(100002, "Mainstream A", latitude=_at(0.5)),
_row(100003, "Mainstream B", latitude=_at(0.6)),
)
assert select_similar(frame, 100001) == []
assert select_similar(frame, 100002) == []
def test_boys_never_meets_girls():
frame = _frame(
_row(100001, "Boys School", gender="Boys"),
_row(100002, "Girls School", gender="Girls", latitude=_at(0.5)),
_row(100003, "Mixed School", gender="Mixed", latitude=_at(0.6)),
_row(100004, "Another Mixed", gender="Mixed", latitude=_at(0.7)),
)
urns = {s["urn"] for s in select_similar(frame, 100001)}
assert 100002 not in urns
assert urns == {100003, 100004}
def test_closed_schools_and_missing_coordinates_are_dropped():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "Closed", status="Closed", latitude=_at(0.5)),
_row(100003, "No coords", latitude=np.nan, longitude=np.nan),
_row(100004, "Good A", latitude=_at(0.6)),
_row(100005, "Good B", latitude=_at(0.7)),
)
assert {s["urn"] for s in select_similar(frame, 100001)} == {100004, 100005}
def test_all_through_is_offered_on_both_phase_sides():
frame = _frame(
_row(100001, "Primary subject", phase="Primary"),
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
_row(100003, "Primary peer", phase="Primary", latitude=_at(0.6)),
)
assert 100002 in {s["urn"] for s in select_similar(frame, 100001)}
secondary = _frame(
_row(100010, "Secondary subject", phase="Secondary"),
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
_row(100011, "Secondary peer", phase="Secondary", latitude=_at(0.6)),
)
assert 100002 in {s["urn"] for s in select_similar(secondary, 100010)}
def test_sixteen_plus_is_matched_against_secondary_not_primary():
"""GIAS phase 6 is "16 plus", and PHASE_GROUPS puts it in the secondary
group — a sixth-form college's peers are secondaries and other colleges,
never primary schools. A substring test for "secondary" misses it silently:
no crash, just a page offering infant schools to a sixth form."""
frame = _frame(
_row(100001, "Sixth Form College", phase="16 plus", age_range="16-19"),
_row(100002, "Nearby Secondary", phase="Secondary", latitude=_at(0.5),
attainment_8_score=52.0),
_row(100003, "Nearby College", phase="16 plus", latitude=_at(0.6)),
_row(100004, "Nearby Primary", phase="Primary", latitude=_at(0.1)),
)
result = select_similar(frame, 100001)
urns = {s["urn"] for s in result}
assert 100004 not in urns, "a primary school is not a peer for a sixth form"
assert urns == {100002, 100003}
assert all(s["metric_key"] == "attainment_8_score" for s in result)
def test_is_secondary_phase_agrees_with_the_phase_groups_it_selects_from():
for phase in ("Secondary", "Middle deemed secondary", "16 plus"):
assert is_secondary_phase(phase) is True, phase
for phase in ("Primary", "Middle deemed primary", "Nursery", "", None):
assert is_secondary_phase(phase) is False, phase
# In PHASE_GROUPS an all-through school is on both sides, but it renders
# with the primary template, and the metric follows the phase side.
assert is_secondary_phase("All-through") is False
# ---------------------------------------------------------------------------
# What the card reports
# ---------------------------------------------------------------------------
def test_shared_lists_only_what_is_actually_shared():
frame = _frame(
_row(100001, "Subject", phase="Secondary", gender="Mixed",
religious_denomination="None", admissions_policy="Non-selective"),
_row(100002, "Full match", phase="Secondary", gender="Mixed",
religious_denomination="None", admissions_policy="Non-selective", latitude=_at(0.5)),
_row(100003, "Faith differs", phase="Secondary", gender="Mixed",
religious_denomination="Church of England", admissions_policy="Non-selective", latitude=_at(0.6)),
)
by_urn = {s["urn"]: s for s in select_similar(frame, 100001)}
assert by_urn[100002]["shared"] == ["Mixed", "Non-selective", "No religious character"]
assert by_urn[100003]["shared"] == ["Mixed", "Non-selective"]
def test_a_shared_faith_is_named():
frame = _frame(
_row(100001, "Subject", religious_denomination="Roman Catholic"),
_row(100002, "Also RC", religious_denomination="Roman Catholic", latitude=_at(0.4)),
_row(100003, "Secular", religious_denomination="None", latitude=_at(0.5)),
)
by_urn = {s["urn"]: s for s in select_similar(frame, 100001)}
assert "Roman Catholic" in by_urn[100002]["shared"]
assert by_urn[100003]["shared"] == ["Mixed"]
def test_shared_is_empty_when_nothing_is_shared():
frame = _frame(
_row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"),
_row(100002, "A", gender="Mixed", religious_denomination="None", latitude=_at(0.4)),
_row(100003, "B", gender="Mixed", religious_denomination="Church of England", latitude=_at(0.5)),
)
assert all(s["shared"] == [] for s in select_similar(frame, 100001))
def test_no_tier_is_reported_because_there_are_no_tiers():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.4)),
_row(100003, "B", latitude=_at(0.5)),
)
assert all("tier" not in s for s in select_similar(frame, 100001))
def test_metric_follows_the_phase_side_not_the_neighbour():
frame = _frame(
_row(100001, "Subject", phase="Secondary", attainment_8_score=50.0),
_row(100002, "A", phase="Secondary", attainment_8_score=52.8, latitude=_at(0.5)),
_row(100003, "B", phase="Secondary", attainment_8_score=np.nan, latitude=_at(0.6)),
)
by_urn = {s["urn"]: s for s in select_similar(frame, 100001)}
assert by_urn[100002]["metric_key"] == "attainment_8_score"
assert by_urn[100002]["metric_value"] == 52.8
assert by_urn[100002]["metric_year"] == 202425
assert by_urn[100003]["metric_value"] is None
def test_values_are_json_safe_native_types():
frame = _frame(
_row(100001, "Subject"),
_row(100002, "A", latitude=_at(0.5)),
_row(100003, "B", latitude=_at(0.6)),
)
for school in select_similar(frame, 100001):
assert isinstance(school["urn"], int)
assert isinstance(school["distance_miles"], float)
assert not isinstance(school["metric_value"], np.generic)
# ---------------------------------------------------------------------------
# The endpoint
# ---------------------------------------------------------------------------
import pytest
from fastapi.testclient import TestClient
def _endpoint_frame():
return _frame(
_row(100001, "Subject Primary"),
_row(100002, "Neighbour A", latitude=_at(0.5)),
_row(100003, "Neighbour B", latitude=_at(0.6)),
)
@pytest.fixture()
def client(monkeypatch):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_latest_school_data", _endpoint_frame)
monkeypatch.setattr(app_module, "load_school_data", _endpoint_frame)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
return TestClient(app_module.app, raise_server_exceptions=False)
def test_detail_payload_carries_nearby_schools(client):
resp = client.get("/api/schools/100001")
assert resp.status_code == 200, resp.text
similar = resp.json()["similar_schools"]
assert [s["school_name"] for s in similar] == ["Neighbour A", "Neighbour B"]
assert similar[0]["metric_key"] == "rwm_expected_pct"
def test_a_failure_in_selection_does_not_break_the_page(client, monkeypatch):
from backend import app as app_module
def _explode(*args, **kwargs):
raise ValueError("selection blew up")
monkeypatch.setattr(app_module, "select_similar", _explode)
resp = client.get("/api/schools/100001")
assert resp.status_code == 200, resp.text
assert resp.json()["similar_schools"] == []