"""Selection rules for the "similar schools nearby" section. The hard filters encode claims the section is not allowed to make — that a selective school is an alternative to a non-selective one, that a special school is comparable to a mainstream one, or that a Girls school is an option for a Boys school's reader. They never relax. The soft preferences describe how close the intake is, and they do — but only far enough to reach a usable set, never far enough to fill the last of the six slots. """ import numpy as np import pandas as pd from backend.similar_schools import is_secondary_phase, select_similar # Roughly 0.7 miles apart in latitude at this longitude. BASE_LAT, BASE_LON = 51.5000, -0.1000 def _row(urn, name, **overrides): base = { "urn": urn, "school_name": name, "local_authority": "Testshire", "school_type": "Community school", "phase": "Primary", "age_range": "4-11", "status": "Open", "gender": "Mixed", "religious_denomination": "None", "admissions_policy": "Not applicable", "latitude": BASE_LAT, "longitude": BASE_LON, "year": 202425, "rwm_expected_pct": 70.0, "attainment_8_score": np.nan, } base.update(overrides) return base def _frame(*rows): return pd.DataFrame(list(rows)) def _at(miles): """A latitude `miles` north of BASE_LAT.""" return BASE_LAT + miles / 69.0 def test_returns_nearest_same_phase_schools(): frame = _frame( _row(100001, "Subject"), _row(100002, "Near", latitude=_at(0.5)), _row(100003, "Mid", latitude=_at(1.0)), _row(100004, "Far", latitude=_at(2.0)), ) result = select_similar(frame, 100001, is_secondary=False) assert [s["urn"] for s in result] == [100002, 100003, 100004] assert result[0]["distance_miles"] == 0.5 def test_excludes_the_subject_school(): frame = _frame( _row(100001, "Subject"), _row(100002, "A", latitude=_at(0.5)), _row(100003, "B", latitude=_at(0.6)), ) assert 100001 not in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} def test_selective_never_meets_non_selective(): frame = _frame( _row(100001, "Grammar", phase="Secondary", admissions_policy="Selective"), _row(100002, "Comp A", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.5)), _row(100003, "Comp B", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.6)), ) assert select_similar(frame, 100001, is_secondary=True) == [] reverse = select_similar(frame, 100002, is_secondary=True) assert 100001 not in {s["urn"] for s in reverse} def test_special_schools_match_only_each_other(): frame = _frame( _row(100001, "Special", school_type="Community special school"), _row(100002, "Mainstream A", latitude=_at(0.5)), _row(100003, "Mainstream B", latitude=_at(0.6)), ) assert select_similar(frame, 100001, is_secondary=False) == [] assert select_similar(frame, 100002, is_secondary=False) == [] def test_boys_never_meets_girls(): frame = _frame( _row(100001, "Boys School", gender="Boys"), _row(100002, "Girls School", gender="Girls", latitude=_at(0.5)), _row(100003, "Mixed School", gender="Mixed", latitude=_at(0.6)), _row(100004, "Another Mixed", gender="Mixed", latitude=_at(0.7)), ) urns = {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} assert 100002 not in urns assert urns == {100003, 100004} def test_closed_schools_and_missing_coordinates_are_dropped(): frame = _frame( _row(100001, "Subject"), _row(100002, "Closed", status="Closed", latitude=_at(0.5)), _row(100003, "No coords", latitude=np.nan, longitude=np.nan), _row(100004, "Good A", latitude=_at(0.6)), _row(100005, "Good B", latitude=_at(0.7)), ) assert {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} == {100004, 100005} def test_tiers_relax_faith_before_gender(): frame = _frame( _row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"), # Tier 1: same gender and same faith. _row(100002, "Tier one", gender="Boys", religious_denomination="Roman Catholic", latitude=_at(2.0)), # Tier 2: same gender, different faith — closer, but a weaker match. _row(100003, "Tier two", gender="Boys", religious_denomination="None", latitude=_at(0.5)), # Tier 3: mixed gender, different faith. _row(100004, "Tier three", gender="Mixed", religious_denomination="None", latitude=_at(0.6)), ) result = select_similar(frame, 100001, is_secondary=False) tier_by_urn = {s["urn"]: s["tier"] for s in result} assert tier_by_urn == {100002: 1, 100003: 2, 100004: 3} # Selected by tier, displayed by distance. assert [s["urn"] for s in result] == [100003, 100004, 100002] def test_caps_at_six_taking_the_nearest(): frame = _frame( _row(100001, "Subject"), *[_row(100010 + n, f"Peer {n}", latitude=_at(0.1 * (n + 1))) for n in range(7)], ) result = select_similar(frame, 100001, is_secondary=False) assert len(result) == 6 # The seventh-nearest is the one dropped, not an arbitrary one. assert 100016 not in {s["urn"] for s in result} def test_tiers_stop_once_enough_are_found(): """Four tier-1 matches are a usable set, so tier 2 is never opened — even though it holds a school that is closer than any of them.""" frame = _frame( _row(100001, "Subject", religious_denomination="Roman Catholic"), _row(100002, "RC one", religious_denomination="Roman Catholic", latitude=_at(0.5)), _row(100003, "RC two", religious_denomination="Roman Catholic", latitude=_at(0.6)), _row(100004, "RC three", religious_denomination="Roman Catholic", latitude=_at(0.7)), _row(100005, "RC four", religious_denomination="Roman Catholic", latitude=_at(0.8)), # Closer than every one of them, but only a tier-2 match. _row(100006, "Secular and nearer", religious_denomination="None", latitude=_at(0.2)), ) result = select_similar(frame, 100001, is_secondary=False) assert 100006 not in {s["urn"] for s in result} assert len(result) == 4 assert all(s["tier"] == 1 for s in result) def test_a_school_is_never_taken_twice(): frame = _frame( _row(100001, "Subject"), _row(100002, "A", latitude=_at(0.5)), _row(100003, "B", latitude=_at(0.6)), ) result = select_similar(frame, 100001, is_secondary=False) assert len(result) == len({s["urn"] for s in result}) def test_fewer_than_two_matches_returns_empty(): frame = _frame( _row(100001, "Subject"), _row(100002, "Only neighbour", latitude=_at(0.5)), ) assert select_similar(frame, 100001, is_secondary=False) == [] def test_beyond_the_widest_radius_is_not_offered(): frame = _frame( _row(100001, "Subject"), _row(100002, "A", latitude=_at(11.0)), _row(100003, "B", latitude=_at(12.0)), ) assert select_similar(frame, 100001, is_secondary=False) == [] def test_all_through_is_offered_on_both_phase_sides(): frame = _frame( _row(100001, "Primary subject", phase="Primary"), _row(100002, "All through", phase="All-through", latitude=_at(0.5)), _row(100003, "Primary peer", phase="Primary", latitude=_at(0.6)), ) assert 100002 in {s["urn"] for s in select_similar(frame, 100001, is_secondary=False)} secondary = _frame( _row(100010, "Secondary subject", phase="Secondary"), _row(100002, "All through", phase="All-through", latitude=_at(0.5)), _row(100011, "Secondary peer", phase="Secondary", latitude=_at(0.6)), ) assert 100002 in {s["urn"] for s in select_similar(secondary, 100010, is_secondary=True)} def test_sixteen_plus_is_matched_against_secondary_not_primary(): """GIAS phase 6 is "16 plus", and PHASE_GROUPS puts it in the secondary group — a sixth-form college's peers are secondaries and other colleges, never primary schools. A substring test for "secondary" misses it silently: no crash, just a page offering infant schools to a sixth form.""" frame = _frame( _row(100001, "Sixth Form College", phase="16 plus", age_range="16-19"), _row(100002, "Nearby Secondary", phase="Secondary", latitude=_at(0.5), attainment_8_score=52.0), _row(100003, "Nearby College", phase="16 plus", latitude=_at(0.6), attainment_8_score=np.nan), _row(100004, "Nearby Primary", phase="Primary", latitude=_at(0.1)), ) result = select_similar(frame, 100001, is_secondary=is_secondary_phase("16 plus")) urns = {s["urn"] for s in result} assert 100004 not in urns, "a primary school is not a peer for a sixth form" assert urns == {100002, 100003} assert all(s["metric_key"] == "attainment_8_score" for s in result) def test_is_secondary_phase_agrees_with_the_phase_groups_it_selects_from(): """The two must not drift: whatever this calls secondary decides which PHASE_GROUPS bucket the candidates come from.""" for phase in ("Secondary", "Middle deemed secondary", "16 plus"): assert is_secondary_phase(phase) is True, phase for phase in ("Primary", "Middle deemed primary", "Nursery", "", None): assert is_secondary_phase(phase) is False, phase # In PHASE_GROUPS an all-through school is on both sides, but it renders # with the primary template, and the metric follows the template. assert is_secondary_phase("All-through") is False def test_chips_state_only_what_the_tier_earned(): frame = _frame( _row(100001, "Subject", phase="Secondary", gender="Mixed", religious_denomination="None", admissions_policy="Non-selective"), _row(100002, "Full match", phase="Secondary", gender="Mixed", religious_denomination="None", admissions_policy="Non-selective", latitude=_at(0.5)), _row(100003, "Faith differs", phase="Secondary", gender="Mixed", religious_denomination="Church of England", admissions_policy="Non-selective", latitude=_at(0.6)), ) by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)} assert by_urn[100002]["shared"] == ["Mixed", "Non-selective", "No religious character"] assert by_urn[100003]["shared"] == ["Mixed", "Non-selective"] def test_tier_three_chip_is_the_plain_phase(): frame = _frame( _row(100001, "Subject", gender="Boys"), _row(100002, "A", gender="Mixed", latitude=_at(0.5)), _row(100003, "B", gender="Mixed", latitude=_at(0.6)), ) result = select_similar(frame, 100001, is_secondary=False) assert all(s["shared"] == ["Primary school"] for s in result) def test_metric_follows_the_template_not_the_neighbour(): frame = _frame( _row(100001, "Subject", phase="Secondary", attainment_8_score=50.0), _row(100002, "A", phase="Secondary", attainment_8_score=52.8, latitude=_at(0.5)), _row(100003, "B", phase="Secondary", attainment_8_score=np.nan, latitude=_at(0.6)), ) by_urn = {s["urn"]: s for s in select_similar(frame, 100001, is_secondary=True)} assert by_urn[100002]["metric_key"] == "attainment_8_score" assert by_urn[100002]["metric_value"] == 52.8 assert by_urn[100002]["metric_year"] == 202425 assert by_urn[100003]["metric_value"] is None def test_values_are_json_safe_native_types(): frame = _frame( _row(100001, "Subject"), _row(100002, "A", latitude=_at(0.5)), _row(100003, "B", latitude=_at(0.6)), ) for school in select_similar(frame, 100001, is_secondary=False): assert isinstance(school["urn"], int) assert isinstance(school["distance_miles"], float) assert not isinstance(school["metric_value"], np.generic) # --------------------------------------------------------------------------- # The endpoint # --------------------------------------------------------------------------- import pytest from fastapi.testclient import TestClient def _endpoint_frame(): return _frame( _row(100001, "Subject Primary"), _row(100002, "Neighbour A", latitude=_at(0.5)), _row(100003, "Neighbour B", latitude=_at(0.6)), ) @pytest.fixture() def client(monkeypatch): from backend import app as app_module monkeypatch.setattr(app_module, "load_latest_school_data", _endpoint_frame) monkeypatch.setattr(app_module, "load_school_data", _endpoint_frame) monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {}) return TestClient(app_module.app, raise_server_exceptions=False) def test_detail_payload_carries_similar_schools(client): resp = client.get("/api/schools/100001") assert resp.status_code == 200, resp.text similar = resp.json()["similar_schools"] assert [s["school_name"] for s in similar] == ["Neighbour A", "Neighbour B"] assert similar[0]["metric_key"] == "rwm_expected_pct" def test_a_failure_in_selection_does_not_break_the_page(client, monkeypatch): from backend import app as app_module def _explode(*args, **kwargs): raise ValueError("selection blew up") monkeypatch.setattr(app_module, "select_similar", _explode) resp = client.get("/api/schools/100001") assert resp.status_code == 200, resp.text assert resp.json()["similar_schools"] == []