Files
school_compare/backend/tests/test_supplementary_batch.py
T
TudorandClaude Opus 5 c9a1892bfb
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m4s
PR Checks / Backend Smoke (pull_request) Successful in 7s
PR Checks / Build Backend (no push) (pull_request) Successful in 16s
PR Checks / Build Frontend (no push) (pull_request) Successful in 45s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 11s
PR Checks / AI Code Review (Claude) (pull_request) Failing after 3m17s
feat(admissions): publish the latest cut-off only, holding history back
Earlier years are to become a paid feature, so they stop being published.

The load-bearing part is that this is a change to the API, not only to the
page. /api/schools/{urn} is public and unauthenticated: leaving
admission_distance_history in the payload while declining to render it would
have handed the whole record to anyone who opened the network tab. It is
withheld at the source, and the page follows.

Nothing changes upstream. The tap, the plausibility band and
fact_admission_distance are untouched and still load every published year, so
restoring history for entitled callers is a change to one function in
data_loader rather than a re-collection.

What the reader now gets is the latest figure on the Admissions tile, and a
Distance section that answers the question the number alone cannot: whether
their own address falls inside it. Retitled to "How far away are you?", which
is what it now does — the previous title described a record that is no longer
there.

Removed with the history: the trend chart, the year-by-year table, the
per-year verdict strip, the trend summary and the coverage note, along with
their CSS. The section goes from 743px to 417px.

One consequence worth naming. A run of years used to soften a single close
call — a home just outside one year's cut-off was usually inside another. With
one year published, the "too close to call" band is the entire safety margin
between a parent and a place they do not have, so the verdict now names its
year, and the three outcomes are tinted apart rather than distinguished by
wording alone.

The existing stylesheet test earned its keep here: the three verdict classes
were referenced before they were written, and it caught them. Unstyled, a
"beyond the cut-off" result would have been indistinguishable from an "inside"
one — the exact failure the longhand class map was written to prevent.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01WDvkyqqHABm4bmth2kjAxE
2026-08-20 18:44:57 +01:00

177 lines
7.1 KiB
Python

"""get_supplementary_data_batch fetches one query per table for all URNs
(not ~5 per school) and returns the same per-URN block shape as the
single-URN function, picking the latest row per URN where relevant."""
import types
from backend import data_loader
from backend.data_loader import get_supplementary_data_batch
def _sort_key(criterion):
"""(column name, descending) for a SQLAlchemy order_by argument.
A bare column (Model.year) arrives as an InstrumentedAttribute carrying
.key; Model.year.desc() wraps it in a UnaryExpression whose column sits on
.element.
"""
name = getattr(criterion, "key", None)
if name is not None:
return name, False
element = getattr(criterion, "element", None)
name = getattr(element, "key", None)
return name, "DESC" in str(criterion).upper()
class _FakeQuery:
"""Records that a query ran and serves canned rows filtered by an in-list.
order_by is honoured rather than ignored. The batch loader picks a row per
URN by position — first for "latest Ofsted", last for "latest cut-off
distance" — which is only correct because the database returned them
sorted. A double that drops the ORDER BY makes those picks depend on
fixture insertion order instead, so the test would pass with the sort
reversed or removed and prove nothing about the query.
"""
def __init__(self, recorder, model_name, rows):
self._rec = recorder
self._model = model_name
self._rows = rows
def filter(self, *args, **kwargs):
return self
def order_by(self, *criteria):
for crit in reversed(criteria): # reversed = stable multi-key sort
name, descending = _sort_key(crit)
if not name:
continue
values = [getattr(r, name, None) for r in self._rows]
# Only sort on plainly comparable values. Some fixtures stand dates
# up as namespace objects, which raise on <; leaving those in their
# given order matches what the real query would produce for them.
if not all(isinstance(v, (int, float, str)) for v in values):
continue
self._rows = sorted(
self._rows, key=lambda r: getattr(r, name), reverse=descending
)
return self
def all(self):
self._rec.append(self._model)
return self._rows
def first(self):
self._rec.append(self._model)
return self._rows[0] if self._rows else None
class _FakeSession:
def __init__(self, rows_by_model):
self.rows_by_model = rows_by_model
self.queries: list[str] = []
def query(self, model):
name = model.__name__
return _FakeQuery(self.queries, name, self.rows_by_model.get(name, []))
def rollback(self):
pass
def _ofsted_row(urn, date, oe):
base = {f: None for f in (
"framework", "inspection_type", "quality_of_education", "behaviour_attitudes",
"personal_development", "leadership_management", "early_years_provision",
"sixth_form_provision", "ungraded_outcome", "ungraded_grade",
"rc_safeguarding_met", "rc_inclusion", "rc_curriculum_teaching", "rc_achievement",
"rc_attendance_behaviour", "rc_personal_development", "rc_leadership_governance",
"rc_early_years", "rc_sixth_form", "report_url",
)}
base.update(urn=urn, inspection_date=types.SimpleNamespace(isoformat=lambda: date),
overall_effectiveness=oe, grade_source=None)
return types.SimpleNamespace(**base)
def _adm_row(urn, year):
return types.SimpleNamespace(
urn=urn, year=year, school_phase="Primary", places_offered=100,
total_applications=200, first_preference_applications=150,
first_preference_offers=140, first_preference_offer_pct=93.3,
oversubscription_ratio=1.5, oversubscribed=True,
total_offers=100, second_preference_offers=5, third_preference_offers=2,
cross_la_applications=10, cross_la_offers=3,
)
def _dist_row(urn, year, distance_m, route_count=1):
return types.SimpleNamespace(
urn=urn, year=year, distance_m=distance_m, route_count=route_count,
la_name="Camden", distance_unit_raw="miles", source_file="camden/guide.pdf",
)
def test_one_query_per_table_and_latest_row_per_urn():
rows = {
# URN 1 has two Ofsted rows; the batch must keep the most recent (2023).
"FactOfstedInspection": [
_ofsted_row(1, "2023-01-01", 2),
_ofsted_row(1, "2019-01-01", 3),
_ofsted_row(2, "2021-06-01", 1),
],
"FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)],
# URN 1 has three years of cut-offs; only the most recent is served.
# Deliberately not in year order — the ordering is the query's job.
"FactAdmissionDistance": [
_dist_row(1, 2026, 529.47),
_dist_row(1, 2024, 772.49),
_dist_row(1, 2025, 1421.05),
_dist_row(2, 2023, 2029.38, route_count=4),
],
"FactPupilCharacteristics": [],
"FactDeprivation": [],
"FactFinance": [],
}
session = _FakeSession(rows)
out = get_supplementary_data_batch(session, [1, 2])
# Exactly one query per table — six total, regardless of two URNs.
assert sorted(session.queries) == [
"FactAdmissionDistance", "FactAdmissions", "FactDeprivation",
"FactFinance", "FactOfstedInspection", "FactPupilCharacteristics",
]
# Latest Ofsted kept per URN
assert out[1]["ofsted"]["overall_effectiveness"] == 2
assert out[2]["ofsted"]["overall_effectiveness"] == 1
# Admissions history grouped per URN, latest exposed as `admissions`
assert [r["year"] for r in out[1]["admissions_history"]] == [202526, 202627]
assert out[1]["admissions"]["year"] == 202627
assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}]
# Cut-off distance: the latest year only. Earlier years stay in the mart
# but are held back as a paid feature, and this API is public — serving
# them here would hand them to anyone reading the response. The fixture
# rows are deliberately out of order, so "latest" only comes out right if
# the query's ORDER BY is doing the work.
assert out[1]["admission_distance"]["year"] == 2026
assert "admission_distance_history" not in out[1]
assert out[1]["admission_distance"]["distance_m"] == 529.47
# route_count travels with the figure — the page needs it to say the
# distance is the furthest of several bands rather than the only one.
assert out[2]["admission_distance"]["route_count"] == 4
# Empty tables degrade to the null block, not a crash
assert out[1]["census"] is None and out[1]["deprivation"] is None
def test_single_wrapper_matches_batch(monkeypatch):
session = _FakeSession({"FactOfstedInspection": [_ofsted_row(5, "2022-01-01", 2)]})
single = data_loader.get_supplementary_data(session, 5)
assert single["ofsted"]["overall_effectiveness"] == 2
assert single["admissions_history"] == []
assert single["admission_distance"] is None
assert "admission_distance_history" not in single