Files
school_compare/backend/tests/test_supplementary_batch.py
T

183 lines
7.4 KiB
Python
Raw Normal View History

"""get_supplementary_data_batch fetches one query per table for all URNs
(not ~5 per school) and returns the same per-URN block shape as the
single-URN function, picking the latest row per URN where relevant."""
import types
from backend import data_loader
from backend.data_loader import get_supplementary_data_batch
def _sort_key(criterion):
"""(column name, descending) for a SQLAlchemy order_by argument.
A bare column (Model.year) arrives as an InstrumentedAttribute carrying
.key; Model.year.desc() wraps it in a UnaryExpression whose column sits on
.element.
"""
name = getattr(criterion, "key", None)
if name is not None:
return name, False
element = getattr(criterion, "element", None)
name = getattr(element, "key", None)
return name, "DESC" in str(criterion).upper()
class _FakeQuery:
"""Records that a query ran and serves canned rows filtered by an in-list.
order_by is honoured rather than ignored. The batch loader picks a row per
URN by position — first for "latest Ofsted", last for "latest cut-off
distance" — which is only correct because the database returned them
sorted. A double that drops the ORDER BY makes those picks depend on
fixture insertion order instead, so the test would pass with the sort
reversed or removed and prove nothing about the query.
"""
def __init__(self, recorder, model_name, rows):
self._rec = recorder
self._model = model_name
self._rows = rows
def filter(self, *args, **kwargs):
return self
def order_by(self, *criteria):
for crit in reversed(criteria): # reversed = stable multi-key sort
name, descending = _sort_key(crit)
if not name:
continue
values = [getattr(r, name, None) for r in self._rows]
# Only sort on plainly comparable values. Some fixtures stand dates
# up as namespace objects, which raise on <; leaving those in their
# given order matches what the real query would produce for them.
if not all(isinstance(v, (int, float, str)) for v in values):
continue
self._rows = sorted(
self._rows, key=lambda r: getattr(r, name), reverse=descending
)
return self
def all(self):
self._rec.append(self._model)
return self._rows
def first(self):
self._rec.append(self._model)
return self._rows[0] if self._rows else None
class _FakeSession:
def __init__(self, rows_by_model):
self.rows_by_model = rows_by_model
self.queries: list[str] = []
def query(self, model):
name = model.__name__
return _FakeQuery(self.queries, name, self.rows_by_model.get(name, []))
def rollback(self):
pass
def _ofsted_row(urn, date, oe):
base = {f: None for f in (
"framework", "inspection_type", "quality_of_education", "behaviour_attitudes",
"personal_development", "leadership_management", "early_years_provision",
"sixth_form_provision", "ungraded_outcome", "ungraded_grade",
"rc_safeguarding_met", "rc_inclusion", "rc_curriculum_teaching", "rc_achievement",
"rc_attendance_behaviour", "rc_personal_development", "rc_leadership_governance",
"rc_early_years", "rc_sixth_form", "report_url",
)}
base.update(urn=urn, inspection_date=types.SimpleNamespace(isoformat=lambda: date),
overall_effectiveness=oe, grade_source=None)
return types.SimpleNamespace(**base)
def _adm_row(urn, year):
return types.SimpleNamespace(
urn=urn, year=year, school_phase="Primary", places_offered=100,
total_applications=200, first_preference_applications=150,
first_preference_offers=140, first_preference_offer_pct=93.3,
oversubscription_ratio=1.5, oversubscribed=True,
total_offers=100, second_preference_offers=5, third_preference_offers=2,
cross_la_applications=10, cross_la_offers=3,
)
def _dist_row(urn, year, distance_m, route_count=1):
return types.SimpleNamespace(
urn=urn, year=year, distance_m=distance_m, route_count=route_count,
la_name="Camden", distance_unit_raw="miles", source_file="camden/guide.pdf",
)
def test_one_query_per_table_and_latest_row_per_urn():
rows = {
# URN 1 has two Ofsted rows; the batch must keep the most recent (2023).
"FactOfstedInspection": [
_ofsted_row(1, "2023-01-01", 2),
_ofsted_row(1, "2019-01-01", 3),
_ofsted_row(2, "2021-06-01", 1),
],
"FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)],
# URN 1 has three years of cut-offs; only the most recent is served.
# Deliberately not in year order — the ordering is the query's job.
"FactAdmissionDistance": [
_dist_row(1, 2026, 529.47),
_dist_row(1, 2024, 772.49),
_dist_row(1, 2025, 1421.05),
_dist_row(2, 2023, 2029.38, route_count=4),
],
"FactPupilCharacteristics": [],
"FactDeprivation": [],
"FactFinance": [],
"FactKs4Destinations": [],
"FactKs5Destinations": [],
}
session = _FakeSession(rows)
out = get_supplementary_data_batch(session, [1, 2])
# Exactly one query per table — eight total, regardless of two URNs.
assert sorted(session.queries) == [
"FactAdmissionDistance", "FactAdmissions", "FactDeprivation",
"FactFinance", "FactKs4Destinations", "FactKs5Destinations",
"FactOfstedInspection", "FactPupilCharacteristics",
]
# A school with no destination rows gets null, not an empty shell — the
# frontend renders the section from the block's presence.
assert out[1]["destinations"] is None
# Latest Ofsted kept per URN
assert out[1]["ofsted"]["overall_effectiveness"] == 2
assert out[2]["ofsted"]["overall_effectiveness"] == 1
# Admissions history grouped per URN, latest exposed as `admissions`
assert [r["year"] for r in out[1]["admissions_history"]] == [202526, 202627]
assert out[1]["admissions"]["year"] == 202627
assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}]
# Cut-off distance: the latest year only. Earlier years stay in the mart
# but are held back as a paid feature, and this API is public — serving
# them here would hand them to anyone reading the response. The fixture
# rows are deliberately out of order, so "latest" only comes out right if
# the query's ORDER BY is doing the work.
assert out[1]["admission_distance"]["year"] == 2026
assert "admission_distance_history" not in out[1]
assert out[1]["admission_distance"]["distance_m"] == 529.47
# route_count travels with the figure — the page needs it to say the
# distance is the furthest of several bands rather than the only one.
assert out[2]["admission_distance"]["route_count"] == 4
# Empty tables degrade to the null block, not a crash
assert out[1]["census"] is None and out[1]["deprivation"] is None
def test_single_wrapper_matches_batch(monkeypatch):
session = _FakeSession({"FactOfstedInspection": [_ofsted_row(5, "2022-01-01", 2)]})
single = data_loader.get_supplementary_data(session, 5)
assert single["ofsted"]["overall_effectiveness"] == 2
assert single["admissions_history"] == []
assert single["admission_distance"] is None
assert "admission_distance_history" not in single