"""get_supplementary_data_batch fetches one query per table for all URNs (not ~5 per school) and returns the same per-URN block shape as the single-URN function, picking the latest row per URN where relevant.""" import types from backend import data_loader from backend.data_loader import get_supplementary_data_batch def _sort_key(criterion): """(column name, descending) for a SQLAlchemy order_by argument. A bare column (Model.year) arrives as an InstrumentedAttribute carrying .key; Model.year.desc() wraps it in a UnaryExpression whose column sits on .element. """ name = getattr(criterion, "key", None) if name is not None: return name, False element = getattr(criterion, "element", None) name = getattr(element, "key", None) return name, "DESC" in str(criterion).upper() class _FakeQuery: """Records that a query ran and serves canned rows filtered by an in-list. order_by is honoured rather than ignored. The batch loader picks a row per URN by position — first for "latest Ofsted", last for "latest cut-off distance" — which is only correct because the database returned them sorted. A double that drops the ORDER BY makes those picks depend on fixture insertion order instead, so the test would pass with the sort reversed or removed and prove nothing about the query. """ def __init__(self, recorder, model_name, rows): self._rec = recorder self._model = model_name self._rows = rows def filter(self, *args, **kwargs): return self def order_by(self, *criteria): for crit in reversed(criteria): # reversed = stable multi-key sort name, descending = _sort_key(crit) if not name: continue values = [getattr(r, name, None) for r in self._rows] # Only sort on plainly comparable values. Some fixtures stand dates # up as namespace objects, which raise on <; leaving those in their # given order matches what the real query would produce for them. if not all(isinstance(v, (int, float, str)) for v in values): continue self._rows = sorted( self._rows, key=lambda r: getattr(r, name), reverse=descending ) return self def all(self): self._rec.append(self._model) return self._rows def first(self): self._rec.append(self._model) return self._rows[0] if self._rows else None class _FakeSession: def __init__(self, rows_by_model): self.rows_by_model = rows_by_model self.queries: list[str] = [] def query(self, model): name = model.__name__ return _FakeQuery(self.queries, name, self.rows_by_model.get(name, [])) def rollback(self): pass def _ofsted_row(urn, date, oe): base = {f: None for f in ( "framework", "inspection_type", "quality_of_education", "behaviour_attitudes", "personal_development", "leadership_management", "early_years_provision", "sixth_form_provision", "ungraded_outcome", "ungraded_grade", "rc_safeguarding_met", "rc_inclusion", "rc_curriculum_teaching", "rc_achievement", "rc_attendance_behaviour", "rc_personal_development", "rc_leadership_governance", "rc_early_years", "rc_sixth_form", "report_url", )} base.update(urn=urn, inspection_date=types.SimpleNamespace(isoformat=lambda: date), overall_effectiveness=oe, grade_source=None) return types.SimpleNamespace(**base) def _adm_row(urn, year): return types.SimpleNamespace( urn=urn, year=year, school_phase="Primary", places_offered=100, total_applications=200, first_preference_applications=150, first_preference_offers=140, first_preference_offer_pct=93.3, oversubscription_ratio=1.5, oversubscribed=True, total_offers=100, second_preference_offers=5, third_preference_offers=2, cross_la_applications=10, cross_la_offers=3, ) def _dist_row(urn, year, distance_m, route_count=1): return types.SimpleNamespace( urn=urn, year=year, distance_m=distance_m, route_count=route_count, la_name="Camden", distance_unit_raw="miles", source_file="camden/guide.pdf", ) def test_one_query_per_table_and_latest_row_per_urn(): rows = { # URN 1 has two Ofsted rows; the batch must keep the most recent (2023). "FactOfstedInspection": [ _ofsted_row(1, "2023-01-01", 2), _ofsted_row(1, "2019-01-01", 3), _ofsted_row(2, "2021-06-01", 1), ], "FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)], # URN 1 has three years of cut-offs; only the most recent is served. # Deliberately not in year order — the ordering is the query's job. "FactAdmissionDistance": [ _dist_row(1, 2026, 529.47), _dist_row(1, 2024, 772.49), _dist_row(1, 2025, 1421.05), _dist_row(2, 2023, 2029.38, route_count=4), ], "FactPupilCharacteristics": [], "FactDeprivation": [], "FactFinance": [], } session = _FakeSession(rows) out = get_supplementary_data_batch(session, [1, 2]) # Exactly one query per table — six total, regardless of two URNs. assert sorted(session.queries) == [ "FactAdmissionDistance", "FactAdmissions", "FactDeprivation", "FactFinance", "FactOfstedInspection", "FactPupilCharacteristics", ] # Latest Ofsted kept per URN assert out[1]["ofsted"]["overall_effectiveness"] == 2 assert out[2]["ofsted"]["overall_effectiveness"] == 1 # Admissions history grouped per URN, latest exposed as `admissions` assert [r["year"] for r in out[1]["admissions_history"]] == [202526, 202627] assert out[1]["admissions"]["year"] == 202627 assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}] # Cut-off distance: full history oldest-first, with the newest year also # exposed on its own. The fixture rows are deliberately out of order, so # this only passes if the query's ORDER BY is doing the work. assert [r["year"] for r in out[1]["admission_distance_history"]] == [2024, 2025, 2026] assert out[1]["admission_distance"]["year"] == 2026 assert out[1]["admission_distance"]["distance_m"] == 529.47 # route_count travels with the figure — the page needs it to say the # distance is the furthest of several bands rather than the only one. assert out[2]["admission_distance"]["route_count"] == 4 # Empty tables degrade to the null block, not a crash assert out[1]["census"] is None and out[1]["deprivation"] is None def test_single_wrapper_matches_batch(monkeypatch): session = _FakeSession({"FactOfstedInspection": [_ofsted_row(5, "2022-01-01", 2)]}) single = data_loader.get_supplementary_data(session, 5) assert single["ofsted"]["overall_effectiveness"] == 2 assert single["admissions_history"] == [] assert single["admission_distance"] is None assert single["admission_distance_history"] == []