Compare commits

..
Author SHA1 Message Date
TudorandClaude Fable 5 315f1feede perf(api): batch supplementary queries — one per table, not five per school
PR Checks / Backend Smoke (pull_request) Canceled after 0s
PR Checks / Build Backend (no push) (pull_request) Canceled after 0s
PR Checks / Build Frontend (no push) (pull_request) Canceled after 0s
PR Checks / Build Pipeline (no push) (pull_request) Canceled after 0s
PR Checks / AI Code Review (Claude) (pull_request) Canceled after 0s
PR Checks / Frontend Typecheck + Tests (pull_request) Canceled after 8m16s
get_supplementary_data ran ~5 sequential DB round-trips per URN, so
/api/compare scaled at ~37ms/school (measured on staging: 1 school 155ms,
3 schools 220ms, 6 schools 340ms). get_supplementary_data_batch fetches
each table once with WHERE urn IN (...) and groups in Python, collapsing
5*N round-trips to a constant 5. get_supplementary_data is now a thin
wrapper so the detail endpoint is unchanged; the compare endpoint makes
one batched call. Each table degrades independently on failure.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0146VHeLAWjDVE2B5uU67jCB
2026-07-14 22:42:10 +01:00
8 changed files with 323 additions and 215 deletions
+4 -1
View File
@@ -30,6 +30,7 @@ from .data_loader import (
load_latest_school_data, load_latest_school_data,
geocode_single_postcode, geocode_single_postcode,
get_supplementary_data, get_supplementary_data,
get_supplementary_data_batch,
search_schools_typesense, search_schools_typesense,
) )
from .data_loader import get_data_info as get_db_info from .data_loader import get_data_info as get_db_info
@@ -679,8 +680,10 @@ async def compare_schools(
db = None db = None
try: try:
db = database.SessionLocal() db = database.SessionLocal()
# One query per table for all schools, not ~5 queries per school.
batch = get_supplementary_data_batch(db, urn_list)
for urn in urn_list: for urn in urn_list:
supp = get_supplementary_data(db, urn) supp = batch.get(urn, {})
supplementary_by_urn[urn] = { supplementary_by_urn[urn] = {
key: supp.get(key, default) key: supp.get(key, default)
for key, default in _EMPTY_SUPPLEMENTARY.items() for key, default in _EMPTY_SUPPLEMENTARY.items()
+123 -65
View File
@@ -662,30 +662,8 @@ def _admissions_row_dict(a) -> dict:
} }
def get_supplementary_data(db: Session, urn: int) -> dict: def _census_dict(pc) -> dict:
"""Fetch all supplementary data for a single school URN.""" return {
result = {}
def safe_query(model, pk_field, latest_field=None):
try:
q = db.query(model).filter(getattr(model, pk_field) == urn)
if latest_field:
q = q.order_by(getattr(model, latest_field).desc())
return q.first()
except Exception as e:
import logging
logging.getLogger(__name__).error("safe_query failed for %s: %s", model.__name__, e)
db.rollback()
return None
# Latest Ofsted inspection
o = safe_query(FactOfstedInspection, "urn", "inspection_date")
result["ofsted"] = _ofsted_block(o, urn) if o else None
# Census (latest year of fact_pupil_characteristics)
pc = safe_query(FactPupilCharacteristics, "urn", "year")
result["census"] = (
{
"year": pc.year, "year": pc.year,
"total_pupils": pc.total_pupils, "total_pupils": pc.total_pupils,
"female_pupils": pc.female_pupils, "female_pupils": pc.female_pupils,
@@ -693,52 +671,18 @@ def get_supplementary_data(db: Session, urn: int) -> dict:
"fsm_pct": pc.fsm_pct, "fsm_pct": pc.fsm_pct,
"eal_pct": pc.eal_pct, "eal_pct": pc.eal_pct,
} }
if pc
else None
)
# Admissions — all years, oldest first (for the multi-year trend view).
try:
admissions_rows = (
db.query(FactAdmissions)
.filter(FactAdmissions.urn == urn)
.order_by(FactAdmissions.year.asc())
.all()
)
except Exception as e:
import logging
logging.getLogger(__name__).error("admissions history query failed: %s", e)
db.rollback()
admissions_rows = []
history = [_admissions_row_dict(a) for a in admissions_rows] def _deprivation_dict(d) -> dict:
result["admissions_history"] = history return {
# Keep the single latest-year object for backwards-compatible consumers
# (hero chips, etc.).
result["admissions"] = history[-1] if history else None
# SEN detail — not available in current marts
result["sen_detail"] = None
# Phonics — no school-level data on EES
result["phonics"] = None
# Deprivation
d = safe_query(FactDeprivation, "urn")
result["deprivation"] = (
{
"lsoa_code": d.lsoa_code, "lsoa_code": d.lsoa_code,
"idaci_score": d.idaci_score, "idaci_score": d.idaci_score,
"idaci_decile": d.idaci_decile, "idaci_decile": d.idaci_decile,
} }
if d
else None
)
# Finance (latest year)
f = safe_query(FactFinance, "urn", "year") def _finance_dict(f) -> dict:
result["finance"] = ( return {
{
"year": f.year, "year": f.year,
"per_pupil_spend": f.per_pupil_spend, "per_pupil_spend": f.per_pupil_spend,
"staff_cost_pct": f.staff_cost_pct, "staff_cost_pct": f.staff_cost_pct,
@@ -746,8 +690,122 @@ def get_supplementary_data(db: Session, urn: int) -> dict:
"support_staff_cost_pct": f.support_staff_cost_pct, "support_staff_cost_pct": f.support_staff_cost_pct,
"premises_cost_pct": f.premises_cost_pct, "premises_cost_pct": f.premises_cost_pct,
} }
if f
else None
def _empty_supplementary() -> dict:
return {
"ofsted": None,
"census": None,
"admissions": None,
"admissions_history": [],
"sen_detail": None,
"phonics": None,
"deprivation": None,
"finance": None,
}
def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict:
"""Fetch supplementary data for many URNs with one query per table
(WHERE urn IN (...)) instead of ~5 queries per school, collapsing the
per-request round-trips from 5*N to a constant 5. Returns {urn: block}
with the same shape get_supplementary_data produces per URN.
Each table is queried independently and failures degrade that table to
empty for every URN — a missing mart never blanks the others.
"""
urns = [int(u) for u in urns]
result = {urn: _empty_supplementary() for urn in urns}
if not urns:
return result
def _safe(fn):
try:
fn()
except Exception as e:
import logging
logging.getLogger(__name__).error("batch supplementary query failed: %s", e)
db.rollback()
# Ofsted — latest inspection per URN. Ordered so the first row seen per
# URN is the most recent.
def _ofsted():
rows = (
db.query(FactOfstedInspection)
.filter(FactOfstedInspection.urn.in_(urns))
.order_by(FactOfstedInspection.urn, FactOfstedInspection.inspection_date.desc())
.all()
) )
seen = set()
for o in rows:
if o.urn in seen:
continue
seen.add(o.urn)
result[o.urn]["ofsted"] = _ofsted_block(o, o.urn)
_safe(_ofsted)
# Census — latest year per URN.
def _census():
rows = (
db.query(FactPupilCharacteristics)
.filter(FactPupilCharacteristics.urn.in_(urns))
.order_by(FactPupilCharacteristics.urn, FactPupilCharacteristics.year.desc())
.all()
)
seen = set()
for pc in rows:
if pc.urn in seen:
continue
seen.add(pc.urn)
result[pc.urn]["census"] = _census_dict(pc)
_safe(_census)
# Admissions — all years per URN, oldest first (multi-year trend view).
def _admissions():
rows = (
db.query(FactAdmissions)
.filter(FactAdmissions.urn.in_(urns))
.order_by(FactAdmissions.urn, FactAdmissions.year.asc())
.all()
)
history: dict = {urn: [] for urn in urns}
for a in rows:
history[a.urn].append(_admissions_row_dict(a))
for urn, rows_for_urn in history.items():
result[urn]["admissions_history"] = rows_for_urn
result[urn]["admissions"] = rows_for_urn[-1] if rows_for_urn else None
_safe(_admissions)
# Deprivation — one row per URN.
def _deprivation():
rows = (
db.query(FactDeprivation)
.filter(FactDeprivation.urn.in_(urns))
.all()
)
for d in rows:
result[d.urn]["deprivation"] = _deprivation_dict(d)
_safe(_deprivation)
# Finance — latest year per URN.
def _finance():
rows = (
db.query(FactFinance)
.filter(FactFinance.urn.in_(urns))
.order_by(FactFinance.urn, FactFinance.year.desc())
.all()
)
seen = set()
for f in rows:
if f.urn in seen:
continue
seen.add(f.urn)
result[f.urn]["finance"] = _finance_dict(f)
_safe(_finance)
return result return result
def get_supplementary_data(db: Session, urn: int) -> dict:
"""Supplementary data for a single URN (thin wrapper over the batch)."""
return get_supplementary_data_batch(db, [urn])[int(urn)]
+5 -3
View File
@@ -67,7 +67,9 @@ def client(monkeypatch):
monkeypatch.setattr(app_module, "load_school_data", _two_primary_schools_df) monkeypatch.setattr(app_module, "load_school_data", _two_primary_schools_df)
monkeypatch.setattr( monkeypatch.setattr(
app_module, "get_supplementary_data", lambda db, urn: dict(CANNED_SUPPLEMENTARY) app_module,
"get_supplementary_data_batch",
lambda db, urns: {int(u): dict(CANNED_SUPPLEMENTARY) for u in urns},
) )
monkeypatch.setattr(database_module, "SessionLocal", _StubSession) monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
return TestClient(app_module.app, raise_server_exceptions=False) return TestClient(app_module.app, raise_server_exceptions=False)
@@ -102,10 +104,10 @@ def test_top_level_national_averages_and_benchmarks(client):
def test_supplementary_failure_degrades_not_500(client, monkeypatch): def test_supplementary_failure_degrades_not_500(client, monkeypatch):
from backend import app as app_module from backend import app as app_module
def _boom(db, urn): def _boom(db, urns):
raise RuntimeError("marts unavailable") raise RuntimeError("marts unavailable")
monkeypatch.setattr(app_module, "get_supplementary_data", _boom) monkeypatch.setattr(app_module, "get_supplementary_data_batch", _boom)
resp = client.get("/api/compare?urns=100140") resp = client.get("/api/compare?urns=100140")
assert resp.status_code == 200 assert resp.status_code == 200
school = resp.json()["comparison"]["100140"] school = resp.json()["comparison"]["100140"]
+111
View File
@@ -0,0 +1,111 @@
"""get_supplementary_data_batch fetches one query per table for all URNs
(not ~5 per school) and returns the same per-URN block shape as the
single-URN function, picking the latest row per URN where relevant."""
import types
from backend import data_loader
from backend.data_loader import get_supplementary_data_batch
class _FakeQuery:
"""Records that a query ran and serves canned rows filtered by an in-list."""
def __init__(self, recorder, model_name, rows):
self._rec = recorder
self._model = model_name
self._rows = rows
def filter(self, *args, **kwargs):
return self
def order_by(self, *args, **kwargs):
return self
def all(self):
self._rec.append(self._model)
return self._rows
def first(self):
self._rec.append(self._model)
return self._rows[0] if self._rows else None
class _FakeSession:
def __init__(self, rows_by_model):
self.rows_by_model = rows_by_model
self.queries: list[str] = []
def query(self, model):
name = model.__name__
return _FakeQuery(self.queries, name, self.rows_by_model.get(name, []))
def rollback(self):
pass
def _ofsted_row(urn, date, oe):
base = {f: None for f in (
"framework", "inspection_type", "quality_of_education", "behaviour_attitudes",
"personal_development", "leadership_management", "early_years_provision",
"sixth_form_provision", "ungraded_outcome", "ungraded_grade",
"rc_safeguarding_met", "rc_inclusion", "rc_curriculum_teaching", "rc_achievement",
"rc_attendance_behaviour", "rc_personal_development", "rc_leadership_governance",
"rc_early_years", "rc_sixth_form", "report_url",
)}
base.update(urn=urn, inspection_date=types.SimpleNamespace(isoformat=lambda: date),
overall_effectiveness=oe, grade_source=None)
return types.SimpleNamespace(**base)
def _adm_row(urn, year):
return types.SimpleNamespace(
urn=urn, year=year, school_phase="Primary", places_offered=100,
total_applications=200, first_preference_applications=150,
first_preference_offers=140, first_preference_offer_pct=93.3,
oversubscription_ratio=1.5, oversubscribed=True,
total_offers=100, second_preference_offers=5, third_preference_offers=2,
cross_la_applications=10, cross_la_offers=3,
)
def test_one_query_per_table_and_latest_row_per_urn():
rows = {
# URN 1 has two Ofsted rows; the batch must keep the most recent (2023).
"FactOfstedInspection": [
_ofsted_row(1, "2023-01-01", 2),
_ofsted_row(1, "2019-01-01", 3),
_ofsted_row(2, "2021-06-01", 1),
],
"FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)],
"FactPupilCharacteristics": [],
"FactDeprivation": [],
"FactFinance": [],
}
session = _FakeSession(rows)
out = get_supplementary_data_batch(session, [1, 2])
# Exactly one query per table — five total, regardless of two URNs.
assert sorted(session.queries) == [
"FactAdmissions", "FactDeprivation", "FactFinance",
"FactOfstedInspection", "FactPupilCharacteristics",
]
# Latest Ofsted kept per URN
assert out[1]["ofsted"]["overall_effectiveness"] == 2
assert out[2]["ofsted"]["overall_effectiveness"] == 1
# Admissions history grouped per URN, latest exposed as `admissions`
assert [r["year"] for r in out[1]["admissions_history"]] == [202526, 202627]
assert out[1]["admissions"]["year"] == 202627
assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}]
# Empty tables degrade to the null block, not a crash
assert out[1]["census"] is None and out[1]["deprivation"] is None
def test_single_wrapper_matches_batch(monkeypatch):
session = _FakeSession({"FactOfstedInspection": [_ofsted_row(5, "2022-01-01", 2)]})
single = data_loader.get_supplementary_data(session, 5)
assert single["ofsted"]["overall_effectiveness"] == 2
assert single["admissions_history"] == []
+11 -26
View File
@@ -19,27 +19,6 @@ function schoolLinks(page: Page) {
return page.locator('a[href^="/school/"]'); return page.locator('a[href^="/school/"]');
} }
/**
* Two URNs guaranteed to be pure-primary (same phase). The compare page's
* phase tabs split all-through schools (which carry KS4 data) onto the
* secondary tab, so picking two arbitrary "primary" search hits can land
* them on different tabs where only the active one renders. Selecting via
* the API by exact phase keeps both on the same tab. Data-invariant: uses
* whatever primaries the environment holds.
*/
async function twoPrimaryUrns(page: Page): Promise<[string, string]> {
const res = await page.request.get('/api/schools?search=primary&per_page=50');
expect(res.ok()).toBeTruthy();
const body = await res.json();
const urns: string[] = (body.schools ?? [])
.filter((s: { phase?: string; rwm_expected_pct?: number | null }) =>
s.phase === 'Primary' && s.rwm_expected_pct != null,
)
.map((s: { urn: number }) => String(s.urn));
expect(urns.length).toBeGreaterThanOrEqual(2);
return [urns[0], urns[1]];
}
test('home page loads with hero search', async ({ page }) => { test('home page loads with hero search', async ({ page }) => {
await page.goto('/'); await page.goto('/');
await expect(page.locator('h1').first()).toBeVisible(); await expect(page.locator('h1').first()).toBeVisible();
@@ -160,13 +139,19 @@ test('results map fullscreen falls back to an overlay on iOS', async ({ page })
}); });
test('comparing two schools shows the parent-first sections side by side', async ({ page }) => { test('comparing two schools shows the parent-first sections side by side', async ({ page }) => {
// Two same-phase (pure primary) schools so both stay on one tab. // Collect two school URNs from search results, then load the share URL
const [urn0, urn1] = await twoPrimaryUrns(page); await searchByName(page, 'primary');
await expect(schoolLinks(page).first()).toBeVisible({ timeout: 15_000 });
const hrefs = await schoolLinks(page).evaluateAll((links) =>
links.map((l) => (l as HTMLAnchorElement).getAttribute('href') || '')
);
const urns = [...new Set(hrefs.map((h) => h.match(/\/school\/(\d+)/)?.[1]).filter(Boolean))];
expect(urns.length).toBeGreaterThanOrEqual(2);
await page.goto(`/compare?urns=${urn0},${urn1}`); await page.goto(`/compare?urns=${urns[0]},${urns[1]}`);
// Both schools' detail links should render in the comparison view // Both schools' detail links should render in the comparison view
await expect(page.locator(`a[href*="${urn0}"]`).first()).toBeVisible({ timeout: 15_000 }); await expect(page.locator(`a[href*="${urns[0]}"]`).first()).toBeVisible({ timeout: 15_000 });
await expect(page.locator(`a[href*="${urn1}"]`).first()).toBeVisible(); await expect(page.locator(`a[href*="${urns[1]}"]`).first()).toBeVisible();
// The parent-first sections render in order (data-invariant: headings only) // The parent-first sections render in order (data-invariant: headings only)
for (const heading of [ for (const heading of [
@@ -1,82 +0,0 @@
/**
* Regression: on refresh, the compare page must show the SSR-rendered data.
*
* The basket hydrates from the URL a beat after mount (selectedSchools is
* empty for the first render), so the fetch effect must not blank the
* SSR payload during that window — and must not refetch data the server
* already provided.
*/
import { render, screen, waitFor } from '@testing-library/react';
import { ComparisonView } from '@/components/ComparisonView';
import { ComparisonProvider } from '@/context/ComparisonProvider';
import type { ComparisonData, School } from '@/lib/types';
const fetchComparison = jest.fn();
jest.mock('@/lib/api', () => ({
fetchComparison: (...args: unknown[]) => fetchComparison(...args),
}));
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
function school(urn: number, name: string): School {
return {
urn,
school_name: name,
local_authority: 'Testshire',
school_type: 'Community school',
rwm_expected_pct: 80,
phase: 'Primary',
} as School;
}
function data(urn: number, name: string): ComparisonData {
return {
school_info: school(urn, name),
yearly_data: [{ year: 202425, rwm_expected_pct: 80 }] as ComparisonData['yearly_data'],
ofsted: null,
census: null,
admissions: null,
admissions_history: [],
deprivation: null,
};
}
const INITIAL_DATA = {
'100': data(100, 'Alpha Primary'),
'200': data(200, 'Beta Primary'),
};
beforeEach(() => {
fetchComparison.mockReset();
});
test('renders SSR data on refresh without wiping it or refetching', async () => {
render(
<ComparisonProvider>
<ComparisonView
initialData={INITIAL_DATA}
initialNationalAverages={{
year: 202425,
primary: { rwm_expected_pct: 62 },
secondary: {},
by_year: [],
}}
initialBenchmarks={undefined}
initialUrns={[100, 200]}
metrics={[]}
selectedMetric="rwm_expected_pct"
/>
</ComparisonProvider>,
);
// Both SSR-provided schools appear (data was not blanked during hydration)
await waitFor(() => {
expect(screen.getAllByText('Alpha Primary').length).toBeGreaterThan(0);
});
expect(screen.getAllByText('Beta Primary').length).toBeGreaterThan(0);
expect(screen.getByRole('heading', { name: 'At a glance' })).toBeInTheDocument();
// …and the client never refetched data the server already rendered.
expect(fetchComparison).not.toHaveBeenCalled();
});
+18 -19
View File
@@ -107,26 +107,24 @@ export function ComparisonView({
router.replace(newUrl, { scroll: false }); router.replace(newUrl, { scroll: false });
}, [urnKey, selectedMetric, pathname, searchParams, router]); }, [urnKey, selectedMetric, pathname, searchParams, router]);
// Fetch when the school set changes, but only for schools we don't already // Fetch only when the school set changes. The very first run is skipped
// have data for. This skips the refetch of SSR-rendered data on load AND // when the SSR payload already covers the current set — no double-fetch
// avoids a network call when a school is merely removed. A ref holds the // of data the server just rendered.
// latest data so the effect can read it without re-running on every fetch. const firstFetchRef = useRef(true);
//
// Correctness note: we must NOT null the data on a transient empty urnKey.
// On mount the basket is empty for a beat before it hydrates from the URL,
// and blanking here (then skipping the refetch because SSR "covers" the set)
// was leaving the page empty on refresh. The render already shows the empty
// state whenever `selectedSchools` is empty, so stale data for deselected
// schools is harmless — it's simply unused.
const comparisonDataRef = useRef(comparisonData);
comparisonDataRef.current = comparisonData;
useEffect(() => { useEffect(() => {
if (!isInitialized || !urnKey) return; if (!urnKey) {
setComparisonData(null);
setNationalAverages(undefined);
setBenchmarks(undefined);
return;
}
const have = comparisonDataRef.current ?? {}; if (firstFetchRef.current) {
const covered = urnKey.split(',').every((urn) => have[urn] != null); firstFetchRef.current = false;
if (covered) return; const ssrUrns = new Set(Object.keys(initialData ?? {}));
const covered = urnKey.split(',').every((urn) => ssrUrns.has(urn));
if (covered && ssrUrns.size > 0) return;
}
fetchComparison(urnKey, { cache: 'no-store' }) fetchComparison(urnKey, { cache: 'no-store' })
.then((data) => { .then((data) => {
@@ -140,7 +138,8 @@ export function ComparisonView({
// destroy a working comparison the user is looking at. // destroy a working comparison the user is looking at.
console.error('Failed to fetch comparison:', err); console.error('Failed to fetch comparison:', err);
}); });
}, [urnKey, isInitialized]); // eslint-disable-next-line react-hooks/exhaustive-deps
}, [urnKey]);
// Classify schools by phase using comparison data // Classify schools by phase using comparison data
const classifySchool = (school: School): 'primary' | 'secondary' => { const classifySchool = (school: School): 'primary' | 'secondary' => {
+41 -9
View File
@@ -1,18 +1,50 @@
/** /**
* Custom hook for managing school comparison state. * Custom hook for managing school comparison state
* * Uses shared context for real-time updates across components
* This hook is mounted on every page via the global Navigation and
* ComparisonToast, so it must stay cheap — it exposes basket state only.
* The compare page fetches `/api/compare` itself (ComparisonView); nothing
* ever read the comparison payload from here, so the previous per-page SWR
* fetch (which fired on every page whenever the basket was non-empty) was
* dead weight and has been removed.
*/ */
'use client'; 'use client';
import useSWR from 'swr';
import { fetcher } from '@/lib/api';
import { useComparisonContext } from '@/context/ComparisonContext'; import { useComparisonContext } from '@/context/ComparisonContext';
import type { ComparisonResponse } from '@/lib/types';
export function useComparison() { export function useComparison() {
return useComparisonContext(); const {
selectedSchools,
addSchool,
removeSchool,
replaceSchools,
clearAll,
isSelected,
canAddMore,
isInitialized,
} = useComparisonContext();
// Fetch comparison data for selected schools
const urns = selectedSchools.map((s) => s.urn).join(',');
const { data, error, isLoading, mutate } = useSWR<ComparisonResponse>(
selectedSchools.length > 0 ? `/compare?urns=${urns}` : null,
fetcher,
{
revalidateOnFocus: false,
dedupingInterval: 10000,
}
);
return {
selectedSchools,
comparisonData: data?.comparison,
isLoading,
error,
addSchool,
removeSchool,
replaceSchools,
clearAll,
isSelected,
canAddMore,
isInitialized,
mutate,
};
} }