Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
870af949ee | ||
|
|
c6ff77f05b | ||
|
|
c7ddf0505d | ||
|
|
e659867590 | ||
|
|
c26b65246f | ||
|
|
3728a63275 | ||
|
|
b6e48c4930 | ||
|
|
9f4f2507cc | ||
|
|
94bfac9caf | ||
|
|
65a2619e1d |
No files matched your search
+54
-8
@@ -1273,17 +1273,63 @@ async def get_filter_options(request: Request):
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _la_averages_payload(df: pd.DataFrame) -> dict:
|
||||||
|
"""Per-LA Attainment 8 for the "vs LA avg" comparison: DfE's own LA
|
||||||
|
averages (fact_ks4_la_averages, all state-funded schools), never a mean of
|
||||||
|
the dataframe. That mean counted independent and special schools and put
|
||||||
|
most LAs about 7 points low (audit H2).
|
||||||
|
|
||||||
|
The year is the latest with any school Attainment 8, so the average and
|
||||||
|
the scores set against it are the same year. Figures are keyed by our LA
|
||||||
|
name through the LA code. An LA without a DfE figure for that year (City of
|
||||||
|
London) is absent; no figures for the year, or no mart, give an empty map,
|
||||||
|
so rows show no comparison, never another year's figure.
|
||||||
|
"""
|
||||||
|
empty = {"year": 0, "secondary": {"attainment_8_by_la": {}}}
|
||||||
|
if df.empty or "attainment_8_score" not in df.columns:
|
||||||
|
return empty
|
||||||
|
scored = df[df["attainment_8_score"].notna()]
|
||||||
|
if scored.empty:
|
||||||
|
return empty
|
||||||
|
year = int(scored["year"].max())
|
||||||
|
|
||||||
|
la = (df[["local_authority_code", "local_authority"]]
|
||||||
|
.dropna()
|
||||||
|
.drop_duplicates("local_authority_code"))
|
||||||
|
name_by_code = {int(code): name for code, name in
|
||||||
|
zip(la["local_authority_code"], la["local_authority"])}
|
||||||
|
|
||||||
|
from . import database
|
||||||
|
from .models import Ks4LaAverage
|
||||||
|
|
||||||
|
rows: list = []
|
||||||
|
db = None
|
||||||
|
try:
|
||||||
|
db = database.SessionLocal()
|
||||||
|
rows = db.query(Ks4LaAverage).filter(Ks4LaAverage.year == year).all()
|
||||||
|
except Exception:
|
||||||
|
import logging
|
||||||
|
logging.getLogger(__name__).warning(
|
||||||
|
"DfE LA averages unavailable for %s", year, exc_info=True)
|
||||||
|
if db is not None:
|
||||||
|
db.rollback()
|
||||||
|
finally:
|
||||||
|
if db is not None:
|
||||||
|
db.close()
|
||||||
|
|
||||||
|
by_la = {
|
||||||
|
name_by_code[row.la_code]: row.attainment_8_score
|
||||||
|
for row in rows
|
||||||
|
if row.attainment_8_score is not None and row.la_code in name_by_code
|
||||||
|
}
|
||||||
|
return {"year": year, "secondary": {"attainment_8_by_la": by_la}}
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/la-averages")
|
@app.get("/api/la-averages")
|
||||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||||
async def get_la_averages(request: Request):
|
async def get_la_averages(request: Request):
|
||||||
"""Get per-LA average Attainment 8 score for secondary schools in the latest year."""
|
"""DfE's per-LA Attainment 8 averages for the latest year with results."""
|
||||||
df = load_school_data()
|
return _la_averages_payload(load_school_data())
|
||||||
if df.empty:
|
|
||||||
return {"year": 0, "secondary": {"attainment_8_by_la": {}}}
|
|
||||||
latest_year = int(df["year"].max())
|
|
||||||
sec_df = df[(df["year"] == latest_year) & df["attainment_8_score"].notna()]
|
|
||||||
la_avg = sec_df.groupby("local_authority")["attainment_8_score"].mean().round(1).to_dict()
|
|
||||||
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
|
||||||
|
|
||||||
|
|
||||||
_KS2_NATIONAL_METRICS = [
|
_KS2_NATIONAL_METRICS = [
|
||||||
|
|||||||
@@ -344,6 +344,18 @@ class Ks4NationalAverage(Base):
|
|||||||
gcse_grade_91_pct = Column(Float)
|
gcse_grade_91_pct = Column(Float)
|
||||||
|
|
||||||
|
|
||||||
|
class Ks4LaAverage(Base):
|
||||||
|
"""Official DfE KS4 local-authority averages (all state-funded schools) —
|
||||||
|
one row per academic year and LA. la_code is the GIAS LA code."""
|
||||||
|
__tablename__ = "fact_ks4_la_averages"
|
||||||
|
__table_args__ = MARTS
|
||||||
|
|
||||||
|
year = Column(Integer, primary_key=True)
|
||||||
|
la_code = Column(Integer, primary_key=True)
|
||||||
|
la_name = Column(String)
|
||||||
|
attainment_8_score = Column(Float)
|
||||||
|
|
||||||
|
|
||||||
class Ks2NationalAverage(Base):
|
class Ks2NationalAverage(Base):
|
||||||
"""Official DfE KS2 national headline averages — one row per academic year."""
|
"""Official DfE KS2 national headline averages — one row per academic year."""
|
||||||
__tablename__ = "fact_ks2_national_averages"
|
__tablename__ = "fact_ks2_national_averages"
|
||||||
|
|||||||
@@ -0,0 +1,119 @@
|
|||||||
|
"""/api/la-averages serves DfE's own LA averages (fact_ks4_la_averages).
|
||||||
|
|
||||||
|
It used to average the dataframe: every school with an Attainment 8,
|
||||||
|
independent and special schools included. Kensington and Chelsea came out at
|
||||||
|
35.2 against DfE's 54.5, and most LAs about 7 points low (audit H2).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pandas as pd
|
||||||
|
import pytest
|
||||||
|
from fastapi.testclient import TestClient
|
||||||
|
|
||||||
|
LATEST = 202425
|
||||||
|
|
||||||
|
|
||||||
|
def _df():
|
||||||
|
return pd.DataFrame([
|
||||||
|
# A state school and an independent: their mean, 37.65, is not DfE's figure.
|
||||||
|
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=54.9),
|
||||||
|
dict(year=LATEST, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=20.4),
|
||||||
|
dict(year=LATEST, local_authority="Bristol, City of", local_authority_code=801, attainment_8_score=45.0),
|
||||||
|
dict(year=LATEST, local_authority="West Sussex", local_authority_code=938, attainment_8_score=48.0),
|
||||||
|
# DfE publishes no LA figure for City of London.
|
||||||
|
dict(year=LATEST, local_authority="City of London", local_authority_code=201, attainment_8_score=30.0),
|
||||||
|
# A newer year with primary results only.
|
||||||
|
dict(year=202526, local_authority="Kensington and Chelsea", local_authority_code=207, attainment_8_score=np.nan),
|
||||||
|
])
|
||||||
|
|
||||||
|
|
||||||
|
class _Row:
|
||||||
|
def __init__(self, year, la_code, la_name, attainment_8_score):
|
||||||
|
self.year = year
|
||||||
|
self.la_code = la_code
|
||||||
|
self.la_name = la_name
|
||||||
|
self.attainment_8_score = attainment_8_score
|
||||||
|
|
||||||
|
|
||||||
|
class _StubSession:
|
||||||
|
rows = [
|
||||||
|
_Row(LATEST, 207, "Kensington and Chelsea", 54.5),
|
||||||
|
_Row(LATEST, 801, "Bristol City", 46.3), # DfE's spelling, not ours
|
||||||
|
_Row(LATEST, 938, "West Sussex", None), # suppressed
|
||||||
|
_Row(LATEST, 330, "Birmingham", 44.0), # no school of ours there
|
||||||
|
_Row(202324, 207, "Kensington and Chelsea", 54.5),
|
||||||
|
]
|
||||||
|
|
||||||
|
def query(self, model):
|
||||||
|
assert model.__name__ == "Ks4LaAverage"
|
||||||
|
return self
|
||||||
|
|
||||||
|
def filter(self, condition):
|
||||||
|
# The payload filters on year == <year>; apply it as Postgres would.
|
||||||
|
self._year = condition.right.value
|
||||||
|
return self
|
||||||
|
|
||||||
|
def all(self):
|
||||||
|
return [r for r in self.rows if r.year == self._year]
|
||||||
|
|
||||||
|
def rollback(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class _OldYearOnly(_StubSession):
|
||||||
|
rows = [_Row(202324, 207, "Kensington and Chelsea", 54.5)]
|
||||||
|
|
||||||
|
|
||||||
|
class _NoMart(_StubSession):
|
||||||
|
def all(self):
|
||||||
|
raise RuntimeError('relation "marts.fact_ks4_la_averages" does not exist')
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture()
|
||||||
|
def payload(monkeypatch):
|
||||||
|
from backend import app as app_module
|
||||||
|
from backend import database as database_module
|
||||||
|
|
||||||
|
def _run(session_cls):
|
||||||
|
monkeypatch.setattr(database_module, "SessionLocal", session_cls)
|
||||||
|
return app_module._la_averages_payload(_df())
|
||||||
|
|
||||||
|
return _run
|
||||||
|
|
||||||
|
|
||||||
|
def test_serves_dfe_figures_keyed_by_our_la_names(payload):
|
||||||
|
out = payload(_StubSession)
|
||||||
|
assert out["year"] == LATEST
|
||||||
|
assert out["secondary"]["attainment_8_by_la"] == {
|
||||||
|
"Kensington and Chelsea": 54.5,
|
||||||
|
"Bristol, City of": 46.3,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_an_la_without_a_dfe_figure_is_absent(payload):
|
||||||
|
by_la = payload(_StubSession)["secondary"]["attainment_8_by_la"]
|
||||||
|
assert "City of London" not in by_la
|
||||||
|
assert "West Sussex" not in by_la
|
||||||
|
assert None not in by_la.values()
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_dfe_figures_for_the_year_give_an_empty_map(payload):
|
||||||
|
assert payload(_OldYearOnly) == {"year": LATEST, "secondary": {"attainment_8_by_la": {}}}
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_missing_mart_gives_an_empty_map(payload):
|
||||||
|
assert payload(_NoMart)["secondary"]["attainment_8_by_la"] == {}
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_endpoint_serves_dfe_figures(monkeypatch):
|
||||||
|
from backend import app as app_module
|
||||||
|
from backend import database as database_module
|
||||||
|
|
||||||
|
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||||
|
monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
|
||||||
|
resp = TestClient(app_module.app).get("/api/la-averages")
|
||||||
|
assert resp.status_code == 200, resp.text
|
||||||
|
assert resp.json()["secondary"]["attainment_8_by_la"]["Kensington and Chelsea"] == 54.5
|
||||||
@@ -417,12 +417,15 @@ test('a secondary search row compares its Attainment 8 with the LA average', asy
|
|||||||
// guards the comparison itself; the unit test pins the cache mode.
|
// guards the comparison itself; the unit test pins the cache mode.
|
||||||
const la = await (await page.request.get('/api/la-averages')).json();
|
const la = await (await page.request.get('/api/la-averages')).json();
|
||||||
const averages: Record<string, number> = la.secondary?.attainment_8_by_la ?? {};
|
const averages: Record<string, number> = la.secondary?.attainment_8_by_la ?? {};
|
||||||
|
// DfE publishes about 152 LA averages. An empty map (a missing mart, or a
|
||||||
|
// year the LA data set has not reached) hides every comparison: fail, not skip.
|
||||||
|
expect(Object.keys(averages).length).toBeGreaterThan(100);
|
||||||
const res = await page.request.get('/api/schools?search=school&phase=secondary&page_size=50');
|
const res = await page.request.get('/api/schools?search=school&phase=secondary&page_size=50');
|
||||||
expect(res.ok()).toBeTruthy();
|
expect(res.ok()).toBeTruthy();
|
||||||
const school = ((await res.json()).schools ?? []).find(
|
const school = ((await res.json()).schools ?? []).find(
|
||||||
(s: { attainment_8_score?: number | null; local_authority?: string; school_type?: string }) =>
|
(s: { attainment_8_score?: number | null; local_authority?: string; school_type?: string }) =>
|
||||||
s.attainment_8_score != null && s.local_authority != null && averages[s.local_authority] != null
|
s.attainment_8_score != null && s.local_authority != null && averages[s.local_authority] != null
|
||||||
&& !/special|pupil referral|alternative provision/i.test(s.school_type ?? ''));
|
&& !/special|pupil referral|alternative provision|independent/i.test(s.school_type ?? ''));
|
||||||
test.skip(!school, 'no mainstream secondary with an LA average here');
|
test.skip(!school, 'no mainstream secondary with an LA average here');
|
||||||
|
|
||||||
await searchByName(page, school.school_name);
|
await searchByName(page, school.school_name);
|
||||||
@@ -433,6 +436,52 @@ test('a secondary search row compares its Attainment 8 with the LA average', asy
|
|||||||
await expect(stats.getByText(/vs LA avg/)).toBeVisible();
|
await expect(stats.getByText(/vs LA avg/)).toBeVisible();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('an independent secondary shows its Attainment 8 without an LA comparison', async ({ page }) => {
|
||||||
|
// DfE's LA averages cover state-funded schools, and an independent school's
|
||||||
|
// Attainment 8 leaves out IGCSEs, so a gap would mislead (audit H2). A state
|
||||||
|
// school on the same page must show its gap first, so the absence is real.
|
||||||
|
const la = await (await page.request.get('/api/la-averages')).json();
|
||||||
|
const averages: Record<string, number> = la.secondary?.attainment_8_by_la ?? {};
|
||||||
|
expect(Object.keys(averages).length).toBeGreaterThan(100);
|
||||||
|
|
||||||
|
type Row = { urn: number; school_type?: string; local_authority?: string; attainment_8_score?: number | null };
|
||||||
|
const compared = (s: Row) => s.attainment_8_score != null && s.local_authority != null
|
||||||
|
&& averages[s.local_authority] != null
|
||||||
|
&& !/special|pupil referral|alternative provision/i.test(s.school_type ?? '');
|
||||||
|
let found: { la: string; state: Row; independent: Row } | null = null;
|
||||||
|
for (const name of ['Kensington and Chelsea', 'Westminster', 'Camden', 'Hammersmith and Fulham', 'Barnet']) {
|
||||||
|
// The search page asks for the same first 50 schools.
|
||||||
|
const res = await page.request.get(`/api/schools?search=${encodeURIComponent(name)}&phase=secondary&page_size=50`);
|
||||||
|
const schools: Row[] = ((await res.json()).schools ?? []).filter(compared);
|
||||||
|
const independent = schools.find(s => /independent/i.test(s.school_type ?? ''));
|
||||||
|
const state = schools.find(s => !/independent/i.test(s.school_type ?? ''));
|
||||||
|
if (independent && state) { found = { la: name, state, independent }; break; }
|
||||||
|
}
|
||||||
|
test.skip(!found, 'no LA here lists a state and an independent secondary on one page');
|
||||||
|
|
||||||
|
await page.goto(`/?search=${encodeURIComponent(found!.la)}&phase=secondary`);
|
||||||
|
const stats = (urn: number) => page.locator(`a[href^="/school/${urn}-"]`).first()
|
||||||
|
.locator('xpath=ancestor::div[contains(@class, "__rowContent")][1]')
|
||||||
|
.locator('[class*="__line3"]');
|
||||||
|
await expect(stats(found!.state.urn).getByText(/vs LA avg/)).toBeVisible({ timeout: 15_000 });
|
||||||
|
const independent = stats(found!.independent.urn);
|
||||||
|
await expect(independent.getByText(found!.independent.attainment_8_score!.toFixed(1))).toBeVisible();
|
||||||
|
await expect(independent.getByText(/vs LA avg/)).toHaveCount(0);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('a secondary shows its 2023/24 GCSE results, the last year DfE published Progress 8', async ({ page }) => {
|
||||||
|
// DfE re-issued its 2023/24 file under older column names, and every
|
||||||
|
// school's 2023/24 row loaded empty (audit C2). These are DfE's final
|
||||||
|
// figures for Bishop Stopford School, so they do not change.
|
||||||
|
await page.goto('/school/137086');
|
||||||
|
const history = page.locator('#history');
|
||||||
|
await history.getByText('View raw year-by-year data').click();
|
||||||
|
const row = history.getByRole('row', { name: /2023\/24/ });
|
||||||
|
await expect(row).toContainText('64.1');
|
||||||
|
await expect(row).toContainText('+1.0');
|
||||||
|
await expect(row).toContainText('91.7%');
|
||||||
|
});
|
||||||
|
|
||||||
test('a phase outside primary/secondary filters to that phase, not to everything', async ({ page }) => {
|
test('a phase outside primary/secondary filters to that phase, not to everything', async ({ page }) => {
|
||||||
// The search page offers every GIAS phase, but the API only knew the grouped
|
// The search page offers every GIAS phase, but the API only knew the grouped
|
||||||
// ones and silently dropped the rest — so "Nursery" returned primaries.
|
// ones and silently dropped the rest — so "Nursery" returned primaries.
|
||||||
|
|||||||
@@ -105,3 +105,20 @@ it('puts the card back when the pins are rebuilt for a reason other than the sch
|
|||||||
expect(container.querySelector('.sc-popup')).toHaveTextContent('Southmead Primary School');
|
expect(container.querySelector('.sc-popup')).toHaveTextContent('Southmead Primary School');
|
||||||
expect(container.querySelectorAll('.sc-pin--selected')).toHaveLength(1);
|
expect(container.querySelectorAll('.sc-pin--selected')).toHaveLength(1);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('compares a state secondary with its LA average on the card, but not an independent one', () => {
|
||||||
|
const state: School = { ...base, urn: 3, school_name: 'Holland Park School', school_type: 'Academy converter',
|
||||||
|
phase: 'Secondary', local_authority: 'Kensington and Chelsea', attainment_8_score: 60,
|
||||||
|
latitude: 51.5, longitude: -0.2, distance: 0.3 };
|
||||||
|
const independent: School = { ...state, urn: 4, school_name: 'Abbey Gate College',
|
||||||
|
school_type: 'Other independent school', attainment_8_score: 20.4 };
|
||||||
|
const laAverages = { 'Kensington and Chelsea': 54.5 };
|
||||||
|
|
||||||
|
const { container, rerender } = renderMap({ schools: [state, independent], laAverages, selectedUrn: 3 });
|
||||||
|
expect(container.querySelector('.sc-popup')).toHaveTextContent('60.0 Att 8 +5.5 vs LA');
|
||||||
|
|
||||||
|
rerender({ schools: [state, independent], laAverages, selectedUrn: 4 });
|
||||||
|
const card = container.querySelector('.sc-popup')!;
|
||||||
|
expect(card).toHaveTextContent('20.4 Att 8');
|
||||||
|
expect(card).not.toHaveTextContent(/vs LA/);
|
||||||
|
});
|
||||||
@@ -116,3 +116,38 @@ describe('SecondarySchoolRow shares the school page flags', () => {
|
|||||||
expect(screen.getByText('Fee-paying')).toBeInTheDocument();
|
expect(screen.getByText('Fee-paying')).toBeInTheDocument();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe('SecondarySchoolRow LA comparison', () => {
|
||||||
|
it('compares a state school with its LA average', () => {
|
||||||
|
render(
|
||||||
|
<SecondarySchoolRow
|
||||||
|
school={{ ...base, school_type: 'Academy converter', attainment_8_score: 60 }}
|
||||||
|
laAvgAttainment8={54.5}
|
||||||
|
/>,
|
||||||
|
);
|
||||||
|
expect(screen.getByText(/\+5\.5 vs LA avg/)).toBeInTheDocument();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('keeps the comparison for a school whose type is unknown', () => {
|
||||||
|
render(
|
||||||
|
<SecondarySchoolRow
|
||||||
|
school={{ ...base, school_type: null, attainment_8_score: 60 } as unknown as School}
|
||||||
|
laAvgAttainment8={54.5}
|
||||||
|
/>,
|
||||||
|
);
|
||||||
|
expect(screen.getByText(/\+5\.5 vs LA avg/)).toBeInTheDocument();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("shows an independent school's Attainment 8 without an LA comparison", () => {
|
||||||
|
// DfE's LA average covers state-funded schools; an independent's
|
||||||
|
// Attainment 8 leaves out IGCSEs (audit H2).
|
||||||
|
render(
|
||||||
|
<SecondarySchoolRow
|
||||||
|
school={{ ...base, school_type: 'Other independent school', attainment_8_score: 20.4 }}
|
||||||
|
laAvgAttainment8={54.5}
|
||||||
|
/>,
|
||||||
|
);
|
||||||
|
expect(screen.getByText('20.4')).toBeInTheDocument();
|
||||||
|
expect(screen.queryByText(/vs LA avg/)).not.toBeInTheDocument();
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -391,3 +391,20 @@ describe('singleSexLabel', () => {
|
|||||||
expect(singleSexLabel(undefined)).toBeNull();
|
expect(singleSexLabel(undefined)).toBeNull();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe('isIndependentSchool', () => {
|
||||||
|
const { isIndependentSchool } = require('@/lib/utils');
|
||||||
|
|
||||||
|
it('matches both GIAS independent types', () => {
|
||||||
|
expect(isIndependentSchool({ school_type: 'Other independent school' })).toBe(true);
|
||||||
|
expect(isIndependentSchool({ school_type: 'Other independent special school' })).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('does not match state-funded types or a missing type', () => {
|
||||||
|
for (const t of ['Academy converter', 'Community school', 'Free schools', 'Non-maintained special school']) {
|
||||||
|
expect(isIndependentSchool({ school_type: t })).toBe(false);
|
||||||
|
}
|
||||||
|
expect(isIndependentSchool({ school_type: null })).toBe(false);
|
||||||
|
expect(isIndependentSchool({})).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -9,7 +9,7 @@ import { useEffect, useRef, useState } from 'react';
|
|||||||
import L from 'leaflet';
|
import L from 'leaflet';
|
||||||
import 'leaflet/dist/leaflet.css';
|
import 'leaflet/dist/leaflet.css';
|
||||||
import type { School } from '@/lib/types';
|
import type { School } from '@/lib/types';
|
||||||
import { schoolUrl, isSpecialSchool, buildOfstedListBadge, listRwmValue } from '@/lib/utils';
|
import { schoolUrl, isSpecialSchool, isIndependentSchool, buildOfstedListBadge, listRwmValue } from '@/lib/utils';
|
||||||
|
|
||||||
interface LeafletMapInnerProps {
|
interface LeafletMapInnerProps {
|
||||||
schools: School[];
|
schools: School[];
|
||||||
@@ -70,7 +70,7 @@ function metricHtml(school: School, { nationalAvgRwm, laAverages }: CardContext)
|
|||||||
const score = school.attainment_8_score;
|
const score = school.attainment_8_score;
|
||||||
const laAvg = school.local_authority ? (laAverages?.[school.local_authority] ?? null) : null;
|
const laAvg = school.local_authority ? (laAverages?.[school.local_authority] ?? null) : null;
|
||||||
let delta = '';
|
let delta = '';
|
||||||
if (!special && laAvg != null) {
|
if (!special && !isIndependentSchool(school) && laAvg != null) {
|
||||||
const diff = Math.round((score - laAvg) * 10) / 10;
|
const diff = Math.round((score - laAvg) * 10) / 10;
|
||||||
// Att8 runs 0–90 in 0.1 steps; ±0.5 is meaningful, where RWM needs ±2.
|
// Att8 runs 0–90 in 0.1 steps; ±0.5 is meaningful, where RWM needs ±2.
|
||||||
const cls = diff >= 0.5 ? 'sc-up' : diff <= -0.5 ? 'sc-down' : '';
|
const cls = diff >= 0.5 ? 'sc-up' : diff <= -0.5 ? 'sc-down' : '';
|
||||||
|
|||||||
@@ -11,7 +11,7 @@
|
|||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import type { School } from '@/lib/types';
|
import type { School } from '@/lib/types';
|
||||||
import { buildOfstedListBadge, getPhaseStyle, schoolUrl, formatAgeRange, isProposedToClose, isSpecialSchool } from '@/lib/utils';
|
import { buildOfstedListBadge, getPhaseStyle, schoolUrl, formatAgeRange, isProposedToClose, isSpecialSchool, isIndependentSchool } from '@/lib/utils';
|
||||||
import { schoolFlags, schoolTypeLabel } from '@/lib/schoolFacts';
|
import { schoolFlags, schoolTypeLabel } from '@/lib/schoolFacts';
|
||||||
import styles from './SecondarySchoolRow.module.css';
|
import styles from './SecondarySchoolRow.module.css';
|
||||||
|
|
||||||
@@ -45,10 +45,12 @@ export function SecondarySchoolRow({
|
|||||||
const att8 = school.attainment_8_score;
|
const att8 = school.attainment_8_score;
|
||||||
// The school's own Attainment 8 is a same-school figure — shown whenever it
|
// The school's own Attainment 8 is a same-school figure — shown whenever it
|
||||||
// exists (special schools included; their type tag on line 2 gives context).
|
// exists (special schools included; their type tag on line 2 gives context).
|
||||||
// Only the vs-LA-average delta, a benchmark comparison, is dropped for
|
// Only the vs-LA-average delta, a benchmark comparison, is dropped: for
|
||||||
// special schools / PRUs / AP, whose pupils aren't measured against it fairly.
|
// special schools / PRUs / AP, whose pupils aren't measured against it
|
||||||
|
// fairly, and for independent schools, because DfE's LA average covers
|
||||||
|
// state-funded schools and an independent's Attainment 8 leaves out IGCSEs.
|
||||||
const laDelta =
|
const laDelta =
|
||||||
att8 != null && !isSpecialSchool(school) && laAvgAttainment8 != null
|
att8 != null && !isSpecialSchool(school) && !isIndependentSchool(school) && laAvgAttainment8 != null
|
||||||
? att8 - laAvgAttainment8
|
? att8 - laAvgAttainment8
|
||||||
: null;
|
: null;
|
||||||
|
|
||||||
|
|||||||
@@ -758,6 +758,16 @@ export function isSpecialSchool(school: { school_type?: string | null }): boolea
|
|||||||
return /\bspecial\b/.test(t) || /pupil referral/.test(t) || /alternative provision/.test(t);
|
return /\bspecial\b/.test(t) || /pupil referral/.test(t) || /alternative provision/.test(t);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Independent (fee-paying) schools: GIAS types "Other independent school" and
|
||||||
|
* "Other independent special school". DfE's Attainment 8 for them leaves out
|
||||||
|
* IGCSEs, and DfE's LA averages cover state-funded schools only, so callers
|
||||||
|
* drop the "vs LA avg" comparison for them (audit H2).
|
||||||
|
*/
|
||||||
|
export function isIndependentSchool(school: { school_type?: string | null }): boolean {
|
||||||
|
return /\bindependent\b/i.test(school.school_type ?? '');
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Whether GIAS records a religious character. "None", "Does not apply" and
|
* Whether GIAS records a religious character. "None", "Does not apply" and
|
||||||
* "Not applicable" are the register's ways of saying it has none; the place
|
* "Not applicable" are the register's ways of saying it has none; the place
|
||||||
|
|||||||
@@ -106,9 +106,12 @@ print(f'Validation passed: {{count}} GIAS rows')
|
|||||||
""",
|
""",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Marts fed by annual EES staging models are rebuilt by the EES DAG, even
|
||||||
|
# when they join dim_school. Selecting them here fails in any database
|
||||||
|
# where that DAG hasn't run (pipeline/tests/test_dag_selectors.py).
|
||||||
dbt_build = BashOperator(
|
dbt_build = BashOperator(
|
||||||
task_id="dbt_build",
|
task_id="dbt_build",
|
||||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_gias_establishments+ stg_gias_links+ gias_code_names+ --exclude int_ks2_with_lineage+ int_ks4_with_lineage+",
|
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_gias_establishments+ stg_gias_links+ gias_code_names+ --exclude int_ks2_with_lineage+ int_ks4_with_lineage+ stg_ees_ks4_destinations+ stg_ees_ks5_destinations+",
|
||||||
)
|
)
|
||||||
|
|
||||||
sync_typesense = BashOperator(
|
sync_typesense = BashOperator(
|
||||||
@@ -143,7 +146,7 @@ with DAG(
|
|||||||
|
|
||||||
dbt_build_ofsted = BashOperator(
|
dbt_build_ofsted = BashOperator(
|
||||||
task_id="dbt_build",
|
task_id="dbt_build",
|
||||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_ofsted_inspections+ int_ofsted_latest+ fact_ofsted_inspection+ dim_school+",
|
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_ofsted_inspections+ int_ofsted_latest+ fact_ofsted_inspection+ dim_school+ --exclude stg_ees_ks4_destinations+ stg_ees_ks5_destinations+",
|
||||||
)
|
)
|
||||||
|
|
||||||
sync_typesense_ofsted = BashOperator(
|
sync_typesense_ofsted = BashOperator(
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
"""Read a GIAS extract from the raw bytes of the download.
|
||||||
|
|
||||||
|
GIAS writes its CSVs in Windows-1252 and sends no charset, so `resp.text`
|
||||||
|
leaves requests to guess the codec. On 3 Oct 2026 it guessed windows-1250 and
|
||||||
|
"à" became "ŕ". Decode the bytes ourselves instead.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
GIAS_ENCODING = "cp1252"
|
||||||
|
|
||||||
|
|
||||||
|
def read_gias_csv(content: bytes, logger=None) -> pd.DataFrame:
|
||||||
|
"""Every column as a string; a blank cell stays ''."""
|
||||||
|
# Windows-1252 leaves five bytes undefined. One stray byte must not stop
|
||||||
|
# the daily refresh of every school, so it becomes U+FFFD and is logged.
|
||||||
|
text = content.decode(GIAS_ENCODING, errors="replace")
|
||||||
|
undecodable = text.count("�")
|
||||||
|
if undecodable and logger is not None:
|
||||||
|
logger.warning("%d byte(s) in the GIAS extract could not be decoded as %s",
|
||||||
|
undecodable, GIAS_ENCODING)
|
||||||
|
return pd.read_csv(io.StringIO(text), dtype=str, keep_default_na=False)
|
||||||
@@ -7,6 +7,8 @@ from datetime import date, timedelta
|
|||||||
from singer_sdk import Stream, Tap
|
from singer_sdk import Stream, Tap
|
||||||
from singer_sdk import typing as th
|
from singer_sdk import typing as th
|
||||||
|
|
||||||
|
from tap_uk_gias.gias_csv import read_gias_csv
|
||||||
|
|
||||||
GIAS_URL_TEMPLATE = (
|
GIAS_URL_TEMPLATE = (
|
||||||
"https://ea-edubase-api-prod.azurewebsites.net"
|
"https://ea-edubase-api-prod.azurewebsites.net"
|
||||||
"/edubase/downloads/public/edubasealldata{date}.csv"
|
"/edubase/downloads/public/edubasealldata{date}.csv"
|
||||||
@@ -74,9 +76,6 @@ class GIASEstablishmentsStream(Stream):
|
|||||||
|
|
||||||
def get_records(self, context):
|
def get_records(self, context):
|
||||||
"""Download GIAS CSV and yield rows."""
|
"""Download GIAS CSV and yield rows."""
|
||||||
import io
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
import requests
|
||||||
|
|
||||||
today = date.today()
|
today = date.today()
|
||||||
@@ -94,12 +93,7 @@ class GIASEstablishmentsStream(Stream):
|
|||||||
|
|
||||||
resp.raise_for_status()
|
resp.raise_for_status()
|
||||||
|
|
||||||
df = pd.read_csv(
|
df = read_gias_csv(resp.content, self.logger)
|
||||||
io.StringIO(resp.text),
|
|
||||||
encoding="latin-1",
|
|
||||||
dtype=str,
|
|
||||||
keep_default_na=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
for _, row in df.iterrows():
|
for _, row in df.iterrows():
|
||||||
record = row.to_dict()
|
record = row.to_dict()
|
||||||
@@ -126,9 +120,6 @@ class GIASLinksStream(Stream):
|
|||||||
|
|
||||||
def get_records(self, context):
|
def get_records(self, context):
|
||||||
"""Download GIAS links CSV and yield rows."""
|
"""Download GIAS links CSV and yield rows."""
|
||||||
import io
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
import requests
|
||||||
|
|
||||||
today = date.today()
|
today = date.today()
|
||||||
@@ -146,12 +137,7 @@ class GIASLinksStream(Stream):
|
|||||||
|
|
||||||
resp.raise_for_status()
|
resp.raise_for_status()
|
||||||
|
|
||||||
df = pd.read_csv(
|
df = read_gias_csv(resp.content, self.logger)
|
||||||
io.StringIO(resp.text),
|
|
||||||
encoding="latin-1",
|
|
||||||
dtype=str,
|
|
||||||
keep_default_na=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
for _, row in df.iterrows():
|
for _, row in df.iterrows():
|
||||||
record = row.to_dict()
|
record = row.to_dict()
|
||||||
|
|||||||
@@ -0,0 +1,98 @@
|
|||||||
|
"""Every scheduled dbt build must only build models whose parents exist.
|
||||||
|
|
||||||
|
The daily GIAS build selects `stg_gias_establishments+`, so any mart that joins
|
||||||
|
dim_school joins the daily build too. When such a mart also reads a staging
|
||||||
|
model that only the manually triggered EES DAG builds, the daily build fails in
|
||||||
|
any database where that DAG has not run since. Sync and cache invalidation then
|
||||||
|
never run either. The destinations marts did this from late August 2026.
|
||||||
|
|
||||||
|
The graph is read from the model SQL, because CI has no dbt.
|
||||||
|
"""
|
||||||
|
import re
|
||||||
|
from collections import defaultdict
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
PIPELINE = Path(__file__).resolve().parents[1]
|
||||||
|
MODELS = PIPELINE / 'transform' / 'models'
|
||||||
|
DAG_FILE = PIPELINE / 'dags' / 'school_data_pipeline.py'
|
||||||
|
|
||||||
|
REF = re.compile(r"ref\(\s*'([a-z0-9_]+)'\s*\)")
|
||||||
|
DBT_BUILD = re.compile(r'dbt_build\w*\s*=\s*BashOperator\(.*?build --profiles-dir \. --target production ([^"]+)"', re.S)
|
||||||
|
DAG_ID = re.compile(r'dag_id="([a-z0-9_]+)"')
|
||||||
|
|
||||||
|
DAILY = 'school_data_daily'
|
||||||
|
|
||||||
|
# dim_school reads int_ofsted_latest only when the relation exists
|
||||||
|
# (adapter.get_relation), so a missing table is not a failure.
|
||||||
|
OPTIONAL_PARENTS = {'int_ofsted_latest'}
|
||||||
|
|
||||||
|
|
||||||
|
def model_parents():
|
||||||
|
"""{model: models it refs}. Seeds are left out: they are loaded once and always exist."""
|
||||||
|
sql = {p.stem: p.read_text() for p in MODELS.rglob('*.sql')}
|
||||||
|
return {name: set(REF.findall(text)) & set(sql) for name, text in sql.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def downstream(node, children):
|
||||||
|
seen, stack = {node}, [node]
|
||||||
|
while stack:
|
||||||
|
for child in children[stack.pop()]:
|
||||||
|
if child not in seen:
|
||||||
|
seen.add(child)
|
||||||
|
stack.append(child)
|
||||||
|
return seen
|
||||||
|
|
||||||
|
|
||||||
|
def expand(tokens, children):
|
||||||
|
out = set()
|
||||||
|
for token in tokens:
|
||||||
|
out |= downstream(token[:-1], children) if token.endswith('+') else {token}
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def scheduled_builds():
|
||||||
|
"""{dag_id: dbt selection arguments} for every dbt build in the DAG file."""
|
||||||
|
text = DAG_FILE.read_text()
|
||||||
|
starts = [(m.start(), m.group(1)) for m in DAG_ID.finditer(text)]
|
||||||
|
builds = {}
|
||||||
|
for i, (start, dag_id) in enumerate(starts):
|
||||||
|
end = starts[i + 1][0] if i + 1 < len(starts) else len(text)
|
||||||
|
found = DBT_BUILD.search(text, start, end)
|
||||||
|
if found:
|
||||||
|
builds[dag_id] = found.group(1)
|
||||||
|
return builds
|
||||||
|
|
||||||
|
|
||||||
|
def selected_models(args, parents):
|
||||||
|
children = defaultdict(set)
|
||||||
|
for model, ps in parents.items():
|
||||||
|
for p in ps:
|
||||||
|
children[p].add(model)
|
||||||
|
select = re.search(r'--select (.+?)(?= --exclude|$)', args).group(1).split()
|
||||||
|
excluded = re.search(r'--exclude (.+)$', args)
|
||||||
|
exclude = excluded.group(1).split() if excluded else []
|
||||||
|
return (expand(select, children) - expand(exclude, children)) & set(parents)
|
||||||
|
|
||||||
|
|
||||||
|
PARENTS = model_parents()
|
||||||
|
BUILDS = scheduled_builds()
|
||||||
|
DAILY_MODELS = selected_models(BUILDS[DAILY], PARENTS)
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_dag_with_a_dbt_build_is_parsed():
|
||||||
|
assert set(BUILDS) == {
|
||||||
|
'school_data_daily', 'school_data_monthly_ofsted', 'school_data_annual_ees',
|
||||||
|
'school_data_annual_idaci', 'school_data_annual_distance',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize('dag_id', sorted(BUILDS))
|
||||||
|
def test_selected_models_only_read_models_that_exist(dag_id):
|
||||||
|
selected = selected_models(BUILDS[dag_id], PARENTS)
|
||||||
|
# The daily build is the base layer: other DAGs may rely on what it builds.
|
||||||
|
available = selected | OPTIONAL_PARENTS | (DAILY_MODELS if dag_id != DAILY else set())
|
||||||
|
missing = {model: sorted(PARENTS[model] - available) for model in sorted(selected)
|
||||||
|
if PARENTS[model] - available}
|
||||||
|
assert missing == {}, f'{dag_id} builds models whose parents it never builds: {missing}'
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
"""GIAS publishes its extracts in Windows-1252 and declares no charset.
|
||||||
|
|
||||||
|
The tap used to hand pandas `resp.text`, so requests guessed the codec.
|
||||||
|
On 3 Oct 2026 it guessed windows-1250, and "St Thomas à Becket" was stored
|
||||||
|
as "St Thomas ŕ Becket". The `encoding=` passed to read_csv did nothing,
|
||||||
|
because the text was already decoded.
|
||||||
|
"""
|
||||||
|
import importlib.util
|
||||||
|
import logging
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-gias'
|
||||||
|
/ 'tap_uk_gias' / 'gias_csv.py')
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def gias_csv():
|
||||||
|
spec = importlib.util.spec_from_file_location('gias_csv', MODULE)
|
||||||
|
module = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(module)
|
||||||
|
return module
|
||||||
|
|
||||||
|
|
||||||
|
# Byte for byte as GIAS writes it: 0xE0 à, 0x92 ’, 0xE9 é, 0xB0 °, 0xE7 ç.
|
||||||
|
EXTRACT = (
|
||||||
|
b'"URN","EstablishmentName","HeadLastName"\r\n'
|
||||||
|
b'"138950","St Thomas \xe0 Becket Catholic Secondary School","Smith"\r\n'
|
||||||
|
b'"100000","The Dean and Chapter of St Paul\x92s Cathedral","Pr\xe9vert"\r\n'
|
||||||
|
b'"140677","North Star 180\xb0","Fran\xe7ois"\r\n'
|
||||||
|
b'"100001","No head recorded",""\r\n'
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_names_decode_as_windows_1252(gias_csv):
|
||||||
|
df = gias_csv.read_gias_csv(EXTRACT)
|
||||||
|
assert list(df['EstablishmentName']) == [
|
||||||
|
'St Thomas à Becket Catholic Secondary School',
|
||||||
|
'The Dean and Chapter of St Paul’s Cathedral',
|
||||||
|
'North Star 180°',
|
||||||
|
'No head recorded',
|
||||||
|
]
|
||||||
|
assert list(df['HeadLastName']) == ['Smith', 'Prévert', 'François', '']
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_codec_requests_guessed_is_not_used(gias_csv):
|
||||||
|
# What the tap stored on 3 Oct: the same bytes read as windows-1250.
|
||||||
|
assert 'ŕ' in EXTRACT.decode('cp1250')
|
||||||
|
names = ' '.join(gias_csv.read_gias_csv(EXTRACT)['EstablishmentName'])
|
||||||
|
assert 'ŕ' not in names
|
||||||
|
|
||||||
|
|
||||||
|
def test_values_stay_strings(gias_csv):
|
||||||
|
df = gias_csv.read_gias_csv(EXTRACT)
|
||||||
|
assert df.loc[0, 'URN'] == '138950'
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_byte_windows_1252_leaves_undefined_does_not_stop_the_load(gias_csv, caplog):
|
||||||
|
# 0x81 has no Windows-1252 character. One odd name must not block the daily
|
||||||
|
# refresh of every school, but it must be visible in the log.
|
||||||
|
extract = b'"URN","EstablishmentName"\r\n"100002","Odd \x81 Name"\r\n'
|
||||||
|
with caplog.at_level(logging.WARNING):
|
||||||
|
df = gias_csv.read_gias_csv(extract, logger=logging.getLogger('gias'))
|
||||||
|
assert df.loc[0, 'EstablishmentName'] == 'Odd � Name'
|
||||||
|
assert 'could not be decoded' in caplog.text
|
||||||
Reference in new issue
Block a user