Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
315f1feede | ||
|
|
8e4ee64140 | ||
|
|
d2dc78aeb5 | ||
|
|
619e3a1189 | ||
|
|
52f8994401 | ||
|
|
9990f540f7 | ||
|
|
6dd9b04b50 | ||
|
|
d0e71e2cf0 | ||
|
|
6138e2b2ee | ||
|
|
4a8e798c64 | ||
|
|
c0f31a5941 | ||
|
|
cec7941b44 | ||
|
|
dbaa15c099 | ||
|
|
b5b47ca135 | ||
|
|
0c89b2c34e | ||
|
|
4bf90b5f09 | ||
|
|
17b4498c80 |
+143
-79
@@ -25,10 +25,12 @@ import asyncio
|
||||
from .config import settings
|
||||
from .data_loader import (
|
||||
clear_cache,
|
||||
compute_benchmarks,
|
||||
load_school_data,
|
||||
load_latest_school_data,
|
||||
geocode_single_postcode,
|
||||
get_supplementary_data,
|
||||
get_supplementary_data_batch,
|
||||
search_schools_typesense,
|
||||
)
|
||||
from .data_loader import get_data_info as get_db_info
|
||||
@@ -662,6 +664,36 @@ async def compare_schools(
|
||||
if comparison_data.empty:
|
||||
raise HTTPException(status_code=404, detail="No schools found")
|
||||
|
||||
# One session for all schools' supplementary blocks; failures degrade
|
||||
# to empty blocks rather than failing a working comparison (mirrors
|
||||
# the detail endpoint's defensive pattern).
|
||||
from . import database
|
||||
|
||||
_EMPTY_SUPPLEMENTARY = {
|
||||
"ofsted": None,
|
||||
"census": None,
|
||||
"admissions": None,
|
||||
"admissions_history": [],
|
||||
"deprivation": None,
|
||||
}
|
||||
supplementary_by_urn: dict = {}
|
||||
db = None
|
||||
try:
|
||||
db = database.SessionLocal()
|
||||
# One query per table for all schools, not ~5 queries per school.
|
||||
batch = get_supplementary_data_batch(db, urn_list)
|
||||
for urn in urn_list:
|
||||
supp = batch.get(urn, {})
|
||||
supplementary_by_urn[urn] = {
|
||||
key: supp.get(key, default)
|
||||
for key, default in _EMPTY_SUPPLEMENTARY.items()
|
||||
}
|
||||
except Exception:
|
||||
supplementary_by_urn = {}
|
||||
finally:
|
||||
if db is not None:
|
||||
db.close()
|
||||
|
||||
result = {}
|
||||
for urn in urn_list:
|
||||
school_data = comparison_data[comparison_data["urn"] == urn].sort_values("year")
|
||||
@@ -677,11 +709,27 @@ async def compare_schools(
|
||||
"phase": latest.get("phase", ""),
|
||||
"attainment_8_score": float(latest["attainment_8_score"]) if pd.notna(latest.get("attainment_8_score")) else None,
|
||||
"rwm_expected_pct": float(latest["rwm_expected_pct"]) if pd.notna(latest.get("rwm_expected_pct")) else None,
|
||||
# GIAS facts the compare "Who goes there" section needs
|
||||
# (same fields the detail endpoint exposes)
|
||||
"religious_denomination": convert_to_native(latest.get("religious_denomination")),
|
||||
"age_range": convert_to_native(latest.get("age_range")),
|
||||
"gender": convert_to_native(latest.get("gender")),
|
||||
"has_sixth_form": convert_to_native(latest.get("has_sixth_form")),
|
||||
"capacity": convert_to_native(latest.get("capacity")),
|
||||
"gias_total_pupils": convert_to_native(latest.get("gias_total_pupils")),
|
||||
"trust_name": convert_to_native(latest.get("trust_name")),
|
||||
},
|
||||
"yearly_data": clean_for_json(school_data),
|
||||
**supplementary_by_urn.get(urn, dict(_EMPTY_SUPPLEMENTARY)),
|
||||
}
|
||||
|
||||
return {"comparison": result}
|
||||
return {
|
||||
"comparison": result,
|
||||
# Official DfE anchors + computed state-school benchmarks so the
|
||||
# compare UI can label provenance correctly (spec §8.6).
|
||||
"national_averages": _national_averages_payload(df),
|
||||
"benchmarks": compute_benchmarks(df),
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/filters")
|
||||
@@ -727,96 +775,101 @@ async def get_la_averages(request: Request):
|
||||
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
||||
|
||||
|
||||
@app.get("/api/national-averages")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_national_averages(request: Request):
|
||||
_KS2_NATIONAL_METRICS = [
|
||||
"rwm_expected_pct", "rwm_high_pct",
|
||||
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
|
||||
"gps_expected_pct", "gps_high_pct", "science_expected_pct",
|
||||
"reading_avg_score", "maths_avg_score", "gps_avg_score",
|
||||
"reading_progress", "writing_progress", "maths_progress",
|
||||
"overall_absence_pct", "persistent_absence_pct",
|
||||
"disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct",
|
||||
]
|
||||
_KS4_NATIONAL_METRICS = [
|
||||
"attainment_8_score", "progress_8_score",
|
||||
"english_maths_standard_pass_pct", "english_maths_strong_pass_pct",
|
||||
"ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct",
|
||||
"ebacc_avg_score", "gcse_grade_91_pct",
|
||||
]
|
||||
|
||||
|
||||
def _national_averages_payload(df: pd.DataFrame) -> dict:
|
||||
"""National-averages payload shared by /api/national-averages and
|
||||
/api/compare.
|
||||
|
||||
Both series are persisted marts computed at import time: official DfE
|
||||
KS2 figures (fact_ks2_national_averages) and dataset-computed KS4
|
||||
averages (fact_ks4_national_averages) — the API never aggregates the
|
||||
performance dataframe per request. If the KS4 mart hasn't been built
|
||||
yet (deploy lands before the next DAG run), fall back to computing the
|
||||
latest year only — a single-year scan, never the historical loop.
|
||||
"""
|
||||
Compute national average for each metric from the latest data year.
|
||||
Returns separate averages for primary (KS2) and secondary (KS4) schools.
|
||||
Values are derived from the loaded DataFrame so they automatically
|
||||
stay current when new data is loaded.
|
||||
"""
|
||||
df = load_school_data()
|
||||
if df.empty:
|
||||
return {"primary": {}, "secondary": {}}
|
||||
|
||||
ks2_metrics = [
|
||||
"rwm_expected_pct", "rwm_high_pct",
|
||||
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
|
||||
"reading_avg_score", "maths_avg_score", "gps_avg_score",
|
||||
"reading_progress", "writing_progress", "maths_progress",
|
||||
"overall_absence_pct", "persistent_absence_pct",
|
||||
"disadvantaged_gap", "disadvantaged_pct", "sen_support_pct", "eal_pct",
|
||||
]
|
||||
ks4_metrics = [
|
||||
"attainment_8_score", "progress_8_score",
|
||||
"english_maths_standard_pass_pct", "english_maths_strong_pass_pct",
|
||||
"ebacc_entry_pct", "ebacc_standard_pass_pct", "ebacc_strong_pass_pct",
|
||||
"ebacc_avg_score", "gcse_grade_91_pct",
|
||||
]
|
||||
latest_year = int(df["year"].max())
|
||||
|
||||
def _means(sub_df, metric_list):
|
||||
from . import database
|
||||
from .models import Ks2NationalAverage, Ks4NationalAverage
|
||||
|
||||
def _row_metrics(row, metric_list):
|
||||
out = {}
|
||||
for col in metric_list:
|
||||
if col in sub_df.columns:
|
||||
val = sub_df[col].dropna()
|
||||
if len(val) > 0:
|
||||
out[col] = round(float(val.mean()), 2)
|
||||
val = getattr(row, col, None)
|
||||
if val is not None:
|
||||
out[col] = val
|
||||
return out
|
||||
|
||||
latest_year = int(df["year"].max())
|
||||
df_latest = df[df["year"] == latest_year]
|
||||
|
||||
# Primary: schools where KS2 data is non-null
|
||||
primary_df = df_latest[df_latest["rwm_expected_pct"].notna()]
|
||||
# Secondary: schools where KS4 data is non-null
|
||||
secondary_df = df_latest[df_latest["attainment_8_score"].notna()]
|
||||
|
||||
latest_primary = _means(primary_df, ks2_metrics)
|
||||
latest_secondary = _means(secondary_df, ks4_metrics)
|
||||
|
||||
# Per-year KS2 primary averages: use official DfE figures from the mart table.
|
||||
# Per-year KS4 secondary averages: computed from our dataset (no DfE dataset yet).
|
||||
from .database import SessionLocal
|
||||
from .models import Ks2NationalAverage
|
||||
|
||||
by_year = []
|
||||
ks2_rows: list = []
|
||||
ks4_rows: list = []
|
||||
db = None
|
||||
try:
|
||||
db = SessionLocal()
|
||||
nat_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
|
||||
# Build a lookup of computed secondary averages per year as fallback
|
||||
secondary_by_year = {}
|
||||
for yr in sorted(df["year"].dropna().unique()):
|
||||
yr = int(yr)
|
||||
df_yr = df[df["year"] == yr]
|
||||
secondary_by_year[yr] = _means(
|
||||
df_yr[df_yr["attainment_8_score"].notna()], ks4_metrics
|
||||
)
|
||||
# Merge: official KS2 figures + computed KS4 figures per year
|
||||
ks2_years = {r.year for r in nat_rows}
|
||||
all_years = sorted(ks2_years | set(secondary_by_year.keys()))
|
||||
nat_lookup = {r.year: r for r in nat_rows}
|
||||
for yr in all_years:
|
||||
primary_yr: dict = {}
|
||||
if yr in nat_lookup:
|
||||
r = nat_lookup[yr]
|
||||
for col in ks2_metrics:
|
||||
val = getattr(r, col, None)
|
||||
if val is not None:
|
||||
primary_yr[col] = val
|
||||
by_year.append({
|
||||
"year": yr,
|
||||
"primary": primary_yr,
|
||||
"secondary": secondary_by_year.get(yr, {}),
|
||||
})
|
||||
db = database.SessionLocal()
|
||||
try:
|
||||
ks2_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
|
||||
except Exception:
|
||||
db.rollback()
|
||||
try:
|
||||
ks4_rows = db.query(Ks4NationalAverage).order_by(Ks4NationalAverage.year).all()
|
||||
except Exception:
|
||||
db.rollback()
|
||||
except Exception:
|
||||
pass
|
||||
finally:
|
||||
db.close()
|
||||
if db is not None:
|
||||
db.close()
|
||||
|
||||
# Update latest_primary with official DfE figure for the latest year if available
|
||||
if by_year:
|
||||
latest_official = next((e["primary"] for e in reversed(by_year) if e["primary"]), None)
|
||||
if latest_official:
|
||||
latest_primary = latest_official
|
||||
primary_by_year = {r.year: _row_metrics(r, _KS2_NATIONAL_METRICS) for r in ks2_rows}
|
||||
secondary_by_year = {r.year: _row_metrics(r, _KS4_NATIONAL_METRICS) for r in ks4_rows}
|
||||
|
||||
if not any(secondary_by_year.values()):
|
||||
# KS4 mart missing/empty: compute the latest year only.
|
||||
df_latest = df[df["year"] == latest_year]
|
||||
sec = (
|
||||
df_latest[df_latest["attainment_8_score"].notna()]
|
||||
if "attainment_8_score" in df_latest.columns
|
||||
else df_latest.iloc[0:0]
|
||||
)
|
||||
vals = {}
|
||||
for col in _KS4_NATIONAL_METRICS:
|
||||
if col in sec.columns:
|
||||
v = sec[col].dropna()
|
||||
if len(v) > 0:
|
||||
vals[col] = round(float(v.mean()), 2)
|
||||
if vals:
|
||||
secondary_by_year[latest_year] = vals
|
||||
|
||||
all_years = sorted(set(primary_by_year) | set(secondary_by_year))
|
||||
by_year = [
|
||||
{
|
||||
"year": yr,
|
||||
"primary": primary_by_year.get(yr, {}),
|
||||
"secondary": secondary_by_year.get(yr, {}),
|
||||
}
|
||||
for yr in all_years
|
||||
]
|
||||
|
||||
latest_primary = next((e["primary"] for e in reversed(by_year) if e["primary"]), {})
|
||||
latest_secondary = next((e["secondary"] for e in reversed(by_year) if e["secondary"]), {})
|
||||
|
||||
return {
|
||||
"year": latest_year,
|
||||
@@ -826,6 +879,17 @@ async def get_national_averages(request: Request):
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/national-averages")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_national_averages(request: Request):
|
||||
"""
|
||||
National averages: official DfE KS2 figures per year plus computed
|
||||
KS4 averages, derived from the loaded DataFrame and the
|
||||
fact_ks2_national_averages mart.
|
||||
"""
|
||||
return _national_averages_payload(load_school_data())
|
||||
|
||||
|
||||
@app.get("/api/metrics")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_available_metrics(request: Request):
|
||||
|
||||
+281
-121
@@ -21,6 +21,7 @@ from .models import (
|
||||
FactOfstedInspection, FactAdmissions,
|
||||
FactDeprivation, FactFinance, FactPupilCharacteristics,
|
||||
)
|
||||
from .ofsted_codes import ofsted_page_url, report_card_labels
|
||||
from .schemas import SCHOOL_TYPE_MAP
|
||||
from .gias_codes import (
|
||||
ADMISSIONS_POLICY,
|
||||
@@ -190,13 +191,20 @@ _MAIN_QUERY = text("""
|
||||
p.reading_high_pct,
|
||||
p.reading_avg_score,
|
||||
p.reading_progress,
|
||||
p.reading_progress_lower_ci,
|
||||
p.reading_progress_upper_ci,
|
||||
p.writing_expected_pct,
|
||||
p.writing_high_pct,
|
||||
p.writing_progress,
|
||||
p.writing_progress_lower_ci,
|
||||
p.writing_progress_upper_ci,
|
||||
p.writing_working_towards_pct,
|
||||
p.maths_expected_pct,
|
||||
p.maths_high_pct,
|
||||
p.maths_avg_score,
|
||||
p.maths_progress,
|
||||
p.maths_progress_lower_ci,
|
||||
p.maths_progress_upper_ci,
|
||||
p.gps_expected_pct,
|
||||
p.gps_high_pct,
|
||||
p.gps_avg_score,
|
||||
@@ -225,6 +233,9 @@ _MAIN_QUERY = text("""
|
||||
p.progress_8_maths,
|
||||
p.progress_8_ebacc,
|
||||
p.progress_8_open,
|
||||
p.progress_8_banding,
|
||||
p.attainment_8_disadvantage_gap,
|
||||
p.progress_8_disadvantage_gap,
|
||||
p.english_maths_strong_pass_pct,
|
||||
p.english_maths_standard_pass_pct,
|
||||
p.ebacc_entry_pct,
|
||||
@@ -514,138 +525,287 @@ def get_data_info(db: Session = None) -> dict:
|
||||
# SUPPLEMENTARY DATA — per-school detail page
|
||||
# =============================================================================
|
||||
|
||||
def get_supplementary_data(db: Session, urn: int) -> dict:
|
||||
"""Fetch all supplementary data for a single school URN."""
|
||||
result = {}
|
||||
def compute_benchmarks(df: pd.DataFrame) -> dict:
|
||||
"""State-school benchmarks computed from our dataset (spec §5/§8.6).
|
||||
|
||||
def safe_query(model, pk_field, latest_field=None):
|
||||
NOT official DfE figures — consumers must label them
|
||||
"state-school average (computed from our dataset)". The disadvantaged
|
||||
attainment average is weighted by cohort size (eligible_pupils) so
|
||||
small schools don't dominate; context measures are medians.
|
||||
"""
|
||||
if df.empty or "year" not in df.columns:
|
||||
return {}
|
||||
latest_year = df["year"].max()
|
||||
if pd.isna(latest_year):
|
||||
return {}
|
||||
d = df[df["year"] == latest_year]
|
||||
if d.empty:
|
||||
return {}
|
||||
is_secondary = (
|
||||
d["attainment_8_score"].notna()
|
||||
if "attainment_8_score" in d.columns
|
||||
else pd.Series(False, index=d.index)
|
||||
)
|
||||
prim, sec = d[~is_secondary], d[is_secondary]
|
||||
|
||||
def _median(sub, col):
|
||||
if col not in sub.columns:
|
||||
return None
|
||||
v = sub[col].median()
|
||||
return round(float(v), 1) if pd.notna(v) else None
|
||||
|
||||
def _weighted_disadvantaged(sub):
|
||||
needed = {"rwm_expected_disadvantaged_pct", "eligible_pupils"}
|
||||
if not needed <= set(sub.columns):
|
||||
return None
|
||||
s = sub.dropna(subset=list(needed))
|
||||
if s.empty or s["eligible_pupils"].sum() == 0:
|
||||
return None
|
||||
w = (
|
||||
(s["rwm_expected_disadvantaged_pct"] * s["eligible_pupils"]).sum()
|
||||
/ s["eligible_pupils"].sum()
|
||||
)
|
||||
return round(float(w), 1)
|
||||
|
||||
def _block(sub, with_disadvantaged):
|
||||
median_pupils = None
|
||||
if "total_pupils" in sub.columns:
|
||||
mp = sub["total_pupils"].median()
|
||||
if pd.notna(mp):
|
||||
median_pupils = int(mp)
|
||||
block = {
|
||||
"eal_pct": _median(sub, "eal_pct"),
|
||||
"sen_support_pct": _median(sub, "sen_support_pct"),
|
||||
"disadvantaged_pct": _median(sub, "disadvantaged_pct"),
|
||||
"median_pupils": median_pupils,
|
||||
}
|
||||
if with_disadvantaged:
|
||||
block["disadvantaged_rwm_expected_pct"] = _weighted_disadvantaged(sub)
|
||||
return block
|
||||
|
||||
return {
|
||||
"source": "state-school average (computed from our dataset)",
|
||||
"year": int(latest_year),
|
||||
"primary": _block(prim, with_disadvantaged=True),
|
||||
"secondary": _block(sec, with_disadvantaged=False),
|
||||
}
|
||||
|
||||
|
||||
def _ofsted_block(o, urn: int) -> dict:
|
||||
"""Serialize the latest Ofsted inspection row for API responses.
|
||||
|
||||
`grade_source` records where the effective overall grade came from:
|
||||
a graded (Section 5) inspection, or carried forward from an ungraded
|
||||
(Section 8) outcome — materially different claims a UI must be able
|
||||
to distinguish. `report_card` holds coded+labelled renewed-framework
|
||||
(Nov 2025) area judgements; safeguarding is a separate boolean and
|
||||
never appears among the graded areas.
|
||||
"""
|
||||
if o.overall_effectiveness is not None:
|
||||
grade_source = "graded"
|
||||
overall = o.overall_effectiveness
|
||||
elif o.ungraded_grade is not None:
|
||||
# Fall back to the grade parsed from an ungraded (Section 8) outcome
|
||||
# (e.g. "School remains Good") so the detail page matches the list badge.
|
||||
grade_source = "ungraded_carried_forward"
|
||||
overall = o.ungraded_grade
|
||||
else:
|
||||
grade_source = None
|
||||
overall = None
|
||||
|
||||
block = {
|
||||
"framework": o.framework,
|
||||
"inspection_date": o.inspection_date.isoformat() if o.inspection_date else None,
|
||||
"inspection_type": o.inspection_type,
|
||||
"overall_effectiveness": overall,
|
||||
"grade_source": grade_source,
|
||||
"quality_of_education": o.quality_of_education,
|
||||
"behaviour_attitudes": o.behaviour_attitudes,
|
||||
"personal_development": o.personal_development,
|
||||
"leadership_management": o.leadership_management,
|
||||
"early_years_provision": o.early_years_provision,
|
||||
"sixth_form_provision": o.sixth_form_provision,
|
||||
"previous_overall": None, # Not available in new schema
|
||||
"rc_safeguarding_met": o.rc_safeguarding_met,
|
||||
"rc_inclusion": o.rc_inclusion,
|
||||
"rc_curriculum_teaching": o.rc_curriculum_teaching,
|
||||
"rc_achievement": o.rc_achievement,
|
||||
"rc_attendance_behaviour": o.rc_attendance_behaviour,
|
||||
"rc_personal_development": o.rc_personal_development,
|
||||
"rc_leadership_governance": o.rc_leadership_governance,
|
||||
"rc_early_years": o.rc_early_years,
|
||||
"rc_sixth_form": o.rc_sixth_form,
|
||||
"report_url": o.report_url,
|
||||
"ofsted_page_url": ofsted_page_url(urn),
|
||||
}
|
||||
block["report_card"] = report_card_labels(block)
|
||||
return block
|
||||
|
||||
|
||||
def _admissions_row_dict(a) -> dict:
|
||||
"""Serialize one fact_admissions row for API responses."""
|
||||
return {
|
||||
"year": a.year,
|
||||
"school_phase": a.school_phase,
|
||||
"places_offered": a.places_offered,
|
||||
"total_applications": a.total_applications,
|
||||
"first_preference_applications": a.first_preference_applications,
|
||||
"first_preference_offers": a.first_preference_offers,
|
||||
"first_preference_offer_pct": a.first_preference_offer_pct,
|
||||
"oversubscription_ratio": a.oversubscription_ratio,
|
||||
"oversubscribed": a.oversubscribed,
|
||||
"total_offers": a.total_offers,
|
||||
"second_preference_offers": a.second_preference_offers,
|
||||
"third_preference_offers": a.third_preference_offers,
|
||||
"cross_la_applications": a.cross_la_applications,
|
||||
"cross_la_offers": a.cross_la_offers,
|
||||
}
|
||||
|
||||
|
||||
def _census_dict(pc) -> dict:
|
||||
return {
|
||||
"year": pc.year,
|
||||
"total_pupils": pc.total_pupils,
|
||||
"female_pupils": pc.female_pupils,
|
||||
"male_pupils": pc.male_pupils,
|
||||
"fsm_pct": pc.fsm_pct,
|
||||
"eal_pct": pc.eal_pct,
|
||||
}
|
||||
|
||||
|
||||
def _deprivation_dict(d) -> dict:
|
||||
return {
|
||||
"lsoa_code": d.lsoa_code,
|
||||
"idaci_score": d.idaci_score,
|
||||
"idaci_decile": d.idaci_decile,
|
||||
}
|
||||
|
||||
|
||||
def _finance_dict(f) -> dict:
|
||||
return {
|
||||
"year": f.year,
|
||||
"per_pupil_spend": f.per_pupil_spend,
|
||||
"staff_cost_pct": f.staff_cost_pct,
|
||||
"teacher_cost_pct": f.teacher_cost_pct,
|
||||
"support_staff_cost_pct": f.support_staff_cost_pct,
|
||||
"premises_cost_pct": f.premises_cost_pct,
|
||||
}
|
||||
|
||||
|
||||
def _empty_supplementary() -> dict:
|
||||
return {
|
||||
"ofsted": None,
|
||||
"census": None,
|
||||
"admissions": None,
|
||||
"admissions_history": [],
|
||||
"sen_detail": None,
|
||||
"phonics": None,
|
||||
"deprivation": None,
|
||||
"finance": None,
|
||||
}
|
||||
|
||||
|
||||
def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict:
|
||||
"""Fetch supplementary data for many URNs with one query per table
|
||||
(WHERE urn IN (...)) instead of ~5 queries per school, collapsing the
|
||||
per-request round-trips from 5*N to a constant 5. Returns {urn: block}
|
||||
with the same shape get_supplementary_data produces per URN.
|
||||
|
||||
Each table is queried independently and failures degrade that table to
|
||||
empty for every URN — a missing mart never blanks the others.
|
||||
"""
|
||||
urns = [int(u) for u in urns]
|
||||
result = {urn: _empty_supplementary() for urn in urns}
|
||||
if not urns:
|
||||
return result
|
||||
|
||||
def _safe(fn):
|
||||
try:
|
||||
q = db.query(model).filter(getattr(model, pk_field) == urn)
|
||||
if latest_field:
|
||||
q = q.order_by(getattr(model, latest_field).desc())
|
||||
return q.first()
|
||||
fn()
|
||||
except Exception as e:
|
||||
import logging
|
||||
logging.getLogger(__name__).error("safe_query failed for %s: %s", model.__name__, e)
|
||||
logging.getLogger(__name__).error("batch supplementary query failed: %s", e)
|
||||
db.rollback()
|
||||
return None
|
||||
|
||||
# Latest Ofsted inspection
|
||||
o = safe_query(FactOfstedInspection, "urn", "inspection_date")
|
||||
result["ofsted"] = (
|
||||
{
|
||||
"framework": o.framework,
|
||||
"inspection_date": o.inspection_date.isoformat() if o.inspection_date else None,
|
||||
"inspection_type": o.inspection_type,
|
||||
# Fall back to the grade parsed from an ungraded (Section 8) outcome
|
||||
# (e.g. "School remains Good") when there's no graded grade, so the
|
||||
# detail page matches the list badge.
|
||||
"overall_effectiveness": (
|
||||
o.overall_effectiveness
|
||||
if o.overall_effectiveness is not None
|
||||
else o.ungraded_grade
|
||||
),
|
||||
"quality_of_education": o.quality_of_education,
|
||||
"behaviour_attitudes": o.behaviour_attitudes,
|
||||
"personal_development": o.personal_development,
|
||||
"leadership_management": o.leadership_management,
|
||||
"early_years_provision": o.early_years_provision,
|
||||
"sixth_form_provision": o.sixth_form_provision,
|
||||
"previous_overall": None, # Not available in new schema
|
||||
"rc_safeguarding_met": o.rc_safeguarding_met,
|
||||
"rc_inclusion": o.rc_inclusion,
|
||||
"rc_curriculum_teaching": o.rc_curriculum_teaching,
|
||||
"rc_achievement": o.rc_achievement,
|
||||
"rc_attendance_behaviour": o.rc_attendance_behaviour,
|
||||
"rc_personal_development": o.rc_personal_development,
|
||||
"rc_leadership_governance": o.rc_leadership_governance,
|
||||
"rc_early_years": o.rc_early_years,
|
||||
"rc_sixth_form": o.rc_sixth_form,
|
||||
"report_url": o.report_url,
|
||||
}
|
||||
if o
|
||||
else None
|
||||
)
|
||||
|
||||
# Census (latest year of fact_pupil_characteristics)
|
||||
pc = safe_query(FactPupilCharacteristics, "urn", "year")
|
||||
result["census"] = (
|
||||
{
|
||||
"year": pc.year,
|
||||
"total_pupils": pc.total_pupils,
|
||||
"female_pupils": pc.female_pupils,
|
||||
"male_pupils": pc.male_pupils,
|
||||
"fsm_pct": pc.fsm_pct,
|
||||
"eal_pct": pc.eal_pct,
|
||||
}
|
||||
if pc
|
||||
else None
|
||||
)
|
||||
|
||||
# Admissions — all years, oldest first (for the multi-year trend view).
|
||||
def _admissions_row(a):
|
||||
return {
|
||||
"year": a.year,
|
||||
"school_phase": a.school_phase,
|
||||
"places_offered": a.places_offered,
|
||||
"total_applications": a.total_applications,
|
||||
"first_preference_applications": a.first_preference_applications,
|
||||
"first_preference_offers": a.first_preference_offers,
|
||||
"first_preference_offer_pct": a.first_preference_offer_pct,
|
||||
"oversubscription_ratio": a.oversubscription_ratio,
|
||||
"oversubscribed": a.oversubscribed,
|
||||
}
|
||||
|
||||
try:
|
||||
admissions_rows = (
|
||||
db.query(FactAdmissions)
|
||||
.filter(FactAdmissions.urn == urn)
|
||||
.order_by(FactAdmissions.year.asc())
|
||||
# Ofsted — latest inspection per URN. Ordered so the first row seen per
|
||||
# URN is the most recent.
|
||||
def _ofsted():
|
||||
rows = (
|
||||
db.query(FactOfstedInspection)
|
||||
.filter(FactOfstedInspection.urn.in_(urns))
|
||||
.order_by(FactOfstedInspection.urn, FactOfstedInspection.inspection_date.desc())
|
||||
.all()
|
||||
)
|
||||
except Exception as e:
|
||||
import logging
|
||||
logging.getLogger(__name__).error("admissions history query failed: %s", e)
|
||||
db.rollback()
|
||||
admissions_rows = []
|
||||
seen = set()
|
||||
for o in rows:
|
||||
if o.urn in seen:
|
||||
continue
|
||||
seen.add(o.urn)
|
||||
result[o.urn]["ofsted"] = _ofsted_block(o, o.urn)
|
||||
_safe(_ofsted)
|
||||
|
||||
history = [_admissions_row(a) for a in admissions_rows]
|
||||
result["admissions_history"] = history
|
||||
# Keep the single latest-year object for backwards-compatible consumers
|
||||
# (hero chips, etc.).
|
||||
result["admissions"] = history[-1] if history else None
|
||||
# Census — latest year per URN.
|
||||
def _census():
|
||||
rows = (
|
||||
db.query(FactPupilCharacteristics)
|
||||
.filter(FactPupilCharacteristics.urn.in_(urns))
|
||||
.order_by(FactPupilCharacteristics.urn, FactPupilCharacteristics.year.desc())
|
||||
.all()
|
||||
)
|
||||
seen = set()
|
||||
for pc in rows:
|
||||
if pc.urn in seen:
|
||||
continue
|
||||
seen.add(pc.urn)
|
||||
result[pc.urn]["census"] = _census_dict(pc)
|
||||
_safe(_census)
|
||||
|
||||
# SEN detail — not available in current marts
|
||||
result["sen_detail"] = None
|
||||
# Admissions — all years per URN, oldest first (multi-year trend view).
|
||||
def _admissions():
|
||||
rows = (
|
||||
db.query(FactAdmissions)
|
||||
.filter(FactAdmissions.urn.in_(urns))
|
||||
.order_by(FactAdmissions.urn, FactAdmissions.year.asc())
|
||||
.all()
|
||||
)
|
||||
history: dict = {urn: [] for urn in urns}
|
||||
for a in rows:
|
||||
history[a.urn].append(_admissions_row_dict(a))
|
||||
for urn, rows_for_urn in history.items():
|
||||
result[urn]["admissions_history"] = rows_for_urn
|
||||
result[urn]["admissions"] = rows_for_urn[-1] if rows_for_urn else None
|
||||
_safe(_admissions)
|
||||
|
||||
# Phonics — no school-level data on EES
|
||||
result["phonics"] = None
|
||||
# Deprivation — one row per URN.
|
||||
def _deprivation():
|
||||
rows = (
|
||||
db.query(FactDeprivation)
|
||||
.filter(FactDeprivation.urn.in_(urns))
|
||||
.all()
|
||||
)
|
||||
for d in rows:
|
||||
result[d.urn]["deprivation"] = _deprivation_dict(d)
|
||||
_safe(_deprivation)
|
||||
|
||||
# Deprivation
|
||||
d = safe_query(FactDeprivation, "urn")
|
||||
result["deprivation"] = (
|
||||
{
|
||||
"lsoa_code": d.lsoa_code,
|
||||
"idaci_score": d.idaci_score,
|
||||
"idaci_decile": d.idaci_decile,
|
||||
}
|
||||
if d
|
||||
else None
|
||||
)
|
||||
|
||||
# Finance (latest year)
|
||||
f = safe_query(FactFinance, "urn", "year")
|
||||
result["finance"] = (
|
||||
{
|
||||
"year": f.year,
|
||||
"per_pupil_spend": f.per_pupil_spend,
|
||||
"staff_cost_pct": f.staff_cost_pct,
|
||||
"teacher_cost_pct": f.teacher_cost_pct,
|
||||
"support_staff_cost_pct": f.support_staff_cost_pct,
|
||||
"premises_cost_pct": f.premises_cost_pct,
|
||||
}
|
||||
if f
|
||||
else None
|
||||
)
|
||||
# Finance — latest year per URN.
|
||||
def _finance():
|
||||
rows = (
|
||||
db.query(FactFinance)
|
||||
.filter(FactFinance.urn.in_(urns))
|
||||
.order_by(FactFinance.urn, FactFinance.year.desc())
|
||||
.all()
|
||||
)
|
||||
seen = set()
|
||||
for f in rows:
|
||||
if f.urn in seen:
|
||||
continue
|
||||
seen.add(f.urn)
|
||||
result[f.urn]["finance"] = _finance_dict(f)
|
||||
_safe(_finance)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def get_supplementary_data(db: Session, urn: int) -> dict:
|
||||
"""Supplementary data for a single URN (thin wrapper over the batch)."""
|
||||
return get_supplementary_data_batch(db, [urn])[int(urn)]
|
||||
|
||||
@@ -88,6 +88,15 @@ class KS2Performance(Base):
|
||||
maths_high_pct = Column(Float)
|
||||
maths_avg_score = Column(Float)
|
||||
maths_progress = Column(Float)
|
||||
# Progress confidence intervals + writing working-towards (published
|
||||
# for years with progress measures, i.e. up to 2022/23)
|
||||
reading_progress_lower_ci = Column(Float)
|
||||
reading_progress_upper_ci = Column(Float)
|
||||
writing_progress_lower_ci = Column(Float)
|
||||
writing_progress_upper_ci = Column(Float)
|
||||
writing_working_towards_pct = Column(Float)
|
||||
maths_progress_lower_ci = Column(Float)
|
||||
maths_progress_upper_ci = Column(Float)
|
||||
gps_expected_pct = Column(Float)
|
||||
gps_high_pct = Column(Float)
|
||||
gps_avg_score = Column(Float)
|
||||
@@ -165,6 +174,11 @@ class FactAdmissions(Base):
|
||||
total_applications = Column(Integer)
|
||||
first_preference_applications = Column(Integer)
|
||||
first_preference_offers = Column(Integer)
|
||||
total_offers = Column(Integer)
|
||||
second_preference_offers = Column(Integer)
|
||||
third_preference_offers = Column(Integer)
|
||||
cross_la_applications = Column(Integer)
|
||||
cross_la_offers = Column(Integer)
|
||||
first_preference_offer_pct = Column(Float)
|
||||
oversubscription_ratio = Column(Float)
|
||||
oversubscribed = Column(Boolean)
|
||||
@@ -217,6 +231,23 @@ class FactFinance(Base):
|
||||
premises_cost_pct = Column(Float)
|
||||
|
||||
|
||||
class Ks4NationalAverage(Base):
|
||||
"""Computed national KS4 averages (from our dataset) — one row per year."""
|
||||
__tablename__ = "fact_ks4_national_averages"
|
||||
__table_args__ = MARTS
|
||||
|
||||
year = Column(Integer, primary_key=True)
|
||||
attainment_8_score = Column(Float)
|
||||
progress_8_score = Column(Float)
|
||||
english_maths_standard_pass_pct = Column(Float)
|
||||
english_maths_strong_pass_pct = Column(Float)
|
||||
ebacc_entry_pct = Column(Float)
|
||||
ebacc_standard_pass_pct = Column(Float)
|
||||
ebacc_strong_pass_pct = Column(Float)
|
||||
ebacc_avg_score = Column(Float)
|
||||
gcse_grade_91_pct = Column(Float)
|
||||
|
||||
|
||||
class Ks2NationalAverage(Base):
|
||||
"""Official DfE KS2 national headline averages — one row per academic year."""
|
||||
__tablename__ = "fact_ks2_national_averages"
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Ofsted renewed-framework (Nov 2025) report-card code translation.
|
||||
|
||||
Scale labels are the live-sampled vocabulary from the Ofsted MI file
|
||||
(see pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE) —
|
||||
verified against real data, not the consultation draft.
|
||||
"""
|
||||
|
||||
REPORT_CARD_GRADE_NAMES = {
|
||||
1: "Exceptional",
|
||||
2: "Strong standard",
|
||||
3: "Expected standard",
|
||||
4: "Needs attention",
|
||||
5: "Urgent improvement",
|
||||
}
|
||||
|
||||
# Graded evaluation areas only — safeguarding is a separate boolean
|
||||
# judgement and must never appear in grade counts or label maps.
|
||||
_RC_AREA_KEYS = (
|
||||
"rc_inclusion",
|
||||
"rc_curriculum_teaching",
|
||||
"rc_achievement",
|
||||
"rc_attendance_behaviour",
|
||||
"rc_personal_development",
|
||||
"rc_leadership_governance",
|
||||
"rc_early_years",
|
||||
"rc_sixth_form",
|
||||
)
|
||||
|
||||
|
||||
def report_card_labels(ofsted: dict) -> dict:
|
||||
"""{area_key: {code, label}} for populated, known-valued rc_* areas."""
|
||||
out = {}
|
||||
for key in _RC_AREA_KEYS:
|
||||
code = ofsted.get(key)
|
||||
label = REPORT_CARD_GRADE_NAMES.get(code)
|
||||
if code is not None and label is not None:
|
||||
out[key] = {"code": code, "label": label}
|
||||
return out
|
||||
|
||||
|
||||
def ofsted_page_url(urn: int) -> str:
|
||||
"""The school's page on ofsted.gov.uk (all its reports live there —
|
||||
we never deep-link an individual report)."""
|
||||
return f"https://reports.ofsted.gov.uk/provider/21/{urn}"
|
||||
@@ -0,0 +1,78 @@
|
||||
"""compute_benchmarks: state-school benchmarks computed from our dataset
|
||||
(spec §5/§8.6). The disadvantaged average must be weighted by cohort size,
|
||||
medians must ignore NaN, and only the latest year counts."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from backend.data_loader import compute_benchmarks
|
||||
|
||||
LATEST = 202425
|
||||
|
||||
|
||||
def _df():
|
||||
rows = [
|
||||
# Six primary schools, latest year. Disadvantaged RWM chosen so the
|
||||
# weighted average differs clearly from the unweighted mean:
|
||||
# weighted = (40*100 + 60*300) / 400 = 55.0 ; unweighted mean = 50.0
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=100,
|
||||
rwm_expected_disadvantaged_pct=40.0, eal_pct=10.0,
|
||||
sen_support_pct=10.0, disadvantaged_pct=20.0, total_pupils=200),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=300,
|
||||
rwm_expected_disadvantaged_pct=60.0, eal_pct=20.0,
|
||||
sen_support_pct=14.0, disadvantaged_pct=24.0, total_pupils=280),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=np.nan,
|
||||
rwm_expected_disadvantaged_pct=99.0, eal_pct=30.0,
|
||||
sen_support_pct=18.0, disadvantaged_pct=30.0, total_pupils=300),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=50,
|
||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=np.nan,
|
||||
sen_support_pct=np.nan, disadvantaged_pct=np.nan, total_pupils=np.nan),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=40,
|
||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=40.0,
|
||||
sen_support_pct=20.0, disadvantaged_pct=40.0, total_pupils=350),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=60,
|
||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=50.0,
|
||||
sen_support_pct=22.0, disadvantaged_pct=44.0, total_pupils=400),
|
||||
# Two secondary schools (attainment_8 non-null)
|
||||
dict(year=LATEST, attainment_8_score=45.0, eligible_pupils=180,
|
||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=15.0,
|
||||
sen_support_pct=12.0, disadvantaged_pct=22.0, total_pupils=1000),
|
||||
dict(year=LATEST, attainment_8_score=50.0, eligible_pupils=200,
|
||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=25.0,
|
||||
sen_support_pct=16.0, disadvantaged_pct=26.0, total_pupils=1200),
|
||||
# An older-year primary row that must NOT influence anything
|
||||
dict(year=202324, attainment_8_score=np.nan, eligible_pupils=500,
|
||||
rwm_expected_disadvantaged_pct=1.0, eal_pct=99.0,
|
||||
sen_support_pct=99.0, disadvantaged_pct=99.0, total_pupils=9999),
|
||||
]
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
|
||||
def test_weighted_disadvantaged_average():
|
||||
b = compute_benchmarks(_df())
|
||||
# Row 3 has NaN eligible_pupils and must be excluded from the weighting.
|
||||
assert b["primary"]["disadvantaged_rwm_expected_pct"] == 55.0
|
||||
|
||||
|
||||
def test_medians_ignore_nan_and_older_years():
|
||||
b = compute_benchmarks(_df())
|
||||
assert b["year"] == LATEST
|
||||
# eal medians over [10,20,30,40,50] = 30
|
||||
assert b["primary"]["eal_pct"] == 30.0
|
||||
# median pupils over [200,280,300,350,400] = 300
|
||||
assert b["primary"]["median_pupils"] == 300
|
||||
|
||||
|
||||
def test_secondary_block_has_no_disadvantaged_rwm():
|
||||
b = compute_benchmarks(_df())
|
||||
assert "disadvantaged_rwm_expected_pct" not in b["secondary"]
|
||||
assert b["secondary"]["median_pupils"] == 1100
|
||||
|
||||
|
||||
def test_provenance_string():
|
||||
b = compute_benchmarks(_df())
|
||||
assert b["source"] == "state-school average (computed from our dataset)"
|
||||
|
||||
|
||||
def test_empty_df():
|
||||
assert compute_benchmarks(pd.DataFrame()) == {}
|
||||
@@ -0,0 +1,123 @@
|
||||
"""/api/compare enrichment for the compare redesign: per-school
|
||||
supplementary blocks, top-level national_averages (shared with the
|
||||
/api/national-averages endpoint) and computed benchmarks — all additive."""
|
||||
|
||||
import types
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
LATEST = 202425
|
||||
|
||||
CANNED_SUPPLEMENTARY = {
|
||||
"ofsted": {"overall_effectiveness": 2, "grade_source": "graded",
|
||||
"report_card": {}, "ofsted_page_url": "https://reports.ofsted.gov.uk/provider/21/100140"},
|
||||
"census": {"year": 202526, "fsm_pct": 29.8},
|
||||
"admissions": {"year": 202627, "second_preference_offers": 4},
|
||||
"admissions_history": [{"year": 202627, "second_preference_offers": 4}],
|
||||
"sen_detail": None,
|
||||
"phonics": None,
|
||||
"deprivation": {"idaci_decile": 4},
|
||||
"finance": None,
|
||||
}
|
||||
|
||||
|
||||
def _two_primary_schools_df() -> pd.DataFrame:
|
||||
rows = []
|
||||
for urn, name, rwm, dis in ((100140, "Plumcroft Primary School", 79.0, 72.0),
|
||||
(138690, "Barclay Primary School", 87.0, 86.0)):
|
||||
rows.append(dict(
|
||||
urn=urn, school_name=name, local_authority="Greenwich",
|
||||
school_type="Community school", address="1 Road", phase="Primary",
|
||||
year=LATEST, rwm_expected_pct=rwm, attainment_8_score=np.nan,
|
||||
eligible_pupils=60, rwm_expected_disadvantaged_pct=dis,
|
||||
eal_pct=20.0, sen_support_pct=14.0, disadvantaged_pct=25.0,
|
||||
total_pupils=1000.0,
|
||||
))
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
|
||||
class _StubNatRow:
|
||||
year = 202425
|
||||
rwm_expected_pct = 62.1
|
||||
gps_expected_pct = 72.0
|
||||
science_expected_pct = 81.0
|
||||
|
||||
|
||||
class _StubSession:
|
||||
def query(self, *a, **k):
|
||||
return self
|
||||
|
||||
def order_by(self, *a, **k):
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
return [_StubNatRow()]
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend import database as database_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _two_primary_schools_df)
|
||||
monkeypatch.setattr(
|
||||
app_module,
|
||||
"get_supplementary_data_batch",
|
||||
lambda db, urns: {int(u): dict(CANNED_SUPPLEMENTARY) for u in urns},
|
||||
)
|
||||
monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_existing_shape_is_preserved(client):
|
||||
body = client.get("/api/compare?urns=100140,138690").json()
|
||||
school = body["comparison"]["100140"]
|
||||
assert school["school_info"]["rwm_expected_pct"] == 79.0
|
||||
assert school["yearly_data"][0]["year"] == LATEST
|
||||
|
||||
|
||||
def test_each_school_gains_supplementary_blocks(client):
|
||||
body = client.get("/api/compare?urns=100140,138690").json()
|
||||
for urn in ("100140", "138690"):
|
||||
school = body["comparison"][urn]
|
||||
assert school["ofsted"]["grade_source"] == "graded"
|
||||
assert school["census"]["fsm_pct"] == 29.8
|
||||
assert school["admissions"]["second_preference_offers"] == 4
|
||||
assert school["admissions_history"][0]["year"] == 202627
|
||||
assert school["deprivation"]["idaci_decile"] == 4
|
||||
|
||||
|
||||
def test_top_level_national_averages_and_benchmarks(client):
|
||||
body = client.get("/api/compare?urns=100140,138690").json()
|
||||
assert body["national_averages"]["year"] == LATEST
|
||||
assert body["benchmarks"]["source"] == "state-school average (computed from our dataset)"
|
||||
# weighted over equal cohorts of 72 and 86 = 79.0
|
||||
assert body["benchmarks"]["primary"]["disadvantaged_rwm_expected_pct"] == 79.0
|
||||
|
||||
|
||||
def test_supplementary_failure_degrades_not_500(client, monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
def _boom(db, urns):
|
||||
raise RuntimeError("marts unavailable")
|
||||
|
||||
monkeypatch.setattr(app_module, "get_supplementary_data_batch", _boom)
|
||||
resp = client.get("/api/compare?urns=100140")
|
||||
assert resp.status_code == 200
|
||||
school = resp.json()["comparison"]["100140"]
|
||||
assert school["ofsted"] is None
|
||||
assert school["admissions_history"] == []
|
||||
|
||||
|
||||
def test_national_averages_endpoint_exposes_gps_science(client):
|
||||
body = client.get("/api/national-averages").json()
|
||||
latest_primary_by_year = [e["primary"] for e in body["by_year"] if e["primary"]]
|
||||
assert latest_primary_by_year, "expected official by_year rows from the stub"
|
||||
assert latest_primary_by_year[-1]["gps_expected_pct"] == 72.0
|
||||
assert latest_primary_by_year[-1]["science_expected_pct"] == 81.0
|
||||
@@ -0,0 +1,93 @@
|
||||
"""_national_averages_payload reads persisted marts (computed at import
|
||||
time) — it must never loop the dataframe per year. The only dataframe work
|
||||
allowed is the single-latest-year KS4 fallback for the window between a
|
||||
deploy and the next DAG run."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
LATEST = 202425
|
||||
|
||||
|
||||
def _df():
|
||||
return pd.DataFrame(
|
||||
[
|
||||
dict(year=202324, attainment_8_score=40.0, rwm_expected_pct=np.nan),
|
||||
dict(year=LATEST, attainment_8_score=50.0, rwm_expected_pct=np.nan),
|
||||
dict(year=LATEST, attainment_8_score=30.0, rwm_expected_pct=np.nan),
|
||||
dict(year=LATEST, attainment_8_score=np.nan, rwm_expected_pct=80.0),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _Ks2Row:
|
||||
year = LATEST
|
||||
rwm_expected_pct = 62.1
|
||||
gps_expected_pct = 72.0
|
||||
|
||||
|
||||
class _Ks4Row:
|
||||
year = LATEST
|
||||
attainment_8_score = 46.5
|
||||
progress_8_score = -0.02
|
||||
|
||||
|
||||
class _StubSession:
|
||||
"""Returns KS2 rows for the first query and KS4 rows for the second —
|
||||
mirroring the payload's query order."""
|
||||
|
||||
def __init__(self):
|
||||
self.calls = 0
|
||||
|
||||
def query(self, model):
|
||||
self._model = model.__name__
|
||||
return self
|
||||
|
||||
def order_by(self, *a):
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
return [_Ks2Row()] if self._model == "Ks2NationalAverage" else [_Ks4Row()]
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
|
||||
class _Ks4MissingSession(_StubSession):
|
||||
def all(self):
|
||||
if self._model == "Ks4NationalAverage":
|
||||
raise RuntimeError("relation does not exist")
|
||||
return [_Ks2Row()]
|
||||
|
||||
def rollback(self):
|
||||
pass
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def payload(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend import database as database_module
|
||||
|
||||
def _run(session_cls):
|
||||
monkeypatch.setattr(database_module, "SessionLocal", session_cls)
|
||||
return app_module._national_averages_payload(_df())
|
||||
|
||||
return _run
|
||||
|
||||
|
||||
def test_ks4_averages_come_from_the_mart_not_the_dataframe(payload):
|
||||
body = payload(_StubSession)
|
||||
# Mart value (46.5), NOT the dataframe mean of (50+30)/2 = 40.0
|
||||
assert body["secondary"]["attainment_8_score"] == 46.5
|
||||
assert body["primary"]["rwm_expected_pct"] == 62.1
|
||||
assert body["by_year"][-1]["secondary"]["progress_8_score"] == -0.02
|
||||
|
||||
|
||||
def test_missing_ks4_mart_falls_back_to_latest_year_only(payload):
|
||||
body = payload(_Ks4MissingSession)
|
||||
# Fallback computes the latest year from the df: mean(50, 30) = 40.0
|
||||
assert body["secondary"]["attainment_8_score"] == 40.0
|
||||
# ...and only the latest year — no historical KS4 loop
|
||||
ks4_years = [e["year"] for e in body["by_year"] if e["secondary"]]
|
||||
assert ks4_years == [LATEST]
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Report-card code translation uses the live-sampled Ofsted vocabulary
|
||||
(pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE):
|
||||
Exceptional / Strong standard / Expected standard / Needs attention /
|
||||
Urgent improvement — never the consultation draft's 'Attention needed'."""
|
||||
|
||||
from backend.ofsted_codes import (
|
||||
REPORT_CARD_GRADE_NAMES,
|
||||
ofsted_page_url,
|
||||
report_card_labels,
|
||||
)
|
||||
|
||||
|
||||
def test_scale_is_sampled_vocabulary():
|
||||
assert REPORT_CARD_GRADE_NAMES == {
|
||||
1: "Exceptional",
|
||||
2: "Strong standard",
|
||||
3: "Expected standard",
|
||||
4: "Needs attention",
|
||||
5: "Urgent improvement",
|
||||
}
|
||||
|
||||
|
||||
def test_labels_only_for_populated_areas_and_never_safeguarding():
|
||||
ofsted = {
|
||||
"rc_achievement": 2,
|
||||
"rc_inclusion": 3,
|
||||
"rc_attendance_behaviour": 4,
|
||||
"rc_early_years": None,
|
||||
"rc_safeguarding_met": True,
|
||||
"overall_effectiveness": None,
|
||||
}
|
||||
labels = report_card_labels(ofsted)
|
||||
assert labels == {
|
||||
"rc_achievement": {"code": 2, "label": "Strong standard"},
|
||||
"rc_inclusion": {"code": 3, "label": "Expected standard"},
|
||||
"rc_attendance_behaviour": {"code": 4, "label": "Needs attention"},
|
||||
}
|
||||
|
||||
|
||||
def test_unknown_code_is_skipped_not_crashed():
|
||||
assert report_card_labels({"rc_achievement": 9}) == {}
|
||||
|
||||
|
||||
def test_provider_url():
|
||||
assert ofsted_page_url(138690) == "https://reports.ofsted.gov.uk/provider/21/138690"
|
||||
@@ -0,0 +1,111 @@
|
||||
"""get_supplementary_data_batch fetches one query per table for all URNs
|
||||
(not ~5 per school) and returns the same per-URN block shape as the
|
||||
single-URN function, picking the latest row per URN where relevant."""
|
||||
|
||||
import types
|
||||
|
||||
from backend import data_loader
|
||||
from backend.data_loader import get_supplementary_data_batch
|
||||
|
||||
|
||||
class _FakeQuery:
|
||||
"""Records that a query ran and serves canned rows filtered by an in-list."""
|
||||
|
||||
def __init__(self, recorder, model_name, rows):
|
||||
self._rec = recorder
|
||||
self._model = model_name
|
||||
self._rows = rows
|
||||
|
||||
def filter(self, *args, **kwargs):
|
||||
return self
|
||||
|
||||
def order_by(self, *args, **kwargs):
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
self._rec.append(self._model)
|
||||
return self._rows
|
||||
|
||||
def first(self):
|
||||
self._rec.append(self._model)
|
||||
return self._rows[0] if self._rows else None
|
||||
|
||||
|
||||
class _FakeSession:
|
||||
def __init__(self, rows_by_model):
|
||||
self.rows_by_model = rows_by_model
|
||||
self.queries: list[str] = []
|
||||
|
||||
def query(self, model):
|
||||
name = model.__name__
|
||||
return _FakeQuery(self.queries, name, self.rows_by_model.get(name, []))
|
||||
|
||||
def rollback(self):
|
||||
pass
|
||||
|
||||
|
||||
def _ofsted_row(urn, date, oe):
|
||||
base = {f: None for f in (
|
||||
"framework", "inspection_type", "quality_of_education", "behaviour_attitudes",
|
||||
"personal_development", "leadership_management", "early_years_provision",
|
||||
"sixth_form_provision", "ungraded_outcome", "ungraded_grade",
|
||||
"rc_safeguarding_met", "rc_inclusion", "rc_curriculum_teaching", "rc_achievement",
|
||||
"rc_attendance_behaviour", "rc_personal_development", "rc_leadership_governance",
|
||||
"rc_early_years", "rc_sixth_form", "report_url",
|
||||
)}
|
||||
base.update(urn=urn, inspection_date=types.SimpleNamespace(isoformat=lambda: date),
|
||||
overall_effectiveness=oe, grade_source=None)
|
||||
return types.SimpleNamespace(**base)
|
||||
|
||||
|
||||
def _adm_row(urn, year):
|
||||
return types.SimpleNamespace(
|
||||
urn=urn, year=year, school_phase="Primary", places_offered=100,
|
||||
total_applications=200, first_preference_applications=150,
|
||||
first_preference_offers=140, first_preference_offer_pct=93.3,
|
||||
oversubscription_ratio=1.5, oversubscribed=True,
|
||||
total_offers=100, second_preference_offers=5, third_preference_offers=2,
|
||||
cross_la_applications=10, cross_la_offers=3,
|
||||
)
|
||||
|
||||
|
||||
def test_one_query_per_table_and_latest_row_per_urn():
|
||||
rows = {
|
||||
# URN 1 has two Ofsted rows; the batch must keep the most recent (2023).
|
||||
"FactOfstedInspection": [
|
||||
_ofsted_row(1, "2023-01-01", 2),
|
||||
_ofsted_row(1, "2019-01-01", 3),
|
||||
_ofsted_row(2, "2021-06-01", 1),
|
||||
],
|
||||
"FactAdmissions": [_adm_row(1, 202526), _adm_row(1, 202627), _adm_row(2, 202627)],
|
||||
"FactPupilCharacteristics": [],
|
||||
"FactDeprivation": [],
|
||||
"FactFinance": [],
|
||||
}
|
||||
session = _FakeSession(rows)
|
||||
out = get_supplementary_data_batch(session, [1, 2])
|
||||
|
||||
# Exactly one query per table — five total, regardless of two URNs.
|
||||
assert sorted(session.queries) == [
|
||||
"FactAdmissions", "FactDeprivation", "FactFinance",
|
||||
"FactOfstedInspection", "FactPupilCharacteristics",
|
||||
]
|
||||
|
||||
# Latest Ofsted kept per URN
|
||||
assert out[1]["ofsted"]["overall_effectiveness"] == 2
|
||||
assert out[2]["ofsted"]["overall_effectiveness"] == 1
|
||||
|
||||
# Admissions history grouped per URN, latest exposed as `admissions`
|
||||
assert [r["year"] for r in out[1]["admissions_history"]] == [202526, 202627]
|
||||
assert out[1]["admissions"]["year"] == 202627
|
||||
assert out[2]["admissions_history"] == [{**out[2]["admissions_history"][0]}]
|
||||
|
||||
# Empty tables degrade to the null block, not a crash
|
||||
assert out[1]["census"] is None and out[1]["deprivation"] is None
|
||||
|
||||
|
||||
def test_single_wrapper_matches_batch(monkeypatch):
|
||||
session = _FakeSession({"FactOfstedInspection": [_ofsted_row(5, "2022-01-01", 2)]})
|
||||
single = data_loader.get_supplementary_data(session, 5)
|
||||
assert single["ofsted"]["overall_effectiveness"] == 2
|
||||
assert single["admissions_history"] == []
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Supplementary-block enrichment for the compare redesign: report-card
|
||||
labels, provider-page URL, graded-vs-carried-forward provenance, and the
|
||||
admissions preference/cross-LA detail promoted in the data-foundation PR."""
|
||||
|
||||
import types
|
||||
|
||||
from backend.data_loader import _admissions_row_dict, _ofsted_block
|
||||
|
||||
|
||||
def _row(**kw):
|
||||
base = dict(
|
||||
framework="RC", inspection_date=None, inspection_type=None,
|
||||
overall_effectiveness=None, quality_of_education=None,
|
||||
behaviour_attitudes=None, personal_development=None,
|
||||
leadership_management=None, early_years_provision=None,
|
||||
sixth_form_provision=None, ungraded_outcome=None, ungraded_grade=None,
|
||||
rc_safeguarding_met=None, rc_inclusion=None, rc_curriculum_teaching=None,
|
||||
rc_achievement=None, rc_attendance_behaviour=None,
|
||||
rc_personal_development=None, rc_leadership_governance=None,
|
||||
rc_early_years=None, rc_sixth_form=None, report_url=None,
|
||||
)
|
||||
base.update(kw)
|
||||
return types.SimpleNamespace(**base)
|
||||
|
||||
|
||||
def test_report_card_block_and_provider_url():
|
||||
o = _row(rc_achievement=2, rc_inclusion=3, rc_safeguarding_met=True)
|
||||
block = _ofsted_block(o, urn=100140)
|
||||
assert block["report_card"]["rc_achievement"]["label"] == "Strong standard"
|
||||
assert "rc_safeguarding_met" not in block["report_card"]
|
||||
assert block["rc_safeguarding_met"] is True
|
||||
assert block["ofsted_page_url"] == "https://reports.ofsted.gov.uk/provider/21/100140"
|
||||
|
||||
|
||||
def test_grade_source_graded_vs_carried_forward():
|
||||
assert _ofsted_block(_row(overall_effectiveness=1), urn=1)["grade_source"] == "graded"
|
||||
carried = _ofsted_block(_row(ungraded_grade=2), urn=1)
|
||||
assert carried["grade_source"] == "ungraded_carried_forward"
|
||||
assert carried["overall_effectiveness"] == 2
|
||||
assert _ofsted_block(_row(), urn=1)["grade_source"] is None
|
||||
|
||||
|
||||
def test_ofsted_block_keeps_existing_keys():
|
||||
block = _ofsted_block(_row(overall_effectiveness=2, quality_of_education=2), urn=1)
|
||||
for key in ("framework", "inspection_date", "overall_effectiveness",
|
||||
"quality_of_education", "rc_inclusion", "report_url"):
|
||||
assert key in block
|
||||
|
||||
|
||||
def test_admissions_row_new_fields():
|
||||
a = types.SimpleNamespace(
|
||||
year=202627, school_phase="Primary", places_offered=80,
|
||||
total_applications=185, first_preference_applications=74,
|
||||
first_preference_offers=74, first_preference_offer_pct=100.0,
|
||||
oversubscription_ratio=0.925, oversubscribed=False,
|
||||
total_offers=80, second_preference_offers=4, third_preference_offers=2,
|
||||
cross_la_applications=12, cross_la_offers=3,
|
||||
)
|
||||
d = _admissions_row_dict(a)
|
||||
for k in ("total_offers", "second_preference_offers", "third_preference_offers",
|
||||
"cross_la_applications", "cross_la_offers"):
|
||||
assert d[k] == getattr(a, k)
|
||||
# Existing keys unchanged
|
||||
assert d["first_preference_offer_pct"] == 100.0
|
||||
assert d["oversubscribed"] is False
|
||||
@@ -0,0 +1,399 @@
|
||||
# Compare API Enrichment (Backend PR) Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Expose the PR #32 data through the API so the redesigned compare screen can be built: enrich `/api/compare` with supplementary blocks + national averages + computed benchmarks, translate Ofsted report-card codes to labels, and surface the new mart columns (spec §6, §8 of `docs/superpowers/specs/2026-07-11-compare-screen-redesign-design.md`).
|
||||
|
||||
**Architecture:** All changes are additive API fields — existing consumers keep working. One small dbt change rides along: `fact_performance` (the combined KS2+KS4 mart the backend's `_MAIN_QUERY` reads) enumerates columns explicitly and was not extended in PR #32, so the new KS2 CI and KS4 banding columns must be threaded through it here. Everything else is backend Python: `models.py` mappings, `data_loader` query/supplementary additions, an Ofsted label dictionary (gias_codes pattern), and `/api/compare` composition.
|
||||
|
||||
**Tech Stack:** FastAPI, SQLAlchemy, pandas; dbt (one model); pytest via `python -m pytest backend/tests -q` (CI installs `requirements.txt pytest "httpx<0.28"`; locally use `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests -q`).
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- **Never push to `main`.** Branch: `feat/compare-api-enrichment`.
|
||||
- **Additive only** to API responses; never rename/remove existing fields (frontend + e2e depend on them).
|
||||
- **Report-card scale labels are the live-sampled vocabulary** (evidence in `pipeline/scripts/diagnose_compare_gaps.py`): `1=Exceptional, 2=Strong standard, 3=Expected standard, 4=Needs attention, 5=Urgent improvement`. Never "Attention needed". Safeguarding is boolean met/not-met, never counted as a graded area.
|
||||
- **Ofsted links** are always the provider page `https://reports.ofsted.gov.uk/provider/21/{urn}` (spec §5) labelled as the school's Ofsted page.
|
||||
- **Benchmark provenance** (spec §8.6): computed values are "state-school average (computed from our dataset)" — the API must expose them under a `benchmarks` key, clearly separate from official `national_averages`.
|
||||
- TDD: each behaviour lands with a failing test first, in `backend/tests/` following the `test_school_details.py` pattern (pandas fixture + monkeypatched `load_school_data` + `TestClient`).
|
||||
- Deploy note for the PR body: the new API fields return NULL/empty until prod's DAGs have run post-promotion.
|
||||
|
||||
---
|
||||
|
||||
### Task 0: Branch
|
||||
|
||||
- [ ] `git checkout main && git pull && git checkout -b feat/compare-api-enrichment` (commit this plan file on the branch).
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Thread PR #32 columns through `fact_performance`
|
||||
|
||||
**Files:**
|
||||
- Modify: `pipeline/transform/models/marts/fact_performance.sql`
|
||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (fact_performance block, if it has one — add the columns wherever the model's other columns are listed; if the model has no column list there, skip the yml)
|
||||
|
||||
**Interfaces:**
|
||||
- Produces (for `_MAIN_QUERY` in Task 4): `ks2.*` CI columns and `ks4.progress_8_banding`, `ks4.attainment_8_disadvantage_gap`, `ks4.progress_8_disadvantage_gap` on `marts.fact_performance`.
|
||||
|
||||
- [ ] **Step 1:** In `fact_performance.sql`, after `ks2.reading_progress,` add `ks2.reading_progress_lower_ci,` and `ks2.reading_progress_upper_ci,`; after `ks2.writing_progress,` add `ks2.writing_progress_lower_ci,`, `ks2.writing_progress_upper_ci,`, `ks2.writing_working_towards_pct,`; after `ks2.maths_progress,` add `ks2.maths_progress_lower_ci,`, `ks2.maths_progress_upper_ci,`. In the KS4 section, after the `ks4.progress_8_upper_ci`-equivalent line (locate the Progress 8 block) add:
|
||||
|
||||
```sql
|
||||
ks4.progress_8_banding,
|
||||
ks4.attainment_8_disadvantage_gap,
|
||||
ks4.progress_8_disadvantage_gap,
|
||||
```
|
||||
|
||||
- [ ] **Step 2:** Parse gate: `cd pipeline/transform && uv run --with dbt-postgres python -m dbt.cli.main parse --profiles-dir .` → exit 0.
|
||||
|
||||
- [ ] **Step 3:** Commit: `feat(pipeline): thread compare-foundation columns through fact_performance`
|
||||
|
||||
---
|
||||
|
||||
### Task 2: ORM mappings for the new mart columns
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/models.py` (`KS2Performance` after `maths_progress`; `FactAdmissions` after `first_preference_offers`)
|
||||
- Test: none (declarative mappings; covered by Task 4's query tests)
|
||||
|
||||
**Interfaces:**
|
||||
- Produces attributes used by Task 4: `KS2Performance.reading_progress_lower_ci` … `maths_progress_upper_ci`, `writing_working_towards_pct` (Float); `FactAdmissions.total_offers`, `.second_preference_offers`, `.third_preference_offers`, `.cross_la_applications`, `.cross_la_offers` (Integer).
|
||||
|
||||
- [ ] **Step 1:** Add to `KS2Performance` (next to the existing progress columns):
|
||||
|
||||
```python
|
||||
reading_progress_lower_ci = Column(Float)
|
||||
reading_progress_upper_ci = Column(Float)
|
||||
writing_progress_lower_ci = Column(Float)
|
||||
writing_progress_upper_ci = Column(Float)
|
||||
writing_working_towards_pct = Column(Float)
|
||||
maths_progress_lower_ci = Column(Float)
|
||||
maths_progress_upper_ci = Column(Float)
|
||||
```
|
||||
|
||||
Add to `FactAdmissions` (after `first_preference_offers`):
|
||||
|
||||
```python
|
||||
total_offers = Column(Integer)
|
||||
second_preference_offers = Column(Integer)
|
||||
third_preference_offers = Column(Integer)
|
||||
cross_la_applications = Column(Integer)
|
||||
cross_la_offers = Column(Integer)
|
||||
```
|
||||
|
||||
(`FactOfstedInspection` already maps all `rc_*` columns with the right types — verify, don't change.)
|
||||
|
||||
- [ ] **Step 2:** Commit: `feat(api): map compare-foundation mart columns`
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Ofsted label dictionary + provider URL (TDD)
|
||||
|
||||
**Files:**
|
||||
- Create: `backend/ofsted_codes.py`
|
||||
- Test: `backend/tests/test_ofsted_codes.py`
|
||||
|
||||
**Interfaces:**
|
||||
- Produces for Task 4: `REPORT_CARD_GRADE_NAMES: dict[int, str]`, `report_card_labels(ofsted: dict) -> dict` (returns `{area_key: {"code": int, "label": str}}` for the non-null `rc_*` grade fields, excluding safeguarding), `ofsted_page_url(urn: int) -> str`.
|
||||
|
||||
- [ ] **Step 1: Failing tests**
|
||||
|
||||
```python
|
||||
"""Report-card code translation uses the live-sampled Ofsted vocabulary
|
||||
(pipeline/scripts/diagnose_compare_gaps.py TASK 7 VALUE SAMPLE):
|
||||
Exceptional / Strong standard / Expected standard / Needs attention /
|
||||
Urgent improvement — never the consultation draft's 'Attention needed'."""
|
||||
from backend.ofsted_codes import (
|
||||
REPORT_CARD_GRADE_NAMES, report_card_labels, ofsted_page_url,
|
||||
)
|
||||
|
||||
|
||||
def test_scale_is_sampled_vocabulary():
|
||||
assert REPORT_CARD_GRADE_NAMES == {
|
||||
1: "Exceptional",
|
||||
2: "Strong standard",
|
||||
3: "Expected standard",
|
||||
4: "Needs attention",
|
||||
5: "Urgent improvement",
|
||||
}
|
||||
|
||||
|
||||
def test_labels_only_for_populated_areas_and_never_safeguarding():
|
||||
ofsted = {
|
||||
"rc_achievement": 2,
|
||||
"rc_inclusion": 3,
|
||||
"rc_attendance_behaviour": 4,
|
||||
"rc_early_years": None,
|
||||
"rc_safeguarding_met": True,
|
||||
"overall_effectiveness": None,
|
||||
}
|
||||
labels = report_card_labels(ofsted)
|
||||
assert labels == {
|
||||
"rc_achievement": {"code": 2, "label": "Strong standard"},
|
||||
"rc_inclusion": {"code": 3, "label": "Expected standard"},
|
||||
"rc_attendance_behaviour": {"code": 4, "label": "Needs attention"},
|
||||
}
|
||||
|
||||
|
||||
def test_unknown_code_is_skipped_not_crashed():
|
||||
assert report_card_labels({"rc_achievement": 9}) == {}
|
||||
|
||||
|
||||
def test_provider_url():
|
||||
assert ofsted_page_url(138690) == "https://reports.ofsted.gov.uk/provider/21/138690"
|
||||
```
|
||||
|
||||
- [ ] **Step 2:** Run `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_ofsted_codes.py -q` → FAIL (module missing).
|
||||
|
||||
- [ ] **Step 3: Implement `backend/ofsted_codes.py`**
|
||||
|
||||
```python
|
||||
"""Ofsted renewed-framework (Nov 2025) report-card code translation.
|
||||
|
||||
Scale labels are the live-sampled vocabulary from the Ofsted MI file
|
||||
(see pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE) —
|
||||
verified against real data, not the consultation draft.
|
||||
"""
|
||||
|
||||
REPORT_CARD_GRADE_NAMES = {
|
||||
1: "Exceptional",
|
||||
2: "Strong standard",
|
||||
3: "Expected standard",
|
||||
4: "Needs attention",
|
||||
5: "Urgent improvement",
|
||||
}
|
||||
|
||||
# Graded evaluation areas only — safeguarding is a separate boolean
|
||||
# judgement and must never appear in grade counts or label maps.
|
||||
_RC_AREA_KEYS = (
|
||||
"rc_inclusion",
|
||||
"rc_curriculum_teaching",
|
||||
"rc_achievement",
|
||||
"rc_attendance_behaviour",
|
||||
"rc_personal_development",
|
||||
"rc_leadership_governance",
|
||||
"rc_early_years",
|
||||
"rc_sixth_form",
|
||||
)
|
||||
|
||||
|
||||
def report_card_labels(ofsted: dict) -> dict:
|
||||
"""{area_key: {code, label}} for populated, known-valued rc_* areas."""
|
||||
out = {}
|
||||
for key in _RC_AREA_KEYS:
|
||||
code = ofsted.get(key)
|
||||
label = REPORT_CARD_GRADE_NAMES.get(code)
|
||||
if code is not None and label is not None:
|
||||
out[key] = {"code": code, "label": label}
|
||||
return out
|
||||
|
||||
|
||||
def ofsted_page_url(urn: int) -> str:
|
||||
"""The school's page on ofsted.gov.uk (all its reports live there —
|
||||
we never deep-link an individual report; spec §5)."""
|
||||
return f"https://reports.ofsted.gov.uk/provider/21/{urn}"
|
||||
```
|
||||
|
||||
- [ ] **Step 4:** Re-run the test file → 4 passed. Run the full suite (same command, `backend/tests -q`) → all pass.
|
||||
|
||||
- [ ] **Step 5:** Commit: `feat(api): Ofsted report-card labels and provider-page URL`
|
||||
|
||||
---
|
||||
|
||||
### Task 4: data_loader — query columns + richer supplementary blocks (TDD)
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/data_loader.py` (`_MAIN_QUERY` ~line 153; `get_supplementary_data` ~line 460)
|
||||
- Test: `backend/tests/test_supplementary_enrichment.py`
|
||||
|
||||
**Interfaces:**
|
||||
- `_MAIN_QUERY` additionally selects (KS2 block, after `p.maths_progress`): `p.reading_progress_lower_ci, p.reading_progress_upper_ci, p.writing_progress_lower_ci, p.writing_progress_upper_ci, p.writing_working_towards_pct, p.maths_progress_lower_ci, p.maths_progress_upper_ci`; (KS4 block, after the Progress 8 CI columns): `p.progress_8_banding, p.attainment_8_disadvantage_gap, p.progress_8_disadvantage_gap`. Note `_MAIN_QUERY_NO_SIXTH_FORM`/`_MAIN_QUERY_LEGACY_NAMES` are string-derived from `_MAIN_QUERY` (lines 259-270) and inherit automatically — verify the assertions there still hold.
|
||||
- `get_supplementary_data(db, urn)["admissions"]` rows additionally carry: `total_offers`, `second_preference_offers`, `third_preference_offers`, `cross_la_applications`, `cross_la_offers` (add to `_admissions_row`).
|
||||
- `get_supplementary_data(db, urn)["ofsted"]` additionally carries: `report_card` (the `report_card_labels(...)` dict, `{}` when no rc data), `ofsted_page_url`, and `grade_source`: `"graded"` when `overall_effectiveness` came from the graded column, `"ungraded_carried_forward"` when the fallback `ungraded_grade` supplied it, `None` when neither.
|
||||
|
||||
- [ ] **Step 1: Failing tests** — construct a fake Ofsted row object (simple `types.SimpleNamespace` with the model's attributes) and call the block-building logic via `get_supplementary_data` with a stubbed session (follow how existing tests stub the db; if none do, factor the ofsted-dict construction into a pure helper `_ofsted_block(o, urn)` and test that directly — preferred):
|
||||
|
||||
```python
|
||||
import types
|
||||
from backend.data_loader import _ofsted_block
|
||||
|
||||
|
||||
def _row(**kw):
|
||||
base = dict(
|
||||
framework="RC", inspection_date=None, inspection_type=None,
|
||||
overall_effectiveness=None, quality_of_education=None,
|
||||
behaviour_attitudes=None, personal_development=None,
|
||||
leadership_management=None, early_years_provision=None,
|
||||
sixth_form_provision=None, ungraded_outcome=None, ungraded_grade=None,
|
||||
rc_safeguarding_met=None, rc_inclusion=None, rc_curriculum_teaching=None,
|
||||
rc_achievement=None, rc_attendance_behaviour=None,
|
||||
rc_personal_development=None, rc_leadership_governance=None,
|
||||
rc_early_years=None, rc_sixth_form=None, report_url=None,
|
||||
)
|
||||
base.update(kw)
|
||||
return types.SimpleNamespace(**base)
|
||||
|
||||
|
||||
def test_report_card_block_and_provider_url():
|
||||
o = _row(rc_achievement=2, rc_inclusion=3, rc_safeguarding_met=True)
|
||||
block = _ofsted_block(o, urn=100140)
|
||||
assert block["report_card"]["rc_achievement"]["label"] == "Strong standard"
|
||||
assert "rc_safeguarding_met" not in block["report_card"]
|
||||
assert block["rc_safeguarding_met"] is True
|
||||
assert block["ofsted_page_url"] == "https://reports.ofsted.gov.uk/provider/21/100140"
|
||||
|
||||
|
||||
def test_grade_source_graded_vs_carried_forward():
|
||||
assert _ofsted_block(_row(overall_effectiveness=1), urn=1)["grade_source"] == "graded"
|
||||
carried = _ofsted_block(_row(ungraded_grade=2), urn=1)
|
||||
assert carried["grade_source"] == "ungraded_carried_forward"
|
||||
assert carried["overall_effectiveness"] == 2
|
||||
assert _ofsted_block(_row(), urn=1)["grade_source"] is None
|
||||
|
||||
|
||||
def test_admissions_row_new_fields():
|
||||
from backend.data_loader import _admissions_row_dict
|
||||
a = types.SimpleNamespace(
|
||||
year=202627, school_phase="Primary", places_offered=80,
|
||||
total_applications=185, first_preference_applications=74,
|
||||
first_preference_offers=74, first_preference_offer_pct=100.0,
|
||||
oversubscription_ratio=0.925, oversubscribed=False,
|
||||
total_offers=80, second_preference_offers=4, third_preference_offers=2,
|
||||
cross_la_applications=12, cross_la_offers=3,
|
||||
)
|
||||
d = _admissions_row_dict(a)
|
||||
for k in ("total_offers", "second_preference_offers", "third_preference_offers",
|
||||
"cross_la_applications", "cross_la_offers"):
|
||||
assert d[k] == getattr(a, k)
|
||||
```
|
||||
|
||||
- [ ] **Step 2:** Run → FAIL (helpers don't exist).
|
||||
|
||||
- [ ] **Step 3: Implement.** Refactor the existing inline ofsted-dict construction in `get_supplementary_data` into a module-level `_ofsted_block(o, urn)` that produces the existing keys **unchanged** plus the three new ones (`report_card` via `report_card_labels(...)` from Task 3, `ofsted_page_url` via `ofsted_page_url(urn)`, `grade_source` per the interface rule — derived from which source supplied `overall_effectiveness`). Rename/extract the local `_admissions_row` into module-level `_admissions_row_dict(a)` and append the five new fields. Add the ten new columns to `_MAIN_QUERY` exactly as the interface lists them. `get_supplementary_data` calls both helpers; its external shape gains only additive keys.
|
||||
|
||||
- [ ] **Step 4:** Full suite → all pass (existing `test_school_details.py` etc. must not break; if a fixture enumerates yearly-data columns, extend it with the new NaN columns as needed).
|
||||
|
||||
- [ ] **Step 5:** Commit: `feat(api): expose progress CIs, KS4 banding/gaps, admissions detail, report-card labels`
|
||||
|
||||
---
|
||||
|
||||
### Task 5: Computed benchmarks helper (TDD)
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/data_loader.py` (new function)
|
||||
- Test: `backend/tests/test_benchmarks.py`
|
||||
|
||||
**Interfaces:**
|
||||
- Produces for Task 6: `compute_benchmarks(df) -> dict` — pure function over the main dataframe (latest year, state schools), shape:
|
||||
|
||||
```python
|
||||
{
|
||||
"source": "state-school average (computed from our dataset)",
|
||||
"year": 202425,
|
||||
"primary": {
|
||||
"disadvantaged_rwm_expected_pct": 46.1, # weighted by eligible_pupils
|
||||
"eal_pct": 22.3, # median
|
||||
"sen_support_pct": 14.0, # median
|
||||
"disadvantaged_pct": 24.8, # median (FSM6 proxy)
|
||||
"median_pupils": 281, # median school size
|
||||
},
|
||||
"secondary": { "median_pupils": 1024, "eal_pct": ..., "sen_support_pct": ..., "disadvantaged_pct": ... },
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 1: Failing tests** — build a small synthetic df (6 primary rows with known eligible_pupils/rwm_expected_disadvantaged_pct so the weighted average is hand-checkable; a couple of secondary rows flagged by non-null `attainment_8_score`), assert: weighted disadvantaged average matches hand computation (not the unweighted mean), medians ignore NaN, secondary block lacks the disadvantaged-RWM key, latest-year filtering (rows from an older year must not affect results), and empty df → `{}`.
|
||||
|
||||
- [ ] **Step 2:** Run → FAIL.
|
||||
|
||||
- [ ] **Step 3: Implement** in `data_loader.py`:
|
||||
|
||||
```python
|
||||
def compute_benchmarks(df: pd.DataFrame) -> dict:
|
||||
"""State-school benchmarks computed from our dataset (spec §5/§8.6).
|
||||
These are NOT official DfE figures — consumers must label them
|
||||
'state-school average (computed from our dataset)'."""
|
||||
if df.empty or "year" not in df.columns:
|
||||
return {}
|
||||
latest_year = df["year"].max()
|
||||
d = df[df["year"] == latest_year]
|
||||
if d.empty:
|
||||
return {}
|
||||
is_secondary = d["attainment_8_score"].notna() if "attainment_8_score" in d.columns else pd.Series(False, index=d.index)
|
||||
prim, sec = d[~is_secondary], d[is_secondary]
|
||||
|
||||
def _median(sub, col):
|
||||
if col not in sub.columns:
|
||||
return None
|
||||
v = sub[col].median()
|
||||
return round(float(v), 1) if pd.notna(v) else None
|
||||
|
||||
def _weighted_disadvantaged(sub):
|
||||
if not {"rwm_expected_disadvantaged_pct", "eligible_pupils"} <= set(sub.columns):
|
||||
return None
|
||||
s = sub.dropna(subset=["rwm_expected_disadvantaged_pct", "eligible_pupils"])
|
||||
if s.empty or s["eligible_pupils"].sum() == 0:
|
||||
return None
|
||||
w = (s["rwm_expected_disadvantaged_pct"] * s["eligible_pupils"]).sum() / s["eligible_pupils"].sum()
|
||||
return round(float(w), 1)
|
||||
|
||||
def _block(sub, with_disadvantaged):
|
||||
block = {
|
||||
"eal_pct": _median(sub, "eal_pct"),
|
||||
"sen_support_pct": _median(sub, "sen_support_pct"),
|
||||
"disadvantaged_pct": _median(sub, "disadvantaged_pct"),
|
||||
"median_pupils": int(sub["total_pupils"].median()) if "total_pupils" in sub.columns and pd.notna(sub["total_pupils"].median()) else None,
|
||||
}
|
||||
if with_disadvantaged:
|
||||
block["disadvantaged_rwm_expected_pct"] = _weighted_disadvantaged(sub)
|
||||
return block
|
||||
|
||||
return {
|
||||
"source": "state-school average (computed from our dataset)",
|
||||
"year": int(latest_year),
|
||||
"primary": _block(prim, with_disadvantaged=True),
|
||||
"secondary": _block(sec, with_disadvantaged=False),
|
||||
}
|
||||
```
|
||||
|
||||
(Adapt column presence to the real df — `sen_support_pct` reaches the df via `_MAIN_QUERY`; confirm and add it there if the KS2 block doesn't already select it, mirroring Task 4's additions.)
|
||||
|
||||
- [ ] **Step 4:** Full suite → pass. **Step 5:** Commit: `feat(api): computed state-school benchmarks`
|
||||
|
||||
---
|
||||
|
||||
### Task 6: Enrich `/api/compare` + expose GPS/science national averages (TDD)
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/app.py` (`compare_schools` ~line 636; `get_national_averages` ~line 730)
|
||||
- Test: `backend/tests/test_compare_enrichment.py`
|
||||
|
||||
**Interfaces (response additions, all additive):**
|
||||
- `/api/compare` top level gains: `"national_averages"` (same payload the `/api/national-averages` endpoint returns — extract the endpoint body into a helper `_national_averages_payload(df)` and reuse; do not duplicate the logic) and `"benchmarks"` (Task 5's `compute_benchmarks(df)`).
|
||||
- Each `comparison[urn]` gains: `"ofsted"`, `"census"`, `"admissions"`, `"admissions_history"`, `"deprivation"` from `get_supplementary_data` (one `SessionLocal()` for the whole request, closed in `finally`; on exception the five keys are `None`/`[]` — mirror the detail endpoint's defensive pattern at app.py:583-590).
|
||||
- `get_national_averages`' KS2 metric list gains `"gps_expected_pct", "gps_high_pct", "science_expected_pct"` so the England ticks for GPS/science flow once the data exists.
|
||||
|
||||
- [ ] **Step 1: Failing tests** — monkeypatch `load_school_data` with a two-school primary df (reuse/extend the fixture style of `test_school_details.py`) and monkeypatch `get_supplementary_data` to a canned dict; assert on `TestClient(app).get("/api/compare?urns=...")`:
|
||||
- response keeps the existing shape (`comparison[urn]["school_info"]["rwm_expected_pct"]` etc.),
|
||||
- each school gains the five supplementary keys (canned values round-tripped),
|
||||
- top-level `national_averages` and `benchmarks` present; `benchmarks["source"]` is the exact provenance string,
|
||||
- a supplementary-layer exception (monkeypatched to raise) degrades to `ofsted: None` etc. with HTTP 200,
|
||||
- `/api/national-averages` includes `gps_expected_pct` in the primary block when the df/national table provides it (monkeypatch the national-averages source the endpoint reads).
|
||||
|
||||
- [ ] **Step 2:** Run → FAIL. **Step 3:** Implement per the interfaces. **Step 4:** Full suite → pass.
|
||||
|
||||
- [ ] **Step 5:** Commit: `feat(api): compare endpoint carries supplementary blocks, national averages and benchmarks`
|
||||
|
||||
---
|
||||
|
||||
### Task 7: PR + verification
|
||||
|
||||
- [ ] **Step 1:** Full suite one more time + `uv run --with pyyaml python3 -c "import yaml; yaml.safe_load(open('.gitea/workflows/deploy.yml'))"` sanity is NOT needed (no workflow changes) — instead run the dbt parse gate again (Task 1 file).
|
||||
- [ ] **Step 2:** Push, open PR via the Gitea API (credential-helper basic auth). PR body: the new response shapes (one JSON sketch), the reused-not-duplicated national-averages helper, the provenance rule for benchmarks, deploy note (fields NULL until prod DAGs run post-promotion), and that no e2e change is needed (no user-facing behaviour changes — the compare UI still reads the old fields; the frontend PR carries the journey updates).
|
||||
- [ ] **Step 3:** After merge + staging deploy: `curl -s https://stx.schoolcompare.co.uk/api/compare?urns=138690,100140 | python3 -m json.tool | head -80` — verify the new keys and that `benchmarks.primary.disadvantaged_rwm_expected_pct` is plausible (~45-47). Verify `/api/national-averages` now carries `gps_expected_pct`/`science_expected_pct` (values or honest nulls if DfE suppresses them at national level).
|
||||
|
||||
---
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Frontend rebuild + e2e journeys (next PR — consumes everything this PR exposes).
|
||||
- `schemas.py` METRIC_DEFINITIONS additions for the trends picker (frontend PR decides which of the new columns become picker metrics).
|
||||
- CI-based progress banding logic (frontend computes Above/Average/Below from the CI columns; historical years only).
|
||||
@@ -32,26 +32,24 @@ export default async function ComparePage({ searchParams }: ComparePageProps) {
|
||||
const selectedMetric = metricParam || 'rwm_expected_pct';
|
||||
|
||||
try {
|
||||
// Fetch comparison data if URNs provided
|
||||
let comparisonData = null;
|
||||
if (urns.length > 0) {
|
||||
try {
|
||||
const response = await fetchComparison(urnsParam!);
|
||||
comparisonData = response.comparison;
|
||||
} catch (error) {
|
||||
console.error('Failed to fetch comparison:', error);
|
||||
}
|
||||
}
|
||||
// Fetch comparison + metrics in parallel — they are independent.
|
||||
const [comparisonResponse, metricsResponse] = await Promise.all([
|
||||
urns.length > 0
|
||||
? fetchComparison(urnsParam!).catch((error) => {
|
||||
console.error('Failed to fetch comparison:', error);
|
||||
return null;
|
||||
})
|
||||
: Promise.resolve(null),
|
||||
fetchMetrics(),
|
||||
]);
|
||||
|
||||
// Fetch available metrics
|
||||
const metricsResponse = await fetchMetrics();
|
||||
|
||||
// Metrics is already an array
|
||||
const metricsArray = metricsResponse?.metrics || [];
|
||||
|
||||
return (
|
||||
<ComparisonView
|
||||
initialData={comparisonData}
|
||||
initialData={comparisonResponse?.comparison ?? null}
|
||||
initialNationalAverages={comparisonResponse?.national_averages}
|
||||
initialBenchmarks={comparisonResponse?.benchmarks}
|
||||
initialUrns={urns}
|
||||
metrics={metricsArray}
|
||||
selectedMetric={selectedMetric}
|
||||
|
||||
@@ -35,6 +35,8 @@ import styles from './ComparisonView.module.css';
|
||||
|
||||
interface ComparisonViewProps {
|
||||
initialData: Record<string, ComparisonData> | null;
|
||||
initialNationalAverages?: NationalAverages;
|
||||
initialBenchmarks?: Benchmarks;
|
||||
initialUrns: number[];
|
||||
metrics: MetricDefinition[];
|
||||
selectedMetric: string;
|
||||
@@ -42,6 +44,8 @@ interface ComparisonViewProps {
|
||||
|
||||
export function ComparisonView({
|
||||
initialData,
|
||||
initialNationalAverages,
|
||||
initialBenchmarks,
|
||||
initialUrns,
|
||||
metrics,
|
||||
selectedMetric: initialMetric,
|
||||
@@ -54,8 +58,10 @@ export function ComparisonView({
|
||||
const [selectedMetric, setSelectedMetric] = useState(initialMetric);
|
||||
const [isModalOpen, setIsModalOpen] = useState(false);
|
||||
const [comparisonData, setComparisonData] = useState(initialData);
|
||||
const [nationalAverages, setNationalAverages] = useState<NationalAverages | undefined>();
|
||||
const [benchmarks, setBenchmarks] = useState<Benchmarks | undefined>();
|
||||
const [nationalAverages, setNationalAverages] = useState<NationalAverages | undefined>(
|
||||
initialNationalAverages,
|
||||
);
|
||||
const [benchmarks, setBenchmarks] = useState<Benchmarks | undefined>(initialBenchmarks);
|
||||
const [shareConfirm, setShareConfirm] = useState(false);
|
||||
const [comparePhase, setComparePhase] = useState<'primary' | 'secondary'>('primary');
|
||||
// Tracks whether the user has explicitly clicked a phase tab.
|
||||
@@ -81,13 +87,16 @@ export function ComparisonView({
|
||||
}
|
||||
}, [isInitialized]); // eslint-disable-line react-hooks/exhaustive-deps
|
||||
|
||||
// Sync URL with selected schools + metric, and (re)fetch the comparison.
|
||||
const urnKey = selectedSchools.map((s) => s.urn).join(',');
|
||||
|
||||
// Sync the URL with the selection + metric. Pure navigation state — no
|
||||
// fetching here: metric changes are presentational (the data is already
|
||||
// client-side) and must not refire the comparison request.
|
||||
useEffect(() => {
|
||||
const urns = selectedSchools.map((s) => s.urn).join(',');
|
||||
const params = new URLSearchParams(searchParams);
|
||||
|
||||
if (urns) {
|
||||
params.set('urns', urns);
|
||||
if (urnKey) {
|
||||
params.set('urns', urnKey);
|
||||
} else {
|
||||
params.delete('urns');
|
||||
}
|
||||
@@ -96,26 +105,41 @@ export function ComparisonView({
|
||||
|
||||
const newUrl = `${pathname}?${params.toString()}`;
|
||||
router.replace(newUrl, { scroll: false });
|
||||
}, [urnKey, selectedMetric, pathname, searchParams, router]);
|
||||
|
||||
if (selectedSchools.length > 0) {
|
||||
fetchComparison(urns, { cache: 'no-store' })
|
||||
.then((data) => {
|
||||
setComparisonData(data.comparison);
|
||||
setNationalAverages(data.national_averages);
|
||||
setBenchmarks(data.benchmarks);
|
||||
})
|
||||
.catch((err) => {
|
||||
// Keep whatever we already have (SSR data or a previous fetch) rather
|
||||
// than blanking the page — a transient refetch failure shouldn't
|
||||
// destroy a working comparison the user is looking at.
|
||||
console.error('Failed to fetch comparison:', err);
|
||||
});
|
||||
} else {
|
||||
// Fetch only when the school set changes. The very first run is skipped
|
||||
// when the SSR payload already covers the current set — no double-fetch
|
||||
// of data the server just rendered.
|
||||
const firstFetchRef = useRef(true);
|
||||
useEffect(() => {
|
||||
if (!urnKey) {
|
||||
setComparisonData(null);
|
||||
setNationalAverages(undefined);
|
||||
setBenchmarks(undefined);
|
||||
return;
|
||||
}
|
||||
}, [selectedSchools, selectedMetric, pathname, searchParams, router]);
|
||||
|
||||
if (firstFetchRef.current) {
|
||||
firstFetchRef.current = false;
|
||||
const ssrUrns = new Set(Object.keys(initialData ?? {}));
|
||||
const covered = urnKey.split(',').every((urn) => ssrUrns.has(urn));
|
||||
if (covered && ssrUrns.size > 0) return;
|
||||
}
|
||||
|
||||
fetchComparison(urnKey, { cache: 'no-store' })
|
||||
.then((data) => {
|
||||
setComparisonData(data.comparison);
|
||||
setNationalAverages(data.national_averages);
|
||||
setBenchmarks(data.benchmarks);
|
||||
})
|
||||
.catch((err) => {
|
||||
// Keep whatever we already have (SSR data or a previous fetch) rather
|
||||
// than blanking the page — a transient refetch failure shouldn't
|
||||
// destroy a working comparison the user is looking at.
|
||||
console.error('Failed to fetch comparison:', err);
|
||||
});
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [urnKey]);
|
||||
|
||||
// Classify schools by phase using comparison data
|
||||
const classifySchool = (school: School): 'primary' | 'secondary' => {
|
||||
|
||||
@@ -160,6 +160,12 @@ models:
|
||||
- name: year
|
||||
tests: [not_null, unique]
|
||||
|
||||
- name: fact_ks4_national_averages
|
||||
description: Computed national KS4 averages (means across state schools in our dataset — not official DfE figures) — one row per academic year
|
||||
columns:
|
||||
- name: year
|
||||
tests: [not_null, unique]
|
||||
|
||||
- name: fact_deprivation
|
||||
description: IDACI deprivation index — one row per URN
|
||||
columns:
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
{{ config(materialized='table') }}
|
||||
|
||||
-- Mart: Computed national KS4 averages — one row per academic year.
|
||||
-- Unlike fact_ks2_national_averages (official DfE figures), DfE publishes no
|
||||
-- KS4 national-headline dataset we ingest yet, so these are means computed
|
||||
-- across the state schools in our dataset. Computed once at build time so the
|
||||
-- API never has to aggregate the full performance table per request.
|
||||
-- Semantics match the API's previous per-request computation: rows where
|
||||
-- attainment_8_score is non-null; per-column means ignore NULLs.
|
||||
|
||||
select
|
||||
year,
|
||||
round(avg(attainment_8_score)::numeric, 2) as attainment_8_score,
|
||||
round(avg(progress_8_score)::numeric, 2) as progress_8_score,
|
||||
round(avg(english_maths_standard_pass_pct)::numeric, 2) as english_maths_standard_pass_pct,
|
||||
round(avg(english_maths_strong_pass_pct)::numeric, 2) as english_maths_strong_pass_pct,
|
||||
round(avg(ebacc_entry_pct)::numeric, 2) as ebacc_entry_pct,
|
||||
round(avg(ebacc_standard_pass_pct)::numeric, 2) as ebacc_standard_pass_pct,
|
||||
round(avg(ebacc_strong_pass_pct)::numeric, 2) as ebacc_strong_pass_pct,
|
||||
round(avg(ebacc_avg_score)::numeric, 2) as ebacc_avg_score,
|
||||
round(avg(gcse_grade_91_pct)::numeric, 2) as gcse_grade_91_pct
|
||||
from {{ ref('fact_ks4_performance') }}
|
||||
where attainment_8_score is not null
|
||||
group by year
|
||||
order by year
|
||||
@@ -25,13 +25,20 @@ select
|
||||
ks2.reading_high_pct,
|
||||
ks2.reading_avg_score,
|
||||
ks2.reading_progress,
|
||||
ks2.reading_progress_lower_ci,
|
||||
ks2.reading_progress_upper_ci,
|
||||
ks2.writing_expected_pct,
|
||||
ks2.writing_high_pct,
|
||||
ks2.writing_progress,
|
||||
ks2.writing_progress_lower_ci,
|
||||
ks2.writing_progress_upper_ci,
|
||||
ks2.writing_working_towards_pct,
|
||||
ks2.maths_expected_pct,
|
||||
ks2.maths_high_pct,
|
||||
ks2.maths_avg_score,
|
||||
ks2.maths_progress,
|
||||
ks2.maths_progress_lower_ci,
|
||||
ks2.maths_progress_upper_ci,
|
||||
ks2.gps_expected_pct,
|
||||
ks2.gps_high_pct,
|
||||
ks2.gps_avg_score,
|
||||
@@ -61,6 +68,9 @@ select
|
||||
ks4.progress_8_maths,
|
||||
ks4.progress_8_ebacc,
|
||||
ks4.progress_8_open,
|
||||
ks4.progress_8_banding,
|
||||
ks4.attainment_8_disadvantage_gap,
|
||||
ks4.progress_8_disadvantage_gap,
|
||||
ks4.english_maths_strong_pass_pct,
|
||||
ks4.english_maths_standard_pass_pct,
|
||||
ks4.ebacc_entry_pct,
|
||||
|
||||
Reference in New Issue
Block a user