Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cc5b6955d8 | ||
|
|
35752a7d53 | ||
|
|
e48469b058 | ||
|
|
660d84502a |
@@ -1,4 +1,4 @@
|
|||||||
name: Stage (build -> staging -> E2E gate)
|
name: Deploy (staging -> E2E gate -> production)
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
@@ -193,5 +193,48 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||||
|
|
||||||
# Production deployment is a second, manual approval: see promote.yml
|
promote-prod:
|
||||||
# ("Promote to Production (manual)") and docs/DEPLOY.md.
|
name: Promote to Production
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: [e2e-staging]
|
||||||
|
steps:
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
|
- name: Log in to Gitea Container Registry
|
||||||
|
uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: ${{ env.REGISTRY }}
|
||||||
|
username: ${{ gitea.actor }}
|
||||||
|
password: ${{ secrets.REGISTRY_TOKEN }}
|
||||||
|
|
||||||
|
- name: Retag verified images as prod
|
||||||
|
run: |
|
||||||
|
SHORT_SHA="sha-$(echo "${{ gitea.sha }}" | cut -c1-7)"
|
||||||
|
for IMAGE in \
|
||||||
|
"${REGISTRY}/${BACKEND_IMAGE_NAME}" \
|
||||||
|
"${REGISTRY}/${FRONTEND_IMAGE_NAME}" \
|
||||||
|
"${REGISTRY}/${PIPELINE_IMAGE_NAME}"; do
|
||||||
|
# Keep a rollback pointer before moving :prod
|
||||||
|
docker buildx imagetools create -t "${IMAGE}:prod-previous" "${IMAGE}:prod" || true
|
||||||
|
docker buildx imagetools create -t "${IMAGE}:prod" "${IMAGE}:${SHORT_SHA}"
|
||||||
|
echo "Promoted ${IMAGE}:${SHORT_SHA} -> :prod"
|
||||||
|
done
|
||||||
|
|
||||||
|
- name: Trigger production stack update
|
||||||
|
run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}"
|
||||||
|
|
||||||
|
- name: Wait for production to become healthy
|
||||||
|
run: |
|
||||||
|
echo "Polling ${PROD_BASE_URL} for up to 5 minutes..."
|
||||||
|
for i in $(seq 1 60); do
|
||||||
|
if curl -fsS -o /dev/null --max-time 10 "${PROD_BASE_URL}/"; then
|
||||||
|
echo "Production is up (attempt $i)"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
echo "Production did not become healthy in time" >&2
|
||||||
|
exit 1
|
||||||
|
env:
|
||||||
|
PROD_BASE_URL: ${{ secrets.PROD_BASE_URL }}
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
# TEMPORARY diagnostic workflow — delete after the rankings year= bug is closed.
|
||||||
|
name: Staging Rankings Diagnostic
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- diag/staging-rankings-year
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
staging-api-diagnostic:
|
||||||
|
name: Staging rankings year diagnostic
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Probe staging rankings year handling
|
||||||
|
env:
|
||||||
|
BASE: ${{ secrets.STAGING_BASE_URL }}
|
||||||
|
run: |
|
||||||
|
probe() {
|
||||||
|
echo "== $1 =="
|
||||||
|
curl -s --max-time 15 -D /tmp/h.txt -o /tmp/b.txt "$BASE$1"
|
||||||
|
echo "--- headers:"; cat /tmp/h.txt
|
||||||
|
echo "--- body (first 400 bytes):"; head -c 400 /tmp/b.txt; echo
|
||||||
|
}
|
||||||
|
probe "/api/data-info"
|
||||||
|
probe "/api/filters"
|
||||||
|
probe "/api/rankings?metric=rwm_expected_pct&limit=3"
|
||||||
|
probe "/api/rankings?metric=rwm_expected_pct&limit=3&year=202425"
|
||||||
|
probe "/"
|
||||||
@@ -12,7 +12,25 @@ env:
|
|||||||
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
|
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
frontend-checks:
|
# TEMPORARY: evidence gathering for the rankings year= empty-list bug on
|
||||||
|
# staging. Remove before merging. Prints status codes and row counts only.
|
||||||
|
staging-api-diagnostic:
|
||||||
|
name: Staging rankings year diagnostic
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Probe staging rankings year handling
|
||||||
|
env:
|
||||||
|
BASE: ${{ secrets.STAGING_BASE_URL }}
|
||||||
|
run: |
|
||||||
|
echo "== /api/filters years =="
|
||||||
|
curl -s --max-time 15 "$BASE/api/filters" -o /tmp/f.json -w "status %{http_code}\n"
|
||||||
|
python3 -c "import json; print(json.load(open('/tmp/f.json')).get('years'))" || head -c 300 /tmp/f.json
|
||||||
|
echo "== /api/rankings probes (metric=rwm_expected_pct, limit=3) =="
|
||||||
|
for Q in "" "&year=202425" "&year=201819" "&year=2024"; do
|
||||||
|
CODE=$(curl -s --max-time 15 -o /tmp/r.json -w "%{http_code}" "$BASE/api/rankings?metric=rwm_expected_pct&limit=3$Q")
|
||||||
|
echo "query [$Q] -> status $CODE"
|
||||||
|
python3 -c "import json; d=json.load(open('/tmp/r.json')); print(' year:', d.get('year'), 'total:', d.get('total'), 'rows:', len(d.get('rankings', [])))" || head -c 300 /tmp/r.json
|
||||||
|
done
|
||||||
name: Frontend Typecheck + Tests
|
name: Frontend Typecheck + Tests
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
@@ -51,14 +69,11 @@ jobs:
|
|||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: pip install -r requirements.txt pytest "httpx<0.28"
|
run: pip install -r requirements.txt
|
||||||
|
|
||||||
- name: Import smoke test
|
- name: Import smoke test
|
||||||
run: python -c "from backend.app import app; print('backend imports OK')"
|
run: python -c "from backend.app import app; print('backend imports OK')"
|
||||||
|
|
||||||
- name: Backend unit tests
|
|
||||||
run: python -m pytest backend/tests -q
|
|
||||||
|
|
||||||
build-backend:
|
build-backend:
|
||||||
name: Build Backend (no push)
|
name: Build Backend (no push)
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
|
|||||||
@@ -1,126 +0,0 @@
|
|||||||
name: Promote to Production (manual)
|
|
||||||
|
|
||||||
# Second approval gate of the deploy model: run this workflow from the
|
|
||||||
# Actions UI after testing the feature on staging. It refuses commits
|
|
||||||
# whose staging E2E gate is not green. See docs/DEPLOY.md.
|
|
||||||
|
|
||||||
on:
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
sha:
|
|
||||||
description: >-
|
|
||||||
Commit SHA on main to promote (full or >=7 chars).
|
|
||||||
Leave empty to promote the latest main commit.
|
|
||||||
required: false
|
|
||||||
default: ""
|
|
||||||
|
|
||||||
# Only one promotion at a time; never cancel an in-flight promotion.
|
|
||||||
concurrency:
|
|
||||||
group: prod-promotion
|
|
||||||
cancel-in-progress: false
|
|
||||||
|
|
||||||
env:
|
|
||||||
REGISTRY: privaterepo.sitaru.org
|
|
||||||
BACKEND_IMAGE_NAME: ${{ gitea.repository }}-backend
|
|
||||||
FRONTEND_IMAGE_NAME: ${{ gitea.repository }}-frontend
|
|
||||||
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
promote-prod:
|
|
||||||
name: Promote approved commit to Production
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Checkout repository (full history for ancestry check)
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
with:
|
|
||||||
fetch-depth: 0
|
|
||||||
|
|
||||||
- name: Resolve and validate target SHA
|
|
||||||
id: resolve
|
|
||||||
# SECURITY: the dispatch input is untrusted — it reaches the shell
|
|
||||||
# only via env (never spliced into `run:` with ${{ }}) and is only
|
|
||||||
# used as a quoted argument. The resolved value is validated as a
|
|
||||||
# 40-hex sha and required to be an ancestor of main before any
|
|
||||||
# later step interpolates it.
|
|
||||||
env:
|
|
||||||
SHA_INPUT: ${{ gitea.event.inputs.sha }}
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
case "$SHA_INPUT" in
|
|
||||||
-*) echo "REFUSED: SHA input may not start with '-'." >&2; exit 1 ;;
|
|
||||||
esac
|
|
||||||
if [ -z "$SHA_INPUT" ]; then
|
|
||||||
SHA_INPUT="$(git rev-parse origin/main)"
|
|
||||||
fi
|
|
||||||
FULL_SHA=$(git rev-parse --verify --quiet "${SHA_INPUT}^{commit}") || {
|
|
||||||
echo "REFUSED: not a commit in this repository." >&2
|
|
||||||
exit 1
|
|
||||||
}
|
|
||||||
echo "$FULL_SHA" | grep -Eq '^[0-9a-f]{40}$'
|
|
||||||
if ! git merge-base --is-ancestor "$FULL_SHA" origin/main; then
|
|
||||||
echo "REFUSED: $FULL_SHA is not on main — only main commits are promotable." >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
SHORT_SHA="sha-$(echo "$FULL_SHA" | cut -c1-7)"
|
|
||||||
echo "full=$FULL_SHA" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "short=$SHORT_SHA" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "Promoting $FULL_SHA (images tagged $SHORT_SHA)"
|
|
||||||
|
|
||||||
- name: Verify the staging E2E gate passed for this commit
|
|
||||||
run: |
|
|
||||||
STATUS_JSON=$(curl -fsS \
|
|
||||||
-H "Authorization: token ${{ secrets.REGISTRY_TOKEN }}" \
|
|
||||||
"https://${REGISTRY}/api/v1/repos/${{ gitea.repository }}/commits/${{ steps.resolve.outputs.full }}/status")
|
|
||||||
echo "$STATUS_JSON" | python3 -c "
|
|
||||||
import json, sys
|
|
||||||
d = json.load(sys.stdin)
|
|
||||||
ok = [s for s in d.get('statuses', [])
|
|
||||||
if 'E2E Journeys against Staging' in s.get('context', '')
|
|
||||||
and s.get('status') == 'success']
|
|
||||||
if not ok:
|
|
||||||
print('REFUSED: no successful \"E2E Journeys against Staging\" status on this commit.')
|
|
||||||
print('Contexts found:', [s.get('context') for s in d.get('statuses', [])])
|
|
||||||
sys.exit(1)
|
|
||||||
print('E2E gate verified green for this commit.')
|
|
||||||
"
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
|
||||||
uses: docker/setup-buildx-action@v3
|
|
||||||
|
|
||||||
- name: Log in to Gitea Container Registry
|
|
||||||
uses: docker/login-action@v3
|
|
||||||
with:
|
|
||||||
registry: ${{ env.REGISTRY }}
|
|
||||||
username: ${{ gitea.actor }}
|
|
||||||
password: ${{ secrets.REGISTRY_TOKEN }}
|
|
||||||
|
|
||||||
- name: Retag approved images as prod (keeping rollback pointer)
|
|
||||||
run: |
|
|
||||||
SHORT_SHA="${{ steps.resolve.outputs.short }}"
|
|
||||||
for IMAGE in \
|
|
||||||
"${REGISTRY}/${BACKEND_IMAGE_NAME}" \
|
|
||||||
"${REGISTRY}/${FRONTEND_IMAGE_NAME}" \
|
|
||||||
"${REGISTRY}/${PIPELINE_IMAGE_NAME}"; do
|
|
||||||
# Keep a rollback pointer before moving :prod
|
|
||||||
docker buildx imagetools create -t "${IMAGE}:prod-previous" "${IMAGE}:prod" || true
|
|
||||||
docker buildx imagetools create -t "${IMAGE}:prod" "${IMAGE}:${SHORT_SHA}"
|
|
||||||
echo "Promoted ${IMAGE}:${SHORT_SHA} -> :prod"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Trigger production stack update
|
|
||||||
run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}"
|
|
||||||
|
|
||||||
- name: Wait for production to become healthy
|
|
||||||
run: |
|
|
||||||
echo "Polling ${PROD_BASE_URL} for up to 5 minutes..."
|
|
||||||
for i in $(seq 1 60); do
|
|
||||||
if curl -fsS -o /dev/null --max-time 10 "${PROD_BASE_URL}/"; then
|
|
||||||
echo "Production is up (attempt $i)"
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
sleep 5
|
|
||||||
done
|
|
||||||
echo "Production did not become healthy in time" >&2
|
|
||||||
exit 1
|
|
||||||
env:
|
|
||||||
PROD_BASE_URL: ${{ secrets.PROD_BASE_URL }}
|
|
||||||
+1
-1
@@ -1,2 +1,2 @@
|
|||||||
venv
|
venv
|
||||||
__pycache__/
|
backend/__pycache__
|
||||||
|
|||||||
+23
-93
@@ -25,7 +25,6 @@ import asyncio
|
|||||||
from .config import settings
|
from .config import settings
|
||||||
from .data_loader import (
|
from .data_loader import (
|
||||||
clear_cache,
|
clear_cache,
|
||||||
compute_benchmarks,
|
|
||||||
load_school_data,
|
load_school_data,
|
||||||
load_latest_school_data,
|
load_latest_school_data,
|
||||||
geocode_single_postcode,
|
geocode_single_postcode,
|
||||||
@@ -34,7 +33,7 @@ from .data_loader import (
|
|||||||
)
|
)
|
||||||
from .data_loader import get_data_info as get_db_info
|
from .data_loader import get_data_info as get_db_info
|
||||||
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||||
from .utils import clean_for_json, convert_to_native
|
from .utils import clean_for_json
|
||||||
|
|
||||||
# Values to exclude from filter dropdowns (empty strings, non-applicable labels)
|
# Values to exclude from filter dropdowns (empty strings, non-applicable labels)
|
||||||
EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"}
|
EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"}
|
||||||
@@ -417,17 +416,10 @@ async def get_schools(
|
|||||||
df_latest = df_latest[df_latest["gender"].str.lower() == gender.lower()]
|
df_latest = df_latest[df_latest["gender"].str.lower() == gender.lower()]
|
||||||
if admissions_policy:
|
if admissions_policy:
|
||||||
df_latest = df_latest[df_latest["admissions_policy"].str.lower() == admissions_policy.lower()]
|
df_latest = df_latest[df_latest["admissions_policy"].str.lower() == admissions_policy.lower()]
|
||||||
# GIAS OfficialSixthForm flag (dim_school.has_sixth_form). NULL (flag not
|
if has_sixth_form == "yes":
|
||||||
# yet populated by the pipeline) is treated as "no sixth form".
|
df_latest = df_latest[df_latest["age_range"].str.contains("18", na=False)]
|
||||||
if has_sixth_form in ("yes", "no"):
|
elif has_sixth_form == "no":
|
||||||
if "has_sixth_form" in df_latest.columns:
|
df_latest = df_latest[~df_latest["age_range"].str.contains("18", na=False)]
|
||||||
flag = df_latest["has_sixth_form"].eq(True)
|
|
||||||
else: # Defensive fallback only — data_loader now always synthesizes
|
|
||||||
# has_sixth_form as NULL when the DB predates the pipeline re-run,
|
|
||||||
# so this branch shouldn't normally trigger. Falls back to age
|
|
||||||
# range if the column is somehow absent anyway.
|
|
||||||
flag = df_latest["age_range"].str.contains("18", na=False)
|
|
||||||
df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag]
|
|
||||||
|
|
||||||
# Include key result metrics for display on cards
|
# Include key result metrics for display on cards
|
||||||
location_cols = ["latitude", "longitude"]
|
location_cols = ["latitude", "longitude"]
|
||||||
@@ -580,7 +572,7 @@ async def get_school_details(request: Request, urn: int):
|
|||||||
# Get latest info for the school
|
# Get latest info for the school
|
||||||
latest = school_data.iloc[-1]
|
latest = school_data.iloc[-1]
|
||||||
|
|
||||||
# Fetch supplementary data (Ofsted, admissions, etc.)
|
# Fetch supplementary data (Ofsted, Parent View, admissions, etc.)
|
||||||
from .database import SessionLocal
|
from .database import SessionLocal
|
||||||
supplementary = {}
|
supplementary = {}
|
||||||
try:
|
try:
|
||||||
@@ -590,13 +582,8 @@ async def get_school_details(request: Request, urn: int):
|
|||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
# Schools with no performance rows (post-16 institutions, PRUs, new
|
return {
|
||||||
# schools) carry NaN in every LEFT-JOINed numeric column; NaN reaching
|
"school_info": {
|
||||||
# JSONResponse raises ValueError, so school_info needs the same
|
|
||||||
# conversion yearly_data gets from clean_for_json.
|
|
||||||
school_info = {
|
|
||||||
k: convert_to_native(v)
|
|
||||||
for k, v in {
|
|
||||||
"urn": urn,
|
"urn": urn,
|
||||||
"school_name": latest.get("school_name", ""),
|
"school_name": latest.get("school_name", ""),
|
||||||
"local_authority": latest.get("local_authority", ""),
|
"local_authority": latest.get("local_authority", ""),
|
||||||
@@ -604,8 +591,6 @@ async def get_school_details(request: Request, urn: int):
|
|||||||
"address": latest.get("address", ""),
|
"address": latest.get("address", ""),
|
||||||
"religious_denomination": latest.get("religious_denomination", ""),
|
"religious_denomination": latest.get("religious_denomination", ""),
|
||||||
"age_range": latest.get("age_range", ""),
|
"age_range": latest.get("age_range", ""),
|
||||||
"has_sixth_form": latest.get("has_sixth_form"),
|
|
||||||
"status": latest.get("status"),
|
|
||||||
"latitude": latest.get("latitude"),
|
"latitude": latest.get("latitude"),
|
||||||
"longitude": latest.get("longitude"),
|
"longitude": latest.get("longitude"),
|
||||||
"phase": latest.get("phase"),
|
"phase": latest.get("phase"),
|
||||||
@@ -616,14 +601,11 @@ async def get_school_details(request: Request, urn: int):
|
|||||||
"total_pupils": latest.get("gias_total_pupils"),
|
"total_pupils": latest.get("gias_total_pupils"),
|
||||||
"trust_name": latest.get("trust_name"),
|
"trust_name": latest.get("trust_name"),
|
||||||
"gender": latest.get("gender"),
|
"gender": latest.get("gender"),
|
||||||
}.items()
|
},
|
||||||
}
|
|
||||||
|
|
||||||
return {
|
|
||||||
"school_info": school_info,
|
|
||||||
"yearly_data": clean_for_json(school_data),
|
"yearly_data": clean_for_json(school_data),
|
||||||
# Supplementary data (null if not yet populated by Kestra)
|
# Supplementary data (null if not yet populated by Kestra)
|
||||||
"ofsted": supplementary.get("ofsted"),
|
"ofsted": supplementary.get("ofsted"),
|
||||||
|
"parent_view": supplementary.get("parent_view"),
|
||||||
"census": supplementary.get("census"),
|
"census": supplementary.get("census"),
|
||||||
"admissions": supplementary.get("admissions"),
|
"admissions": supplementary.get("admissions"),
|
||||||
"admissions_history": supplementary.get("admissions_history") or [],
|
"admissions_history": supplementary.get("admissions_history") or [],
|
||||||
@@ -663,34 +645,6 @@ async def compare_schools(
|
|||||||
if comparison_data.empty:
|
if comparison_data.empty:
|
||||||
raise HTTPException(status_code=404, detail="No schools found")
|
raise HTTPException(status_code=404, detail="No schools found")
|
||||||
|
|
||||||
# One session for all schools' supplementary blocks; failures degrade
|
|
||||||
# to empty blocks rather than failing a working comparison (mirrors
|
|
||||||
# the detail endpoint's defensive pattern).
|
|
||||||
from . import database
|
|
||||||
|
|
||||||
_EMPTY_SUPPLEMENTARY = {
|
|
||||||
"ofsted": None,
|
|
||||||
"census": None,
|
|
||||||
"admissions": None,
|
|
||||||
"admissions_history": [],
|
|
||||||
"deprivation": None,
|
|
||||||
}
|
|
||||||
supplementary_by_urn: dict = {}
|
|
||||||
db = None
|
|
||||||
try:
|
|
||||||
db = database.SessionLocal()
|
|
||||||
for urn in urn_list:
|
|
||||||
supp = get_supplementary_data(db, urn)
|
|
||||||
supplementary_by_urn[urn] = {
|
|
||||||
key: supp.get(key, default)
|
|
||||||
for key, default in _EMPTY_SUPPLEMENTARY.items()
|
|
||||||
}
|
|
||||||
except Exception:
|
|
||||||
supplementary_by_urn = {}
|
|
||||||
finally:
|
|
||||||
if db is not None:
|
|
||||||
db.close()
|
|
||||||
|
|
||||||
result = {}
|
result = {}
|
||||||
for urn in urn_list:
|
for urn in urn_list:
|
||||||
school_data = comparison_data[comparison_data["urn"] == urn].sort_values("year")
|
school_data = comparison_data[comparison_data["urn"] == urn].sort_values("year")
|
||||||
@@ -706,27 +660,11 @@ async def compare_schools(
|
|||||||
"phase": latest.get("phase", ""),
|
"phase": latest.get("phase", ""),
|
||||||
"attainment_8_score": float(latest["attainment_8_score"]) if pd.notna(latest.get("attainment_8_score")) else None,
|
"attainment_8_score": float(latest["attainment_8_score"]) if pd.notna(latest.get("attainment_8_score")) else None,
|
||||||
"rwm_expected_pct": float(latest["rwm_expected_pct"]) if pd.notna(latest.get("rwm_expected_pct")) else None,
|
"rwm_expected_pct": float(latest["rwm_expected_pct"]) if pd.notna(latest.get("rwm_expected_pct")) else None,
|
||||||
# GIAS facts the compare "Who goes there" section needs
|
|
||||||
# (same fields the detail endpoint exposes)
|
|
||||||
"religious_denomination": convert_to_native(latest.get("religious_denomination")),
|
|
||||||
"age_range": convert_to_native(latest.get("age_range")),
|
|
||||||
"gender": convert_to_native(latest.get("gender")),
|
|
||||||
"has_sixth_form": convert_to_native(latest.get("has_sixth_form")),
|
|
||||||
"capacity": convert_to_native(latest.get("capacity")),
|
|
||||||
"gias_total_pupils": convert_to_native(latest.get("gias_total_pupils")),
|
|
||||||
"trust_name": convert_to_native(latest.get("trust_name")),
|
|
||||||
},
|
},
|
||||||
"yearly_data": clean_for_json(school_data),
|
"yearly_data": clean_for_json(school_data),
|
||||||
**supplementary_by_urn.get(urn, dict(_EMPTY_SUPPLEMENTARY)),
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return {
|
return {"comparison": result}
|
||||||
"comparison": result,
|
|
||||||
# Official DfE anchors + computed state-school benchmarks so the
|
|
||||||
# compare UI can label provenance correctly (spec §8.6).
|
|
||||||
"national_averages": _national_averages_payload(df),
|
|
||||||
"benchmarks": compute_benchmarks(df),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/filters")
|
@app.get("/api/filters")
|
||||||
@@ -772,17 +710,22 @@ async def get_la_averages(request: Request):
|
|||||||
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
return {"year": latest_year, "secondary": {"attainment_8_by_la": la_avg}}
|
||||||
|
|
||||||
|
|
||||||
def _national_averages_payload(df: pd.DataFrame) -> dict:
|
@app.get("/api/national-averages")
|
||||||
"""National-averages payload shared by /api/national-averages and
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||||
/api/compare. Official DfE KS2 figures come from the mart table;
|
async def get_national_averages(request: Request):
|
||||||
KS4 figures are computed from our dataset (no DfE dataset yet)."""
|
"""
|
||||||
|
Compute national average for each metric from the latest data year.
|
||||||
|
Returns separate averages for primary (KS2) and secondary (KS4) schools.
|
||||||
|
Values are derived from the loaded DataFrame so they automatically
|
||||||
|
stay current when new data is loaded.
|
||||||
|
"""
|
||||||
|
df = load_school_data()
|
||||||
if df.empty:
|
if df.empty:
|
||||||
return {"primary": {}, "secondary": {}}
|
return {"primary": {}, "secondary": {}}
|
||||||
|
|
||||||
ks2_metrics = [
|
ks2_metrics = [
|
||||||
"rwm_expected_pct", "rwm_high_pct",
|
"rwm_expected_pct", "rwm_high_pct",
|
||||||
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
|
"reading_expected_pct", "writing_expected_pct", "maths_expected_pct",
|
||||||
"gps_expected_pct", "gps_high_pct", "science_expected_pct",
|
|
||||||
"reading_avg_score", "maths_avg_score", "gps_avg_score",
|
"reading_avg_score", "maths_avg_score", "gps_avg_score",
|
||||||
"reading_progress", "writing_progress", "maths_progress",
|
"reading_progress", "writing_progress", "maths_progress",
|
||||||
"overall_absence_pct", "persistent_absence_pct",
|
"overall_absence_pct", "persistent_absence_pct",
|
||||||
@@ -817,13 +760,12 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
|
|||||||
|
|
||||||
# Per-year KS2 primary averages: use official DfE figures from the mart table.
|
# Per-year KS2 primary averages: use official DfE figures from the mart table.
|
||||||
# Per-year KS4 secondary averages: computed from our dataset (no DfE dataset yet).
|
# Per-year KS4 secondary averages: computed from our dataset (no DfE dataset yet).
|
||||||
from . import database
|
from .database import SessionLocal
|
||||||
from .models import Ks2NationalAverage
|
from .models import Ks2NationalAverage
|
||||||
|
|
||||||
by_year = []
|
by_year = []
|
||||||
db = None
|
|
||||||
try:
|
try:
|
||||||
db = database.SessionLocal()
|
db = SessionLocal()
|
||||||
nat_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
|
nat_rows = db.query(Ks2NationalAverage).order_by(Ks2NationalAverage.year).all()
|
||||||
# Build a lookup of computed secondary averages per year as fallback
|
# Build a lookup of computed secondary averages per year as fallback
|
||||||
secondary_by_year = {}
|
secondary_by_year = {}
|
||||||
@@ -851,7 +793,6 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
|
|||||||
"secondary": secondary_by_year.get(yr, {}),
|
"secondary": secondary_by_year.get(yr, {}),
|
||||||
})
|
})
|
||||||
finally:
|
finally:
|
||||||
if db is not None:
|
|
||||||
db.close()
|
db.close()
|
||||||
|
|
||||||
# Update latest_primary with official DfE figure for the latest year if available
|
# Update latest_primary with official DfE figure for the latest year if available
|
||||||
@@ -868,17 +809,6 @@ def _national_averages_payload(df: pd.DataFrame) -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/national-averages")
|
|
||||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
|
||||||
async def get_national_averages(request: Request):
|
|
||||||
"""
|
|
||||||
National averages: official DfE KS2 figures per year plus computed
|
|
||||||
KS4 averages, derived from the loaded DataFrame and the
|
|
||||||
fact_ks2_national_averages mart.
|
|
||||||
"""
|
|
||||||
return _national_averages_payload(load_school_data())
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/metrics")
|
@app.get("/api/metrics")
|
||||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||||
async def get_available_metrics(request: Request):
|
async def get_available_metrics(request: Request):
|
||||||
|
|||||||
+77
-276
@@ -3,58 +3,21 @@ Data loading module — reads from marts.* tables built by dbt.
|
|||||||
Provides efficient queries with caching.
|
Provides efficient queries with caching.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
|
||||||
import re
|
|
||||||
|
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from typing import Optional, Dict, Tuple, List
|
from typing import Optional, Dict, Tuple, List
|
||||||
import requests
|
import requests
|
||||||
from sqlalchemy import text
|
from sqlalchemy import text
|
||||||
import sqlalchemy.exc
|
|
||||||
from sqlalchemy.orm import Session
|
from sqlalchemy.orm import Session
|
||||||
|
|
||||||
from .config import settings
|
from .config import settings
|
||||||
from .database import SessionLocal, engine
|
from .database import SessionLocal, engine
|
||||||
from .models import (
|
from .models import (
|
||||||
DimSchool, DimLocation, KS2Performance,
|
DimSchool, DimLocation, KS2Performance,
|
||||||
FactOfstedInspection, FactAdmissions,
|
FactOfstedInspection, FactParentView, FactAdmissions,
|
||||||
FactDeprivation, FactFinance, FactPupilCharacteristics,
|
FactDeprivation, FactFinance, FactPupilCharacteristics,
|
||||||
)
|
)
|
||||||
from .ofsted_codes import ofsted_page_url, report_card_labels
|
|
||||||
from .schemas import SCHOOL_TYPE_MAP
|
from .schemas import SCHOOL_TYPE_MAP
|
||||||
from .gias_codes import (
|
|
||||||
ADMISSIONS_POLICY,
|
|
||||||
ESTABLISHMENT_STATUS,
|
|
||||||
PHASE_OF_EDUCATION,
|
|
||||||
RELIGIOUS_CHARACTER,
|
|
||||||
SCHOOL_TYPE,
|
|
||||||
translate,
|
|
||||||
)
|
|
||||||
|
|
||||||
# mart code column -> (API name column, dictionary)
|
|
||||||
_GIAS_CODE_COLUMNS = {
|
|
||||||
"phase_code": ("phase", PHASE_OF_EDUCATION),
|
|
||||||
"school_type_code": ("school_type", SCHOOL_TYPE),
|
|
||||||
"status_code": ("status", ESTABLISHMENT_STATUS),
|
|
||||||
"religious_character_code": ("religious_denomination", RELIGIOUS_CHARACTER),
|
|
||||||
"admissions_policy_code": ("admissions_policy", ADMISSIONS_POLICY),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def translate_gias_code_columns(df: pd.DataFrame) -> pd.DataFrame:
|
|
||||||
"""Map GIAS code columns to today's name columns (API contract).
|
|
||||||
|
|
||||||
Runs immediately after pd.read_sql so every downstream consumer —
|
|
||||||
filters, PHASE_GROUPS, payloads, /api/filters — keeps seeing names.
|
|
||||||
DataFrames without the code columns (old schema, test fixtures) pass
|
|
||||||
through unchanged.
|
|
||||||
"""
|
|
||||||
for code_col, (name_col, mapping) in _GIAS_CODE_COLUMNS.items():
|
|
||||||
if code_col in df.columns:
|
|
||||||
df[name_col] = df[code_col].map(lambda c: translate(c, mapping))
|
|
||||||
return df
|
|
||||||
|
|
||||||
|
|
||||||
_postcode_cache: Dict[str, Tuple[float, float]] = {}
|
_postcode_cache: Dict[str, Tuple[float, float]] = {}
|
||||||
_typesense_client = None
|
_typesense_client = None
|
||||||
@@ -155,16 +118,14 @@ _MAIN_QUERY = text("""
|
|||||||
SELECT
|
SELECT
|
||||||
s.urn,
|
s.urn,
|
||||||
s.school_name,
|
s.school_name,
|
||||||
s.phase_code,
|
s.phase,
|
||||||
s.school_type_code,
|
s.school_type,
|
||||||
s.academy_trust_name AS trust_name,
|
s.academy_trust_name AS trust_name,
|
||||||
s.academy_trust_uid AS trust_uid,
|
s.academy_trust_uid AS trust_uid,
|
||||||
s.religious_character_code,
|
s.religious_character AS religious_denomination,
|
||||||
s.gender,
|
s.gender,
|
||||||
s.age_range,
|
s.age_range,
|
||||||
s.has_sixth_form,
|
s.admissions_policy,
|
||||||
s.status_code,
|
|
||||||
s.admissions_policy_code,
|
|
||||||
s.capacity,
|
s.capacity,
|
||||||
s.total_pupils AS gias_total_pupils,
|
s.total_pupils AS gias_total_pupils,
|
||||||
s.headteacher_name,
|
s.headteacher_name,
|
||||||
@@ -191,20 +152,13 @@ _MAIN_QUERY = text("""
|
|||||||
p.reading_high_pct,
|
p.reading_high_pct,
|
||||||
p.reading_avg_score,
|
p.reading_avg_score,
|
||||||
p.reading_progress,
|
p.reading_progress,
|
||||||
p.reading_progress_lower_ci,
|
|
||||||
p.reading_progress_upper_ci,
|
|
||||||
p.writing_expected_pct,
|
p.writing_expected_pct,
|
||||||
p.writing_high_pct,
|
p.writing_high_pct,
|
||||||
p.writing_progress,
|
p.writing_progress,
|
||||||
p.writing_progress_lower_ci,
|
|
||||||
p.writing_progress_upper_ci,
|
|
||||||
p.writing_working_towards_pct,
|
|
||||||
p.maths_expected_pct,
|
p.maths_expected_pct,
|
||||||
p.maths_high_pct,
|
p.maths_high_pct,
|
||||||
p.maths_avg_score,
|
p.maths_avg_score,
|
||||||
p.maths_progress,
|
p.maths_progress,
|
||||||
p.maths_progress_lower_ci,
|
|
||||||
p.maths_progress_upper_ci,
|
|
||||||
p.gps_expected_pct,
|
p.gps_expected_pct,
|
||||||
p.gps_high_pct,
|
p.gps_high_pct,
|
||||||
p.gps_avg_score,
|
p.gps_avg_score,
|
||||||
@@ -233,9 +187,6 @@ _MAIN_QUERY = text("""
|
|||||||
p.progress_8_maths,
|
p.progress_8_maths,
|
||||||
p.progress_8_ebacc,
|
p.progress_8_ebacc,
|
||||||
p.progress_8_open,
|
p.progress_8_open,
|
||||||
p.progress_8_banding,
|
|
||||||
p.attainment_8_disadvantage_gap,
|
|
||||||
p.progress_8_disadvantage_gap,
|
|
||||||
p.english_maths_strong_pass_pct,
|
p.english_maths_strong_pass_pct,
|
||||||
p.english_maths_standard_pass_pct,
|
p.english_maths_standard_pass_pct,
|
||||||
p.ebacc_entry_pct,
|
p.ebacc_entry_pct,
|
||||||
@@ -263,92 +214,11 @@ _MAIN_QUERY = text("""
|
|||||||
ORDER BY s.school_name, p.year
|
ORDER BY s.school_name, p.year
|
||||||
""")
|
""")
|
||||||
|
|
||||||
# Fallback used when marts.dim_school predates the has_sixth_form column
|
|
||||||
# (i.e. the nightly dbt pipeline hasn't rebuilt the mart yet on this DB).
|
|
||||||
# Keeps the column present as NULL so downstream code — including the
|
|
||||||
# app.py fallback branch — behaves as designed instead of KeyError-ing.
|
|
||||||
_MAIN_QUERY_NO_SIXTH_FORM = text(
|
|
||||||
str(_MAIN_QUERY).replace("s.has_sixth_form,", "NULL AS has_sixth_form,")
|
|
||||||
)
|
|
||||||
assert "NULL AS has_sixth_form" in str(_MAIN_QUERY_NO_SIXTH_FORM), (
|
|
||||||
"expected replacement of 's.has_sixth_form,' to have taken effect"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Fallback used when marts.dim_school predates the GIAS code-dictionary
|
|
||||||
# migration (i.e. the nightly dbt pipeline hasn't rebuilt the mart yet on
|
|
||||||
# this DB, so it still has the old name columns instead of *_code columns).
|
|
||||||
_MAIN_QUERY_LEGACY_NAMES = str(_MAIN_QUERY)
|
|
||||||
_LEGACY_NAME_REPLACEMENTS = [
|
|
||||||
("s.phase_code,", "s.phase,"),
|
|
||||||
("s.school_type_code,", "s.school_type,"),
|
|
||||||
(
|
|
||||||
"s.religious_character_code,",
|
|
||||||
"s.religious_character AS religious_denomination,",
|
|
||||||
),
|
|
||||||
("s.status_code,", "s.status,"),
|
|
||||||
("s.admissions_policy_code,", "s.admissions_policy,"),
|
|
||||||
]
|
|
||||||
for _old, _new in _LEGACY_NAME_REPLACEMENTS:
|
|
||||||
assert _old in _MAIN_QUERY_LEGACY_NAMES, (
|
|
||||||
f"expected {_old!r} to be present in _MAIN_QUERY before replacement"
|
|
||||||
)
|
|
||||||
_MAIN_QUERY_LEGACY_NAMES = _MAIN_QUERY_LEGACY_NAMES.replace(_old, _new)
|
|
||||||
_MAIN_QUERY_LEGACY_NAMES = text(_MAIN_QUERY_LEGACY_NAMES)
|
|
||||||
|
|
||||||
_GIAS_CODE_COLUMN_NAMES = (
|
|
||||||
"phase_code",
|
|
||||||
"school_type_code",
|
|
||||||
"religious_character_code",
|
|
||||||
"status_code",
|
|
||||||
"admissions_policy_code",
|
|
||||||
)
|
|
||||||
|
|
||||||
_MISSING_COLUMN_RE = re.compile(r'column "?(?:s\.)?(\w+)"? does not exist')
|
|
||||||
|
|
||||||
|
|
||||||
def _missing_column_name(exc: Exception) -> Optional[str]:
|
|
||||||
"""Name of the missing column from a psycopg2 UndefinedColumn error.
|
|
||||||
|
|
||||||
Inspects exc.orig (the DBAPI error), whose message names only the
|
|
||||||
offending column — str(exc) also embeds the full SQL statement, which
|
|
||||||
contains every column name and therefore must not be matched against.
|
|
||||||
"""
|
|
||||||
orig = getattr(exc, "orig", None)
|
|
||||||
match = _MISSING_COLUMN_RE.search(str(orig) if orig is not None else str(exc))
|
|
||||||
return match.group(1) if match else None
|
|
||||||
|
|
||||||
|
|
||||||
def load_school_data_as_dataframe() -> pd.DataFrame:
|
def load_school_data_as_dataframe() -> pd.DataFrame:
|
||||||
"""Load all school + KS2 data as a pandas DataFrame."""
|
"""Load all school + KS2 data as a pandas DataFrame."""
|
||||||
try:
|
try:
|
||||||
df = pd.read_sql(_MAIN_QUERY, engine)
|
df = pd.read_sql(_MAIN_QUERY, engine)
|
||||||
except sqlalchemy.exc.ProgrammingError as exc:
|
|
||||||
missing = _missing_column_name(exc)
|
|
||||||
if missing in _GIAS_CODE_COLUMN_NAMES:
|
|
||||||
logging.getLogger(__name__).warning(
|
|
||||||
"marts predate the GIAS code migration — falling back to "
|
|
||||||
"legacy name-column query: %s",
|
|
||||||
exc,
|
|
||||||
)
|
|
||||||
try:
|
|
||||||
df = pd.read_sql(_MAIN_QUERY_LEGACY_NAMES, engine)
|
|
||||||
except Exception as exc2:
|
|
||||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
|
||||||
return pd.DataFrame()
|
|
||||||
elif missing == "has_sixth_form":
|
|
||||||
logging.getLogger(__name__).warning(
|
|
||||||
"marts.dim_school is missing has_sixth_form (pipeline hasn't "
|
|
||||||
"rebuilt the mart yet on this DB) — retrying without it: %s",
|
|
||||||
exc,
|
|
||||||
)
|
|
||||||
try:
|
|
||||||
df = pd.read_sql(_MAIN_QUERY_NO_SIXTH_FORM, engine)
|
|
||||||
except Exception as exc2:
|
|
||||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
|
||||||
return pd.DataFrame()
|
|
||||||
else:
|
|
||||||
print(f"Warning: Could not load school data from marts: {exc}")
|
|
||||||
return pd.DataFrame()
|
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
print(f"Warning: Could not load school data from marts: {exc}")
|
print(f"Warning: Could not load school data from marts: {exc}")
|
||||||
return pd.DataFrame()
|
return pd.DataFrame()
|
||||||
@@ -356,8 +226,6 @@ def load_school_data_as_dataframe() -> pd.DataFrame:
|
|||||||
if df.empty:
|
if df.empty:
|
||||||
return df
|
return df
|
||||||
|
|
||||||
df = translate_gias_code_columns(df)
|
|
||||||
|
|
||||||
# Build address string
|
# Build address string
|
||||||
df["address"] = df.apply(
|
df["address"] = df.apply(
|
||||||
lambda r: ", ".join(
|
lambda r: ", ".join(
|
||||||
@@ -525,143 +393,6 @@ def get_data_info(db: Session = None) -> dict:
|
|||||||
# SUPPLEMENTARY DATA — per-school detail page
|
# SUPPLEMENTARY DATA — per-school detail page
|
||||||
# =============================================================================
|
# =============================================================================
|
||||||
|
|
||||||
def compute_benchmarks(df: pd.DataFrame) -> dict:
|
|
||||||
"""State-school benchmarks computed from our dataset (spec §5/§8.6).
|
|
||||||
|
|
||||||
NOT official DfE figures — consumers must label them
|
|
||||||
"state-school average (computed from our dataset)". The disadvantaged
|
|
||||||
attainment average is weighted by cohort size (eligible_pupils) so
|
|
||||||
small schools don't dominate; context measures are medians.
|
|
||||||
"""
|
|
||||||
if df.empty or "year" not in df.columns:
|
|
||||||
return {}
|
|
||||||
latest_year = df["year"].max()
|
|
||||||
if pd.isna(latest_year):
|
|
||||||
return {}
|
|
||||||
d = df[df["year"] == latest_year]
|
|
||||||
if d.empty:
|
|
||||||
return {}
|
|
||||||
is_secondary = (
|
|
||||||
d["attainment_8_score"].notna()
|
|
||||||
if "attainment_8_score" in d.columns
|
|
||||||
else pd.Series(False, index=d.index)
|
|
||||||
)
|
|
||||||
prim, sec = d[~is_secondary], d[is_secondary]
|
|
||||||
|
|
||||||
def _median(sub, col):
|
|
||||||
if col not in sub.columns:
|
|
||||||
return None
|
|
||||||
v = sub[col].median()
|
|
||||||
return round(float(v), 1) if pd.notna(v) else None
|
|
||||||
|
|
||||||
def _weighted_disadvantaged(sub):
|
|
||||||
needed = {"rwm_expected_disadvantaged_pct", "eligible_pupils"}
|
|
||||||
if not needed <= set(sub.columns):
|
|
||||||
return None
|
|
||||||
s = sub.dropna(subset=list(needed))
|
|
||||||
if s.empty or s["eligible_pupils"].sum() == 0:
|
|
||||||
return None
|
|
||||||
w = (
|
|
||||||
(s["rwm_expected_disadvantaged_pct"] * s["eligible_pupils"]).sum()
|
|
||||||
/ s["eligible_pupils"].sum()
|
|
||||||
)
|
|
||||||
return round(float(w), 1)
|
|
||||||
|
|
||||||
def _block(sub, with_disadvantaged):
|
|
||||||
median_pupils = None
|
|
||||||
if "total_pupils" in sub.columns:
|
|
||||||
mp = sub["total_pupils"].median()
|
|
||||||
if pd.notna(mp):
|
|
||||||
median_pupils = int(mp)
|
|
||||||
block = {
|
|
||||||
"eal_pct": _median(sub, "eal_pct"),
|
|
||||||
"sen_support_pct": _median(sub, "sen_support_pct"),
|
|
||||||
"disadvantaged_pct": _median(sub, "disadvantaged_pct"),
|
|
||||||
"median_pupils": median_pupils,
|
|
||||||
}
|
|
||||||
if with_disadvantaged:
|
|
||||||
block["disadvantaged_rwm_expected_pct"] = _weighted_disadvantaged(sub)
|
|
||||||
return block
|
|
||||||
|
|
||||||
return {
|
|
||||||
"source": "state-school average (computed from our dataset)",
|
|
||||||
"year": int(latest_year),
|
|
||||||
"primary": _block(prim, with_disadvantaged=True),
|
|
||||||
"secondary": _block(sec, with_disadvantaged=False),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _ofsted_block(o, urn: int) -> dict:
|
|
||||||
"""Serialize the latest Ofsted inspection row for API responses.
|
|
||||||
|
|
||||||
`grade_source` records where the effective overall grade came from:
|
|
||||||
a graded (Section 5) inspection, or carried forward from an ungraded
|
|
||||||
(Section 8) outcome — materially different claims a UI must be able
|
|
||||||
to distinguish. `report_card` holds coded+labelled renewed-framework
|
|
||||||
(Nov 2025) area judgements; safeguarding is a separate boolean and
|
|
||||||
never appears among the graded areas.
|
|
||||||
"""
|
|
||||||
if o.overall_effectiveness is not None:
|
|
||||||
grade_source = "graded"
|
|
||||||
overall = o.overall_effectiveness
|
|
||||||
elif o.ungraded_grade is not None:
|
|
||||||
# Fall back to the grade parsed from an ungraded (Section 8) outcome
|
|
||||||
# (e.g. "School remains Good") so the detail page matches the list badge.
|
|
||||||
grade_source = "ungraded_carried_forward"
|
|
||||||
overall = o.ungraded_grade
|
|
||||||
else:
|
|
||||||
grade_source = None
|
|
||||||
overall = None
|
|
||||||
|
|
||||||
block = {
|
|
||||||
"framework": o.framework,
|
|
||||||
"inspection_date": o.inspection_date.isoformat() if o.inspection_date else None,
|
|
||||||
"inspection_type": o.inspection_type,
|
|
||||||
"overall_effectiveness": overall,
|
|
||||||
"grade_source": grade_source,
|
|
||||||
"quality_of_education": o.quality_of_education,
|
|
||||||
"behaviour_attitudes": o.behaviour_attitudes,
|
|
||||||
"personal_development": o.personal_development,
|
|
||||||
"leadership_management": o.leadership_management,
|
|
||||||
"early_years_provision": o.early_years_provision,
|
|
||||||
"sixth_form_provision": o.sixth_form_provision,
|
|
||||||
"previous_overall": None, # Not available in new schema
|
|
||||||
"rc_safeguarding_met": o.rc_safeguarding_met,
|
|
||||||
"rc_inclusion": o.rc_inclusion,
|
|
||||||
"rc_curriculum_teaching": o.rc_curriculum_teaching,
|
|
||||||
"rc_achievement": o.rc_achievement,
|
|
||||||
"rc_attendance_behaviour": o.rc_attendance_behaviour,
|
|
||||||
"rc_personal_development": o.rc_personal_development,
|
|
||||||
"rc_leadership_governance": o.rc_leadership_governance,
|
|
||||||
"rc_early_years": o.rc_early_years,
|
|
||||||
"rc_sixth_form": o.rc_sixth_form,
|
|
||||||
"report_url": o.report_url,
|
|
||||||
"ofsted_page_url": ofsted_page_url(urn),
|
|
||||||
}
|
|
||||||
block["report_card"] = report_card_labels(block)
|
|
||||||
return block
|
|
||||||
|
|
||||||
|
|
||||||
def _admissions_row_dict(a) -> dict:
|
|
||||||
"""Serialize one fact_admissions row for API responses."""
|
|
||||||
return {
|
|
||||||
"year": a.year,
|
|
||||||
"school_phase": a.school_phase,
|
|
||||||
"places_offered": a.places_offered,
|
|
||||||
"total_applications": a.total_applications,
|
|
||||||
"first_preference_applications": a.first_preference_applications,
|
|
||||||
"first_preference_offers": a.first_preference_offers,
|
|
||||||
"first_preference_offer_pct": a.first_preference_offer_pct,
|
|
||||||
"oversubscription_ratio": a.oversubscription_ratio,
|
|
||||||
"oversubscribed": a.oversubscribed,
|
|
||||||
"total_offers": a.total_offers,
|
|
||||||
"second_preference_offers": a.second_preference_offers,
|
|
||||||
"third_preference_offers": a.third_preference_offers,
|
|
||||||
"cross_la_applications": a.cross_la_applications,
|
|
||||||
"cross_la_offers": a.cross_la_offers,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def get_supplementary_data(db: Session, urn: int) -> dict:
|
def get_supplementary_data(db: Session, urn: int) -> dict:
|
||||||
"""Fetch all supplementary data for a single school URN."""
|
"""Fetch all supplementary data for a single school URN."""
|
||||||
result = {}
|
result = {}
|
||||||
@@ -680,7 +411,64 @@ def get_supplementary_data(db: Session, urn: int) -> dict:
|
|||||||
|
|
||||||
# Latest Ofsted inspection
|
# Latest Ofsted inspection
|
||||||
o = safe_query(FactOfstedInspection, "urn", "inspection_date")
|
o = safe_query(FactOfstedInspection, "urn", "inspection_date")
|
||||||
result["ofsted"] = _ofsted_block(o, urn) if o else None
|
result["ofsted"] = (
|
||||||
|
{
|
||||||
|
"framework": o.framework,
|
||||||
|
"inspection_date": o.inspection_date.isoformat() if o.inspection_date else None,
|
||||||
|
"inspection_type": o.inspection_type,
|
||||||
|
# Fall back to the grade parsed from an ungraded (Section 8) outcome
|
||||||
|
# (e.g. "School remains Good") when there's no graded grade, so the
|
||||||
|
# detail page matches the list badge.
|
||||||
|
"overall_effectiveness": (
|
||||||
|
o.overall_effectiveness
|
||||||
|
if o.overall_effectiveness is not None
|
||||||
|
else o.ungraded_grade
|
||||||
|
),
|
||||||
|
"quality_of_education": o.quality_of_education,
|
||||||
|
"behaviour_attitudes": o.behaviour_attitudes,
|
||||||
|
"personal_development": o.personal_development,
|
||||||
|
"leadership_management": o.leadership_management,
|
||||||
|
"early_years_provision": o.early_years_provision,
|
||||||
|
"sixth_form_provision": o.sixth_form_provision,
|
||||||
|
"previous_overall": None, # Not available in new schema
|
||||||
|
"rc_safeguarding_met": o.rc_safeguarding_met,
|
||||||
|
"rc_inclusion": o.rc_inclusion,
|
||||||
|
"rc_curriculum_teaching": o.rc_curriculum_teaching,
|
||||||
|
"rc_achievement": o.rc_achievement,
|
||||||
|
"rc_attendance_behaviour": o.rc_attendance_behaviour,
|
||||||
|
"rc_personal_development": o.rc_personal_development,
|
||||||
|
"rc_leadership_governance": o.rc_leadership_governance,
|
||||||
|
"rc_early_years": o.rc_early_years,
|
||||||
|
"rc_sixth_form": o.rc_sixth_form,
|
||||||
|
"report_url": o.report_url,
|
||||||
|
}
|
||||||
|
if o
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
|
||||||
|
# Parent View
|
||||||
|
pv = safe_query(FactParentView, "urn")
|
||||||
|
result["parent_view"] = (
|
||||||
|
{
|
||||||
|
"survey_date": pv.survey_date.isoformat() if pv.survey_date else None,
|
||||||
|
"total_responses": pv.total_responses,
|
||||||
|
"q_happy_pct": pv.q_happy_pct,
|
||||||
|
"q_safe_pct": pv.q_safe_pct,
|
||||||
|
"q_behaviour_pct": pv.q_behaviour_pct,
|
||||||
|
"q_bullying_pct": pv.q_bullying_pct,
|
||||||
|
"q_communication_pct": pv.q_communication_pct,
|
||||||
|
"q_progress_pct": pv.q_progress_pct,
|
||||||
|
"q_teaching_pct": pv.q_teaching_pct,
|
||||||
|
"q_information_pct": pv.q_information_pct,
|
||||||
|
"q_curriculum_pct": pv.q_curriculum_pct,
|
||||||
|
"q_future_pct": pv.q_future_pct,
|
||||||
|
"q_leadership_pct": pv.q_leadership_pct,
|
||||||
|
"q_wellbeing_pct": pv.q_wellbeing_pct,
|
||||||
|
"q_recommend_pct": pv.q_recommend_pct,
|
||||||
|
}
|
||||||
|
if pv
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
|
||||||
# Census (latest year of fact_pupil_characteristics)
|
# Census (latest year of fact_pupil_characteristics)
|
||||||
pc = safe_query(FactPupilCharacteristics, "urn", "year")
|
pc = safe_query(FactPupilCharacteristics, "urn", "year")
|
||||||
@@ -698,6 +486,19 @@ def get_supplementary_data(db: Session, urn: int) -> dict:
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Admissions — all years, oldest first (for the multi-year trend view).
|
# Admissions — all years, oldest first (for the multi-year trend view).
|
||||||
|
def _admissions_row(a):
|
||||||
|
return {
|
||||||
|
"year": a.year,
|
||||||
|
"school_phase": a.school_phase,
|
||||||
|
"places_offered": a.places_offered,
|
||||||
|
"total_applications": a.total_applications,
|
||||||
|
"first_preference_applications": a.first_preference_applications,
|
||||||
|
"first_preference_offers": a.first_preference_offers,
|
||||||
|
"first_preference_offer_pct": a.first_preference_offer_pct,
|
||||||
|
"oversubscription_ratio": a.oversubscription_ratio,
|
||||||
|
"oversubscribed": a.oversubscribed,
|
||||||
|
}
|
||||||
|
|
||||||
try:
|
try:
|
||||||
admissions_rows = (
|
admissions_rows = (
|
||||||
db.query(FactAdmissions)
|
db.query(FactAdmissions)
|
||||||
@@ -711,7 +512,7 @@ def get_supplementary_data(db: Session, urn: int) -> dict:
|
|||||||
db.rollback()
|
db.rollback()
|
||||||
admissions_rows = []
|
admissions_rows = []
|
||||||
|
|
||||||
history = [_admissions_row_dict(a) for a in admissions_rows]
|
history = [_admissions_row(a) for a in admissions_rows]
|
||||||
result["admissions_history"] = history
|
result["admissions_history"] = history
|
||||||
# Keep the single latest-year object for backwards-compatible consumers
|
# Keep the single latest-year object for backwards-compatible consumers
|
||||||
# (hero chips, etc.).
|
# (hero chips, etc.).
|
||||||
|
|||||||
@@ -1,155 +0,0 @@
|
|||||||
"""GIAS code -> name dictionaries.
|
|
||||||
|
|
||||||
GENERATED by pipeline/scripts/generate_gias_codes.py from the GIAS bulk CSV
|
|
||||||
— do not edit by hand; rerun the script when the dbt drift test warns.
|
|
||||||
The canonical file is backend/gias_codes.py; pipeline/scripts/gias_codes.py
|
|
||||||
must be byte-identical (enforced by backend/tests/test_gias_codes.py).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import math
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
SCHOOL_TYPE: dict[int, str] = {
|
|
||||||
1: "Community school",
|
|
||||||
2: "Voluntary aided school",
|
|
||||||
3: "Voluntary controlled school",
|
|
||||||
5: "Foundation school",
|
|
||||||
6: "City technology college",
|
|
||||||
7: "Community special school",
|
|
||||||
8: "Non-maintained special school",
|
|
||||||
10: "Other independent special school",
|
|
||||||
11: "Other independent school",
|
|
||||||
12: "Foundation special school",
|
|
||||||
14: "Pupil referral unit",
|
|
||||||
15: "Local authority nursery school",
|
|
||||||
18: "Further education",
|
|
||||||
24: "Secure units",
|
|
||||||
25: "Offshore schools",
|
|
||||||
26: "Service children's education",
|
|
||||||
27: "Miscellaneous",
|
|
||||||
28: "Academy sponsor led",
|
|
||||||
29: "Higher education institutions",
|
|
||||||
30: "Welsh establishment",
|
|
||||||
31: "Sixth form centres",
|
|
||||||
32: "Special post 16 institution",
|
|
||||||
33: "Academy special sponsor led",
|
|
||||||
34: "Academy converter",
|
|
||||||
35: "Free schools",
|
|
||||||
36: "Free schools special",
|
|
||||||
37: "British schools overseas",
|
|
||||||
38: "Free schools alternative provision",
|
|
||||||
39: "Free schools 16 to 19",
|
|
||||||
40: "University technical college",
|
|
||||||
41: "Studio schools",
|
|
||||||
42: "Academy alternative provision converter",
|
|
||||||
43: "Academy alternative provision sponsor led",
|
|
||||||
44: "Academy special converter",
|
|
||||||
45: "Academy 16-19 converter",
|
|
||||||
46: "Academy 16 to 19 sponsor led",
|
|
||||||
49: "Online provider",
|
|
||||||
56: "Institution funded by other government department",
|
|
||||||
57: "Academy secure 16 to 19",
|
|
||||||
}
|
|
||||||
|
|
||||||
ESTABLISHMENT_STATUS: dict[int, str] = {
|
|
||||||
1: "Open",
|
|
||||||
2: "Closed",
|
|
||||||
3: "Open, but proposed to close",
|
|
||||||
4: "Proposed to open",
|
|
||||||
}
|
|
||||||
|
|
||||||
PHASE_OF_EDUCATION: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
1: "Nursery",
|
|
||||||
2: "Primary",
|
|
||||||
3: "Middle deemed primary",
|
|
||||||
4: "Secondary",
|
|
||||||
5: "Middle deemed secondary",
|
|
||||||
6: "16 plus",
|
|
||||||
7: "All-through",
|
|
||||||
}
|
|
||||||
|
|
||||||
OFFICIAL_SIXTH_FORM: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
1: "Has a sixth form",
|
|
||||||
2: "Does not have a sixth form",
|
|
||||||
9: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
RELIGIOUS_CHARACTER: dict[int, str] = {
|
|
||||||
0: "Does not apply",
|
|
||||||
2: "Church of England",
|
|
||||||
3: "Roman Catholic",
|
|
||||||
4: "Methodist",
|
|
||||||
5: "Jewish",
|
|
||||||
6: "None",
|
|
||||||
7: "Muslim",
|
|
||||||
8: "Seventh Day Adventist",
|
|
||||||
9: "Church of England/Methodist",
|
|
||||||
10: "Methodist/Church of England",
|
|
||||||
11: "Church of England/Roman Catholic",
|
|
||||||
12: "Church of England/United Reformed Church",
|
|
||||||
13: "Roman Catholic/Church of England",
|
|
||||||
14: "Quaker",
|
|
||||||
15: "Christian",
|
|
||||||
16: "United Reformed Church",
|
|
||||||
17: "Congregational Church",
|
|
||||||
18: "Free Church",
|
|
||||||
19: "Church of England/Free Church",
|
|
||||||
20: "Church of England/Christian",
|
|
||||||
21: "Sikh",
|
|
||||||
22: "Greek Orthodox",
|
|
||||||
24: "Buddhist",
|
|
||||||
25: "Hindu",
|
|
||||||
26: "Moravian",
|
|
||||||
28: "Inter- / non- denominational",
|
|
||||||
29: "Multi-faith",
|
|
||||||
30: "Church of England/Methodist/United Reform Church/Baptist",
|
|
||||||
31: "Anglican",
|
|
||||||
32: "Anglican/Christian",
|
|
||||||
33: "Anglican/Evangelical",
|
|
||||||
34: "Anglican/Church of England",
|
|
||||||
35: "Catholic",
|
|
||||||
36: "Charadi Jewish",
|
|
||||||
37: "Christian/Evangelical",
|
|
||||||
38: "Christian Science",
|
|
||||||
39: "Christian/Methodist",
|
|
||||||
40: "Christian/non-denominational",
|
|
||||||
41: "Church of England/Evangelical",
|
|
||||||
42: "Islam",
|
|
||||||
43: "Orthodox Jewish",
|
|
||||||
44: "Plymouth Brethren Christian Church",
|
|
||||||
45: "Protestant",
|
|
||||||
46: "Protestant/Evangelical",
|
|
||||||
47: "Reformed Baptist",
|
|
||||||
48: "Roman Catholic/Anglican",
|
|
||||||
49: "Sunni Deobandi",
|
|
||||||
99: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
ADMISSIONS_POLICY: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
2: "Selective",
|
|
||||||
4: "Non-selective",
|
|
||||||
9: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def translate(code, mapping: dict[int, str]) -> str | None:
|
|
||||||
"""Translate a GIAS code to its display name.
|
|
||||||
|
|
||||||
None/NaN -> None (column absent or suppressed). Unknown codes degrade to
|
|
||||||
"Unknown (<code>)" with a warning so a new DfE value never blanks the UI.
|
|
||||||
"""
|
|
||||||
if code is None or (isinstance(code, float) and math.isnan(code)):
|
|
||||||
return None
|
|
||||||
code = int(code)
|
|
||||||
if code not in mapping:
|
|
||||||
logger.warning("Unknown GIAS code %s (not in dictionary)", code)
|
|
||||||
return f"Unknown ({code})"
|
|
||||||
return mapping[code]
|
|
||||||
@@ -433,25 +433,6 @@ def _apply_schema_alterations():
|
|||||||
conn.commit()
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
def _apply_schema_drops():
|
|
||||||
"""
|
|
||||||
Drop tables retired from the schema. Idempotent (DROP … IF EXISTS), so it's
|
|
||||||
safe to run on every migration. Add entries here when a model is removed.
|
|
||||||
"""
|
|
||||||
drops = [
|
|
||||||
# v6: Ofsted Parent View feature removed
|
|
||||||
"DROP TABLE IF EXISTS marts.fact_parent_view CASCADE",
|
|
||||||
]
|
|
||||||
from sqlalchemy import text as sa_text
|
|
||||||
with engine.connect() as conn:
|
|
||||||
for stmt in drops:
|
|
||||||
try:
|
|
||||||
conn.execute(sa_text(stmt))
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: drop skipped ({e})")
|
|
||||||
conn.commit()
|
|
||||||
|
|
||||||
|
|
||||||
def run_full_migration(geocode: bool = False) -> bool:
|
def run_full_migration(geocode: bool = False) -> bool:
|
||||||
"""
|
"""
|
||||||
Run a complete migration: drop all tables and reimport from CSV.
|
Run a complete migration: drop all tables and reimport from CSV.
|
||||||
@@ -498,9 +479,6 @@ def run_full_migration(geocode: bool = False) -> bool:
|
|||||||
print("Applying column additions to supplementary tables...")
|
print("Applying column additions to supplementary tables...")
|
||||||
_apply_schema_alterations()
|
_apply_schema_alterations()
|
||||||
|
|
||||||
print("Dropping retired tables...")
|
|
||||||
_apply_schema_drops()
|
|
||||||
|
|
||||||
print("\nLoading CSV data...")
|
print("\nLoading CSV data...")
|
||||||
df = load_csv_data(settings.data_dir)
|
df = load_csv_data(settings.data_dir)
|
||||||
|
|
||||||
|
|||||||
+28
-20
@@ -17,22 +17,21 @@ class DimSchool(Base):
|
|||||||
|
|
||||||
urn = Column(Integer, primary_key=True)
|
urn = Column(Integer, primary_key=True)
|
||||||
school_name = Column(String(255), nullable=False)
|
school_name = Column(String(255), nullable=False)
|
||||||
phase_code = Column(Integer)
|
phase = Column(String(100))
|
||||||
school_type_code = Column(Integer)
|
school_type = Column(String(100))
|
||||||
academy_trust_name = Column(String(255))
|
academy_trust_name = Column(String(255))
|
||||||
academy_trust_uid = Column(String(20))
|
academy_trust_uid = Column(String(20))
|
||||||
religious_character_code = Column(Integer)
|
religious_character = Column(String(100))
|
||||||
gender = Column(String(20))
|
gender = Column(String(20))
|
||||||
age_range = Column(String(20))
|
age_range = Column(String(20))
|
||||||
has_sixth_form = Column(Boolean)
|
|
||||||
capacity = Column(Integer)
|
capacity = Column(Integer)
|
||||||
total_pupils = Column(Integer)
|
total_pupils = Column(Integer)
|
||||||
headteacher_name = Column(String(200))
|
headteacher_name = Column(String(200))
|
||||||
website = Column(String(255))
|
website = Column(String(255))
|
||||||
telephone = Column(String(30))
|
telephone = Column(String(30))
|
||||||
status_code = Column(Integer)
|
status = Column(String(50))
|
||||||
nursery_provision = Column(Boolean)
|
nursery_provision = Column(Boolean)
|
||||||
admissions_policy_code = Column(Integer)
|
admissions_policy = Column(String(50))
|
||||||
# Denormalised Ofsted summary (updated by monthly pipeline)
|
# Denormalised Ofsted summary (updated by monthly pipeline)
|
||||||
ofsted_grade = Column(Integer)
|
ofsted_grade = Column(Integer)
|
||||||
ofsted_date = Column(Date)
|
ofsted_date = Column(Date)
|
||||||
@@ -88,15 +87,6 @@ class KS2Performance(Base):
|
|||||||
maths_high_pct = Column(Float)
|
maths_high_pct = Column(Float)
|
||||||
maths_avg_score = Column(Float)
|
maths_avg_score = Column(Float)
|
||||||
maths_progress = Column(Float)
|
maths_progress = Column(Float)
|
||||||
# Progress confidence intervals + writing working-towards (published
|
|
||||||
# for years with progress measures, i.e. up to 2022/23)
|
|
||||||
reading_progress_lower_ci = Column(Float)
|
|
||||||
reading_progress_upper_ci = Column(Float)
|
|
||||||
writing_progress_lower_ci = Column(Float)
|
|
||||||
writing_progress_upper_ci = Column(Float)
|
|
||||||
writing_working_towards_pct = Column(Float)
|
|
||||||
maths_progress_lower_ci = Column(Float)
|
|
||||||
maths_progress_upper_ci = Column(Float)
|
|
||||||
gps_expected_pct = Column(Float)
|
gps_expected_pct = Column(Float)
|
||||||
gps_high_pct = Column(Float)
|
gps_high_pct = Column(Float)
|
||||||
gps_avg_score = Column(Float)
|
gps_avg_score = Column(Float)
|
||||||
@@ -159,6 +149,29 @@ class FactOfstedInspection(Base):
|
|||||||
report_url = Column(Text)
|
report_url = Column(Text)
|
||||||
|
|
||||||
|
|
||||||
|
class FactParentView(Base):
|
||||||
|
"""Ofsted Parent View survey — latest per school."""
|
||||||
|
__tablename__ = "fact_parent_view"
|
||||||
|
__table_args__ = MARTS
|
||||||
|
|
||||||
|
urn = Column(Integer, primary_key=True)
|
||||||
|
survey_date = Column(Date)
|
||||||
|
total_responses = Column(Integer)
|
||||||
|
q_happy_pct = Column(Float)
|
||||||
|
q_safe_pct = Column(Float)
|
||||||
|
q_behaviour_pct = Column(Float)
|
||||||
|
q_bullying_pct = Column(Float)
|
||||||
|
q_communication_pct = Column(Float)
|
||||||
|
q_progress_pct = Column(Float)
|
||||||
|
q_teaching_pct = Column(Float)
|
||||||
|
q_information_pct = Column(Float)
|
||||||
|
q_curriculum_pct = Column(Float)
|
||||||
|
q_future_pct = Column(Float)
|
||||||
|
q_leadership_pct = Column(Float)
|
||||||
|
q_wellbeing_pct = Column(Float)
|
||||||
|
q_recommend_pct = Column(Float)
|
||||||
|
|
||||||
|
|
||||||
class FactAdmissions(Base):
|
class FactAdmissions(Base):
|
||||||
"""School admissions — one row per URN per year."""
|
"""School admissions — one row per URN per year."""
|
||||||
__tablename__ = "fact_admissions"
|
__tablename__ = "fact_admissions"
|
||||||
@@ -174,11 +187,6 @@ class FactAdmissions(Base):
|
|||||||
total_applications = Column(Integer)
|
total_applications = Column(Integer)
|
||||||
first_preference_applications = Column(Integer)
|
first_preference_applications = Column(Integer)
|
||||||
first_preference_offers = Column(Integer)
|
first_preference_offers = Column(Integer)
|
||||||
total_offers = Column(Integer)
|
|
||||||
second_preference_offers = Column(Integer)
|
|
||||||
third_preference_offers = Column(Integer)
|
|
||||||
cross_la_applications = Column(Integer)
|
|
||||||
cross_la_offers = Column(Integer)
|
|
||||||
first_preference_offer_pct = Column(Float)
|
first_preference_offer_pct = Column(Float)
|
||||||
oversubscription_ratio = Column(Float)
|
oversubscription_ratio = Column(Float)
|
||||||
oversubscribed = Column(Boolean)
|
oversubscribed = Column(Boolean)
|
||||||
|
|||||||
@@ -1,44 +0,0 @@
|
|||||||
"""Ofsted renewed-framework (Nov 2025) report-card code translation.
|
|
||||||
|
|
||||||
Scale labels are the live-sampled vocabulary from the Ofsted MI file
|
|
||||||
(see pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE) —
|
|
||||||
verified against real data, not the consultation draft.
|
|
||||||
"""
|
|
||||||
|
|
||||||
REPORT_CARD_GRADE_NAMES = {
|
|
||||||
1: "Exceptional",
|
|
||||||
2: "Strong standard",
|
|
||||||
3: "Expected standard",
|
|
||||||
4: "Needs attention",
|
|
||||||
5: "Urgent improvement",
|
|
||||||
}
|
|
||||||
|
|
||||||
# Graded evaluation areas only — safeguarding is a separate boolean
|
|
||||||
# judgement and must never appear in grade counts or label maps.
|
|
||||||
_RC_AREA_KEYS = (
|
|
||||||
"rc_inclusion",
|
|
||||||
"rc_curriculum_teaching",
|
|
||||||
"rc_achievement",
|
|
||||||
"rc_attendance_behaviour",
|
|
||||||
"rc_personal_development",
|
|
||||||
"rc_leadership_governance",
|
|
||||||
"rc_early_years",
|
|
||||||
"rc_sixth_form",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def report_card_labels(ofsted: dict) -> dict:
|
|
||||||
"""{area_key: {code, label}} for populated, known-valued rc_* areas."""
|
|
||||||
out = {}
|
|
||||||
for key in _RC_AREA_KEYS:
|
|
||||||
code = ofsted.get(key)
|
|
||||||
label = REPORT_CARD_GRADE_NAMES.get(code)
|
|
||||||
if code is not None and label is not None:
|
|
||||||
out[key] = {"code": code, "label": label}
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def ofsted_page_url(urn: int) -> str:
|
|
||||||
"""The school's page on ofsted.gov.uk (all its reports live there —
|
|
||||||
we never deep-link an individual report)."""
|
|
||||||
return f"https://reports.ofsted.gov.uk/provider/21/{urn}"
|
|
||||||
@@ -543,8 +543,6 @@ SCHOOL_COLUMNS = [
|
|||||||
"postcode",
|
"postcode",
|
||||||
"religious_denomination",
|
"religious_denomination",
|
||||||
"age_range",
|
"age_range",
|
||||||
"has_sixth_form",
|
|
||||||
"status",
|
|
||||||
"gender",
|
"gender",
|
||||||
"admissions_policy",
|
"admissions_policy",
|
||||||
"ofsted_grade",
|
"ofsted_grade",
|
||||||
|
|||||||
@@ -1,78 +0,0 @@
|
|||||||
"""compute_benchmarks: state-school benchmarks computed from our dataset
|
|
||||||
(spec §5/§8.6). The disadvantaged average must be weighted by cohort size,
|
|
||||||
medians must ignore NaN, and only the latest year counts."""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from backend.data_loader import compute_benchmarks
|
|
||||||
|
|
||||||
LATEST = 202425
|
|
||||||
|
|
||||||
|
|
||||||
def _df():
|
|
||||||
rows = [
|
|
||||||
# Six primary schools, latest year. Disadvantaged RWM chosen so the
|
|
||||||
# weighted average differs clearly from the unweighted mean:
|
|
||||||
# weighted = (40*100 + 60*300) / 400 = 55.0 ; unweighted mean = 50.0
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=100,
|
|
||||||
rwm_expected_disadvantaged_pct=40.0, eal_pct=10.0,
|
|
||||||
sen_support_pct=10.0, disadvantaged_pct=20.0, total_pupils=200),
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=300,
|
|
||||||
rwm_expected_disadvantaged_pct=60.0, eal_pct=20.0,
|
|
||||||
sen_support_pct=14.0, disadvantaged_pct=24.0, total_pupils=280),
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=np.nan,
|
|
||||||
rwm_expected_disadvantaged_pct=99.0, eal_pct=30.0,
|
|
||||||
sen_support_pct=18.0, disadvantaged_pct=30.0, total_pupils=300),
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=50,
|
|
||||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=np.nan,
|
|
||||||
sen_support_pct=np.nan, disadvantaged_pct=np.nan, total_pupils=np.nan),
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=40,
|
|
||||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=40.0,
|
|
||||||
sen_support_pct=20.0, disadvantaged_pct=40.0, total_pupils=350),
|
|
||||||
dict(year=LATEST, attainment_8_score=np.nan, eligible_pupils=60,
|
|
||||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=50.0,
|
|
||||||
sen_support_pct=22.0, disadvantaged_pct=44.0, total_pupils=400),
|
|
||||||
# Two secondary schools (attainment_8 non-null)
|
|
||||||
dict(year=LATEST, attainment_8_score=45.0, eligible_pupils=180,
|
|
||||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=15.0,
|
|
||||||
sen_support_pct=12.0, disadvantaged_pct=22.0, total_pupils=1000),
|
|
||||||
dict(year=LATEST, attainment_8_score=50.0, eligible_pupils=200,
|
|
||||||
rwm_expected_disadvantaged_pct=np.nan, eal_pct=25.0,
|
|
||||||
sen_support_pct=16.0, disadvantaged_pct=26.0, total_pupils=1200),
|
|
||||||
# An older-year primary row that must NOT influence anything
|
|
||||||
dict(year=202324, attainment_8_score=np.nan, eligible_pupils=500,
|
|
||||||
rwm_expected_disadvantaged_pct=1.0, eal_pct=99.0,
|
|
||||||
sen_support_pct=99.0, disadvantaged_pct=99.0, total_pupils=9999),
|
|
||||||
]
|
|
||||||
return pd.DataFrame(rows)
|
|
||||||
|
|
||||||
|
|
||||||
def test_weighted_disadvantaged_average():
|
|
||||||
b = compute_benchmarks(_df())
|
|
||||||
# Row 3 has NaN eligible_pupils and must be excluded from the weighting.
|
|
||||||
assert b["primary"]["disadvantaged_rwm_expected_pct"] == 55.0
|
|
||||||
|
|
||||||
|
|
||||||
def test_medians_ignore_nan_and_older_years():
|
|
||||||
b = compute_benchmarks(_df())
|
|
||||||
assert b["year"] == LATEST
|
|
||||||
# eal medians over [10,20,30,40,50] = 30
|
|
||||||
assert b["primary"]["eal_pct"] == 30.0
|
|
||||||
# median pupils over [200,280,300,350,400] = 300
|
|
||||||
assert b["primary"]["median_pupils"] == 300
|
|
||||||
|
|
||||||
|
|
||||||
def test_secondary_block_has_no_disadvantaged_rwm():
|
|
||||||
b = compute_benchmarks(_df())
|
|
||||||
assert "disadvantaged_rwm_expected_pct" not in b["secondary"]
|
|
||||||
assert b["secondary"]["median_pupils"] == 1100
|
|
||||||
|
|
||||||
|
|
||||||
def test_provenance_string():
|
|
||||||
b = compute_benchmarks(_df())
|
|
||||||
assert b["source"] == "state-school average (computed from our dataset)"
|
|
||||||
|
|
||||||
|
|
||||||
def test_empty_df():
|
|
||||||
assert compute_benchmarks(pd.DataFrame()) == {}
|
|
||||||
@@ -1,121 +0,0 @@
|
|||||||
"""/api/compare enrichment for the compare redesign: per-school
|
|
||||||
supplementary blocks, top-level national_averages (shared with the
|
|
||||||
/api/national-averages endpoint) and computed benchmarks — all additive."""
|
|
||||||
|
|
||||||
import types
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import pytest
|
|
||||||
from fastapi.testclient import TestClient
|
|
||||||
|
|
||||||
LATEST = 202425
|
|
||||||
|
|
||||||
CANNED_SUPPLEMENTARY = {
|
|
||||||
"ofsted": {"overall_effectiveness": 2, "grade_source": "graded",
|
|
||||||
"report_card": {}, "ofsted_page_url": "https://reports.ofsted.gov.uk/provider/21/100140"},
|
|
||||||
"census": {"year": 202526, "fsm_pct": 29.8},
|
|
||||||
"admissions": {"year": 202627, "second_preference_offers": 4},
|
|
||||||
"admissions_history": [{"year": 202627, "second_preference_offers": 4}],
|
|
||||||
"sen_detail": None,
|
|
||||||
"phonics": None,
|
|
||||||
"deprivation": {"idaci_decile": 4},
|
|
||||||
"finance": None,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _two_primary_schools_df() -> pd.DataFrame:
|
|
||||||
rows = []
|
|
||||||
for urn, name, rwm, dis in ((100140, "Plumcroft Primary School", 79.0, 72.0),
|
|
||||||
(138690, "Barclay Primary School", 87.0, 86.0)):
|
|
||||||
rows.append(dict(
|
|
||||||
urn=urn, school_name=name, local_authority="Greenwich",
|
|
||||||
school_type="Community school", address="1 Road", phase="Primary",
|
|
||||||
year=LATEST, rwm_expected_pct=rwm, attainment_8_score=np.nan,
|
|
||||||
eligible_pupils=60, rwm_expected_disadvantaged_pct=dis,
|
|
||||||
eal_pct=20.0, sen_support_pct=14.0, disadvantaged_pct=25.0,
|
|
||||||
total_pupils=1000.0,
|
|
||||||
))
|
|
||||||
return pd.DataFrame(rows)
|
|
||||||
|
|
||||||
|
|
||||||
class _StubNatRow:
|
|
||||||
year = 202425
|
|
||||||
rwm_expected_pct = 62.1
|
|
||||||
gps_expected_pct = 72.0
|
|
||||||
science_expected_pct = 81.0
|
|
||||||
|
|
||||||
|
|
||||||
class _StubSession:
|
|
||||||
def query(self, *a, **k):
|
|
||||||
return self
|
|
||||||
|
|
||||||
def order_by(self, *a, **k):
|
|
||||||
return self
|
|
||||||
|
|
||||||
def all(self):
|
|
||||||
return [_StubNatRow()]
|
|
||||||
|
|
||||||
def close(self):
|
|
||||||
pass
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def client(monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
from backend import database as database_module
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", _two_primary_schools_df)
|
|
||||||
monkeypatch.setattr(
|
|
||||||
app_module, "get_supplementary_data", lambda db, urn: dict(CANNED_SUPPLEMENTARY)
|
|
||||||
)
|
|
||||||
monkeypatch.setattr(database_module, "SessionLocal", _StubSession)
|
|
||||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
|
|
||||||
def test_existing_shape_is_preserved(client):
|
|
||||||
body = client.get("/api/compare?urns=100140,138690").json()
|
|
||||||
school = body["comparison"]["100140"]
|
|
||||||
assert school["school_info"]["rwm_expected_pct"] == 79.0
|
|
||||||
assert school["yearly_data"][0]["year"] == LATEST
|
|
||||||
|
|
||||||
|
|
||||||
def test_each_school_gains_supplementary_blocks(client):
|
|
||||||
body = client.get("/api/compare?urns=100140,138690").json()
|
|
||||||
for urn in ("100140", "138690"):
|
|
||||||
school = body["comparison"][urn]
|
|
||||||
assert school["ofsted"]["grade_source"] == "graded"
|
|
||||||
assert school["census"]["fsm_pct"] == 29.8
|
|
||||||
assert school["admissions"]["second_preference_offers"] == 4
|
|
||||||
assert school["admissions_history"][0]["year"] == 202627
|
|
||||||
assert school["deprivation"]["idaci_decile"] == 4
|
|
||||||
|
|
||||||
|
|
||||||
def test_top_level_national_averages_and_benchmarks(client):
|
|
||||||
body = client.get("/api/compare?urns=100140,138690").json()
|
|
||||||
assert body["national_averages"]["year"] == LATEST
|
|
||||||
assert body["benchmarks"]["source"] == "state-school average (computed from our dataset)"
|
|
||||||
# weighted over equal cohorts of 72 and 86 = 79.0
|
|
||||||
assert body["benchmarks"]["primary"]["disadvantaged_rwm_expected_pct"] == 79.0
|
|
||||||
|
|
||||||
|
|
||||||
def test_supplementary_failure_degrades_not_500(client, monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
def _boom(db, urn):
|
|
||||||
raise RuntimeError("marts unavailable")
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "get_supplementary_data", _boom)
|
|
||||||
resp = client.get("/api/compare?urns=100140")
|
|
||||||
assert resp.status_code == 200
|
|
||||||
school = resp.json()["comparison"]["100140"]
|
|
||||||
assert school["ofsted"] is None
|
|
||||||
assert school["admissions_history"] == []
|
|
||||||
|
|
||||||
|
|
||||||
def test_national_averages_endpoint_exposes_gps_science(client):
|
|
||||||
body = client.get("/api/national-averages").json()
|
|
||||||
latest_primary_by_year = [e["primary"] for e in body["by_year"] if e["primary"]]
|
|
||||||
assert latest_primary_by_year, "expected official by_year rows from the stub"
|
|
||||||
assert latest_primary_by_year[-1]["gps_expected_pct"] == 72.0
|
|
||||||
assert latest_primary_by_year[-1]["science_expected_pct"] == 81.0
|
|
||||||
@@ -1,95 +0,0 @@
|
|||||||
"""Tests for the GIAS code->name dictionaries (spec 2026-07-09).
|
|
||||||
|
|
||||||
The dictionaries are generated from the live GIAS bulk CSV by
|
|
||||||
pipeline/scripts/generate_gias_codes.py — these tests assert the module's
|
|
||||||
contract, key sentinel values the marts/UI depend on, and that the pipeline
|
|
||||||
copy has not drifted from the canonical backend module.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import math
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from backend.gias_codes import (
|
|
||||||
ADMISSIONS_POLICY,
|
|
||||||
ESTABLISHMENT_STATUS,
|
|
||||||
OFFICIAL_SIXTH_FORM,
|
|
||||||
PHASE_OF_EDUCATION,
|
|
||||||
RELIGIOUS_CHARACTER,
|
|
||||||
SCHOOL_TYPE,
|
|
||||||
translate,
|
|
||||||
)
|
|
||||||
|
|
||||||
REPO = Path(__file__).resolve().parents[2]
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_known_code():
|
|
||||||
open_code = next(c for c, n in ESTABLISHMENT_STATUS.items() if n == "Open")
|
|
||||||
assert translate(open_code, ESTABLISHMENT_STATUS) == "Open"
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_unknown_code_degrades_gracefully():
|
|
||||||
assert translate(9999, ESTABLISHMENT_STATUS) == "Unknown (9999)"
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_none_and_nan_return_none():
|
|
||||||
assert translate(None, ESTABLISHMENT_STATUS) is None
|
|
||||||
assert translate(float("nan"), ESTABLISHMENT_STATUS) is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_accepts_float_codes():
|
|
||||||
# pd.read_sql yields float columns when NULLs are present
|
|
||||||
open_code = next(c for c, n in ESTABLISHMENT_STATUS.items() if n == "Open")
|
|
||||||
assert translate(float(open_code), ESTABLISHMENT_STATUS) == "Open"
|
|
||||||
|
|
||||||
|
|
||||||
def test_sentinel_names_present():
|
|
||||||
"""Names the marts/UI compare against must exist verbatim."""
|
|
||||||
assert "Open" in ESTABLISHMENT_STATUS.values()
|
|
||||||
assert "Open, but proposed to close" in ESTABLISHMENT_STATUS.values()
|
|
||||||
assert "Has a sixth form" in OFFICIAL_SIXTH_FORM.values()
|
|
||||||
assert "Primary" in PHASE_OF_EDUCATION.values()
|
|
||||||
assert "Secondary" in PHASE_OF_EDUCATION.values()
|
|
||||||
assert "Does not apply" in RELIGIOUS_CHARACTER.values()
|
|
||||||
assert all(len(d) > 0 for d in (
|
|
||||||
SCHOOL_TYPE, ESTABLISHMENT_STATUS, PHASE_OF_EDUCATION,
|
|
||||||
OFFICIAL_SIXTH_FORM, RELIGIOUS_CHARACTER, ADMISSIONS_POLICY,
|
|
||||||
))
|
|
||||||
|
|
||||||
|
|
||||||
def test_pipeline_copy_is_identical():
|
|
||||||
canonical = (REPO / "backend" / "gias_codes.py").read_text()
|
|
||||||
copy = (REPO / "pipeline" / "scripts" / "gias_codes.py").read_text()
|
|
||||||
assert canonical == copy, (
|
|
||||||
"pipeline/scripts/gias_codes.py has drifted from backend/gias_codes.py — "
|
|
||||||
"regenerate with pipeline/scripts/generate_gias_codes.py and copy the file"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_seed_matches_dictionaries():
|
|
||||||
import csv
|
|
||||||
fields = {
|
|
||||||
"school_type": SCHOOL_TYPE,
|
|
||||||
"establishment_status": ESTABLISHMENT_STATUS,
|
|
||||||
"phase_of_education": PHASE_OF_EDUCATION,
|
|
||||||
"official_sixth_form": OFFICIAL_SIXTH_FORM,
|
|
||||||
"religious_character": RELIGIOUS_CHARACTER,
|
|
||||||
"admissions_policy": ADMISSIONS_POLICY,
|
|
||||||
}
|
|
||||||
seed_path = REPO / "pipeline" / "transform" / "seeds" / "gias_code_names.csv"
|
|
||||||
seed: dict[str, dict[int, str]] = {k: {} for k in fields}
|
|
||||||
with open(seed_path, newline="") as fh:
|
|
||||||
for row in csv.DictReader(fh):
|
|
||||||
seed[row["field"]][int(row["code"])] = row["name"]
|
|
||||||
assert seed == fields
|
|
||||||
|
|
||||||
|
|
||||||
def test_blank_name_sentinel_codes_map_to_empty_string():
|
|
||||||
"""GIAS carries codes whose (name) column is blank — e.g. ReligiousCharacter
|
|
||||||
99 (~4k schools) and AdmissionsPolicy 9 (~5.6k schools). The old name
|
|
||||||
pipeline served these as empty strings; the dictionaries must reproduce
|
|
||||||
that ("" is falsy, so UI tag heuristics stay silent) rather than letting
|
|
||||||
them hit the "Unknown (<code>)" path meant for genuinely new codes."""
|
|
||||||
assert RELIGIOUS_CHARACTER[99] == ""
|
|
||||||
assert ADMISSIONS_POLICY[9] == ""
|
|
||||||
assert translate(99, RELIGIOUS_CHARACTER) == ""
|
|
||||||
assert translate(9, ADMISSIONS_POLICY) == ""
|
|
||||||
@@ -1,132 +0,0 @@
|
|||||||
"""API-boundary translation: marts now carry GIAS codes; the DataFrame the
|
|
||||||
rest of the backend sees must carry today's name strings."""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from backend.data_loader import _missing_column_name, translate_gias_code_columns
|
|
||||||
from backend.gias_codes import ESTABLISHMENT_STATUS, PHASE_OF_EDUCATION
|
|
||||||
|
|
||||||
|
|
||||||
def _code_for(mapping, name):
|
|
||||||
return next(c for c, n in mapping.items() if n == name)
|
|
||||||
|
|
||||||
|
|
||||||
def test_codes_become_todays_names():
|
|
||||||
df = pd.DataFrame([{
|
|
||||||
"urn": 1,
|
|
||||||
"phase_code": float(_code_for(PHASE_OF_EDUCATION, "Primary")),
|
|
||||||
"school_type_code": np.nan,
|
|
||||||
"status_code": float(_code_for(ESTABLISHMENT_STATUS, "Open, but proposed to close")),
|
|
||||||
"religious_character_code": np.nan,
|
|
||||||
"admissions_policy_code": np.nan,
|
|
||||||
}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
row = out.iloc[0]
|
|
||||||
assert row["phase"] == "Primary"
|
|
||||||
assert row["status"] == "Open, but proposed to close"
|
|
||||||
assert row["school_type"] is None
|
|
||||||
assert row["religious_denomination"] is None
|
|
||||||
assert row["admissions_policy"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_unknown_code_degrades_not_blanks():
|
|
||||||
df = pd.DataFrame([{"urn": 1, "phase_code": 9999.0}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
assert out.iloc[0]["phase"] == "Unknown (9999)"
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_code_columns_are_a_noop():
|
|
||||||
"""Old-schema DataFrames (tests, pre-pipeline DBs) pass through untouched."""
|
|
||||||
df = pd.DataFrame([{"urn": 1, "phase": "Primary", "status": "Open"}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
assert out.iloc[0]["phase"] == "Primary"
|
|
||||||
assert out.iloc[0]["status"] == "Open"
|
|
||||||
|
|
||||||
|
|
||||||
def _fake_exc(orig_message):
|
|
||||||
"""A stand-in for sqlalchemy.exc.ProgrammingError: str(exc) embeds the
|
|
||||||
full SQL statement (deliberately containing every column name below, to
|
|
||||||
prove the matcher doesn't fall back to it), while .orig carries the real
|
|
||||||
DBAPI error message naming only the offending column."""
|
|
||||||
exc = Exception(
|
|
||||||
"SELECT s.phase_code, s.school_type_code, s.religious_character_code, "
|
|
||||||
"s.status_code, s.admissions_policy_code, s.has_sixth_form FROM ... "
|
|
||||||
f"[SQL: ...] (Background on this error at: https://...)"
|
|
||||||
)
|
|
||||||
exc.orig = Exception(orig_message) if orig_message is not None else None
|
|
||||||
return exc
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_column_name_quoted():
|
|
||||||
assert _missing_column_name(_fake_exc('column "phase_code" does not exist')) == "phase_code"
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_column_name_unquoted():
|
|
||||||
assert _missing_column_name(_fake_exc("column phase_code does not exist")) == "phase_code"
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_column_name_table_prefixed():
|
|
||||||
assert (
|
|
||||||
_missing_column_name(_fake_exc("column s.has_sixth_form does not exist"))
|
|
||||||
== "has_sixth_form"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_column_name_no_match_returns_none():
|
|
||||||
assert _missing_column_name(_fake_exc("relation \"marts.dim_school\" does not exist")) is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_load_school_data_survives_premigration_marts(monkeypatch):
|
|
||||||
"""Real prod state until the nightly pipeline first rebuilds the mart with
|
|
||||||
the GIAS code columns: marts.dim_school still has the old name columns
|
|
||||||
(phase, school_type, religious_character, status, admissions_policy)
|
|
||||||
instead of the new *_code columns. The first query raises UndefinedColumn
|
|
||||||
on s.phase_code; load_school_data_as_dataframe must retry with the
|
|
||||||
legacy name-column query rather than swallow the error and return (and
|
|
||||||
then have load_school_data cache) an empty DataFrame."""
|
|
||||||
import sqlalchemy.exc
|
|
||||||
from backend import data_loader
|
|
||||||
|
|
||||||
data_loader._df_cache = None
|
|
||||||
data_loader._df_latest_cache = None
|
|
||||||
|
|
||||||
good_df = pd.DataFrame(
|
|
||||||
[
|
|
||||||
{
|
|
||||||
"urn": 1,
|
|
||||||
"school_name": "Legacy School",
|
|
||||||
"phase": "Primary",
|
|
||||||
"school_type": "Academy",
|
|
||||||
"status": "Open",
|
|
||||||
}
|
|
||||||
]
|
|
||||||
)
|
|
||||||
calls = []
|
|
||||||
|
|
||||||
def fake_read_sql(query, con):
|
|
||||||
calls.append(query)
|
|
||||||
if len(calls) == 1:
|
|
||||||
raise sqlalchemy.exc.ProgrammingError(
|
|
||||||
statement=str(data_loader._MAIN_QUERY),
|
|
||||||
params=None,
|
|
||||||
orig=Exception(
|
|
||||||
"(psycopg2.errors.UndefinedColumn) column s.phase_code "
|
|
||||||
"does not exist\nLINE 5: s.phase_code,"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
return good_df.copy()
|
|
||||||
|
|
||||||
monkeypatch.setattr(data_loader.pd, "read_sql", fake_read_sql)
|
|
||||||
|
|
||||||
try:
|
|
||||||
df = data_loader.load_school_data_as_dataframe()
|
|
||||||
finally:
|
|
||||||
data_loader._df_cache = None
|
|
||||||
data_loader._df_latest_cache = None
|
|
||||||
|
|
||||||
assert len(calls) == 2, "must retry with the legacy name-column query variant"
|
|
||||||
assert calls[1] is data_loader._MAIN_QUERY_LEGACY_NAMES
|
|
||||||
assert not df.empty
|
|
||||||
assert df["phase"].iloc[0] == "Primary"
|
|
||||||
assert df["status"].iloc[0] == "Open"
|
|
||||||
@@ -1,45 +0,0 @@
|
|||||||
"""Report-card code translation uses the live-sampled Ofsted vocabulary
|
|
||||||
(pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE):
|
|
||||||
Exceptional / Strong standard / Expected standard / Needs attention /
|
|
||||||
Urgent improvement — never the consultation draft's 'Attention needed'."""
|
|
||||||
|
|
||||||
from backend.ofsted_codes import (
|
|
||||||
REPORT_CARD_GRADE_NAMES,
|
|
||||||
ofsted_page_url,
|
|
||||||
report_card_labels,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_scale_is_sampled_vocabulary():
|
|
||||||
assert REPORT_CARD_GRADE_NAMES == {
|
|
||||||
1: "Exceptional",
|
|
||||||
2: "Strong standard",
|
|
||||||
3: "Expected standard",
|
|
||||||
4: "Needs attention",
|
|
||||||
5: "Urgent improvement",
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_labels_only_for_populated_areas_and_never_safeguarding():
|
|
||||||
ofsted = {
|
|
||||||
"rc_achievement": 2,
|
|
||||||
"rc_inclusion": 3,
|
|
||||||
"rc_attendance_behaviour": 4,
|
|
||||||
"rc_early_years": None,
|
|
||||||
"rc_safeguarding_met": True,
|
|
||||||
"overall_effectiveness": None,
|
|
||||||
}
|
|
||||||
labels = report_card_labels(ofsted)
|
|
||||||
assert labels == {
|
|
||||||
"rc_achievement": {"code": 2, "label": "Strong standard"},
|
|
||||||
"rc_inclusion": {"code": 3, "label": "Expected standard"},
|
|
||||||
"rc_attendance_behaviour": {"code": 4, "label": "Needs attention"},
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_unknown_code_is_skipped_not_crashed():
|
|
||||||
assert report_card_labels({"rc_achievement": 9}) == {}
|
|
||||||
|
|
||||||
|
|
||||||
def test_provider_url():
|
|
||||||
assert ofsted_page_url(138690) == "https://reports.ofsted.gov.uk/provider/21/138690"
|
|
||||||
@@ -1,71 +0,0 @@
|
|||||||
"""Regression tests for GET /api/schools/{urn}.
|
|
||||||
|
|
||||||
Schools with no performance rows (special post-16 institutions, sixth-form
|
|
||||||
centres, PRUs, brand-new schools) come back from the marts LEFT JOIN with
|
|
||||||
NaN in every numeric column. The endpoint must still serialize them — a NaN
|
|
||||||
that reaches Starlette's JSONResponse raises ValueError (allow_nan=False)
|
|
||||||
and the route 500s, which the frontend then renders as a 404.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import pytest
|
|
||||||
from fastapi.testclient import TestClient
|
|
||||||
|
|
||||||
|
|
||||||
def _no_results_school_df() -> pd.DataFrame:
|
|
||||||
"""One school row as produced by the marts query for a school with no
|
|
||||||
performance data: GIAS/location fields partly populated, every
|
|
||||||
results-linked column NaN (including year)."""
|
|
||||||
return pd.DataFrame(
|
|
||||||
[
|
|
||||||
{
|
|
||||||
"urn": 150275,
|
|
||||||
"school_name": "West London Performing Arts Academy",
|
|
||||||
"phase": "Secondary",
|
|
||||||
"school_type": "Special post 16 institution",
|
|
||||||
"trust_name": None,
|
|
||||||
"religious_denomination": "Does not apply",
|
|
||||||
"gender": None,
|
|
||||||
"age_range": "16-25",
|
|
||||||
"admissions_policy": None,
|
|
||||||
"capacity": np.nan,
|
|
||||||
"gias_total_pupils": np.nan,
|
|
||||||
"headteacher_name": None,
|
|
||||||
"website": None,
|
|
||||||
"ofsted_grade": np.nan,
|
|
||||||
"local_authority": "Ealing",
|
|
||||||
"address": "268 Northfield Avenue, London, W5 4UB",
|
|
||||||
"postcode": "W5 4UB",
|
|
||||||
"latitude": 51.4986,
|
|
||||||
"longitude": -0.3148,
|
|
||||||
"year": np.nan,
|
|
||||||
"total_pupils": np.nan,
|
|
||||||
"eligible_pupils": np.nan,
|
|
||||||
"rwm_expected_pct": np.nan,
|
|
||||||
}
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def client(monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", _no_results_school_df)
|
|
||||||
monkeypatch.setattr(
|
|
||||||
app_module, "get_supplementary_data", lambda db, urn: {}
|
|
||||||
)
|
|
||||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
|
|
||||||
def test_school_without_performance_rows_returns_200(client):
|
|
||||||
resp = client.get("/api/schools/150275")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
|
|
||||||
|
|
||||||
def test_nan_gias_fields_serialize_as_null(client):
|
|
||||||
info = client.get("/api/schools/150275").json()["school_info"]
|
|
||||||
assert info["capacity"] is None
|
|
||||||
assert info["total_pupils"] is None
|
|
||||||
assert info["school_name"] == "West London Performing Arts Academy"
|
|
||||||
@@ -1,70 +0,0 @@
|
|||||||
"""Tests for GIAS establishment status exposure.
|
|
||||||
|
|
||||||
"Open, but proposed to close" schools are now kept by the dims; the API must
|
|
||||||
surface `status` on list items and school_info so the UI can render the
|
|
||||||
proposed-to-close marker (listing tag) and notice strip (detail page).
|
|
||||||
"""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import pytest
|
|
||||||
from fastapi.testclient import TestClient
|
|
||||||
|
|
||||||
PROPOSED = "Open, but proposed to close"
|
|
||||||
|
|
||||||
|
|
||||||
def _schools_df() -> pd.DataFrame:
|
|
||||||
base = {
|
|
||||||
"local_authority": "Testshire",
|
|
||||||
"school_type": "Academy",
|
|
||||||
"phase": "Secondary",
|
|
||||||
"address": "1 Test Street",
|
|
||||||
"town": "Testtown",
|
|
||||||
"postcode": "TS1 1AA",
|
|
||||||
"religious_denomination": None,
|
|
||||||
"gender": "Mixed",
|
|
||||||
"age_range": "11-16",
|
|
||||||
"admissions_policy": None,
|
|
||||||
"has_sixth_form": False,
|
|
||||||
"ofsted_grade": np.nan,
|
|
||||||
"ofsted_date": None,
|
|
||||||
"ofsted_framework": None,
|
|
||||||
"latitude": 51.5,
|
|
||||||
"longitude": -0.1,
|
|
||||||
"year": 202425,
|
|
||||||
"total_pupils": 800,
|
|
||||||
"rwm_expected_pct": np.nan,
|
|
||||||
"attainment_8_score": 48.0,
|
|
||||||
}
|
|
||||||
return pd.DataFrame(
|
|
||||||
[
|
|
||||||
{**base, "urn": 200001, "school_name": "Alpha Academy",
|
|
||||||
"status": "Open"},
|
|
||||||
{**base, "urn": 200002, "school_name": "Sarson High School",
|
|
||||||
"status": PROPOSED},
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def client(monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
|
||||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
|
|
||||||
def test_list_payload_includes_status(client):
|
|
||||||
resp = client.get("/api/schools")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
by_urn = {s["urn"]: s for s in resp.json()["schools"]}
|
|
||||||
assert by_urn[200001]["status"] == "Open"
|
|
||||||
assert by_urn[200002]["status"] == PROPOSED
|
|
||||||
|
|
||||||
|
|
||||||
def test_detail_payload_includes_status(client):
|
|
||||||
resp = client.get("/api/schools/200002")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
assert resp.json()["school_info"]["status"] == PROPOSED
|
|
||||||
@@ -1,177 +0,0 @@
|
|||||||
"""Tests for the GIAS-driven has_sixth_form flag (spec 2026-07-07 §3).
|
|
||||||
|
|
||||||
The filter and payloads must use dim_school.has_sixth_form, not the old
|
|
||||||
age_range-contains-"18" substring heuristic. The key regression case is a
|
|
||||||
16-19 sixth-form college: flag true, but "16-19" contains no "18".
|
|
||||||
"""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import pytest
|
|
||||||
from fastapi.testclient import TestClient
|
|
||||||
|
|
||||||
|
|
||||||
def _schools_df() -> pd.DataFrame:
|
|
||||||
"""Latest-year snapshot rows as produced by load_latest_school_data."""
|
|
||||||
base = {
|
|
||||||
"local_authority": "Testshire",
|
|
||||||
"school_type": "Academy",
|
|
||||||
"phase": "Secondary",
|
|
||||||
"address": "1 Test Street",
|
|
||||||
"town": "Testtown",
|
|
||||||
"postcode": "TS1 1AA",
|
|
||||||
"religious_denomination": None,
|
|
||||||
"gender": "Mixed",
|
|
||||||
"admissions_policy": None,
|
|
||||||
"ofsted_grade": np.nan,
|
|
||||||
"ofsted_date": None,
|
|
||||||
"ofsted_framework": None,
|
|
||||||
"latitude": 51.5,
|
|
||||||
"longitude": -0.1,
|
|
||||||
"year": 202425,
|
|
||||||
"total_pupils": 1000,
|
|
||||||
"rwm_expected_pct": np.nan,
|
|
||||||
"attainment_8_score": 50.0,
|
|
||||||
}
|
|
||||||
return pd.DataFrame(
|
|
||||||
[
|
|
||||||
# 11-18 school WITH a registered sixth form
|
|
||||||
{**base, "urn": 100001, "school_name": "Alpha High",
|
|
||||||
"age_range": "11-18", "has_sixth_form": True},
|
|
||||||
# 16-19 college: old heuristic said NO ("16-19" has no "18"),
|
|
||||||
# GIAS flag says YES — must appear in the yes-filter results
|
|
||||||
{**base, "urn": 100002, "school_name": "Beta Sixth Form College",
|
|
||||||
"age_range": "16-19", "has_sixth_form": True},
|
|
||||||
# 11-18 age range on paper but NO registered sixth form:
|
|
||||||
# old heuristic said YES, GIAS flag says NO
|
|
||||||
{**base, "urn": 100003, "school_name": "Gamma Academy",
|
|
||||||
"age_range": "11-18", "has_sixth_form": False},
|
|
||||||
# Missing flag (pipeline not yet re-run) — must not crash,
|
|
||||||
# must not match the yes-filter
|
|
||||||
{**base, "urn": 100004, "school_name": "Delta School",
|
|
||||||
"age_range": "11-16", "has_sixth_form": None},
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def client(monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
|
||||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
|
|
||||||
def _urns(resp):
|
|
||||||
return sorted(s["urn"] for s in resp.json()["schools"])
|
|
||||||
|
|
||||||
|
|
||||||
def test_filter_yes_uses_flag_not_age_range(client):
|
|
||||||
resp = client.get("/api/schools?has_sixth_form=yes")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
# 16-19 college included; 11-18-without-sixth-form excluded
|
|
||||||
assert _urns(resp) == [100001, 100002]
|
|
||||||
|
|
||||||
|
|
||||||
def test_filter_no_uses_flag_not_age_range(client):
|
|
||||||
resp = client.get("/api/schools?has_sixth_form=no")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
# Gamma (flag false) and Delta (flag missing => not true)
|
|
||||||
assert _urns(resp) == [100003, 100004]
|
|
||||||
|
|
||||||
|
|
||||||
def test_list_payload_includes_flag(client):
|
|
||||||
resp = client.get("/api/schools")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
by_urn = {s["urn"]: s for s in resp.json()["schools"]}
|
|
||||||
assert by_urn[100002]["has_sixth_form"] is True
|
|
||||||
assert by_urn[100003]["has_sixth_form"] is False
|
|
||||||
assert by_urn[100004]["has_sixth_form"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_detail_payload_includes_flag(client):
|
|
||||||
resp = client.get("/api/schools/100002")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
assert resp.json()["school_info"]["has_sixth_form"] is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_detail_payload_serializes_numpy_bool(monkeypatch):
|
|
||||||
"""Once the pipeline has run, has_sixth_form is a real bool dtype column
|
|
||||||
(dbt not_null test guarantees no NULLs), so row access yields
|
|
||||||
numpy.bool_ rather than a Python bool. convert_to_native must handle it —
|
|
||||||
otherwise FastAPI's jsonable_encoder raises ValueError and the detail
|
|
||||||
endpoint 500s (C2)."""
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
df = _schools_df()
|
|
||||||
# Drop the row with a None flag — this fixture models the post-pipeline
|
|
||||||
# state where the column is a genuine, fully-populated bool dtype.
|
|
||||||
df = df[df["has_sixth_form"].notna()].reset_index(drop=True)
|
|
||||||
df["has_sixth_form"] = df["has_sixth_form"].astype(bool)
|
|
||||||
assert df["has_sixth_form"].dtype == bool
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
|
||||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
|
||||||
client = TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
resp = client.get("/api/schools/100002")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
assert resp.json()["school_info"]["has_sixth_form"] is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_load_school_data_survives_missing_has_sixth_form_column(monkeypatch):
|
|
||||||
"""Real prod state until the nightly pipeline first rebuilds the mart:
|
|
||||||
marts.dim_school lacks has_sixth_form entirely. The first query raises
|
|
||||||
UndefinedColumn; load_school_data_as_dataframe must retry without the
|
|
||||||
column (synthesizing it as None) rather than swallow the error and
|
|
||||||
return (and then have load_school_data cache) an empty DataFrame (C1)."""
|
|
||||||
import sqlalchemy.exc
|
|
||||||
from backend import data_loader
|
|
||||||
|
|
||||||
data_loader._df_cache = None
|
|
||||||
data_loader._df_latest_cache = None
|
|
||||||
|
|
||||||
good_df = pd.DataFrame(
|
|
||||||
[
|
|
||||||
{
|
|
||||||
"urn": 1,
|
|
||||||
"school_name": "Fallback School",
|
|
||||||
"school_type": "Academy",
|
|
||||||
"has_sixth_form": None,
|
|
||||||
}
|
|
||||||
]
|
|
||||||
)
|
|
||||||
calls = []
|
|
||||||
|
|
||||||
def fake_read_sql(query, con):
|
|
||||||
calls.append(query)
|
|
||||||
if len(calls) == 1:
|
|
||||||
# The statement text still contains phase_code, school_type_code,
|
|
||||||
# etc. (it's the full _MAIN_QUERY SELECT list) — that's exactly
|
|
||||||
# the collision this test guards against: matching must be done
|
|
||||||
# against exc.orig (the DBAPI error), not str(exc)/the statement.
|
|
||||||
raise sqlalchemy.exc.ProgrammingError(
|
|
||||||
statement=str(data_loader._MAIN_QUERY),
|
|
||||||
params=None,
|
|
||||||
orig=Exception(
|
|
||||||
"(psycopg2.errors.UndefinedColumn) column s.has_sixth_form "
|
|
||||||
"does not exist"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
return good_df.copy()
|
|
||||||
|
|
||||||
monkeypatch.setattr(data_loader.pd, "read_sql", fake_read_sql)
|
|
||||||
|
|
||||||
try:
|
|
||||||
df = data_loader.load_school_data_as_dataframe()
|
|
||||||
finally:
|
|
||||||
data_loader._df_cache = None
|
|
||||||
data_loader._df_latest_cache = None
|
|
||||||
|
|
||||||
assert len(calls) == 2, "must retry with the no-sixth-form query variant"
|
|
||||||
assert calls[1] is data_loader._MAIN_QUERY_NO_SIXTH_FORM
|
|
||||||
assert not df.empty
|
|
||||||
assert "has_sixth_form" in df.columns
|
|
||||||
assert df["has_sixth_form"].iloc[0] is None
|
|
||||||
@@ -1,65 +0,0 @@
|
|||||||
"""Supplementary-block enrichment for the compare redesign: report-card
|
|
||||||
labels, provider-page URL, graded-vs-carried-forward provenance, and the
|
|
||||||
admissions preference/cross-LA detail promoted in the data-foundation PR."""
|
|
||||||
|
|
||||||
import types
|
|
||||||
|
|
||||||
from backend.data_loader import _admissions_row_dict, _ofsted_block
|
|
||||||
|
|
||||||
|
|
||||||
def _row(**kw):
|
|
||||||
base = dict(
|
|
||||||
framework="RC", inspection_date=None, inspection_type=None,
|
|
||||||
overall_effectiveness=None, quality_of_education=None,
|
|
||||||
behaviour_attitudes=None, personal_development=None,
|
|
||||||
leadership_management=None, early_years_provision=None,
|
|
||||||
sixth_form_provision=None, ungraded_outcome=None, ungraded_grade=None,
|
|
||||||
rc_safeguarding_met=None, rc_inclusion=None, rc_curriculum_teaching=None,
|
|
||||||
rc_achievement=None, rc_attendance_behaviour=None,
|
|
||||||
rc_personal_development=None, rc_leadership_governance=None,
|
|
||||||
rc_early_years=None, rc_sixth_form=None, report_url=None,
|
|
||||||
)
|
|
||||||
base.update(kw)
|
|
||||||
return types.SimpleNamespace(**base)
|
|
||||||
|
|
||||||
|
|
||||||
def test_report_card_block_and_provider_url():
|
|
||||||
o = _row(rc_achievement=2, rc_inclusion=3, rc_safeguarding_met=True)
|
|
||||||
block = _ofsted_block(o, urn=100140)
|
|
||||||
assert block["report_card"]["rc_achievement"]["label"] == "Strong standard"
|
|
||||||
assert "rc_safeguarding_met" not in block["report_card"]
|
|
||||||
assert block["rc_safeguarding_met"] is True
|
|
||||||
assert block["ofsted_page_url"] == "https://reports.ofsted.gov.uk/provider/21/100140"
|
|
||||||
|
|
||||||
|
|
||||||
def test_grade_source_graded_vs_carried_forward():
|
|
||||||
assert _ofsted_block(_row(overall_effectiveness=1), urn=1)["grade_source"] == "graded"
|
|
||||||
carried = _ofsted_block(_row(ungraded_grade=2), urn=1)
|
|
||||||
assert carried["grade_source"] == "ungraded_carried_forward"
|
|
||||||
assert carried["overall_effectiveness"] == 2
|
|
||||||
assert _ofsted_block(_row(), urn=1)["grade_source"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_ofsted_block_keeps_existing_keys():
|
|
||||||
block = _ofsted_block(_row(overall_effectiveness=2, quality_of_education=2), urn=1)
|
|
||||||
for key in ("framework", "inspection_date", "overall_effectiveness",
|
|
||||||
"quality_of_education", "rc_inclusion", "report_url"):
|
|
||||||
assert key in block
|
|
||||||
|
|
||||||
|
|
||||||
def test_admissions_row_new_fields():
|
|
||||||
a = types.SimpleNamespace(
|
|
||||||
year=202627, school_phase="Primary", places_offered=80,
|
|
||||||
total_applications=185, first_preference_applications=74,
|
|
||||||
first_preference_offers=74, first_preference_offer_pct=100.0,
|
|
||||||
oversubscription_ratio=0.925, oversubscribed=False,
|
|
||||||
total_offers=80, second_preference_offers=4, third_preference_offers=2,
|
|
||||||
cross_la_applications=12, cross_la_offers=3,
|
|
||||||
)
|
|
||||||
d = _admissions_row_dict(a)
|
|
||||||
for k in ("total_offers", "second_preference_offers", "third_preference_offers",
|
|
||||||
"cross_la_applications", "cross_la_offers"):
|
|
||||||
assert d[k] == getattr(a, k)
|
|
||||||
# Existing keys unchanged
|
|
||||||
assert d["first_preference_offer_pct"] == 100.0
|
|
||||||
assert d["oversubscribed"] is False
|
|
||||||
@@ -11,8 +11,6 @@ def convert_to_native(value: Any) -> Any:
|
|||||||
"""Convert numpy types to native Python types for JSON serialization."""
|
"""Convert numpy types to native Python types for JSON serialization."""
|
||||||
if pd.isna(value):
|
if pd.isna(value):
|
||||||
return None
|
return None
|
||||||
if isinstance(value, np.bool_):
|
|
||||||
return bool(value)
|
|
||||||
if isinstance(value, (np.integer,)):
|
if isinstance(value, (np.integer,)):
|
||||||
return int(value)
|
return int(value)
|
||||||
if isinstance(value, (np.floating,)):
|
if isinstance(value, (np.floating,)):
|
||||||
|
|||||||
+1
-2
@@ -13,7 +13,7 @@ WHEN TO BUMP:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
# Current schema version - increment when models change
|
# Current schema version - increment when models change
|
||||||
SCHEMA_VERSION = 6
|
SCHEMA_VERSION = 5
|
||||||
|
|
||||||
# Changelog for documentation
|
# Changelog for documentation
|
||||||
SCHEMA_CHANGELOG = {
|
SCHEMA_CHANGELOG = {
|
||||||
@@ -22,5 +22,4 @@ SCHEMA_CHANGELOG = {
|
|||||||
3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools",
|
3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools",
|
||||||
4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)",
|
4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)",
|
||||||
5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables",
|
5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables",
|
||||||
6: "Removed the Ofsted Parent View feature: dropped fact_parent_view table and model",
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -112,15 +112,11 @@ Full details in `docs/DEPLOY.md`. The short version:
|
|||||||
- **Never push to `main` directly.** Work on a feature branch and open a PR;
|
- **Never push to `main` directly.** Work on a feature branch and open a PR;
|
||||||
branch protection requires the PR checks (typecheck, tests, builds, AI review)
|
branch protection requires the PR checks (typecheck, tests, builds, AI review)
|
||||||
to pass before merge.
|
to pass before merge.
|
||||||
- Merging to `main` deploys automatically **to staging only**: images are
|
- Merging to `main` deploys automatically: images are built once, deployed to
|
||||||
built once, deployed to the staging Portainer stack, and verified by the
|
the **staging** Portainer stack, verified by the Playwright journeys in
|
||||||
Playwright journeys in `e2e/`. Production is a second, manual approval:
|
`e2e/`, and only then retagged `:prod` and rolled out to production.
|
||||||
the "Promote to Production (manual)" workflow in Gitea Actions, run after
|
|
||||||
testing the feature on staging. It refuses commits whose staging E2E gate
|
|
||||||
isn't green. Never trigger it yourself — promotion is the human's call.
|
|
||||||
- If you change user-facing behaviour, update or extend the `e2e/` journey
|
- If you change user-facing behaviour, update or extend the `e2e/` journey
|
||||||
tests in the same PR — they gate whether staging is fit for human testing
|
tests in the same PR — they are the promotion gate.
|
||||||
and whether a commit is promotable.
|
|
||||||
|
|
||||||
## Recent Changes
|
## Recent Changes
|
||||||
|
|
||||||
|
|||||||
+17
-37
@@ -1,61 +1,41 @@
|
|||||||
# SDLC & Deployment Pipeline
|
# SDLC & Deployment Pipeline
|
||||||
|
|
||||||
SchoolCompare uses a two-stage deploy model on Gitea Actions with two human
|
SchoolCompare uses a fully automated staging → production pipeline on Gitea
|
||||||
approvals. AI writes the code on feature branches; the first approval merges
|
Actions. AI writes the code on feature branches; the pipeline verifies every
|
||||||
the PR, which deploys to staging and runs the E2E gate; the second approval —
|
change on a staging environment before promoting the exact same images to
|
||||||
after manual testing on staging — promotes the exact same images to
|
production. Human input is directional only: feature requests, PR review if
|
||||||
production via a manual workflow.
|
desired, and intervention when a gate fails.
|
||||||
|
|
||||||
## The flow
|
## The flow
|
||||||
|
|
||||||
```
|
```
|
||||||
feature branch (AI-authored)
|
feature branch (AI-authored)
|
||||||
│ PR to main ← approval #1
|
│ PR to main
|
||||||
▼
|
▼
|
||||||
PR checks (.gitea/workflows/pr-checks.yml)
|
PR checks (.gitea/workflows/pr-checks.yml)
|
||||||
typecheck + unit tests + backend smoke + image builds (no push)
|
typecheck + unit tests + backend smoke + image builds (no push)
|
||||||
+ Claude code review posted as a PR comment (severe findings fail the check)
|
+ Claude code review posted as a PR comment (severe findings fail the check)
|
||||||
│ merge (branch protection requires green checks)
|
│ merge (branch protection requires green checks)
|
||||||
▼
|
▼
|
||||||
Stage pipeline (.gitea/workflows/deploy.yml) — automatic
|
Deploy pipeline (.gitea/workflows/deploy.yml)
|
||||||
1. build & push images → tags sha-<sha>, staging
|
1. build & push images → tags sha-<sha>, staging
|
||||||
2. staging Portainer webhook → wait for staging health
|
2. staging Portainer webhook → wait for staging health
|
||||||
3. Playwright E2E journeys against staging ← gate before human testing
|
3. Playwright E2E journeys against staging
|
||||||
▼
|
4. retag sha-<sha> → :prod (same bytes — build once, promote the image)
|
||||||
Manual testing on staging (stx.schoolcompare.co.uk)
|
|
||||||
│ Actions → "Promote to Production (manual)" ← approval #2
|
|
||||||
▼
|
|
||||||
Promote pipeline (.gitea/workflows/promote.yml) — manual dispatch
|
|
||||||
1. resolve target sha (input, or latest main if empty)
|
|
||||||
2. REFUSE unless that commit's "E2E Journeys against Staging" status is green
|
|
||||||
3. retag sha-<sha> → :prod (same bytes — build once, promote the image)
|
|
||||||
previous :prod saved as :prod-previous
|
previous :prod saved as :prod-previous
|
||||||
4. prod Portainer webhook → wait for prod health
|
5. prod Portainer webhook → wait for prod health
|
||||||
```
|
```
|
||||||
|
|
||||||
Key principle: **build once, promote the exact image**. Production pins `:prod`,
|
Key principle: **build once, promote the exact image**. Production pins `:prod`,
|
||||||
which only moves when a human runs the promote workflow — and the workflow
|
which only moves after the E2E gate passes on staging. Nothing tags `:latest`
|
||||||
only accepts commits that passed the staging E2E gate. Nothing tags `:latest`
|
|
||||||
anymore.
|
anymore.
|
||||||
|
|
||||||
## Branch & PR workflow
|
## Branch & PR workflow
|
||||||
|
|
||||||
- `main` is protected: no direct pushes, PRs require green status checks.
|
- `main` is protected: no direct pushes, PRs require green status checks.
|
||||||
- All work (human or AI) happens on feature branches → PR to `main`.
|
- All work (human or AI) happens on feature branches → PR to `main`.
|
||||||
- Merging to `main` releases **to staging only**. Production moves only on
|
- Merging to `main` **is** the release action. If staging or the E2E gate
|
||||||
the second approval. If staging or the E2E gate fails, fix forward —
|
fails, production is untouched.
|
||||||
production is untouched either way.
|
|
||||||
|
|
||||||
## Promotion granularity
|
|
||||||
|
|
||||||
Staging always runs the latest `main`. Promoting approves a *state of main*,
|
|
||||||
not a single PR — if two PRs merged since the last promotion, they ship
|
|
||||||
together. Test staging accordingly. To promote an older state, pass its
|
|
||||||
commit SHA to the promote workflow (its images must still exist in the
|
|
||||||
registry).
|
|
||||||
|
|
||||||
Staging quirk for manual testing: external `/api` is broken at the staging
|
|
||||||
proxy — exercise API endpoints from the host, not via the public staging URL.
|
|
||||||
|
|
||||||
## Environments
|
## Environments
|
||||||
|
|
||||||
@@ -99,7 +79,8 @@ fail the E2E gate. That's the point: staging absorbs the risk.
|
|||||||
5. **Bootstrap staging data via Airflow** (no prod dump — staging populates
|
5. **Bootstrap staging data via Airflow** (no prod dump — staging populates
|
||||||
itself from source, exercising the pipeline image end-to-end):
|
itself from source, exercising the pipeline image end-to-end):
|
||||||
- Open the staging Airflow UI (`http://<host>:8081`) and trigger, in order:
|
- Open the staging Airflow UI (`http://<host>:8081`) and trigger, in order:
|
||||||
`school_data_daily`, `school_data_monthly_ofsted`, then the manual-schedule
|
`school_data_daily`, `school_data_monthly_ofsted`,
|
||||||
|
`school_data_monthly_parent_view`, then the manual-schedule
|
||||||
`school_data_annual_ees` and `school_data_annual_idaci`.
|
`school_data_annual_ees` and `school_data_annual_idaci`.
|
||||||
- First runs download from government sources (GIAS, Ofsted, EES, IDACI),
|
- First runs download from government sources (GIAS, Ofsted, EES, IDACI),
|
||||||
run dbt, and sync Typesense — expect the initial backfill to take a while.
|
run dbt, and sync Typesense — expect the initial backfill to take a while.
|
||||||
@@ -112,9 +93,8 @@ fail the E2E gate. That's the point: staging absorbs the risk.
|
|||||||
|
|
||||||
## Rollback
|
## Rollback
|
||||||
|
|
||||||
Re-run "Promote to Production (manual)" with the SHA of the last good commit
|
Every promotion first re-points `:prod-previous` at the outgoing `:prod`.
|
||||||
(fastest, fully gated), or manually re-point the tags — every promotion first
|
To roll back:
|
||||||
saves the outgoing `:prod` as `:prod-previous`:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
for img in backend frontend pipeline; do
|
for img in backend frontend pipeline; do
|
||||||
|
|||||||
@@ -1,560 +0,0 @@
|
|||||||
# GIAS OfficialSixthForm Flag Implementation Plan
|
|
||||||
|
|
||||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
|
||||||
|
|
||||||
**Goal:** Ingest GIAS's authoritative `OfficialSixthForm` flag into `marts.dim_school.has_sixth_form` and replace every `age_range contains "18"` heuristic in the backend and frontend with it.
|
|
||||||
|
|
||||||
**Architecture:** Data flows tap → raw → dbt staging → dbt mart → backend SQL → API payload → Next.js components. The GIAS Singer tap must declare the new CSV column (target-postgres only persists declared columns); the dbt staging model renames it; `dim_school` derives a boolean (with a statutory-age fallback for blank GIAS values); the backend exposes it on list + detail payloads and uses it for the `has_sixth_form=yes|no` filter; the frontend badge/note/filter-labels switch from the age-range substring check to the flag.
|
|
||||||
|
|
||||||
**Tech Stack:** Singer SDK (tap), dbt (Postgres), FastAPI + pandas, Next.js + TypeScript, pytest, Jest/RTL.
|
|
||||||
|
|
||||||
**Spec:** `docs/superpowers/specs/2026-07-07-exam-phase-taxonomy-design.md` §3.
|
|
||||||
|
|
||||||
## Global Constraints
|
|
||||||
|
|
||||||
- A school **has a sixth form** iff GIAS `OfficialSixthForm (name)` = `"Has a sixth form"`. `"Does not have a sixth form"` and `"Not applicable"` → false. Blank/NULL (rare) → fall back to `statutory_high_age >= 18`.
|
|
||||||
- The public API filter parameter stays `has_sixth_form=yes|no` (unchanged contract).
|
|
||||||
- Filter dropdown labels must drop the age-range parentheticals: "With sixth form" / "Without sixth form" (sixth form ≠ age range).
|
|
||||||
- Never push to `main`; work stays on branch `feat/gias-sixth-form-flag` (create from `docs/exam-phase-taxonomy` so the spec is included, or from `main` if that branch has merged).
|
|
||||||
- The dbt models cannot be run locally (no pipeline DB); dbt changes are verified by review + `python -c` schema asserts + existing CI. Do NOT attempt to start a local server.
|
|
||||||
- The backend marts tables are dbt `table` materializations — rebuilt on every pipeline run, so **no ALTER TABLE migration is needed** for `marts.dim_school`.
|
|
||||||
- Deployment ordering: the tap must run before dbt on the first pipeline run after deploy (this is already the DAG order: extract → transform). Until that run happens, `has_sixth_form` is absent from the DB; the backend must treat a missing column as "flag false / fallback", never crash.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 1: Ingest `OfficialSixthForm (name)` — tap schema + dbt staging
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py:31-66` (Singer schema)
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_gias_establishments.sql` (add renamed column)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: raw column `"OfficialSixthForm (name)"` in `raw.gias_establishments`; staging column `official_sixth_form` (text: `Has a sixth form` / `Does not have a sixth form` / `Not applicable` / NULL) consumed by Task 2.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add the property to the Singer schema**
|
|
||||||
|
|
||||||
In `tap.py`, inside `GIASEstablishmentsStream.schema = th.PropertiesList(...)`, add after the `th.Property("PhaseOfEducation (name)", th.StringType),` line:
|
|
||||||
|
|
||||||
```python
|
|
||||||
th.Property("OfficialSixthForm (name)", th.StringType),
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Verify the tap module still imports and declares the column**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
cd /Users/tudor/projects/school_compare/pipeline/plugins/extractors/tap-uk-gias && \
|
|
||||||
python3 -c "
|
|
||||||
import ast, sys
|
|
||||||
src = open('tap_uk_gias/tap.py').read()
|
|
||||||
ast.parse(src)
|
|
||||||
assert '\"OfficialSixthForm (name)\"' in src.replace(\"'\", '\"')
|
|
||||||
print('OK: tap declares OfficialSixthForm (name)')
|
|
||||||
"
|
|
||||||
```
|
|
||||||
Expected: `OK: tap declares OfficialSixthForm (name)`
|
|
||||||
(Uses `ast.parse` instead of importing because `singer_sdk` is not installed locally.)
|
|
||||||
|
|
||||||
- [ ] **Step 3: Add the column to the staging model**
|
|
||||||
|
|
||||||
In `stg_gias_establishments.sql`, in the `renamed` CTE, add after the `"PhaseOfEducation (name)" as phase,` line:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
nullif(trim("OfficialSixthForm (name)"), '') as official_sixth_form,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Sanity-check the SQL edit**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
grep -n "official_sixth_form" /Users/tudor/projects/school_compare/pipeline/transform/models/staging/stg_gias_establishments.sql
|
|
||||||
```
|
|
||||||
Expected: one line showing the new column inside the `renamed` CTE (before `from source`).
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py pipeline/transform/models/staging/stg_gias_establishments.sql
|
|
||||||
git commit -m "feat(pipeline): ingest GIAS OfficialSixthForm into staging
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 2: Derive `dim_school.has_sixth_form` (dbt mart + schema tests + SQLAlchemy model)
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/marts/dim_school.sql` (add derived column)
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (document + test the column)
|
|
||||||
- Modify: `backend/models.py:13-38` (`DimSchool` — add column)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `official_sixth_form` text column from Task 1's staging model.
|
|
||||||
- Produces: `marts.dim_school.has_sixth_form boolean not null`, and `DimSchool.has_sixth_form = Column(Boolean)` for the backend. Task 3 selects it as `s.has_sixth_form`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add the derived column to `dim_school.sql`**
|
|
||||||
|
|
||||||
In the `select`, add after the `s.age_range` line (`s.statutory_low_age || '-' || s.statutory_high_age as age_range,`):
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Authoritative sixth-form flag (spec §3): GIAS OfficialSixthForm.
|
|
||||||
-- "Not applicable" (nurseries, primaries, PRUs) => false. Blank GIAS
|
|
||||||
-- value (rare, new establishments) falls back to the statutory age range.
|
|
||||||
case
|
|
||||||
when s.official_sixth_form = 'Has a sixth form' then true
|
|
||||||
when s.official_sixth_form in ('Does not have a sixth form', 'Not applicable') then false
|
|
||||||
else coalesce(s.statutory_high_age >= 18, false)
|
|
||||||
end as has_sixth_form,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Add schema documentation + tests in `_marts_schema.yml`**
|
|
||||||
|
|
||||||
Under `- name: dim_school` → `columns:`, add after the `phase` column block:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: has_sixth_form
|
|
||||||
description: >
|
|
||||||
Authoritative sixth-form flag from GIAS OfficialSixthForm.
|
|
||||||
"Has a sixth form" => true; "Does not have a sixth form" and
|
|
||||||
"Not applicable" => false; blank GIAS value falls back to
|
|
||||||
statutory_high_age >= 18. Replaces the age_range-contains-"18"
|
|
||||||
heuristic (spec 2026-07-07 §3).
|
|
||||||
tests:
|
|
||||||
- not_null
|
|
||||||
- accepted_values:
|
|
||||||
values: [true, false]
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Add the column to the `DimSchool` SQLAlchemy model**
|
|
||||||
|
|
||||||
In `backend/models.py`, in `class DimSchool`, add after `age_range = Column(String(20))`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
has_sixth_form = Column(Boolean)
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Verify SQL/YAML/Python all parse**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
cd /Users/tudor/projects/school_compare && \
|
|
||||||
python3 -c "
|
|
||||||
import yaml
|
|
||||||
y = yaml.safe_load(open('pipeline/transform/models/marts/_marts_schema.yml'))
|
|
||||||
dim = [m for m in y['models'] if m['name'] == 'dim_school'][0]
|
|
||||||
cols = [c['name'] for c in dim['columns']]
|
|
||||||
assert 'has_sixth_form' in cols, cols
|
|
||||||
print('OK: schema yml documents has_sixth_form')
|
|
||||||
" && \
|
|
||||||
grep -c "has_sixth_form" pipeline/transform/models/marts/dim_school.sql && \
|
|
||||||
python3 -c "import ast; ast.parse(open('backend/models.py').read()); print('OK: models.py parses')"
|
|
||||||
```
|
|
||||||
Expected: `OK: schema yml documents has_sixth_form`, grep count `>= 1`, `OK: models.py parses`.
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/marts/dim_school.sql pipeline/transform/models/marts/_marts_schema.yml backend/models.py
|
|
||||||
git commit -m "feat(pipeline): derive dim_school.has_sixth_form from GIAS flag
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 3: Backend — expose `has_sixth_form` and replace the filter heuristic
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/data_loader.py:117-215` (`_MAIN_QUERY` — select the column)
|
|
||||||
- Modify: `backend/schemas.py:536-553` (`SCHOOL_COLUMNS` — include in list payloads)
|
|
||||||
- Modify: `backend/app.py:419-422` (filter) and `backend/app.py:589-610` (detail `school_info`)
|
|
||||||
- Test: `backend/tests/test_sixth_form_flag.py` (new)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `marts.dim_school.has_sixth_form` (Task 2).
|
|
||||||
- Produces: `has_sixth_form: bool | null` field on `GET /api/schools` items and on `GET /api/schools/{urn}` → `school_info`. Filter `GET /api/schools?has_sixth_form=yes|no` now driven by the flag. Frontend (Task 4) reads `school.has_sixth_form`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the failing tests**
|
|
||||||
|
|
||||||
Create `backend/tests/test_sixth_form_flag.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Tests for the GIAS-driven has_sixth_form flag (spec 2026-07-07 §3).
|
|
||||||
|
|
||||||
The filter and payloads must use dim_school.has_sixth_form, not the old
|
|
||||||
age_range-contains-"18" substring heuristic. The key regression case is a
|
|
||||||
16-19 sixth-form college: flag true, but "16-19" contains no "18".
|
|
||||||
"""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import pytest
|
|
||||||
from fastapi.testclient import TestClient
|
|
||||||
|
|
||||||
|
|
||||||
def _schools_df() -> pd.DataFrame:
|
|
||||||
"""Latest-year snapshot rows as produced by load_latest_school_data."""
|
|
||||||
base = {
|
|
||||||
"local_authority": "Testshire",
|
|
||||||
"school_type": "Academy",
|
|
||||||
"phase": "Secondary",
|
|
||||||
"address": "1 Test Street",
|
|
||||||
"town": "Testtown",
|
|
||||||
"postcode": "TS1 1AA",
|
|
||||||
"religious_denomination": None,
|
|
||||||
"gender": "Mixed",
|
|
||||||
"admissions_policy": None,
|
|
||||||
"ofsted_grade": np.nan,
|
|
||||||
"ofsted_date": None,
|
|
||||||
"ofsted_framework": None,
|
|
||||||
"latitude": 51.5,
|
|
||||||
"longitude": -0.1,
|
|
||||||
"year": 202425,
|
|
||||||
"total_pupils": 1000,
|
|
||||||
"rwm_expected_pct": np.nan,
|
|
||||||
"attainment_8_score": 50.0,
|
|
||||||
}
|
|
||||||
return pd.DataFrame(
|
|
||||||
[
|
|
||||||
# 11-18 school WITH a registered sixth form
|
|
||||||
{**base, "urn": 100001, "school_name": "Alpha High",
|
|
||||||
"age_range": "11-18", "has_sixth_form": True},
|
|
||||||
# 16-19 college: old heuristic said NO ("16-19" has no "18"),
|
|
||||||
# GIAS flag says YES — must appear in the yes-filter results
|
|
||||||
{**base, "urn": 100002, "school_name": "Beta Sixth Form College",
|
|
||||||
"age_range": "16-19", "has_sixth_form": True},
|
|
||||||
# 11-18 age range on paper but NO registered sixth form:
|
|
||||||
# old heuristic said YES, GIAS flag says NO
|
|
||||||
{**base, "urn": 100003, "school_name": "Gamma Academy",
|
|
||||||
"age_range": "11-18", "has_sixth_form": False},
|
|
||||||
# Missing flag (pipeline not yet re-run) — must not crash,
|
|
||||||
# must not match the yes-filter
|
|
||||||
{**base, "urn": 100004, "school_name": "Delta School",
|
|
||||||
"age_range": "11-16", "has_sixth_form": None},
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture()
|
|
||||||
def client(monkeypatch):
|
|
||||||
from backend import app as app_module
|
|
||||||
|
|
||||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
|
||||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
|
||||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
|
||||||
|
|
||||||
|
|
||||||
def _urns(resp):
|
|
||||||
return sorted(s["urn"] for s in resp.json()["schools"])
|
|
||||||
|
|
||||||
|
|
||||||
def test_filter_yes_uses_flag_not_age_range(client):
|
|
||||||
resp = client.get("/api/schools?has_sixth_form=yes")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
# 16-19 college included; 11-18-without-sixth-form excluded
|
|
||||||
assert _urns(resp) == [100001, 100002]
|
|
||||||
|
|
||||||
|
|
||||||
def test_filter_no_uses_flag_not_age_range(client):
|
|
||||||
resp = client.get("/api/schools?has_sixth_form=no")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
# Gamma (flag false) and Delta (flag missing => not true)
|
|
||||||
assert _urns(resp) == [100003, 100004]
|
|
||||||
|
|
||||||
|
|
||||||
def test_list_payload_includes_flag(client):
|
|
||||||
resp = client.get("/api/schools")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
by_urn = {s["urn"]: s for s in resp.json()["schools"]}
|
|
||||||
assert by_urn[100002]["has_sixth_form"] is True
|
|
||||||
assert by_urn[100003]["has_sixth_form"] is False
|
|
||||||
assert by_urn[100004]["has_sixth_form"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_detail_payload_includes_flag(client):
|
|
||||||
resp = client.get("/api/schools/100002")
|
|
||||||
assert resp.status_code == 200, resp.text
|
|
||||||
assert resp.json()["school_info"]["has_sixth_form"] is True
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Run tests to verify they fail**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && python3 -m pytest backend/tests/test_sixth_form_flag.py -v`
|
|
||||||
Expected: FAIL — `test_filter_yes_uses_flag_not_age_range` asserts `[100001, 100002]` but the age-range heuristic returns `[100001, 100003]`; the payload tests fail with `KeyError: 'has_sixth_form'`.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Select the column in `_MAIN_QUERY`**
|
|
||||||
|
|
||||||
In `backend/data_loader.py`, in `_MAIN_QUERY`, add after `s.age_range,`:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
s.has_sixth_form,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Include it in list payloads**
|
|
||||||
|
|
||||||
In `backend/schemas.py`, in `SCHOOL_COLUMNS`, add after `"age_range",`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"has_sixth_form",
|
|
||||||
```
|
|
||||||
|
|
||||||
(`app.py` builds list responses from `SCHOOL_COLUMNS ∩ df.columns`, so a DB that predates the pipeline re-run simply omits the field — no crash.)
|
|
||||||
|
|
||||||
- [ ] **Step 5: Replace the filter heuristic in `app.py`**
|
|
||||||
|
|
||||||
Replace lines 419-422:
|
|
||||||
|
|
||||||
```python
|
|
||||||
if has_sixth_form == "yes":
|
|
||||||
df_latest = df_latest[df_latest["age_range"].str.contains("18", na=False)]
|
|
||||||
elif has_sixth_form == "no":
|
|
||||||
df_latest = df_latest[~df_latest["age_range"].str.contains("18", na=False)]
|
|
||||||
```
|
|
||||||
|
|
||||||
with:
|
|
||||||
|
|
||||||
```python
|
|
||||||
# GIAS OfficialSixthForm flag (dim_school.has_sixth_form). NULL (flag not
|
|
||||||
# yet populated by the pipeline) is treated as "no sixth form".
|
|
||||||
if has_sixth_form in ("yes", "no"):
|
|
||||||
if "has_sixth_form" in df_latest.columns:
|
|
||||||
flag = df_latest["has_sixth_form"].eq(True)
|
|
||||||
else: # DB predates the pipeline re-run — fall back to age range
|
|
||||||
flag = df_latest["age_range"].str.contains("18", na=False)
|
|
||||||
df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag]
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 6: Add the flag to the detail payload**
|
|
||||||
|
|
||||||
In `backend/app.py` `school_info` dict (line ~598), add after `"age_range": latest.get("age_range", ""),`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"has_sixth_form": latest.get("has_sixth_form"),
|
|
||||||
```
|
|
||||||
|
|
||||||
(`convert_to_native` already maps NaN/None → null and numpy bools → bool.)
|
|
||||||
|
|
||||||
- [ ] **Step 7: Run the new tests**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && python3 -m pytest backend/tests/test_sixth_form_flag.py -v`
|
|
||||||
Expected: 4 passed.
|
|
||||||
|
|
||||||
- [ ] **Step 8: Run the full backend suite**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && python3 -m pytest backend/tests -v`
|
|
||||||
Expected: all pass (the pre-existing `test_school_details.py` df has no `has_sixth_form` column — `latest.get()` returns None, serialized as null).
|
|
||||||
|
|
||||||
- [ ] **Step 9: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add backend/data_loader.py backend/schemas.py backend/app.py backend/tests/test_sixth_form_flag.py
|
|
||||||
git commit -m "feat(api): drive has_sixth_form filter and payloads from GIAS flag
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 4: Frontend — badge, note, row tag, and filter labels use the flag
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `nextjs-app/lib/types.ts:10-30` (`School` interface)
|
|
||||||
- Modify: `nextjs-app/components/SecondarySchoolDetailView.tsx:101` (badge + coming-soon note)
|
|
||||||
- Modify: `nextjs-app/components/SecondarySchoolRow.tsx:25-27` (row tag)
|
|
||||||
- Modify: `nextjs-app/components/FilterBar.tsx:370-372` (labels only — param name unchanged)
|
|
||||||
- Test: `nextjs-app/__tests__/components/SecondarySchoolRow.test.tsx` (new)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `has_sixth_form: boolean | null` on both list items and `school_info` (Task 3; both are typed as `School`).
|
|
||||||
- Produces: no new exports — behavior change only.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the failing test**
|
|
||||||
|
|
||||||
Create `nextjs-app/__tests__/components/SecondarySchoolRow.test.tsx`:
|
|
||||||
|
|
||||||
```tsx
|
|
||||||
/**
|
|
||||||
* SecondarySchoolRow — sixth-form tag must come from the GIAS
|
|
||||||
* has_sixth_form flag, not the age_range-contains-"18" heuristic.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import '@testing-library/jest-dom';
|
|
||||||
import { render, screen } from '@testing-library/react';
|
|
||||||
import { SecondarySchoolRow } from '@/components/SecondarySchoolRow';
|
|
||||||
import type { School } from '@/lib/types';
|
|
||||||
|
|
||||||
const base = {
|
|
||||||
urn: 100002,
|
|
||||||
school_name: 'Beta Sixth Form College',
|
|
||||||
local_authority: 'Testshire',
|
|
||||||
school_type: 'Academy',
|
|
||||||
phase: 'Secondary',
|
|
||||||
gender: 'Mixed',
|
|
||||||
attainment_8_score: 50.0,
|
|
||||||
} as unknown as School;
|
|
||||||
|
|
||||||
describe('SecondarySchoolRow sixth-form tag', () => {
|
|
||||||
it('shows the tag for a 16-19 college with the GIAS flag set', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow
|
|
||||||
school={{ ...base, age_range: '16-19', has_sixth_form: true }}
|
|
||||||
/>,
|
|
||||||
);
|
|
||||||
expect(screen.getByText('Sixth form')).toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag for an 11-18 school without a registered sixth form', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow
|
|
||||||
school={{ ...base, age_range: '11-18', has_sixth_form: false }}
|
|
||||||
/>,
|
|
||||||
);
|
|
||||||
expect(screen.queryByText('Sixth form')).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag when the flag is missing (pipeline not yet re-run)', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow school={{ ...base, age_range: '11-18' }} />,
|
|
||||||
);
|
|
||||||
expect(screen.queryByText('Sixth form')).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Run it to verify it fails**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare/nextjs-app && npx jest __tests__/components/SecondarySchoolRow.test.tsx`
|
|
||||||
Expected: FAIL — first test can't find "Sixth form" ("16-19" fails the substring check), second test finds an unexpected "Sixth form" tag. (If TS complains that `has_sixth_form` is not on `School`, that is the same failure — proceed.)
|
|
||||||
|
|
||||||
- [ ] **Step 3: Add the field to the `School` type**
|
|
||||||
|
|
||||||
In `nextjs-app/lib/types.ts`, in `export interface School`, add after `age_range: string | null;`:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
has_sixth_form?: boolean | null;
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Switch `SecondarySchoolRow` to the flag**
|
|
||||||
|
|
||||||
Replace the helper at `SecondarySchoolRow.tsx:25-27`:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
function hasSixthForm(school: School): boolean {
|
|
||||||
return school.age_range?.includes('18') ?? false;
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
with:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
function hasSixthForm(school: School): boolean {
|
|
||||||
// GIAS OfficialSixthForm flag; missing (pipeline not yet re-run) => false.
|
|
||||||
return school.has_sixth_form ?? false;
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 5: Switch `SecondarySchoolDetailView` to the flag**
|
|
||||||
|
|
||||||
Replace line 101:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
const hasSixthForm = schoolInfo.age_range?.includes('18') ?? false;
|
|
||||||
```
|
|
||||||
|
|
||||||
with:
|
|
||||||
|
|
||||||
```ts
|
|
||||||
// GIAS OfficialSixthForm flag; missing (pipeline not yet re-run) => false.
|
|
||||||
const hasSixthForm = schoolInfo.has_sixth_form ?? false;
|
|
||||||
```
|
|
||||||
|
|
||||||
(This drives both the header "Sixth form" badge at line ~230 and the "Post-16 destination data coming soon" note at line ~715 — no changes needed there.)
|
|
||||||
|
|
||||||
- [ ] **Step 6: Fix the filter labels in `FilterBar.tsx`**
|
|
||||||
|
|
||||||
Replace:
|
|
||||||
|
|
||||||
```tsx
|
|
||||||
<option value="yes">With sixth form (11-18)</option>
|
|
||||||
<option value="no">Without sixth form (11-16)</option>
|
|
||||||
```
|
|
||||||
|
|
||||||
with:
|
|
||||||
|
|
||||||
```tsx
|
|
||||||
<option value="yes">With sixth form</option>
|
|
||||||
<option value="no">Without sixth form</option>
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 7: Run the new test and verify it passes**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare/nextjs-app && npx jest __tests__/components/SecondarySchoolRow.test.tsx`
|
|
||||||
Expected: 3 passed.
|
|
||||||
|
|
||||||
- [ ] **Step 8: Run the full frontend checks**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare/nextjs-app && npx tsc --noEmit && npx jest`
|
|
||||||
Expected: typecheck clean, all Jest suites pass.
|
|
||||||
|
|
||||||
- [ ] **Step 9: Verify no heuristic remains**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
grep -rn "includes('18')\|contains(\"18\")" /Users/tudor/projects/school_compare/nextjs-app/components /Users/tudor/projects/school_compare/backend --include="*.tsx" --include="*.ts" --include="*.py" | grep -v test
|
|
||||||
```
|
|
||||||
Expected: only the documented fallback inside `app.py` (DB-predates-pipeline branch); no other hits.
|
|
||||||
|
|
||||||
- [ ] **Step 10: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add nextjs-app/lib/types.ts nextjs-app/components/SecondarySchoolRow.tsx nextjs-app/components/SecondarySchoolDetailView.tsx nextjs-app/components/FilterBar.tsx nextjs-app/__tests__/components/SecondarySchoolRow.test.tsx
|
|
||||||
git commit -m "feat(ui): sixth-form badge, note and filter labels use GIAS flag
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 5: Update the spec status + PR
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `docs/superpowers/specs/2026-07-07-exam-phase-taxonomy-design.md` (§3 "Pipeline change (future work)" → implemented)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: everything above merged into the branch.
|
|
||||||
- Produces: PR ready for review; e2e journeys are the promotion gate (no journey currently exercises the sixth-form filter, and the API contract is unchanged, so no e2e change is required — state this in the PR body).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Mark spec §3 pipeline change as implemented**
|
|
||||||
|
|
||||||
In the spec, change the §3 heading `### Pipeline change (future work)` to `### Pipeline change (implemented 2026-07-07)` and append one line at the end of that subsection:
|
|
||||||
|
|
||||||
```markdown
|
|
||||||
Implemented in `feat/gias-sixth-form-flag` — see
|
|
||||||
`docs/superpowers/plans/2026-07-07-gias-sixth-form-flag.md`.
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add docs/superpowers/specs/2026-07-07-exam-phase-taxonomy-design.md
|
|
||||||
git commit -m "docs: mark sixth-form flag pipeline change implemented
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Push and open the PR (Gitea)**
|
|
||||||
|
|
||||||
Push the branch, then create the PR against `main` using the Gitea API via the git credential helper (token-header auth 401s on this Gitea; basic auth from `git credential fill` works):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git push -u origin feat/gias-sixth-form-flag
|
|
||||||
```
|
|
||||||
|
|
||||||
PR title: `feat: drive sixth-form separation from GIAS OfficialSixthForm flag`
|
|
||||||
PR body must note: (1) API contract unchanged (`has_sixth_form=yes|no`), (2) flag is NULL until the next pipeline run — backend and frontend degrade to "no sixth form" / age-range fallback, (3) no e2e journey change needed, and end with the standard generation footer.
|
|
||||||
|
|
||||||
- [ ] **Step 4: Verify CI passes**
|
|
||||||
|
|
||||||
Watch the PR checks (typecheck, tests, builds, AI review). All must pass before merge; merging deploys to staging automatically.
|
|
||||||
@@ -1,799 +0,0 @@
|
|||||||
# GIAS Code Dictionaries Implementation Plan
|
|
||||||
|
|
||||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
|
||||||
|
|
||||||
**Goal:** Store the six GIAS classification fields as official DfE integer codes in the marts and translate code → name in application code, leaving the API contract (name strings) unchanged.
|
|
||||||
|
|
||||||
**Architecture:** A generation script downloads the public GIAS bulk CSV and emits the dictionaries (Python dicts + a dbt seed) from real data. The tap ingests the `(code)` columns, staging casts them, `dim_school`/`dim_location` keep only codes, and translation happens in exactly two places: `backend/data_loader.py` right after `pd.read_sql`, and `pipeline/scripts/sync_typesense.py` before indexing. A dbt seed test warns when DfE adds/renames a value; a parity test keeps the backend and pipeline dictionary copies identical.
|
|
||||||
|
|
||||||
**Tech Stack:** Singer SDK tap, dbt (Postgres), FastAPI + pandas, Typesense sync script, pytest.
|
|
||||||
|
|
||||||
**Spec:** `docs/superpowers/specs/2026-07-09-gias-code-dictionaries-design.md`
|
|
||||||
|
|
||||||
## Global Constraints
|
|
||||||
|
|
||||||
- **Numeric code values are never assumed.** Every literal code used in SQL or yml (status filter, sixth-form derivation, phase cascade) must be verified against `pipeline/transform/seeds/gias_code_names.csv` generated in Task 1 from the live CSV. The literals written in this plan are best-current-knowledge and each carries a verification step.
|
|
||||||
- **Names served by the API must stay byte-identical** to today's strings (e.g. `Does not apply`, `Open, but proposed to close`) — UI heuristics compare exact strings.
|
|
||||||
- The `(name)` columns stay declared in the tap and present in raw; staging stops exposing them.
|
|
||||||
- `dim_school` and `dim_location` status filters must stay identical (API inner-joins them).
|
|
||||||
- Backend tests run via: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests -v` (no local pytest exists).
|
|
||||||
- dbt cannot run locally — dbt changes are verified statically (grep / yaml parse) + CI.
|
|
||||||
- Never push to `main`. Work on branch `feat/gias-code-dictionaries` (branch off `docs/gias-code-dictionaries` so the spec is included, or off `main` if that has merged).
|
|
||||||
- Commits end with: `Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>`
|
|
||||||
- Deploy runbook (accepted window, spec §7): merge → deploy → trigger `school_data_daily` immediately. No code-level fallback for the old-schema window.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 1: Dictionary generation script, canonical module, pipeline copy, seed
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Create: `pipeline/scripts/generate_gias_codes.py`
|
|
||||||
- Create: `backend/gias_codes.py` (content generated by the script)
|
|
||||||
- Create: `pipeline/scripts/gias_codes.py` (byte-identical copy)
|
|
||||||
- Create: `pipeline/transform/seeds/gias_code_names.csv` (generated)
|
|
||||||
- Test: `backend/tests/test_gias_codes.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: `backend/gias_codes.py` exporting `SCHOOL_TYPE`, `ESTABLISHMENT_STATUS`, `PHASE_OF_EDUCATION`, `OFFICIAL_SIXTH_FORM`, `RELIGIOUS_CHARACTER`, `ADMISSIONS_POLICY` (each `dict[int, str]`) and `translate(code, mapping) -> str | None`. Task 4 imports these; Task 5 imports the pipeline copy; Task 3 reads code literals from the seed CSV.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the failing tests**
|
|
||||||
|
|
||||||
Create `backend/tests/test_gias_codes.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Tests for the GIAS code->name dictionaries (spec 2026-07-09).
|
|
||||||
|
|
||||||
The dictionaries are generated from the live GIAS bulk CSV by
|
|
||||||
pipeline/scripts/generate_gias_codes.py — these tests assert the module's
|
|
||||||
contract, key sentinel values the marts/UI depend on, and that the pipeline
|
|
||||||
copy has not drifted from the canonical backend module.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import math
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from backend.gias_codes import (
|
|
||||||
ADMISSIONS_POLICY,
|
|
||||||
ESTABLISHMENT_STATUS,
|
|
||||||
OFFICIAL_SIXTH_FORM,
|
|
||||||
PHASE_OF_EDUCATION,
|
|
||||||
RELIGIOUS_CHARACTER,
|
|
||||||
SCHOOL_TYPE,
|
|
||||||
translate,
|
|
||||||
)
|
|
||||||
|
|
||||||
REPO = Path(__file__).resolve().parents[2]
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_known_code():
|
|
||||||
open_code = next(c for c, n in ESTABLISHMENT_STATUS.items() if n == "Open")
|
|
||||||
assert translate(open_code, ESTABLISHMENT_STATUS) == "Open"
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_unknown_code_degrades_gracefully():
|
|
||||||
assert translate(9999, ESTABLISHMENT_STATUS) == "Unknown (9999)"
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_none_and_nan_return_none():
|
|
||||||
assert translate(None, ESTABLISHMENT_STATUS) is None
|
|
||||||
assert translate(float("nan"), ESTABLISHMENT_STATUS) is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_translate_accepts_float_codes():
|
|
||||||
# pd.read_sql yields float columns when NULLs are present
|
|
||||||
open_code = next(c for c, n in ESTABLISHMENT_STATUS.items() if n == "Open")
|
|
||||||
assert translate(float(open_code), ESTABLISHMENT_STATUS) == "Open"
|
|
||||||
|
|
||||||
|
|
||||||
def test_sentinel_names_present():
|
|
||||||
"""Names the marts/UI compare against must exist verbatim."""
|
|
||||||
assert "Open" in ESTABLISHMENT_STATUS.values()
|
|
||||||
assert "Open, but proposed to close" in ESTABLISHMENT_STATUS.values()
|
|
||||||
assert "Has a sixth form" in OFFICIAL_SIXTH_FORM.values()
|
|
||||||
assert "Primary" in PHASE_OF_EDUCATION.values()
|
|
||||||
assert "Secondary" in PHASE_OF_EDUCATION.values()
|
|
||||||
assert "Does not apply" in RELIGIOUS_CHARACTER.values()
|
|
||||||
assert all(len(d) > 0 for d in (
|
|
||||||
SCHOOL_TYPE, ESTABLISHMENT_STATUS, PHASE_OF_EDUCATION,
|
|
||||||
OFFICIAL_SIXTH_FORM, RELIGIOUS_CHARACTER, ADMISSIONS_POLICY,
|
|
||||||
))
|
|
||||||
|
|
||||||
|
|
||||||
def test_pipeline_copy_is_identical():
|
|
||||||
canonical = (REPO / "backend" / "gias_codes.py").read_text()
|
|
||||||
copy = (REPO / "pipeline" / "scripts" / "gias_codes.py").read_text()
|
|
||||||
assert canonical == copy, (
|
|
||||||
"pipeline/scripts/gias_codes.py has drifted from backend/gias_codes.py — "
|
|
||||||
"regenerate with pipeline/scripts/generate_gias_codes.py and copy the file"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_seed_matches_dictionaries():
|
|
||||||
import csv
|
|
||||||
fields = {
|
|
||||||
"school_type": SCHOOL_TYPE,
|
|
||||||
"establishment_status": ESTABLISHMENT_STATUS,
|
|
||||||
"phase_of_education": PHASE_OF_EDUCATION,
|
|
||||||
"official_sixth_form": OFFICIAL_SIXTH_FORM,
|
|
||||||
"religious_character": RELIGIOUS_CHARACTER,
|
|
||||||
"admissions_policy": ADMISSIONS_POLICY,
|
|
||||||
}
|
|
||||||
seed_path = REPO / "pipeline" / "transform" / "seeds" / "gias_code_names.csv"
|
|
||||||
seed: dict[str, dict[int, str]] = {k: {} for k in fields}
|
|
||||||
with open(seed_path, newline="") as fh:
|
|
||||||
for row in csv.DictReader(fh):
|
|
||||||
seed[row["field"]][int(row["code"])] = row["name"]
|
|
||||||
assert seed == fields
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Run tests to verify they fail**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_gias_codes.py -v`
|
|
||||||
Expected: FAIL at import — `ModuleNotFoundError: No module named 'backend.gias_codes'`.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Write the generation script**
|
|
||||||
|
|
||||||
Create `pipeline/scripts/generate_gias_codes.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Generate GIAS code->name dictionaries from the live bulk CSV.
|
|
||||||
|
|
||||||
Writes:
|
|
||||||
- backend/gias_codes.py (canonical Python module)
|
|
||||||
- pipeline/scripts/gias_codes.py (byte-identical copy)
|
|
||||||
- pipeline/transform/seeds/gias_code_names.csv (dbt seed for drift test)
|
|
||||||
|
|
||||||
Run from the repo root whenever the dbt drift test warns that DfE
|
|
||||||
added/renamed a value: python pipeline/scripts/generate_gias_codes.py
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import io
|
|
||||||
import sys
|
|
||||||
from datetime import date, timedelta
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
|
|
||||||
GIAS_URL = (
|
|
||||||
"https://ea-edubase-api-prod.azurewebsites.net"
|
|
||||||
"/edubase/downloads/public/edubasealldata{date}.csv"
|
|
||||||
)
|
|
||||||
|
|
||||||
# (CSV code column, CSV name column, python dict name, seed field key)
|
|
||||||
FIELDS = [
|
|
||||||
("TypeOfEstablishment (code)", "TypeOfEstablishment (name)", "SCHOOL_TYPE", "school_type"),
|
|
||||||
("EstablishmentStatus (code)", "EstablishmentStatus (name)", "ESTABLISHMENT_STATUS", "establishment_status"),
|
|
||||||
("PhaseOfEducation (code)", "PhaseOfEducation (name)", "PHASE_OF_EDUCATION", "phase_of_education"),
|
|
||||||
("OfficialSixthForm (code)", "OfficialSixthForm (name)", "OFFICIAL_SIXTH_FORM", "official_sixth_form"),
|
|
||||||
("ReligiousCharacter (code)", "ReligiousCharacter (name)", "RELIGIOUS_CHARACTER", "religious_character"),
|
|
||||||
("AdmissionsPolicy (code)", "AdmissionsPolicy (name)", "ADMISSIONS_POLICY", "admissions_policy"),
|
|
||||||
]
|
|
||||||
|
|
||||||
MODULE_HEADER = '''"""GIAS code -> name dictionaries.
|
|
||||||
|
|
||||||
GENERATED by pipeline/scripts/generate_gias_codes.py from the GIAS bulk CSV
|
|
||||||
— do not edit by hand; rerun the script when the dbt drift test warns.
|
|
||||||
The canonical file is backend/gias_codes.py; pipeline/scripts/gias_codes.py
|
|
||||||
must be byte-identical (enforced by backend/tests/test_gias_codes.py).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import math
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
'''
|
|
||||||
|
|
||||||
MODULE_FOOTER = '''
|
|
||||||
|
|
||||||
def translate(code, mapping: dict[int, str]) -> str | None:
|
|
||||||
"""Translate a GIAS code to its display name.
|
|
||||||
|
|
||||||
None/NaN -> None (column absent or suppressed). Unknown codes degrade to
|
|
||||||
"Unknown (<code>)" with a warning so a new DfE value never blanks the UI.
|
|
||||||
"""
|
|
||||||
if code is None or (isinstance(code, float) and math.isnan(code)):
|
|
||||||
return None
|
|
||||||
code = int(code)
|
|
||||||
if code not in mapping:
|
|
||||||
logger.warning("Unknown GIAS code %s (not in dictionary)", code)
|
|
||||||
return f"Unknown ({code})"
|
|
||||||
return mapping[code]
|
|
||||||
'''
|
|
||||||
|
|
||||||
|
|
||||||
def download_csv() -> pd.DataFrame:
|
|
||||||
for day in (date.today(), date.today() - timedelta(days=1)):
|
|
||||||
url = GIAS_URL.format(date=day.strftime("%Y%m%d"))
|
|
||||||
print(f"Downloading {url}")
|
|
||||||
resp = requests.get(url, timeout=300)
|
|
||||||
if resp.status_code == 404:
|
|
||||||
continue
|
|
||||||
resp.raise_for_status()
|
|
||||||
return pd.read_csv(
|
|
||||||
io.StringIO(resp.content.decode("latin-1")),
|
|
||||||
dtype=str, keep_default_na=False,
|
|
||||||
)
|
|
||||||
sys.exit("GIAS CSV not available for today or yesterday")
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
|
||||||
repo = Path(__file__).resolve().parents[2]
|
|
||||||
df = download_csv()
|
|
||||||
|
|
||||||
module_parts = [MODULE_HEADER]
|
|
||||||
seed_rows: list[tuple[str, int, str]] = []
|
|
||||||
|
|
||||||
for code_col, name_col, dict_name, field_key in FIELDS:
|
|
||||||
pairs = (
|
|
||||||
df[[code_col, name_col]]
|
|
||||||
.loc[lambda d: (d[code_col] != "") & (d[name_col] != "")]
|
|
||||||
.drop_duplicates()
|
|
||||||
)
|
|
||||||
mapping = sorted((int(c), n) for c, n in pairs.itertuples(index=False))
|
|
||||||
dupes = len(mapping) - len({c for c, _ in mapping})
|
|
||||||
if dupes:
|
|
||||||
sys.exit(f"{code_col}: {dupes} codes map to multiple names — investigate before generating")
|
|
||||||
lines = [f"{dict_name}: dict[int, str] = {{"]
|
|
||||||
for code, name in mapping:
|
|
||||||
escaped = name.replace('"', '\\"')
|
|
||||||
lines.append(f' {code}: "{escaped}",')
|
|
||||||
lines.append("}\n")
|
|
||||||
module_parts.append("\n".join(lines))
|
|
||||||
seed_rows += [(field_key, code, name) for code, name in mapping]
|
|
||||||
|
|
||||||
module = "\n".join(module_parts) + MODULE_FOOTER
|
|
||||||
|
|
||||||
(repo / "backend" / "gias_codes.py").write_text(module)
|
|
||||||
(repo / "pipeline" / "scripts" / "gias_codes.py").write_text(module)
|
|
||||||
|
|
||||||
seed_path = repo / "pipeline" / "transform" / "seeds" / "gias_code_names.csv"
|
|
||||||
with open(seed_path, "w", newline="") as fh:
|
|
||||||
import csv
|
|
||||||
w = csv.writer(fh)
|
|
||||||
w.writerow(["field", "code", "name"])
|
|
||||||
w.writerows(seed_rows)
|
|
||||||
|
|
||||||
print(f"Wrote backend/gias_codes.py, pipeline/scripts/gias_codes.py, {seed_path.name}")
|
|
||||||
print("\nKey codes for the dbt work (Task 3):")
|
|
||||||
for field in ("establishment_status", "phase_of_education", "official_sixth_form"):
|
|
||||||
print(f" {field}:")
|
|
||||||
for f, code, name in seed_rows:
|
|
||||||
if f == field:
|
|
||||||
print(f" {code} = {name}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Run the generator**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && uv run --with pandas --with requests python pipeline/scripts/generate_gias_codes.py`
|
|
||||||
Expected: downloads the CSV (~100MB, may take a minute), writes the three files, and prints the status/phase/sixth-form code tables. **Record the printed code tables — Task 3 needs them.** If the download fails twice, report BLOCKED (no network or GIAS outage) rather than inventing dictionary content.
|
|
||||||
|
|
||||||
- [ ] **Step 5: Run the tests again**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_gias_codes.py -v`
|
|
||||||
Expected: 7 passed. If `test_sentinel_names_present` fails, the GIAS vocabulary differs from expectations — inspect the generated module and report DONE_WITH_CONCERNS naming the differing value; do not edit the generated names.
|
|
||||||
|
|
||||||
- [ ] **Step 6: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/scripts/generate_gias_codes.py backend/gias_codes.py pipeline/scripts/gias_codes.py pipeline/transform/seeds/gias_code_names.csv backend/tests/test_gias_codes.py
|
|
||||||
git commit -m "feat: GIAS code->name dictionaries generated from live bulk CSV
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 2: Tap ingests the (code) columns; staging exposes codes, drops names
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py` (Singer schema)
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_gias_establishments.sql`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: staging columns `school_type_code`, `status_code`, `phase_code`, `official_sixth_form_code`, `religious_character_code`, `admissions_policy_code` (all int) consumed by Task 3. Staging **stops exposing** `school_type`, `status`, `phase`, `official_sixth_form`, `religious_character`, `admissions_policy` (names stay in raw only).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add the six (code) properties to the Singer schema**
|
|
||||||
|
|
||||||
In `tap.py`, `GIASEstablishmentsStream.schema`, add each `(code)` property directly above its existing `(name)` sibling:
|
|
||||||
|
|
||||||
```python
|
|
||||||
th.Property("TypeOfEstablishment (code)", th.StringType),
|
|
||||||
th.Property("PhaseOfEducation (code)", th.StringType),
|
|
||||||
th.Property("EstablishmentStatus (code)", th.StringType),
|
|
||||||
th.Property("Gender (name)", ...) # existing line — for placement reference only
|
|
||||||
th.Property("ReligiousCharacter (code)", th.StringType),
|
|
||||||
th.Property("AdmissionsPolicy (code)", th.StringType),
|
|
||||||
th.Property("OfficialSixthForm (code)", th.StringType),
|
|
||||||
```
|
|
||||||
|
|
||||||
(The exact insertion order doesn't matter — the schema is a dict — but keep each `(code)` adjacent to its `(name)` for readability. Do NOT remove any `(name)` property.)
|
|
||||||
|
|
||||||
- [ ] **Step 2: Rewrite the six columns in staging**
|
|
||||||
|
|
||||||
In `stg_gias_establishments.sql` `renamed` CTE, replace:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
"TypeOfEstablishment (name)" as school_type,
|
|
||||||
"PhaseOfEducation (name)" as phase,
|
|
||||||
nullif(trim("OfficialSixthForm (name)"), '') as official_sixth_form,
|
|
||||||
"ReligiousCharacter (name)" as religious_character,
|
|
||||||
"AdmissionsPolicy (name)" as admissions_policy,
|
|
||||||
"EstablishmentStatus (name)" as status,
|
|
||||||
```
|
|
||||||
|
|
||||||
with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
cast(nullif(trim("TypeOfEstablishment (code)"), '') as integer) as school_type_code,
|
|
||||||
cast(nullif(trim("PhaseOfEducation (code)"), '') as integer) as phase_code,
|
|
||||||
cast(nullif(trim("OfficialSixthForm (code)"), '') as integer) as official_sixth_form_code,
|
|
||||||
cast(nullif(trim("ReligiousCharacter (code)"), '') as integer) as religious_character_code,
|
|
||||||
cast(nullif(trim("AdmissionsPolicy (code)"), '') as integer) as admissions_policy_code,
|
|
||||||
cast(nullif(trim("EstablishmentStatus (code)"), '') as integer) as status_code,
|
|
||||||
```
|
|
||||||
|
|
||||||
(The name lines are scattered through the CTE — replace each in place; the six name aliases must no longer appear in the model.)
|
|
||||||
|
|
||||||
- [ ] **Step 3: Verify statically**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
cd /Users/tudor/projects/school_compare && \
|
|
||||||
python3 -c "import ast; ast.parse(open('pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py').read()); print('tap OK')" && \
|
|
||||||
grep -c "(code)" pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py && \
|
|
||||||
grep -E "as (school_type|status|phase|official_sixth_form|religious_character|admissions_policy)," pipeline/transform/models/staging/stg_gias_establishments.sql; echo "name-alias grep exit=$? (want 1 = none found)"
|
|
||||||
```
|
|
||||||
Expected: `tap OK`, code-column count `6`, and the final grep finds nothing (exit 1).
|
|
||||||
|
|
||||||
- [ ] **Step 4: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py pipeline/transform/models/staging/stg_gias_establishments.sql
|
|
||||||
git commit -m "feat(pipeline): ingest GIAS code columns; staging exposes codes not names
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 3: Marts store codes; dbt tests + drift test
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/marts/dim_school.sql`
|
|
||||||
- Modify: `pipeline/transform/models/marts/dim_location.sql`
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml`
|
|
||||||
- Create: `pipeline/transform/tests/assert_gias_code_names_match_seed.sql`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: staging code columns from Task 2; code literals from `pipeline/transform/seeds/gias_code_names.csv` (Task 1).
|
|
||||||
- Produces: `dim_school` columns `school_type_code`, `status_code`, `phase_code`, `religious_character_code`, `admissions_policy_code` (int) replacing their string columns; `has_sixth_form` unchanged (bool). Task 4's `_MAIN_QUERY` selects these.
|
|
||||||
|
|
||||||
**Before writing SQL: open `pipeline/transform/seeds/gias_code_names.csv` and confirm the literals below.** Best-current-knowledge values (VERIFY EACH):
|
|
||||||
`establishment_status`: 1 = Open, 3 = "Open, but proposed to close" (2 = Closed, 4 = Proposed to open).
|
|
||||||
`phase_of_education`: 0 = Not applicable, 2 = Primary, 4 = Secondary, 7 = All-through.
|
|
||||||
`official_sixth_form`: 1 = Has a sixth form, 2 = Does not have a sixth form, 0 = Not applicable.
|
|
||||||
If any differ, use the seed's values everywhere below and say so in your report.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Rewrite dim_school.sql derivations in code space**
|
|
||||||
|
|
||||||
Replace the phase cascade block (`case ... end as phase,`) with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Phase in GIAS code space (see seeds/gias_code_names.csv):
|
|
||||||
-- 2 = Primary, 4 = Secondary, 7 = All-through, 0 = Not applicable.
|
|
||||||
case
|
|
||||||
-- 1. Trust GIAS phase when it's a real value (0 = the catch-all "Not Applicable")
|
|
||||||
when s.phase_code is not null and s.phase_code != 0
|
|
||||||
then s.phase_code
|
|
||||||
-- 2. Infer from statutory age range (independent schools still publish these)
|
|
||||||
when s.statutory_high_age is not null and s.statutory_high_age <= 11 then 2
|
|
||||||
when s.statutory_low_age is not null and s.statutory_low_age >= 11 then 4
|
|
||||||
when s.statutory_low_age is not null and s.statutory_high_age is not null
|
|
||||||
and s.statutory_low_age < 11 and s.statutory_high_age > 11 then 7
|
|
||||||
-- 3. Fallback: infer from school name (covers independents with missing ages)
|
|
||||||
when s.school_name ilike '%primary%'
|
|
||||||
or s.school_name ilike '%infant%'
|
|
||||||
or s.school_name ilike '%junior%'
|
|
||||||
or s.school_name ilike '%preparatory%'
|
|
||||||
or s.school_name ilike '% prep school%'
|
|
||||||
or s.school_name ilike '% prep %'
|
|
||||||
then 2
|
|
||||||
when s.school_name ilike '%secondary%'
|
|
||||||
or s.school_name ilike '%high school%'
|
|
||||||
or s.school_name ilike '%grammar%'
|
|
||||||
or s.school_name ilike '%senior school%'
|
|
||||||
or s.school_name ilike '%upper school%'
|
|
||||||
then 4
|
|
||||||
-- 4. Give up — null renders no phase pill
|
|
||||||
else null
|
|
||||||
end as phase_code,
|
|
||||||
```
|
|
||||||
|
|
||||||
Replace `s.school_type,` with `s.school_type_code,`; `s.religious_character,` with `s.religious_character_code,`; `s.admissions_policy,` with `s.admissions_policy_code,`; `s.status,` with `s.status_code,`.
|
|
||||||
|
|
||||||
Replace the has_sixth_form case with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- GIAS OfficialSixthForm in code space: 1 = has, 2 = does not, 0 = N/A.
|
|
||||||
-- Null (rare, new establishments) falls back to the statutory age range.
|
|
||||||
case
|
|
||||||
when s.official_sixth_form_code = 1 then true
|
|
||||||
when s.official_sixth_form_code in (0, 2) then false
|
|
||||||
else coalesce(s.statutory_high_age >= 18, false)
|
|
||||||
end as has_sixth_form,
|
|
||||||
```
|
|
||||||
|
|
||||||
Replace the status filter with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- 1 = Open; 3 = Open, but proposed to close (still operating; drops out when
|
|
||||||
-- GIAS flips to Closed — marts fully rebuild each run).
|
|
||||||
where s.status_code in (1, 3)
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Same filter in dim_location.sql**
|
|
||||||
|
|
||||||
Replace its `where s.status in ('Open', 'Open, but proposed to close')` (and the comment above it) with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Must match dim_school's status filter exactly (the API inner-joins the two).
|
|
||||||
where s.status_code in (1, 3)
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Update _marts_schema.yml**
|
|
||||||
|
|
||||||
Under `dim_school` columns: rename `phase` → `phase_code` (keep the warn-severity not_null, reword description to mention codes); replace the `status` accepted_values block with:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: status_code
|
|
||||||
description: GIAS EstablishmentStatus code (1 = Open, 3 = Open but proposed to close)
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
values: [1, 3]
|
|
||||||
```
|
|
||||||
|
|
||||||
Add warn-severity accepted_values for the other codes, values copied from the seed (school_type/religious/admissions lists are long — paste the full code list from `gias_code_names.csv` for each):
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: school_type_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [<all school_type codes from the seed>]
|
|
||||||
- name: religious_character_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [<all religious_character codes from the seed>]
|
|
||||||
- name: admissions_policy_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [<all admissions_policy codes from the seed>]
|
|
||||||
```
|
|
||||||
|
|
||||||
(`<...>` here means: paste the actual comma-separated integers from the seed file — the lists exist by the time this task runs. Leaving a literal `<...>` in the yml is a task failure.)
|
|
||||||
|
|
||||||
`has_sixth_form` tests stay unchanged.
|
|
||||||
|
|
||||||
- [ ] **Step 4: Write the drift test**
|
|
||||||
|
|
||||||
Create `pipeline/transform/tests/assert_gias_code_names_match_seed.sql`:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Warn when the live GIAS CSV carries a (code, name) pair we don't have in
|
|
||||||
-- the dictionary seed — i.e. DfE added or renamed a value. Fix by rerunning
|
|
||||||
-- pipeline/scripts/generate_gias_codes.py and committing the regenerated
|
|
||||||
-- dictionaries + seed together.
|
|
||||||
{{ config(severity='warn') }}
|
|
||||||
|
|
||||||
with raw_pairs as (
|
|
||||||
{% for field_key, code_col, name_col in [
|
|
||||||
('school_type', 'TypeOfEstablishment (code)', 'TypeOfEstablishment (name)'),
|
|
||||||
('establishment_status', 'EstablishmentStatus (code)', 'EstablishmentStatus (name)'),
|
|
||||||
('phase_of_education', 'PhaseOfEducation (code)', 'PhaseOfEducation (name)'),
|
|
||||||
('official_sixth_form', 'OfficialSixthForm (code)', 'OfficialSixthForm (name)'),
|
|
||||||
('religious_character', 'ReligiousCharacter (code)', 'ReligiousCharacter (name)'),
|
|
||||||
('admissions_policy', 'AdmissionsPolicy (code)', 'AdmissionsPolicy (name)')
|
|
||||||
] %}
|
|
||||||
select distinct
|
|
||||||
'{{ field_key }}' as field,
|
|
||||||
cast(nullif(trim("{{ code_col }}"), '') as integer) as code,
|
|
||||||
nullif(trim("{{ name_col }}"), '') as name
|
|
||||||
from {{ source('raw', 'gias_establishments') }}
|
|
||||||
where nullif(trim("{{ code_col }}"), '') is not null
|
|
||||||
and nullif(trim("{{ name_col }}"), '') is not null
|
|
||||||
{% if not loop.last %}union all{% endif %}
|
|
||||||
{% endfor %}
|
|
||||||
)
|
|
||||||
|
|
||||||
select r.*
|
|
||||||
from raw_pairs r
|
|
||||||
left join {{ ref('gias_code_names') }} s
|
|
||||||
on s.field = r.field
|
|
||||||
and s.code = r.code
|
|
||||||
and s.name = r.name
|
|
||||||
where s.field is null
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 5: Verify statically**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
cd /Users/tudor/projects/school_compare && \
|
|
||||||
uv run --with pyyaml python -c "import yaml; yaml.safe_load(open('pipeline/transform/models/marts/_marts_schema.yml')); print('yml OK')" && \
|
|
||||||
grep -c "_code" pipeline/transform/models/marts/dim_school.sql && \
|
|
||||||
grep -n "status_code in (1, 3)" pipeline/transform/models/marts/dim_school.sql pipeline/transform/models/marts/dim_location.sql && \
|
|
||||||
grep -rn "s\.status\b\|s\.phase\b\|s\.school_type\b\|s\.religious_character\b\|s\.admissions_policy\b\|official_sixth_form\b" pipeline/transform/models/marts/dim_school.sql | grep -v "_code"; echo "stale-name grep exit=$? (want 1)"
|
|
||||||
```
|
|
||||||
Expected: `yml OK`, both filters matched, and no stale name-column references (final grep exits 1).
|
|
||||||
|
|
||||||
- [ ] **Step 6: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/marts/dim_school.sql pipeline/transform/models/marts/dim_location.sql pipeline/transform/models/marts/_marts_schema.yml pipeline/transform/tests/assert_gias_code_names_match_seed.sql
|
|
||||||
git commit -m "feat(pipeline): dim_school/dim_location store GIAS codes; seed drift test
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 4: Backend translates at the API boundary
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/models.py` (DimSchool columns)
|
|
||||||
- Modify: `backend/data_loader.py` (`_MAIN_QUERY` + translation)
|
|
||||||
- Test: `backend/tests/test_gias_translation.py` (new)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `backend/gias_codes.py` dictionaries + `translate` (Task 1); mart code columns (Task 3).
|
|
||||||
- Produces: `translate_gias_code_columns(df) -> df` in `backend/data_loader.py`; after `load_school_data_as_dataframe()` the DataFrame carries today's name columns (`phase`, `school_type`, `status`, `religious_denomination`, `admissions_policy`) — every downstream consumer unchanged.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the failing tests**
|
|
||||||
|
|
||||||
Create `backend/tests/test_gias_translation.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""API-boundary translation: marts now carry GIAS codes; the DataFrame the
|
|
||||||
rest of the backend sees must carry today's name strings."""
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from backend.data_loader import translate_gias_code_columns
|
|
||||||
from backend.gias_codes import ESTABLISHMENT_STATUS, PHASE_OF_EDUCATION
|
|
||||||
|
|
||||||
|
|
||||||
def _code_for(mapping, name):
|
|
||||||
return next(c for c, n in mapping.items() if n == name)
|
|
||||||
|
|
||||||
|
|
||||||
def test_codes_become_todays_names():
|
|
||||||
df = pd.DataFrame([{
|
|
||||||
"urn": 1,
|
|
||||||
"phase_code": float(_code_for(PHASE_OF_EDUCATION, "Primary")),
|
|
||||||
"school_type_code": np.nan,
|
|
||||||
"status_code": float(_code_for(ESTABLISHMENT_STATUS, "Open, but proposed to close")),
|
|
||||||
"religious_character_code": np.nan,
|
|
||||||
"admissions_policy_code": np.nan,
|
|
||||||
}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
row = out.iloc[0]
|
|
||||||
assert row["phase"] == "Primary"
|
|
||||||
assert row["status"] == "Open, but proposed to close"
|
|
||||||
assert row["school_type"] is None
|
|
||||||
assert row["religious_denomination"] is None
|
|
||||||
assert row["admissions_policy"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_unknown_code_degrades_not_blanks():
|
|
||||||
df = pd.DataFrame([{"urn": 1, "phase_code": 9999.0}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
assert out.iloc[0]["phase"] == "Unknown (9999)"
|
|
||||||
|
|
||||||
|
|
||||||
def test_missing_code_columns_are_a_noop():
|
|
||||||
"""Old-schema DataFrames (tests, pre-pipeline DBs) pass through untouched."""
|
|
||||||
df = pd.DataFrame([{"urn": 1, "phase": "Primary", "status": "Open"}])
|
|
||||||
out = translate_gias_code_columns(df)
|
|
||||||
assert out.iloc[0]["phase"] == "Primary"
|
|
||||||
assert out.iloc[0]["status"] == "Open"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Run to verify failure**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_gias_translation.py -v`
|
|
||||||
Expected: FAIL — `ImportError: cannot import name 'translate_gias_code_columns'`.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Implement translation in data_loader.py**
|
|
||||||
|
|
||||||
Add near the top of `backend/data_loader.py` (after existing imports):
|
|
||||||
|
|
||||||
```python
|
|
||||||
from .gias_codes import (
|
|
||||||
ADMISSIONS_POLICY,
|
|
||||||
ESTABLISHMENT_STATUS,
|
|
||||||
PHASE_OF_EDUCATION,
|
|
||||||
RELIGIOUS_CHARACTER,
|
|
||||||
SCHOOL_TYPE,
|
|
||||||
translate,
|
|
||||||
)
|
|
||||||
|
|
||||||
# mart code column -> (API name column, dictionary)
|
|
||||||
_GIAS_CODE_COLUMNS = {
|
|
||||||
"phase_code": ("phase", PHASE_OF_EDUCATION),
|
|
||||||
"school_type_code": ("school_type", SCHOOL_TYPE),
|
|
||||||
"status_code": ("status", ESTABLISHMENT_STATUS),
|
|
||||||
"religious_character_code": ("religious_denomination", RELIGIOUS_CHARACTER),
|
|
||||||
"admissions_policy_code": ("admissions_policy", ADMISSIONS_POLICY),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def translate_gias_code_columns(df: pd.DataFrame) -> pd.DataFrame:
|
|
||||||
"""Map GIAS code columns to today's name columns (API contract).
|
|
||||||
|
|
||||||
Runs immediately after pd.read_sql so every downstream consumer —
|
|
||||||
filters, PHASE_GROUPS, payloads, /api/filters — keeps seeing names.
|
|
||||||
DataFrames without the code columns (old schema, test fixtures) pass
|
|
||||||
through unchanged.
|
|
||||||
"""
|
|
||||||
for code_col, (name_col, mapping) in _GIAS_CODE_COLUMNS.items():
|
|
||||||
if code_col in df.columns:
|
|
||||||
df[name_col] = df[code_col].map(lambda c: translate(c, mapping))
|
|
||||||
return df
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Switch `_MAIN_QUERY` to code columns and call the translation**
|
|
||||||
|
|
||||||
In `_MAIN_QUERY` replace:
|
|
||||||
`s.phase,` → `s.phase_code,` · `s.school_type,` → `s.school_type_code,` · `s.religious_character AS religious_denomination,` → `s.religious_character_code,` · `s.admissions_policy,` → `s.admissions_policy_code,` · `s.status,` → `s.status_code,`
|
|
||||||
|
|
||||||
In `load_school_data_as_dataframe()`, insert the call immediately after the empty-check and **before** the existing `normalize_school_type` line:
|
|
||||||
|
|
||||||
```python
|
|
||||||
if df.empty:
|
|
||||||
return df
|
|
||||||
|
|
||||||
df = translate_gias_code_columns(df)
|
|
||||||
|
|
||||||
# Build address string
|
|
||||||
...
|
|
||||||
# Normalize school type (existing line — now normalises the translated name)
|
|
||||||
df["school_type"] = df["school_type"].apply(normalize_school_type)
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 5: Update DimSchool in models.py**
|
|
||||||
|
|
||||||
Replace `phase = Column(String(100))`, `school_type = Column(String(100))`, `religious_character = Column(String(100))`, `admissions_policy = Column(String(50))`, `status = Column(String(50))` with:
|
|
||||||
|
|
||||||
```python
|
|
||||||
phase_code = Column(Integer)
|
|
||||||
school_type_code = Column(Integer)
|
|
||||||
religious_character_code = Column(Integer)
|
|
||||||
admissions_policy_code = Column(Integer)
|
|
||||||
status_code = Column(Integer)
|
|
||||||
```
|
|
||||||
|
|
||||||
Then check nothing else references the removed attributes:
|
|
||||||
```bash
|
|
||||||
grep -rn "\.phase\b\|\.school_type\b\|\.religious_character\b\|\.admissions_policy\b\|\.status\b" backend/*.py | grep -i "dimschool\|DimSchool"
|
|
||||||
```
|
|
||||||
Expected: no hits (the backend reads via `_MAIN_QUERY`, not ORM attributes). If there are hits, update them to the `_code` columns + translation and note it in your report.
|
|
||||||
|
|
||||||
- [ ] **Step 6: Run the new tests and the whole backend suite**
|
|
||||||
|
|
||||||
Run: `cd /Users/tudor/projects/school_compare && uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests -v`
|
|
||||||
Expected: all pass — 3 new + all pre-existing (their fixtures carry name columns; translation is a no-op on them).
|
|
||||||
|
|
||||||
- [ ] **Step 7: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add backend/models.py backend/data_loader.py backend/tests/test_gias_translation.py
|
|
||||||
git commit -m "feat(api): translate GIAS codes to names at the query boundary
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 5: Typesense sync translates before indexing
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/scripts/sync_typesense.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `pipeline/scripts/gias_codes.py` (Task 1), mart code columns (Task 3).
|
|
||||||
- Produces: identical Typesense documents to today (facet values are names).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Switch the SELECT and translate**
|
|
||||||
|
|
||||||
In `sync_typesense.py`: add at the top (the DAG runs `python scripts/sync_typesense.py`, so `scripts/` is `sys.path[0]` and a plain import works):
|
|
||||||
|
|
||||||
```python
|
|
||||||
from gias_codes import PHASE_OF_EDUCATION, RELIGIOUS_CHARACTER, SCHOOL_TYPE, translate
|
|
||||||
```
|
|
||||||
|
|
||||||
In the SQL, replace `s.phase,` → `s.phase_code,`, `s.school_type,` → `s.school_type_code,`, `s.religious_character,` → `s.religious_character_code,`.
|
|
||||||
|
|
||||||
In the document builder, replace:
|
|
||||||
|
|
||||||
```python
|
|
||||||
"phase": row["phase"] or "",
|
|
||||||
"school_type": row["school_type"] or "",
|
|
||||||
```
|
|
||||||
with:
|
|
||||||
```python
|
|
||||||
"phase": translate(row["phase_code"], PHASE_OF_EDUCATION) or "",
|
|
||||||
"school_type": translate(row["school_type_code"], SCHOOL_TYPE) or "",
|
|
||||||
```
|
|
||||||
and:
|
|
||||||
```python
|
|
||||||
if row.get("religious_character"):
|
|
||||||
doc["religious_character"] = row["religious_character"]
|
|
||||||
```
|
|
||||||
with:
|
|
||||||
```python
|
|
||||||
religious_character = translate(row.get("religious_character_code"), RELIGIOUS_CHARACTER)
|
|
||||||
if religious_character:
|
|
||||||
doc["religious_character"] = religious_character
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Verify statically**
|
|
||||||
|
|
||||||
Run:
|
|
||||||
```bash
|
|
||||||
cd /Users/tudor/projects/school_compare && \
|
|
||||||
python3 -c "import ast; ast.parse(open('pipeline/scripts/sync_typesense.py').read()); print('sync OK')" && \
|
|
||||||
grep -n "row\[\"phase\"\]\|row\[\"school_type\"\]\|row\[\"religious_character\"\]" pipeline/scripts/sync_typesense.py; echo "stale grep exit=$? (want 1)"
|
|
||||||
```
|
|
||||||
Expected: `sync OK`, no stale name-column row accesses.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/scripts/sync_typesense.py
|
|
||||||
git commit -m "feat(pipeline): typesense sync translates GIAS codes before indexing
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 6: Spec status, PR, deploy runbook
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `docs/superpowers/specs/2026-07-09-gias-code-dictionaries-design.md` (status line)
|
|
||||||
|
|
||||||
- [ ] **Step 1: Mark the spec implemented**
|
|
||||||
|
|
||||||
Change `**Status:** Approved design` to `**Status:** Implemented 2026-07-09 — see docs/superpowers/plans/2026-07-09-gias-code-dictionaries.md`.
|
|
||||||
|
|
||||||
- [ ] **Step 2: Commit and push**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add docs/superpowers/specs/2026-07-09-gias-code-dictionaries-design.md
|
|
||||||
git commit -m "docs: mark GIAS code dictionaries spec implemented
|
|
||||||
|
|
||||||
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>"
|
|
||||||
git push -u origin feat/gias-code-dictionaries
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Open the PR (Gitea API via git credential fill — token-header auth 401s)**
|
|
||||||
|
|
||||||
Title: `feat: GIAS classification fields stored as codes, translated in code`
|
|
||||||
Body must include: (1) API contract unchanged — names still served, translation at the query boundary; (2) the **deploy runbook: merge → deploy → trigger `school_data_daily` immediately** (accepted empty-API window until the marts rebuild — spec §7); (3) dictionary maintenance loop (dbt drift test warns → rerun `generate_gias_codes.py` → commit regenerated files); (4) no frontend/e2e changes. End with the standard generation footer.
|
|
||||||
|
|
||||||
- [ ] **Step 4: Watch CI**
|
|
||||||
|
|
||||||
All PR checks must pass. Do not merge — merging triggers the deploy window; the human runs the runbook.
|
|
||||||
@@ -1,591 +0,0 @@
|
|||||||
# Compare-Screen Data Foundation (Pipeline PR) Implementation Plan
|
|
||||||
|
|
||||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
|
||||||
|
|
||||||
**Goal:** Land every pipeline/dbt change the compare-screen redesign needs (spec §5 + §8 of `docs/superpowers/specs/2026-07-11-compare-screen-redesign-design.md`): promote raw-but-unstored fields to marts, close the national-averages gaps, and wire the Ofsted report-card columns.
|
|
||||||
|
|
||||||
**Architecture:** Meltano Singer taps load `raw.*` tables; dbt builds `staging` → `marts` (read-only for the backend). All changes here are additive columns/rows — no breaking changes to existing marts. The full `dbt build` runs on the server via the Airflow DAGs; locally we gate with `dbt parse` (no DB needed) plus network-only diagnostic scripts.
|
|
||||||
|
|
||||||
**Tech Stack:** Python (Singer SDK taps), dbt-postgres ~1.10 (invoked as `python -m dbt.cli.main`), Meltano, PostgreSQL.
|
|
||||||
|
|
||||||
## Global Constraints
|
|
||||||
|
|
||||||
- **No new external sources** (spec §5): only fields already in the `raw` schema or in files the taps already download. The one sanctioned tap change is the Ofsted MI report-card columns (spec §5, §8.4) and the legacy-KS2 year addition (same DfE performance-tables source).
|
|
||||||
- **Additive only:** never rename or drop existing mart columns; the backend maps them 1:1 in `backend/models.py`.
|
|
||||||
- **Never push to `main`.** Branch: `feat/compare-data-foundation`; PR checks must pass.
|
|
||||||
- Backend `models.py` changes belong to the follow-up backend PR, not this one.
|
|
||||||
- dbt invocation is always `python -m dbt.cli.main` (a bare `dbt` resolves to the wrong binary — see `pipeline/dags/school_data_pipeline.py:27`).
|
|
||||||
- EES suppression codes `z`/`c`/`x` must go through the `safe_numeric` macro.
|
|
||||||
- Computed benchmarks (FSM/EAL/SEN medians, disadvantaged national average) are **backend work** (spec §5) — explicitly out of scope here.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 0: Create the branch
|
|
||||||
|
|
||||||
**Files:** none
|
|
||||||
|
|
||||||
- [ ] **Step 1:** `git checkout main && git pull && git checkout -b feat/compare-data-foundation`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 1: Diagnostics — pin the three unknowns
|
|
||||||
|
|
||||||
The spec flags three facts we must confirm from the actual files before wiring code: (a) why `gps_expected_pct`/`science_expected_pct` are NULL in `marts.fact_ks2_national_averages` despite being mapped end-to-end; (b) what the KS2 attainment long file calls its subjects/years for 2021/22 and 2022/23 (subject-level 2022/23 is NULL in prod; school-level 2021/22 is absent); (c) the exact report-card column headers in the current Ofsted MI CSV.
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Create: `pipeline/scripts/diagnose_compare_gaps.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: a printed findings report; Tasks 5, 6, 7 consume the confirmed column/label names. Precedent: `pipeline/scripts/diagnose_ees_ks4.py`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the diagnostic script**
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Diagnose the three data gaps blocking the compare-screen redesign.
|
|
||||||
|
|
||||||
Run from repo root (network access required, no DB needed):
|
|
||||||
python pipeline/scripts/diagnose_compare_gaps.py
|
|
||||||
"""
|
|
||||||
import io
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
import zipfile
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
|
|
||||||
sys.path.insert(0, "pipeline/plugins/extractors/tap-uk-ees")
|
|
||||||
sys.path.insert(0, "pipeline/plugins/extractors/tap-uk-ofsted")
|
|
||||||
from tap_uk_ees.tap import ( # noqa: E402
|
|
||||||
_KS2_NATIONAL_COL_MAP,
|
|
||||||
_KS2_NATIONAL_CSV_URL,
|
|
||||||
download_release_zip,
|
|
||||||
get_all_releases,
|
|
||||||
)
|
|
||||||
from tap_uk_ofsted.tap import discover_csv_url # noqa: E402
|
|
||||||
|
|
||||||
TIMEOUT = 120
|
|
||||||
|
|
||||||
|
|
||||||
def check_national_gps_science():
|
|
||||||
print("\n=== (a) National catalogue CSV: GPS/science columns ===")
|
|
||||||
resp = requests.get(_KS2_NATIONAL_CSV_URL, timeout=TIMEOUT)
|
|
||||||
resp.raise_for_status()
|
|
||||||
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
|
|
||||||
df.columns = [c.strip().lower() for c in df.columns]
|
|
||||||
for csv_col in ("pt_gps_exp", "pt_scita_exp", "avg_readscore", "avg_matscore", "avg_gpsscore"):
|
|
||||||
status = "PRESENT" if csv_col in df.columns else "MISSING"
|
|
||||||
print(f" {csv_col}: {status}")
|
|
||||||
gps_like = [c for c in df.columns if "gps" in c or "scita" in c or "sci" in c]
|
|
||||||
print(f" all gps/science-ish columns: {gps_like}")
|
|
||||||
nat = df[df.get("geographic_level", "").str.strip().str.lower() == "national"]
|
|
||||||
print(f" national rows time_periods: {sorted(nat['time_period'].unique())}")
|
|
||||||
# Sample the values our map would read for the latest year
|
|
||||||
latest = nat[nat["time_period"] == nat["time_period"].max()]
|
|
||||||
for csv_col, field in _KS2_NATIONAL_COL_MAP.items():
|
|
||||||
val = latest.iloc[0].get(csv_col, "<col missing>") if len(latest) else "<no row>"
|
|
||||||
print(f" {field} <- {csv_col} = {val!r}")
|
|
||||||
|
|
||||||
|
|
||||||
def check_ks2_attainment_years_subjects():
|
|
||||||
print("\n=== (b) EES KS2 attainment: years & subject labels ===")
|
|
||||||
releases = get_all_releases("key-stage-2-attainment")
|
|
||||||
print(f" releases found: {[r['time_period'] for r in releases]}")
|
|
||||||
for release in releases:
|
|
||||||
zf = download_release_zip(release["id"])
|
|
||||||
name = next((n for n in zf.namelist()
|
|
||||||
if "ks2_school_attainment_data" in n and n.endswith(".csv")), None)
|
|
||||||
if not name:
|
|
||||||
print(f" {release['time_period']}: NO school attainment CSV in ZIP")
|
|
||||||
continue
|
|
||||||
with zf.open(name) as f:
|
|
||||||
df = pd.read_csv(f, dtype=str, keep_default_na=False, nrows=200000)
|
|
||||||
years = sorted(df["time_period"].unique())
|
|
||||||
subjects = sorted(df["subject"].unique())
|
|
||||||
print(f" release {release['time_period']}: time_periods={years}")
|
|
||||||
print(f" subjects={subjects}")
|
|
||||||
|
|
||||||
|
|
||||||
def check_ofsted_report_card_columns():
|
|
||||||
print("\n=== (c) Ofsted MI CSV: report-card columns ===")
|
|
||||||
url = discover_csv_url()
|
|
||||||
print(f" MI file: {url}")
|
|
||||||
resp = requests.get(url, timeout=TIMEOUT)
|
|
||||||
resp.raise_for_status()
|
|
||||||
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False, nrows=5)
|
|
||||||
rc_like = [c for c in df.columns
|
|
||||||
if re.search(r"report card|inclusion|curriculum|achievement|safeguard|well.?being|governance", c, re.I)]
|
|
||||||
print(f" candidate report-card columns ({len(rc_like)}):")
|
|
||||||
for c in rc_like:
|
|
||||||
print(f" - {c!r}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
check_national_gps_science()
|
|
||||||
check_ks2_attainment_years_subjects()
|
|
||||||
check_ofsted_report_card_columns()
|
|
||||||
```
|
|
||||||
|
|
||||||
Note: if `_KS2_NATIONAL_CSV_URL` is named differently in `tap_uk_ees/tap.py` (it is defined near the `_KS2_NATIONAL_COL_MAP` around line ~490), import whatever constant holds the catalogue CSV URL.
|
|
||||||
|
|
||||||
- [ ] **Step 2: Run it and record findings**
|
|
||||||
|
|
||||||
Run: `python pipeline/scripts/diagnose_compare_gaps.py 2>&1 | tee /tmp/compare-gaps-findings.txt`
|
|
||||||
Expected: three sections printed. Paste the findings as a comment block at the bottom of the script (so they're committed evidence), e.g. `# FINDINGS 2026-07-12: pt_gps_exp MISSING (actual col: ...), 202122 present in release X, rc columns: [...]`.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/scripts/diagnose_compare_gaps.py
|
|
||||||
git commit -m "chore(pipeline): diagnostic for compare-screen data gaps"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 2: Admissions preference detail → mart
|
|
||||||
|
|
||||||
Staging already extracts `second_preference_offers`, `third_preference_offers`, `total_offers` (`stg_ees_admissions.sql:26-29`) — the mart drops them. The cross-LA fields are declared in the tap (`all_applications_from_another_LA`, `offers_to_applicants_from_another_LA`) but not selected in staging.
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_ees_admissions.sql` (after line 33, in `renamed`)
|
|
||||||
- Modify: `pipeline/transform/models/marts/fact_admissions.sql`
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (fact_admissions block, ~line 120)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces mart columns: `total_offers int`, `second_preference_offers int`, `third_preference_offers int`, `cross_la_applications int`, `cross_la_offers int`. The backend PR will map these in `FactAdmissions`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add cross-LA columns to staging**
|
|
||||||
|
|
||||||
In `stg_ees_admissions.sql`, after the `first_preference_applications` line (line 33):
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Cross-borough demand: applications naming this school from families
|
|
||||||
-- living in another local authority, and offers made to them.
|
|
||||||
{{ safe_numeric('"all_applications_from_another_LA"') }}::integer as cross_la_applications,
|
|
||||||
{{ safe_numeric('"offers_to_applicants_from_another_LA"') }}::integer as cross_la_offers,
|
|
||||||
```
|
|
||||||
|
|
||||||
(Quote the identifiers — the tap emits them with mixed case, same trap as `FSM_eligible_percent`, see the header comment in that file. If `dbt parse` or the DAG run later shows the raw columns are lower-cased in Postgres, drop the double quotes.)
|
|
||||||
|
|
||||||
- [ ] **Step 2: Pass everything through the mart**
|
|
||||||
|
|
||||||
Replace the full select list in `fact_admissions.sql`:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Mart: School admissions — one row per URN per year
|
|
||||||
|
|
||||||
select
|
|
||||||
urn,
|
|
||||||
year,
|
|
||||||
school_phase,
|
|
||||||
places_offered,
|
|
||||||
total_offers,
|
|
||||||
total_applications,
|
|
||||||
first_preference_applications,
|
|
||||||
first_preference_offers,
|
|
||||||
second_preference_offers,
|
|
||||||
third_preference_offers,
|
|
||||||
cross_la_applications,
|
|
||||||
cross_la_offers,
|
|
||||||
first_preference_offer_pct,
|
|
||||||
oversubscription_ratio,
|
|
||||||
oversubscribed,
|
|
||||||
admissions_policy
|
|
||||||
from {{ ref('stg_ees_admissions') }}
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Add schema tests**
|
|
||||||
|
|
||||||
In `_marts_schema.yml` under `fact_admissions.columns`, append:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: second_preference_offers
|
|
||||||
- name: third_preference_offers
|
|
||||||
- name: cross_la_applications
|
|
||||||
- name: cross_la_offers
|
|
||||||
- name: total_offers
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Parse gate**
|
|
||||||
|
|
||||||
Run: `cd pipeline/transform && python -m dbt.cli.main parse --profiles-dir .`
|
|
||||||
Expected: `Done.` with no compilation errors.
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/staging/stg_ees_admissions.sql pipeline/transform/models/marts/fact_admissions.sql pipeline/transform/models/marts/_marts_schema.yml
|
|
||||||
git commit -m "feat(pipeline): admissions preference breakdown and cross-LA demand in marts"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 3: KS2 progress confidence intervals + writing working-towards
|
|
||||||
|
|
||||||
The tap already emits `progress_measure_lower_conf_interval`, `progress_measure_upper_conf_interval`, `working_towards_expected_standard_pupil_percent` (tap.py:203-206). The staging pivot drops them. These power the CI-based Above/Average/Below progress chips (spec §8, first-review item on statistical honesty).
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_ees_ks2.sql` (inside the `pivoted` CTE, next to each subject's `progress_measure_score` case, lines ~41/55/72, and in the final select ~lines 145-152)
|
|
||||||
- Modify: `pipeline/transform/models/marts/fact_ks2_performance.sql`
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (fact_ks2_performance block, ~line 82)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces mart columns: `reading_progress_lower_ci`, `reading_progress_upper_ci`, `writing_progress_lower_ci`, `writing_progress_upper_ci`, `maths_progress_lower_ci`, `maths_progress_upper_ci` (float), `writing_working_towards_pct` (float).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add pivot cases in staging**
|
|
||||||
|
|
||||||
After the `reading_progress` case (line ~41), add:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
max(case when subject = 'Reading'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as reading_progress_lower_ci,
|
|
||||||
max(case when subject = 'Reading'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as reading_progress_upper_ci,
|
|
||||||
```
|
|
||||||
|
|
||||||
After the `writing_progress` case (line ~55), add:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as writing_progress_lower_ci,
|
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as writing_progress_upper_ci,
|
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('working_towards_expected_standard_pupil_percent') }} end) as writing_working_towards_pct,
|
|
||||||
```
|
|
||||||
|
|
||||||
After the `maths_progress` case (line ~72), add:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
max(case when subject = 'Maths'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as maths_progress_lower_ci,
|
|
||||||
max(case when subject = 'Maths'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as maths_progress_upper_ci,
|
|
||||||
```
|
|
||||||
|
|
||||||
Then add the seven new columns to the model's final select (next to the existing `p.reading_progress` / `p.writing_progress` / `p.maths_progress` lines ~145-152):
|
|
||||||
|
|
||||||
```sql
|
|
||||||
p.reading_progress_lower_ci,
|
|
||||||
p.reading_progress_upper_ci,
|
|
||||||
p.writing_progress_lower_ci,
|
|
||||||
p.writing_progress_upper_ci,
|
|
||||||
p.writing_working_towards_pct,
|
|
||||||
p.maths_progress_lower_ci,
|
|
||||||
p.maths_progress_upper_ci,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Pass through the mart**
|
|
||||||
|
|
||||||
In `fact_ks2_performance.sql`, add the same seven column names to the select list immediately after the existing `maths_progress` line (this mart selects staging columns by name; match the file's existing alias style — if columns are selected bare, add them bare).
|
|
||||||
|
|
||||||
- [ ] **Step 3: Schema tests**
|
|
||||||
|
|
||||||
In `_marts_schema.yml` under `fact_ks2_performance.columns`, append the seven names (no tests beyond presence — values are legitimately NULL for 2023/24+ since progress measures ended with 2022/23, spec §4.3):
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: reading_progress_lower_ci
|
|
||||||
- name: reading_progress_upper_ci
|
|
||||||
- name: writing_progress_lower_ci
|
|
||||||
- name: writing_progress_upper_ci
|
|
||||||
- name: writing_working_towards_pct
|
|
||||||
- name: maths_progress_lower_ci
|
|
||||||
- name: maths_progress_upper_ci
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4: Parse gate**
|
|
||||||
|
|
||||||
Run: `cd pipeline/transform && python -m dbt.cli.main parse --profiles-dir .`
|
|
||||||
Expected: `Done.`
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/staging/stg_ees_ks2.sql pipeline/transform/models/marts/fact_ks2_performance.sql pipeline/transform/models/marts/_marts_schema.yml
|
|
||||||
git commit -m "feat(pipeline): KS2 progress confidence intervals and writing working-towards"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 4: KS4 — Progress 8 banding and disadvantage gaps
|
|
||||||
|
|
||||||
The tap's `ees_ks4_info` stream already declares `progress8_banding` (DfE's own "well above average … well below average" label — the ready-made secondary chip), `attainment8_diffn` and `progress8_diffn` (tap.py:338-340). Wire them through staging into the mart.
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_ees_ks4.sql` (the CTE that reads `ees_ks4_info` — the same one that already surfaces `sen_pct`; add three columns to its select and to the final joined select)
|
|
||||||
- Modify: `pipeline/transform/models/marts/fact_ks4_performance.sql` (add after `progress_8_upper_ci`)
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (fact_ks4_performance block, ~line 93)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces mart columns: `progress_8_banding text`, `attainment_8_disadvantage_gap float`, `progress_8_disadvantage_gap float`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Staging — select from the info source**
|
|
||||||
|
|
||||||
In the info CTE of `stg_ees_ks4.sql` add:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
nullif(trim(progress8_banding), '') as progress_8_banding,
|
|
||||||
{{ safe_numeric('attainment8_diffn') }} as attainment_8_disadvantage_gap,
|
|
||||||
{{ safe_numeric('progress8_diffn') }} as progress_8_disadvantage_gap,
|
|
||||||
```
|
|
||||||
|
|
||||||
and add the three names to the model's final select (aliased the same way the CTE's other columns are).
|
|
||||||
|
|
||||||
- [ ] **Step 2: Mart passthrough**
|
|
||||||
|
|
||||||
In `fact_ks4_performance.sql`, after the `progress_8_upper_ci,` line:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
progress_8_banding,
|
|
||||||
attainment_8_disadvantage_gap,
|
|
||||||
progress_8_disadvantage_gap,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Schema tests** — append the three names under `fact_ks4_performance.columns`, plus an accepted-values guard that tolerates NULL:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- name: progress_8_banding
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
values: ['Well above average', 'Above average', 'Average', 'Below average', 'Well below average']
|
|
||||||
config:
|
|
||||||
where: "progress_8_banding is not null"
|
|
||||||
- name: attainment_8_disadvantage_gap
|
|
||||||
- name: progress_8_disadvantage_gap
|
|
||||||
```
|
|
||||||
|
|
||||||
(If the DAG run later shows different capitalisation in the data, fix the accepted values to match the data, not vice versa.)
|
|
||||||
|
|
||||||
- [ ] **Step 4: Parse gate** — `cd pipeline/transform && python -m dbt.cli.main parse --profiles-dir .` → `Done.`
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/staging/stg_ees_ks4.sql pipeline/transform/models/marts/fact_ks4_performance.sql pipeline/transform/models/marts/_marts_schema.yml
|
|
||||||
git commit -m "feat(pipeline): Progress 8 banding and KS4 disadvantage gaps in marts"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 5: National averages — 2015/16 row and GPS/science/scaled-score fix
|
|
||||||
|
|
||||||
Two changes. (1) `stg_ees_ks2_national.sql:34` filters `>= 201617`, which is exactly why the England line starts a year late (2015/16 RWM = 53% exists in the catalogue). (2) GPS/science expected are NULL in prod despite full end-to-end mapping — Task 1's findings say whether the catalogue CSV column names differ from `_KS2_NATIONAL_COL_MAP` (`pt_gps_exp`, `pt_scita_exp`) or whether values are suppressed at source.
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_ees_ks2_national.sql:34`
|
|
||||||
- Modify (conditional on Task 1 findings): `pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py` (`_KS2_NATIONAL_COL_MAP`)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: a 201516 row in `marts.fact_ks2_national_averages`; non-NULL `gps_expected_pct`, `science_expected_pct`, `reading_avg_score`, `maths_avg_score`, `gps_avg_score` for years the DfE publishes them. Backend/frontend consume via `/api/national-averages` unchanged (additive year + newly non-NULL fields).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Widen the year filter**
|
|
||||||
|
|
||||||
In `stg_ees_ks2_national.sql`, change line 34:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
and cast(trim(time_period) as integer) >= 201516
|
|
||||||
```
|
|
||||||
|
|
||||||
(2015/16 was the first year of the current expected-standard tests; nothing earlier is comparable, so keep a floor.)
|
|
||||||
|
|
||||||
- [ ] **Step 2: Fix the column map per Task 1 findings**
|
|
||||||
|
|
||||||
If Task 1 reported the actual CSV column names for GPS/science/scaled scores differ, update `_KS2_NATIONAL_COL_MAP` in `tap.py` accordingly, e.g. (illustrative — use the diagnosed names):
|
|
||||||
|
|
||||||
```python
|
|
||||||
_KS2_NATIONAL_COL_MAP = {
|
|
||||||
# ... existing entries ...
|
|
||||||
"pt_gps_exp": "gps_expected_pct", # replace key with diagnosed name
|
|
||||||
"pt_scita_exp": "science_expected_pct", # replace key with diagnosed name
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
If Task 1 showed the columns are present but suppressed (`x`) at national level for all years, instead delete the two entries from the map, delete the corresponding lines from `stg_ees_ks2_national.sql` and `fact_ks2_national_averages.sql`, and record in the PR description that GPS/science England ticks stay "not in dataset" (the mockups already carry that caveat).
|
|
||||||
|
|
||||||
- [ ] **Step 3: Parse gate** — `cd pipeline/transform && python -m dbt.cli.main parse --profiles-dir .` → `Done.`
|
|
||||||
|
|
||||||
- [ ] **Step 4: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/transform/models/staging/stg_ees_ks2_national.sql pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py
|
|
||||||
git commit -m "fix(pipeline): include 2015/16 national averages; fix GPS/science national mapping"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 6: Legacy KS2 — load the 2021/22 school-level year
|
|
||||||
|
|
||||||
School-level 2021/22 exists in DfE performance-tables archives (same source as the four legacy years already loaded) but in neither our legacy config (stops at 201819, `pipeline/meltano.yml:33-37`) nor EES (starts 2022/23) — unless Task 1's finding (b) showed an EES release carrying 202122, in which case skip this task and note why in the PR.
|
|
||||||
|
|
||||||
The legacy URLs point at the self-hosted filebrowser (`10.0.1.224:8081`) — **the 2021/22 DfE archive must be uploaded there first; this is the one human dependency in this plan.**
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/meltano.yml` (legacy_ks2_urls block, line ~33)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: `raw.legacy_ks2` rows with `year = '202122'`, flowing through `stg_legacy_ks2` → `fact_ks2_performance` unchanged (the stream maps old column names already; 2021/22 CSVs use the same `PTRWM_EXP`-style headers as 2018/19).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Verify the 2021/22 CSV headers match `_LEGACY_KS2_COLUMN_MAP`**
|
|
||||||
|
|
||||||
Download the DfE 2021/22 KS2 revised archive (gov.uk "Compare School Performance data download": 2021-2022 all-schools ZIP), then:
|
|
||||||
|
|
||||||
Run: `python -c "import zipfile,io,pandas as pd; zf=zipfile.ZipFile('/path/to/2021-2022.zip'); n=[x for x in zf.namelist() if 'ks2final' in x.lower() and x.endswith('.csv')][0]; df=pd.read_csv(zf.open(n), dtype=str, nrows=5); import sys; sys.path.insert(0,'pipeline/plugins/extractors/tap-uk-ees'); from tap_uk_ees.tap import _LEGACY_KS2_COLUMN_MAP as m; missing=[c for c in m if c not in df.columns]; print('missing legacy columns:', missing)"`
|
|
||||||
Expected: `missing legacy columns: []` (progress columns `READPROG` etc. may legitimately be missing/blank in 2021/22 — acceptable, they load as NULL).
|
|
||||||
|
|
||||||
- [ ] **Step 2: Upload the archive to the filebrowser and add the config entry**
|
|
||||||
|
|
||||||
In `pipeline/meltano.yml` under `legacy_ks2_urls`, add (with the real share URL from the filebrowser upload):
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
"202122": "http://10.0.1.224:8081/filebrowser/api/public/dl/<SHARE_ID>?inline=true"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/meltano.yml
|
|
||||||
git commit -m "feat(pipeline): load 2021/22 school-level KS2 from legacy performance tables"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4 (only if Task 1(b) showed 2022/23 subject labels differ):** widen the subject matchers in `stg_ees_ks2.sql` the same way GPS already is (`subject ilike '%grammar%' or subject = 'GPS'`), e.g. `subject in ('Reading', 'reading')` → use the diagnosed labels. Parse-gate and commit as `fix(pipeline): match 2022/23 KS2 subject labels`.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 7: Ofsted report-card columns (rc_*)
|
|
||||||
|
|
||||||
Resolves the tap TODO (`stg_ofsted_inspections.sql:37`). The marts/backed columns already exist as stubs; this wires real values. Uses Task 1(c)'s confirmed MI column names — the candidates below follow the MI file's existing naming style and must be corrected against the diagnostic output.
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/plugins/extractors/tap-uk-ofsted/tap_uk_ofsted/tap.py` (COLUMN_PRIORITY ~line 19-72, schema ~line 100-114)
|
|
||||||
- Create: `pipeline/transform/macros/parse_report_card_grade.sql`
|
|
||||||
- Modify: `pipeline/transform/models/staging/stg_ofsted_inspections.sql:36-46`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces mart columns (already declared in `fact_ofsted_inspection`): `rc_safeguarding_met boolean`, and `rc_inclusion` … `rc_sixth_form` as integers on the 5-point scale `1=Exceptional, 2=Strong standard, 3=Expected standard, 4=Needs attention/Attention needed, 5=Urgent improvement`. The backend translates codes to labels (same pattern as `gias_codes.py`), verifying wording against Ofsted's published toolkit (spec §8.4).
|
|
||||||
|
|
||||||
- [ ] **Step 1: Add tap column mappings**
|
|
||||||
|
|
||||||
In `COLUMN_PRIORITY` add (replace candidate strings with Task 1(c)'s exact headers — keep them as priority lists so older files degrade to blank):
|
|
||||||
|
|
||||||
```python
|
|
||||||
"rc_safeguarding_met": ["Report card safeguarding", "Safeguarding"],
|
|
||||||
"rc_inclusion": ["Report card inclusion", "Inclusion"],
|
|
||||||
"rc_curriculum_teaching": ["Report card curriculum and teaching", "Curriculum and teaching"],
|
|
||||||
"rc_achievement": ["Report card achievement", "Achievement"],
|
|
||||||
"rc_attendance_behaviour": ["Report card attendance and behaviour", "Attendance and behaviour"],
|
|
||||||
"rc_personal_development": ["Report card personal development and well-being", "Personal development and well-being"],
|
|
||||||
"rc_leadership_governance": ["Report card leadership and governance", "Leadership and governance"],
|
|
||||||
"rc_early_years": ["Report card early years", "Early years"],
|
|
||||||
"rc_sixth_form": ["Report card sixth form", "Sixth form"],
|
|
||||||
```
|
|
||||||
|
|
||||||
And in the stream schema (next to `report_url`, ~line 114):
|
|
||||||
|
|
||||||
```python
|
|
||||||
th.Property("rc_safeguarding_met", th.StringType),
|
|
||||||
th.Property("rc_inclusion", th.StringType),
|
|
||||||
th.Property("rc_curriculum_teaching", th.StringType),
|
|
||||||
th.Property("rc_achievement", th.StringType),
|
|
||||||
th.Property("rc_attendance_behaviour", th.StringType),
|
|
||||||
th.Property("rc_personal_development", th.StringType),
|
|
||||||
th.Property("rc_leadership_governance", th.StringType),
|
|
||||||
th.Property("rc_early_years", th.StringType),
|
|
||||||
th.Property("rc_sixth_form", th.StringType),
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Write the grade-parsing macro**
|
|
||||||
|
|
||||||
`pipeline/transform/macros/parse_report_card_grade.sql`:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
{% macro parse_report_card_grade(column_name) %}
|
|
||||||
case lower(trim(nullif({{ column_name }}, 'NULL')))
|
|
||||||
when 'exceptional' then 1
|
|
||||||
when 'strong standard' then 2
|
|
||||||
when 'expected standard' then 3
|
|
||||||
when 'needs attention' then 4
|
|
||||||
when 'attention needed' then 4
|
|
||||||
when 'urgent improvement' then 5
|
|
||||||
end
|
|
||||||
{% endmacro %}
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Wire staging**
|
|
||||||
|
|
||||||
Replace `stg_ofsted_inspections.sql` lines 36-46 (the NULL stubs) with:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- Report Card fields (post-Nov 2025 framework), 5-point scale:
|
|
||||||
-- 1 Exceptional · 2 Strong standard · 3 Expected standard
|
|
||||||
-- · 4 Needs attention · 5 Urgent improvement
|
|
||||||
(lower(trim(nullif(rc_safeguarding_met, 'NULL'))) = 'met') as rc_safeguarding_met,
|
|
||||||
{{ parse_report_card_grade('rc_inclusion') }}::integer as rc_inclusion,
|
|
||||||
{{ parse_report_card_grade('rc_curriculum_teaching') }}::integer as rc_curriculum_teaching,
|
|
||||||
{{ parse_report_card_grade('rc_achievement') }}::integer as rc_achievement,
|
|
||||||
{{ parse_report_card_grade('rc_attendance_behaviour') }}::integer as rc_attendance_behaviour,
|
|
||||||
{{ parse_report_card_grade('rc_personal_development') }}::integer as rc_personal_development,
|
|
||||||
{{ parse_report_card_grade('rc_leadership_governance') }}::integer as rc_leadership_governance,
|
|
||||||
{{ parse_report_card_grade('rc_early_years') }}::integer as rc_early_years,
|
|
||||||
{{ parse_report_card_grade('rc_sixth_form') }}::integer as rc_sixth_form,
|
|
||||||
```
|
|
||||||
|
|
||||||
Note `rc_safeguarding_met` becomes boolean (NULL when blank) — matching `fact_ofsted_inspection`'s `rc_safeguarding_met` Boolean column. If `fact_ofsted_inspection.sql` casts these columns, align its casts too (inspect that model; it currently passes the text stubs through).
|
|
||||||
|
|
||||||
- [ ] **Step 4: Parse gate + tap smoke test**
|
|
||||||
|
|
||||||
Run: `cd pipeline/transform && python -m dbt.cli.main parse --profiles-dir .` → `Done.`
|
|
||||||
Run: `python -c "import sys; sys.path.insert(0,'pipeline/plugins/extractors/tap-uk-ofsted'); from tap_uk_ofsted.tap import COLUMN_PRIORITY; assert 'rc_inclusion' in COLUMN_PRIORITY; print('ok')"` → `ok`
|
|
||||||
|
|
||||||
- [ ] **Step 5: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add pipeline/plugins/extractors/tap-uk-ofsted/tap_uk_ofsted/tap.py pipeline/transform/macros/parse_report_card_grade.sql pipeline/transform/models/staging/stg_ofsted_inspections.sql
|
|
||||||
git commit -m "feat(pipeline): extract Ofsted report-card judgements (rc_* columns)"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 8: PR + post-merge verification
|
|
||||||
|
|
||||||
**Files:** none new
|
|
||||||
|
|
||||||
- [ ] **Step 1: Push and open the PR** (Gitea — use the git credential helper + basic-auth API pattern; token-header auth 401s):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git push -u origin feat/compare-data-foundation
|
|
||||||
# then create the PR via the Gitea API with basic auth from `git credential fill`
|
|
||||||
```
|
|
||||||
|
|
||||||
PR body: link spec §5/§8, list the new mart columns, note the Task 6 human dependency (filebrowser upload) and the Task 1 findings file.
|
|
||||||
|
|
||||||
- [ ] **Step 2: After merge, verify the DAG run picked everything up**
|
|
||||||
|
|
||||||
The daily/monthly DAGs rebuild the affected models (`pipeline/dags/school_data_pipeline.py`). Spot-check via the public API (production after promotion, staging first at stx.schoolcompare.co.uk — note external /api is broken at the staging proxy, so check staging from the host):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# 2015/16 national row exists
|
|
||||||
curl -sL "https://www.schoolcompare.co.uk/api/national-averages" | python3 -c "import json,sys; d=json.load(sys.stdin); assert any(r['year']==201516 and r['primary'] for r in d['by_year']), '2015/16 missing'; print('201516 ok')"
|
|
||||||
# 2021/22 school rows exist (Barclay)
|
|
||||||
curl -sL "https://www.schoolcompare.co.uk/api/schools/138690" | python3 -c "import json,sys; d=json.load(sys.stdin); ys=[r['year'] for r in d['yearly_data']]; assert 202122 in [int(y) for y in ys], ys; print('202122 ok')"
|
|
||||||
```
|
|
||||||
|
|
||||||
(The admissions/CI/KS4/rc_* columns aren't API-visible until the backend PR maps them — verify those directly in Postgres from the pipeline host: `select count(*) from marts.fact_admissions where second_preference_offers is not null;` etc.)
|
|
||||||
|
|
||||||
- [ ] **Step 3: Update the spec** — tick off the §5 promotions this PR delivered (edit the spec's promotion list to note "landed in PR #NN") and commit to main via a docs PR or alongside the backend PR.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Out of scope (next plans)
|
|
||||||
|
|
||||||
1. **Backend PR:** map new columns in `backend/models.py`, extend `/api/compare` with supplementary blocks + `national_averages`, computed benchmarks (FSM/EAL/SEN/size medians, disadvantaged national average), CI-based progress banding, report-card label translation (verify against Ofsted toolkit), Ofsted provider-page URLs, graded-vs-ungraded surfacing.
|
|
||||||
2. **Frontend PR:** rebuild `/compare` per the mockups + e2e journeys (promotion gate).
|
|
||||||
3. **Separate bug fix:** third school's series not rendering on the current production chart.
|
|
||||||
4. **Post-v1 (spec):** census ethnicity/young-carer promotion, IDACI display, attendance section, gender-split/absence tier-2 measures.
|
|
||||||
5. **Already in marts, no work needed:** KS4 EBacc entry/APS, grade 5+ English & maths, Progress 8 CIs — `fact_ks4_performance` carries them today; only the backend needs to expose them.
|
|
||||||
@@ -1,399 +0,0 @@
|
|||||||
# Compare API Enrichment (Backend PR) Implementation Plan
|
|
||||||
|
|
||||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
|
||||||
|
|
||||||
**Goal:** Expose the PR #32 data through the API so the redesigned compare screen can be built: enrich `/api/compare` with supplementary blocks + national averages + computed benchmarks, translate Ofsted report-card codes to labels, and surface the new mart columns (spec §6, §8 of `docs/superpowers/specs/2026-07-11-compare-screen-redesign-design.md`).
|
|
||||||
|
|
||||||
**Architecture:** All changes are additive API fields — existing consumers keep working. One small dbt change rides along: `fact_performance` (the combined KS2+KS4 mart the backend's `_MAIN_QUERY` reads) enumerates columns explicitly and was not extended in PR #32, so the new KS2 CI and KS4 banding columns must be threaded through it here. Everything else is backend Python: `models.py` mappings, `data_loader` query/supplementary additions, an Ofsted label dictionary (gias_codes pattern), and `/api/compare` composition.
|
|
||||||
|
|
||||||
**Tech Stack:** FastAPI, SQLAlchemy, pandas; dbt (one model); pytest via `python -m pytest backend/tests -q` (CI installs `requirements.txt pytest "httpx<0.28"`; locally use `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests -q`).
|
|
||||||
|
|
||||||
## Global Constraints
|
|
||||||
|
|
||||||
- **Never push to `main`.** Branch: `feat/compare-api-enrichment`.
|
|
||||||
- **Additive only** to API responses; never rename/remove existing fields (frontend + e2e depend on them).
|
|
||||||
- **Report-card scale labels are the live-sampled vocabulary** (evidence in `pipeline/scripts/diagnose_compare_gaps.py`): `1=Exceptional, 2=Strong standard, 3=Expected standard, 4=Needs attention, 5=Urgent improvement`. Never "Attention needed". Safeguarding is boolean met/not-met, never counted as a graded area.
|
|
||||||
- **Ofsted links** are always the provider page `https://reports.ofsted.gov.uk/provider/21/{urn}` (spec §5) labelled as the school's Ofsted page.
|
|
||||||
- **Benchmark provenance** (spec §8.6): computed values are "state-school average (computed from our dataset)" — the API must expose them under a `benchmarks` key, clearly separate from official `national_averages`.
|
|
||||||
- TDD: each behaviour lands with a failing test first, in `backend/tests/` following the `test_school_details.py` pattern (pandas fixture + monkeypatched `load_school_data` + `TestClient`).
|
|
||||||
- Deploy note for the PR body: the new API fields return NULL/empty until prod's DAGs have run post-promotion.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 0: Branch
|
|
||||||
|
|
||||||
- [ ] `git checkout main && git pull && git checkout -b feat/compare-api-enrichment` (commit this plan file on the branch).
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 1: Thread PR #32 columns through `fact_performance`
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `pipeline/transform/models/marts/fact_performance.sql`
|
|
||||||
- Modify: `pipeline/transform/models/marts/_marts_schema.yml` (fact_performance block, if it has one — add the columns wherever the model's other columns are listed; if the model has no column list there, skip the yml)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces (for `_MAIN_QUERY` in Task 4): `ks2.*` CI columns and `ks4.progress_8_banding`, `ks4.attainment_8_disadvantage_gap`, `ks4.progress_8_disadvantage_gap` on `marts.fact_performance`.
|
|
||||||
|
|
||||||
- [ ] **Step 1:** In `fact_performance.sql`, after `ks2.reading_progress,` add `ks2.reading_progress_lower_ci,` and `ks2.reading_progress_upper_ci,`; after `ks2.writing_progress,` add `ks2.writing_progress_lower_ci,`, `ks2.writing_progress_upper_ci,`, `ks2.writing_working_towards_pct,`; after `ks2.maths_progress,` add `ks2.maths_progress_lower_ci,`, `ks2.maths_progress_upper_ci,`. In the KS4 section, after the `ks4.progress_8_upper_ci`-equivalent line (locate the Progress 8 block) add:
|
|
||||||
|
|
||||||
```sql
|
|
||||||
ks4.progress_8_banding,
|
|
||||||
ks4.attainment_8_disadvantage_gap,
|
|
||||||
ks4.progress_8_disadvantage_gap,
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Parse gate: `cd pipeline/transform && uv run --with dbt-postgres python -m dbt.cli.main parse --profiles-dir .` → exit 0.
|
|
||||||
|
|
||||||
- [ ] **Step 3:** Commit: `feat(pipeline): thread compare-foundation columns through fact_performance`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 2: ORM mappings for the new mart columns
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/models.py` (`KS2Performance` after `maths_progress`; `FactAdmissions` after `first_preference_offers`)
|
|
||||||
- Test: none (declarative mappings; covered by Task 4's query tests)
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces attributes used by Task 4: `KS2Performance.reading_progress_lower_ci` … `maths_progress_upper_ci`, `writing_working_towards_pct` (Float); `FactAdmissions.total_offers`, `.second_preference_offers`, `.third_preference_offers`, `.cross_la_applications`, `.cross_la_offers` (Integer).
|
|
||||||
|
|
||||||
- [ ] **Step 1:** Add to `KS2Performance` (next to the existing progress columns):
|
|
||||||
|
|
||||||
```python
|
|
||||||
reading_progress_lower_ci = Column(Float)
|
|
||||||
reading_progress_upper_ci = Column(Float)
|
|
||||||
writing_progress_lower_ci = Column(Float)
|
|
||||||
writing_progress_upper_ci = Column(Float)
|
|
||||||
writing_working_towards_pct = Column(Float)
|
|
||||||
maths_progress_lower_ci = Column(Float)
|
|
||||||
maths_progress_upper_ci = Column(Float)
|
|
||||||
```
|
|
||||||
|
|
||||||
Add to `FactAdmissions` (after `first_preference_offers`):
|
|
||||||
|
|
||||||
```python
|
|
||||||
total_offers = Column(Integer)
|
|
||||||
second_preference_offers = Column(Integer)
|
|
||||||
third_preference_offers = Column(Integer)
|
|
||||||
cross_la_applications = Column(Integer)
|
|
||||||
cross_la_offers = Column(Integer)
|
|
||||||
```
|
|
||||||
|
|
||||||
(`FactOfstedInspection` already maps all `rc_*` columns with the right types — verify, don't change.)
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Commit: `feat(api): map compare-foundation mart columns`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 3: Ofsted label dictionary + provider URL (TDD)
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Create: `backend/ofsted_codes.py`
|
|
||||||
- Test: `backend/tests/test_ofsted_codes.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces for Task 4: `REPORT_CARD_GRADE_NAMES: dict[int, str]`, `report_card_labels(ofsted: dict) -> dict` (returns `{area_key: {"code": int, "label": str}}` for the non-null `rc_*` grade fields, excluding safeguarding), `ofsted_page_url(urn: int) -> str`.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Failing tests**
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Report-card code translation uses the live-sampled Ofsted vocabulary
|
|
||||||
(pipeline/scripts/diagnose_compare_gaps.py TASK 7 VALUE SAMPLE):
|
|
||||||
Exceptional / Strong standard / Expected standard / Needs attention /
|
|
||||||
Urgent improvement — never the consultation draft's 'Attention needed'."""
|
|
||||||
from backend.ofsted_codes import (
|
|
||||||
REPORT_CARD_GRADE_NAMES, report_card_labels, ofsted_page_url,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def test_scale_is_sampled_vocabulary():
|
|
||||||
assert REPORT_CARD_GRADE_NAMES == {
|
|
||||||
1: "Exceptional",
|
|
||||||
2: "Strong standard",
|
|
||||||
3: "Expected standard",
|
|
||||||
4: "Needs attention",
|
|
||||||
5: "Urgent improvement",
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_labels_only_for_populated_areas_and_never_safeguarding():
|
|
||||||
ofsted = {
|
|
||||||
"rc_achievement": 2,
|
|
||||||
"rc_inclusion": 3,
|
|
||||||
"rc_attendance_behaviour": 4,
|
|
||||||
"rc_early_years": None,
|
|
||||||
"rc_safeguarding_met": True,
|
|
||||||
"overall_effectiveness": None,
|
|
||||||
}
|
|
||||||
labels = report_card_labels(ofsted)
|
|
||||||
assert labels == {
|
|
||||||
"rc_achievement": {"code": 2, "label": "Strong standard"},
|
|
||||||
"rc_inclusion": {"code": 3, "label": "Expected standard"},
|
|
||||||
"rc_attendance_behaviour": {"code": 4, "label": "Needs attention"},
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def test_unknown_code_is_skipped_not_crashed():
|
|
||||||
assert report_card_labels({"rc_achievement": 9}) == {}
|
|
||||||
|
|
||||||
|
|
||||||
def test_provider_url():
|
|
||||||
assert ofsted_page_url(138690) == "https://reports.ofsted.gov.uk/provider/21/138690"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Run `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_ofsted_codes.py -q` → FAIL (module missing).
|
|
||||||
|
|
||||||
- [ ] **Step 3: Implement `backend/ofsted_codes.py`**
|
|
||||||
|
|
||||||
```python
|
|
||||||
"""Ofsted renewed-framework (Nov 2025) report-card code translation.
|
|
||||||
|
|
||||||
Scale labels are the live-sampled vocabulary from the Ofsted MI file
|
|
||||||
(see pipeline/scripts/diagnose_compare_gaps.py, TASK 7 VALUE SAMPLE) —
|
|
||||||
verified against real data, not the consultation draft.
|
|
||||||
"""
|
|
||||||
|
|
||||||
REPORT_CARD_GRADE_NAMES = {
|
|
||||||
1: "Exceptional",
|
|
||||||
2: "Strong standard",
|
|
||||||
3: "Expected standard",
|
|
||||||
4: "Needs attention",
|
|
||||||
5: "Urgent improvement",
|
|
||||||
}
|
|
||||||
|
|
||||||
# Graded evaluation areas only — safeguarding is a separate boolean
|
|
||||||
# judgement and must never appear in grade counts or label maps.
|
|
||||||
_RC_AREA_KEYS = (
|
|
||||||
"rc_inclusion",
|
|
||||||
"rc_curriculum_teaching",
|
|
||||||
"rc_achievement",
|
|
||||||
"rc_attendance_behaviour",
|
|
||||||
"rc_personal_development",
|
|
||||||
"rc_leadership_governance",
|
|
||||||
"rc_early_years",
|
|
||||||
"rc_sixth_form",
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def report_card_labels(ofsted: dict) -> dict:
|
|
||||||
"""{area_key: {code, label}} for populated, known-valued rc_* areas."""
|
|
||||||
out = {}
|
|
||||||
for key in _RC_AREA_KEYS:
|
|
||||||
code = ofsted.get(key)
|
|
||||||
label = REPORT_CARD_GRADE_NAMES.get(code)
|
|
||||||
if code is not None and label is not None:
|
|
||||||
out[key] = {"code": code, "label": label}
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def ofsted_page_url(urn: int) -> str:
|
|
||||||
"""The school's page on ofsted.gov.uk (all its reports live there —
|
|
||||||
we never deep-link an individual report; spec §5)."""
|
|
||||||
return f"https://reports.ofsted.gov.uk/provider/21/{urn}"
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 4:** Re-run the test file → 4 passed. Run the full suite (same command, `backend/tests -q`) → all pass.
|
|
||||||
|
|
||||||
- [ ] **Step 5:** Commit: `feat(api): Ofsted report-card labels and provider-page URL`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 4: data_loader — query columns + richer supplementary blocks (TDD)
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/data_loader.py` (`_MAIN_QUERY` ~line 153; `get_supplementary_data` ~line 460)
|
|
||||||
- Test: `backend/tests/test_supplementary_enrichment.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- `_MAIN_QUERY` additionally selects (KS2 block, after `p.maths_progress`): `p.reading_progress_lower_ci, p.reading_progress_upper_ci, p.writing_progress_lower_ci, p.writing_progress_upper_ci, p.writing_working_towards_pct, p.maths_progress_lower_ci, p.maths_progress_upper_ci`; (KS4 block, after the Progress 8 CI columns): `p.progress_8_banding, p.attainment_8_disadvantage_gap, p.progress_8_disadvantage_gap`. Note `_MAIN_QUERY_NO_SIXTH_FORM`/`_MAIN_QUERY_LEGACY_NAMES` are string-derived from `_MAIN_QUERY` (lines 259-270) and inherit automatically — verify the assertions there still hold.
|
|
||||||
- `get_supplementary_data(db, urn)["admissions"]` rows additionally carry: `total_offers`, `second_preference_offers`, `third_preference_offers`, `cross_la_applications`, `cross_la_offers` (add to `_admissions_row`).
|
|
||||||
- `get_supplementary_data(db, urn)["ofsted"]` additionally carries: `report_card` (the `report_card_labels(...)` dict, `{}` when no rc data), `ofsted_page_url`, and `grade_source`: `"graded"` when `overall_effectiveness` came from the graded column, `"ungraded_carried_forward"` when the fallback `ungraded_grade` supplied it, `None` when neither.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Failing tests** — construct a fake Ofsted row object (simple `types.SimpleNamespace` with the model's attributes) and call the block-building logic via `get_supplementary_data` with a stubbed session (follow how existing tests stub the db; if none do, factor the ofsted-dict construction into a pure helper `_ofsted_block(o, urn)` and test that directly — preferred):
|
|
||||||
|
|
||||||
```python
|
|
||||||
import types
|
|
||||||
from backend.data_loader import _ofsted_block
|
|
||||||
|
|
||||||
|
|
||||||
def _row(**kw):
|
|
||||||
base = dict(
|
|
||||||
framework="RC", inspection_date=None, inspection_type=None,
|
|
||||||
overall_effectiveness=None, quality_of_education=None,
|
|
||||||
behaviour_attitudes=None, personal_development=None,
|
|
||||||
leadership_management=None, early_years_provision=None,
|
|
||||||
sixth_form_provision=None, ungraded_outcome=None, ungraded_grade=None,
|
|
||||||
rc_safeguarding_met=None, rc_inclusion=None, rc_curriculum_teaching=None,
|
|
||||||
rc_achievement=None, rc_attendance_behaviour=None,
|
|
||||||
rc_personal_development=None, rc_leadership_governance=None,
|
|
||||||
rc_early_years=None, rc_sixth_form=None, report_url=None,
|
|
||||||
)
|
|
||||||
base.update(kw)
|
|
||||||
return types.SimpleNamespace(**base)
|
|
||||||
|
|
||||||
|
|
||||||
def test_report_card_block_and_provider_url():
|
|
||||||
o = _row(rc_achievement=2, rc_inclusion=3, rc_safeguarding_met=True)
|
|
||||||
block = _ofsted_block(o, urn=100140)
|
|
||||||
assert block["report_card"]["rc_achievement"]["label"] == "Strong standard"
|
|
||||||
assert "rc_safeguarding_met" not in block["report_card"]
|
|
||||||
assert block["rc_safeguarding_met"] is True
|
|
||||||
assert block["ofsted_page_url"] == "https://reports.ofsted.gov.uk/provider/21/100140"
|
|
||||||
|
|
||||||
|
|
||||||
def test_grade_source_graded_vs_carried_forward():
|
|
||||||
assert _ofsted_block(_row(overall_effectiveness=1), urn=1)["grade_source"] == "graded"
|
|
||||||
carried = _ofsted_block(_row(ungraded_grade=2), urn=1)
|
|
||||||
assert carried["grade_source"] == "ungraded_carried_forward"
|
|
||||||
assert carried["overall_effectiveness"] == 2
|
|
||||||
assert _ofsted_block(_row(), urn=1)["grade_source"] is None
|
|
||||||
|
|
||||||
|
|
||||||
def test_admissions_row_new_fields():
|
|
||||||
from backend.data_loader import _admissions_row_dict
|
|
||||||
a = types.SimpleNamespace(
|
|
||||||
year=202627, school_phase="Primary", places_offered=80,
|
|
||||||
total_applications=185, first_preference_applications=74,
|
|
||||||
first_preference_offers=74, first_preference_offer_pct=100.0,
|
|
||||||
oversubscription_ratio=0.925, oversubscribed=False,
|
|
||||||
total_offers=80, second_preference_offers=4, third_preference_offers=2,
|
|
||||||
cross_la_applications=12, cross_la_offers=3,
|
|
||||||
)
|
|
||||||
d = _admissions_row_dict(a)
|
|
||||||
for k in ("total_offers", "second_preference_offers", "third_preference_offers",
|
|
||||||
"cross_la_applications", "cross_la_offers"):
|
|
||||||
assert d[k] == getattr(a, k)
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Run → FAIL (helpers don't exist).
|
|
||||||
|
|
||||||
- [ ] **Step 3: Implement.** Refactor the existing inline ofsted-dict construction in `get_supplementary_data` into a module-level `_ofsted_block(o, urn)` that produces the existing keys **unchanged** plus the three new ones (`report_card` via `report_card_labels(...)` from Task 3, `ofsted_page_url` via `ofsted_page_url(urn)`, `grade_source` per the interface rule — derived from which source supplied `overall_effectiveness`). Rename/extract the local `_admissions_row` into module-level `_admissions_row_dict(a)` and append the five new fields. Add the ten new columns to `_MAIN_QUERY` exactly as the interface lists them. `get_supplementary_data` calls both helpers; its external shape gains only additive keys.
|
|
||||||
|
|
||||||
- [ ] **Step 4:** Full suite → all pass (existing `test_school_details.py` etc. must not break; if a fixture enumerates yearly-data columns, extend it with the new NaN columns as needed).
|
|
||||||
|
|
||||||
- [ ] **Step 5:** Commit: `feat(api): expose progress CIs, KS4 banding/gaps, admissions detail, report-card labels`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 5: Computed benchmarks helper (TDD)
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/data_loader.py` (new function)
|
|
||||||
- Test: `backend/tests/test_benchmarks.py`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces for Task 6: `compute_benchmarks(df) -> dict` — pure function over the main dataframe (latest year, state schools), shape:
|
|
||||||
|
|
||||||
```python
|
|
||||||
{
|
|
||||||
"source": "state-school average (computed from our dataset)",
|
|
||||||
"year": 202425,
|
|
||||||
"primary": {
|
|
||||||
"disadvantaged_rwm_expected_pct": 46.1, # weighted by eligible_pupils
|
|
||||||
"eal_pct": 22.3, # median
|
|
||||||
"sen_support_pct": 14.0, # median
|
|
||||||
"disadvantaged_pct": 24.8, # median (FSM6 proxy)
|
|
||||||
"median_pupils": 281, # median school size
|
|
||||||
},
|
|
||||||
"secondary": { "median_pupils": 1024, "eal_pct": ..., "sen_support_pct": ..., "disadvantaged_pct": ... },
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 1: Failing tests** — build a small synthetic df (6 primary rows with known eligible_pupils/rwm_expected_disadvantaged_pct so the weighted average is hand-checkable; a couple of secondary rows flagged by non-null `attainment_8_score`), assert: weighted disadvantaged average matches hand computation (not the unweighted mean), medians ignore NaN, secondary block lacks the disadvantaged-RWM key, latest-year filtering (rows from an older year must not affect results), and empty df → `{}`.
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Run → FAIL.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Implement** in `data_loader.py`:
|
|
||||||
|
|
||||||
```python
|
|
||||||
def compute_benchmarks(df: pd.DataFrame) -> dict:
|
|
||||||
"""State-school benchmarks computed from our dataset (spec §5/§8.6).
|
|
||||||
These are NOT official DfE figures — consumers must label them
|
|
||||||
'state-school average (computed from our dataset)'."""
|
|
||||||
if df.empty or "year" not in df.columns:
|
|
||||||
return {}
|
|
||||||
latest_year = df["year"].max()
|
|
||||||
d = df[df["year"] == latest_year]
|
|
||||||
if d.empty:
|
|
||||||
return {}
|
|
||||||
is_secondary = d["attainment_8_score"].notna() if "attainment_8_score" in d.columns else pd.Series(False, index=d.index)
|
|
||||||
prim, sec = d[~is_secondary], d[is_secondary]
|
|
||||||
|
|
||||||
def _median(sub, col):
|
|
||||||
if col not in sub.columns:
|
|
||||||
return None
|
|
||||||
v = sub[col].median()
|
|
||||||
return round(float(v), 1) if pd.notna(v) else None
|
|
||||||
|
|
||||||
def _weighted_disadvantaged(sub):
|
|
||||||
if not {"rwm_expected_disadvantaged_pct", "eligible_pupils"} <= set(sub.columns):
|
|
||||||
return None
|
|
||||||
s = sub.dropna(subset=["rwm_expected_disadvantaged_pct", "eligible_pupils"])
|
|
||||||
if s.empty or s["eligible_pupils"].sum() == 0:
|
|
||||||
return None
|
|
||||||
w = (s["rwm_expected_disadvantaged_pct"] * s["eligible_pupils"]).sum() / s["eligible_pupils"].sum()
|
|
||||||
return round(float(w), 1)
|
|
||||||
|
|
||||||
def _block(sub, with_disadvantaged):
|
|
||||||
block = {
|
|
||||||
"eal_pct": _median(sub, "eal_pct"),
|
|
||||||
"sen_support_pct": _median(sub, "sen_support_pct"),
|
|
||||||
"disadvantaged_pct": _median(sub, "disadvantaged_pct"),
|
|
||||||
"median_pupils": int(sub["total_pupils"].median()) if "total_pupils" in sub.columns and pd.notna(sub["total_pupils"].median()) else None,
|
|
||||||
}
|
|
||||||
if with_disadvantaged:
|
|
||||||
block["disadvantaged_rwm_expected_pct"] = _weighted_disadvantaged(sub)
|
|
||||||
return block
|
|
||||||
|
|
||||||
return {
|
|
||||||
"source": "state-school average (computed from our dataset)",
|
|
||||||
"year": int(latest_year),
|
|
||||||
"primary": _block(prim, with_disadvantaged=True),
|
|
||||||
"secondary": _block(sec, with_disadvantaged=False),
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
(Adapt column presence to the real df — `sen_support_pct` reaches the df via `_MAIN_QUERY`; confirm and add it there if the KS2 block doesn't already select it, mirroring Task 4's additions.)
|
|
||||||
|
|
||||||
- [ ] **Step 4:** Full suite → pass. **Step 5:** Commit: `feat(api): computed state-school benchmarks`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 6: Enrich `/api/compare` + expose GPS/science national averages (TDD)
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `backend/app.py` (`compare_schools` ~line 636; `get_national_averages` ~line 730)
|
|
||||||
- Test: `backend/tests/test_compare_enrichment.py`
|
|
||||||
|
|
||||||
**Interfaces (response additions, all additive):**
|
|
||||||
- `/api/compare` top level gains: `"national_averages"` (same payload the `/api/national-averages` endpoint returns — extract the endpoint body into a helper `_national_averages_payload(df)` and reuse; do not duplicate the logic) and `"benchmarks"` (Task 5's `compute_benchmarks(df)`).
|
|
||||||
- Each `comparison[urn]` gains: `"ofsted"`, `"census"`, `"admissions"`, `"admissions_history"`, `"deprivation"` from `get_supplementary_data` (one `SessionLocal()` for the whole request, closed in `finally`; on exception the five keys are `None`/`[]` — mirror the detail endpoint's defensive pattern at app.py:583-590).
|
|
||||||
- `get_national_averages`' KS2 metric list gains `"gps_expected_pct", "gps_high_pct", "science_expected_pct"` so the England ticks for GPS/science flow once the data exists.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Failing tests** — monkeypatch `load_school_data` with a two-school primary df (reuse/extend the fixture style of `test_school_details.py`) and monkeypatch `get_supplementary_data` to a canned dict; assert on `TestClient(app).get("/api/compare?urns=...")`:
|
|
||||||
- response keeps the existing shape (`comparison[urn]["school_info"]["rwm_expected_pct"]` etc.),
|
|
||||||
- each school gains the five supplementary keys (canned values round-tripped),
|
|
||||||
- top-level `national_averages` and `benchmarks` present; `benchmarks["source"]` is the exact provenance string,
|
|
||||||
- a supplementary-layer exception (monkeypatched to raise) degrades to `ofsted: None` etc. with HTTP 200,
|
|
||||||
- `/api/national-averages` includes `gps_expected_pct` in the primary block when the df/national table provides it (monkeypatch the national-averages source the endpoint reads).
|
|
||||||
|
|
||||||
- [ ] **Step 2:** Run → FAIL. **Step 3:** Implement per the interfaces. **Step 4:** Full suite → pass.
|
|
||||||
|
|
||||||
- [ ] **Step 5:** Commit: `feat(api): compare endpoint carries supplementary blocks, national averages and benchmarks`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 7: PR + verification
|
|
||||||
|
|
||||||
- [ ] **Step 1:** Full suite one more time + `uv run --with pyyaml python3 -c "import yaml; yaml.safe_load(open('.gitea/workflows/deploy.yml'))"` sanity is NOT needed (no workflow changes) — instead run the dbt parse gate again (Task 1 file).
|
|
||||||
- [ ] **Step 2:** Push, open PR via the Gitea API (credential-helper basic auth). PR body: the new response shapes (one JSON sketch), the reused-not-duplicated national-averages helper, the provenance rule for benchmarks, deploy note (fields NULL until prod DAGs run post-promotion), and that no e2e change is needed (no user-facing behaviour changes — the compare UI still reads the old fields; the frontend PR carries the journey updates).
|
|
||||||
- [ ] **Step 3:** After merge + staging deploy: `curl -s https://stx.schoolcompare.co.uk/api/compare?urns=138690,100140 | python3 -m json.tool | head -80` — verify the new keys and that `benchmarks.primary.disadvantaged_rwm_expected_pct` is plausible (~45-47). Verify `/api/national-averages` now carries `gps_expected_pct`/`science_expected_pct` (values or honest nulls if DfE suppresses them at national level).
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Out of scope
|
|
||||||
|
|
||||||
- Frontend rebuild + e2e journeys (next PR — consumes everything this PR exposes).
|
|
||||||
- `schemas.py` METRIC_DEFINITIONS additions for the trends picker (frontend PR decides which of the new columns become picker metrics).
|
|
||||||
- CI-based progress banding logic (frontend computes Above/Average/Below from the CI columns; historical years only).
|
|
||||||
@@ -1,275 +0,0 @@
|
|||||||
# Staged Production Promotion (Manual Gate) Implementation Plan
|
|
||||||
|
|
||||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
|
||||||
|
|
||||||
**Goal:** Merging a PR deploys to staging only; production deployment requires a second, explicit human approval after manual testing on staging.
|
|
||||||
|
|
||||||
**Architecture:** Split the existing single `deploy.yml` pipeline in two. The push-to-main workflow keeps build → staging deploy → e2e gate and **stops there**. A new `promote.yml` runs only on `workflow_dispatch` (the "Run workflow" button in Gitea's Actions UI, supported on this server — Gitea 1.26.4): it verifies the chosen commit passed the staging e2e gate, retags its `:sha-*` images to `:prod` (keeping `:prod-previous` for rollback), and triggers the Portainer prod webhook. Promotion granularity is a main-branch commit: staging always runs the latest main, so you approve a *state of main*, not an individual PR.
|
|
||||||
|
|
||||||
**Tech Stack:** Gitea Actions (1.26.4), Docker buildx imagetools, Portainer webhooks, Gitea commit-status API.
|
|
||||||
|
|
||||||
## Global Constraints
|
|
||||||
|
|
||||||
- **Never push to `main` directly** — this change itself goes through a PR (`chore/staged-prod-promotion` branch).
|
|
||||||
- Existing image tagging scheme is unchanged: `type=sha` (e.g. `sha-6f925ab`) + `:staging`; promotion still retags `:sha-*` → `:prod` with `:prod-previous` kept as the rollback pointer.
|
|
||||||
- The e2e journeys remain a **hard gate before human testing** (a red staging never reaches the promote button) and the promote workflow must refuse to promote a commit whose staging e2e did not succeed.
|
|
||||||
- Secrets already exist and are reused: `REGISTRY_TOKEN` (also a Gitea API token), `PORTAINER_STAGING_WEBHOOK`, `PORTAINER_PROD_WEBHOOK`, `STAGING_BASE_URL`, `PROD_BASE_URL`.
|
|
||||||
- Staging quirk (memory): external `/api` is broken at the staging proxy — manual API testing happens from the host, not through stx.schoolcompare.co.uk; note it in the runbook, don't try to fix it in this plan.
|
|
||||||
|
|
||||||
## Considered approaches (context for the reviewer)
|
|
||||||
|
|
||||||
1. **Manual `workflow_dispatch` promote workflow (chosen).** Native on Gitea 1.26; the second approval is clicking "Run workflow" (or one API call) after testing staging. Least machinery, auditable via the Actions run history.
|
|
||||||
2. *Tag-driven promotion* (`push: tags: promote-*`): works on any Gitea version; approval = pushing a tag. Slightly more scriptable, less discoverable; kept as documented fallback only.
|
|
||||||
3. *GitOps `production` branch + promotion PR:* approval literally reuses the PR-review UI, but adds a second long-lived branch to keep in sync — too much ceremony for a solo project. Rejected.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 0: Branch
|
|
||||||
|
|
||||||
- [ ] `git checkout main && git pull && git checkout -b chore/staged-prod-promotion`
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 1: Stop the push-to-main workflow after the e2e gate
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `.gitea/workflows/deploy.yml`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Produces: images tagged `:sha-<short>` + `:staging` (unchanged), a green `E2E Journeys against Staging` commit status that Task 2's promote workflow checks by name. **Do not rename the `e2e-staging` job's `name:` without updating Task 2's status check.**
|
|
||||||
|
|
||||||
- [ ] **Step 1: Remove the auto-promotion**
|
|
||||||
|
|
||||||
In `.gitea/workflows/deploy.yml`:
|
|
||||||
1. Change line 1 to: `name: Stage (build -> staging -> E2E gate)`
|
|
||||||
2. Delete the entire `promote-prod` job (lines 196–240 in the current file: from ` promote-prod:` to the end of the file).
|
|
||||||
3. Leave `build-*`, `deploy-staging`, and `e2e-staging` untouched.
|
|
||||||
|
|
||||||
- [ ] **Step 2: Sanity-check the YAML**
|
|
||||||
|
|
||||||
Run: `python3 -c "import yaml; yaml.safe_load(open('.gitea/workflows/deploy.yml')); print('yaml ok')"`
|
|
||||||
Expected: `yaml ok`
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add .gitea/workflows/deploy.yml
|
|
||||||
git commit -m "ci: stop deploy pipeline at staging; production promotion becomes manual"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 2: Manual promote workflow
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Create: `.gitea/workflows/promote.yml`
|
|
||||||
|
|
||||||
**Interfaces:**
|
|
||||||
- Consumes: `:sha-<short>` images built by deploy.yml; the `E2E Journeys against Staging` commit status.
|
|
||||||
- Produces: `:prod` and `:prod-previous` tags; prod stack update.
|
|
||||||
|
|
||||||
- [ ] **Step 1: Write the workflow**
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
name: Promote to Production (manual)
|
|
||||||
|
|
||||||
on:
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
sha:
|
|
||||||
description: >-
|
|
||||||
Commit SHA on main to promote (full or >=7 chars).
|
|
||||||
Leave empty to promote the latest main commit.
|
|
||||||
required: false
|
|
||||||
default: ""
|
|
||||||
|
|
||||||
env:
|
|
||||||
REGISTRY: privaterepo.sitaru.org
|
|
||||||
BACKEND_IMAGE_NAME: ${{ gitea.repository }}-backend
|
|
||||||
FRONTEND_IMAGE_NAME: ${{ gitea.repository }}-frontend
|
|
||||||
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
promote-prod:
|
|
||||||
name: Promote approved commit to Production
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Resolve target SHA
|
|
||||||
id: resolve
|
|
||||||
run: |
|
|
||||||
SHA_INPUT="${{ gitea.event.inputs.sha }}"
|
|
||||||
if [ -z "$SHA_INPUT" ]; then
|
|
||||||
SHA_INPUT="${{ gitea.sha }}"
|
|
||||||
fi
|
|
||||||
# Normalise to the full sha via the API so short inputs work
|
|
||||||
FULL_SHA=$(curl -fsS \
|
|
||||||
-H "Authorization: token ${{ secrets.REGISTRY_TOKEN }}" \
|
|
||||||
"https://${REGISTRY}/api/v1/repos/${{ gitea.repository }}/git/commits/${SHA_INPUT}" \
|
|
||||||
| python3 -c "import json,sys; print(json.load(sys.stdin)['sha'])")
|
|
||||||
SHORT_SHA="sha-$(echo "$FULL_SHA" | cut -c1-7)"
|
|
||||||
echo "full=$FULL_SHA" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "short=$SHORT_SHA" >> "$GITHUB_OUTPUT"
|
|
||||||
echo "Promoting $FULL_SHA (images tagged $SHORT_SHA)"
|
|
||||||
|
|
||||||
- name: Verify the staging E2E gate passed for this commit
|
|
||||||
run: |
|
|
||||||
STATUS_JSON=$(curl -fsS \
|
|
||||||
-H "Authorization: token ${{ secrets.REGISTRY_TOKEN }}" \
|
|
||||||
"https://${REGISTRY}/api/v1/repos/${{ gitea.repository }}/commits/${{ steps.resolve.outputs.full }}/status")
|
|
||||||
echo "$STATUS_JSON" | python3 -c "
|
|
||||||
import json, sys
|
|
||||||
d = json.load(sys.stdin)
|
|
||||||
ok = [s for s in d.get('statuses', [])
|
|
||||||
if 'E2E Journeys against Staging' in s.get('context', '')
|
|
||||||
and s.get('status') == 'success']
|
|
||||||
if not ok:
|
|
||||||
print('REFUSED: no successful \"E2E Journeys against Staging\" status on this commit.')
|
|
||||||
print('Contexts found:', [s.get('context') for s in d.get('statuses', [])])
|
|
||||||
sys.exit(1)
|
|
||||||
print('E2E gate verified green for this commit.')
|
|
||||||
"
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
|
||||||
uses: docker/setup-buildx-action@v3
|
|
||||||
|
|
||||||
- name: Log in to Gitea Container Registry
|
|
||||||
uses: docker/login-action@v3
|
|
||||||
with:
|
|
||||||
registry: ${{ env.REGISTRY }}
|
|
||||||
username: ${{ gitea.actor }}
|
|
||||||
password: ${{ secrets.REGISTRY_TOKEN }}
|
|
||||||
|
|
||||||
- name: Retag approved images as prod (keeping rollback pointer)
|
|
||||||
run: |
|
|
||||||
SHORT_SHA="${{ steps.resolve.outputs.short }}"
|
|
||||||
for IMAGE in \
|
|
||||||
"${REGISTRY}/${BACKEND_IMAGE_NAME}" \
|
|
||||||
"${REGISTRY}/${FRONTEND_IMAGE_NAME}" \
|
|
||||||
"${REGISTRY}/${PIPELINE_IMAGE_NAME}"; do
|
|
||||||
docker buildx imagetools create -t "${IMAGE}:prod-previous" "${IMAGE}:prod" || true
|
|
||||||
docker buildx imagetools create -t "${IMAGE}:prod" "${IMAGE}:${SHORT_SHA}"
|
|
||||||
echo "Promoted ${IMAGE}:${SHORT_SHA} -> :prod"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Trigger production stack update
|
|
||||||
run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}"
|
|
||||||
|
|
||||||
- name: Wait for production to become healthy
|
|
||||||
run: |
|
|
||||||
echo "Polling ${PROD_BASE_URL} for up to 5 minutes..."
|
|
||||||
for i in $(seq 1 60); do
|
|
||||||
if curl -fsS -o /dev/null --max-time 10 "${PROD_BASE_URL}/"; then
|
|
||||||
echo "Production is up (attempt $i)"
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
sleep 5
|
|
||||||
done
|
|
||||||
echo "Production did not become healthy in time" >&2
|
|
||||||
exit 1
|
|
||||||
env:
|
|
||||||
PROD_BASE_URL: ${{ secrets.PROD_BASE_URL }}
|
|
||||||
```
|
|
||||||
|
|
||||||
Implementation notes for the engineer:
|
|
||||||
- Gitea Actions uses the GitHub-compatible `$GITHUB_OUTPUT` file for step outputs; if the runner image doesn't populate it, fall back to `$GITEA_OUTPUT` (check the runner's docs/output at first run).
|
|
||||||
- The retag step is copied verbatim from the old `promote-prod` job except the SHA comes from the resolved input instead of `gitea.sha` — behaviour for the default (empty input on latest main) is identical to before.
|
|
||||||
- If `docker buildx imagetools create` fails with "not found" for `${IMAGE}:${SHORT_SHA}`, the chosen commit predates the registry's retention or never built — the error message is the desired behaviour (refuse loudly).
|
|
||||||
|
|
||||||
- [ ] **Step 2: YAML sanity check**
|
|
||||||
|
|
||||||
Run: `python3 -c "import yaml; yaml.safe_load(open('.gitea/workflows/promote.yml')); print('yaml ok')"`
|
|
||||||
Expected: `yaml ok`
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add .gitea/workflows/promote.yml
|
|
||||||
git commit -m "ci: manual production promotion workflow with e2e-gate verification"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 3: Documentation — deploy model + runbook
|
|
||||||
|
|
||||||
**Files:**
|
|
||||||
- Modify: `docs/DEPLOY.md`
|
|
||||||
- Modify: `claude.md` (the SDLC section)
|
|
||||||
|
|
||||||
- [ ] **Step 1: Rewrite the flow description in `docs/DEPLOY.md`**
|
|
||||||
|
|
||||||
Replace the staging→prod description with the new model (adapt to the file's existing structure; the substance to convey):
|
|
||||||
|
|
||||||
```markdown
|
|
||||||
## Deploy model
|
|
||||||
|
|
||||||
1. **PR → main (first approval).** Branch-protected merge; PR checks
|
|
||||||
(typecheck, tests, builds, AI review) must pass.
|
|
||||||
2. **Merge → staging (automatic).** Images are built once and tagged
|
|
||||||
`sha-<short>` + `staging`; the staging stack updates; Playwright
|
|
||||||
journeys in `e2e/` run against staging. A red e2e run means staging
|
|
||||||
is not fit for testing — fix forward before considering promotion.
|
|
||||||
3. **Manual testing on staging.** stx.schoolcompare.co.uk. Note:
|
|
||||||
external `/api` is broken at the staging proxy — exercise API
|
|
||||||
endpoints from the host.
|
|
||||||
4. **Promote → production (second approval).** Actions → "Promote to
|
|
||||||
Production (manual)" → Run workflow. Leave the SHA empty to promote
|
|
||||||
the latest main, or paste a specific commit SHA. The workflow
|
|
||||||
refuses commits whose staging e2e gate is not green, retags the
|
|
||||||
images `:prod` (keeping `:prod-previous`), and updates the prod
|
|
||||||
stack.
|
|
||||||
|
|
||||||
### Promotion granularity
|
|
||||||
|
|
||||||
Staging always runs the latest `main`. Promoting approves a *state of
|
|
||||||
main*, not a single PR — if two PRs merged since the last promotion,
|
|
||||||
they ship together. Test staging accordingly.
|
|
||||||
|
|
||||||
### Rollback
|
|
||||||
|
|
||||||
Re-run "Promote to Production (manual)" with the SHA of the last good
|
|
||||||
commit (or retag manually: `docker buildx imagetools create -t
|
|
||||||
<image>:prod <image>:prod-previous` for each of the three images, then
|
|
||||||
POST the prod Portainer webhook).
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 2: Update the SDLC bullet in `claude.md`**
|
|
||||||
|
|
||||||
Replace the sentence "Merging to `main` deploys automatically: … retagged `:prod` and rolled out to production." with:
|
|
||||||
|
|
||||||
```markdown
|
|
||||||
- Merging to `main` deploys automatically **to staging only**: images
|
|
||||||
are built once, deployed to the staging Portainer stack, and verified
|
|
||||||
by the Playwright journeys in `e2e/`. Production is a second, manual
|
|
||||||
approval: the "Promote to Production (manual)" workflow in Gitea
|
|
||||||
Actions, run after testing the feature on staging. It refuses commits
|
|
||||||
whose staging e2e gate isn't green.
|
|
||||||
```
|
|
||||||
|
|
||||||
- [ ] **Step 3: Commit**
|
|
||||||
|
|
||||||
```bash
|
|
||||||
git add docs/DEPLOY.md claude.md
|
|
||||||
git commit -m "docs: two-stage deploy model (staging auto, production manual)"
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
### Task 4: PR + live validation
|
|
||||||
|
|
||||||
- [ ] **Step 1: Push and open the PR** (Gitea API with credential-helper basic auth, as usual). PR body: the new model in three lines, the rollback recipe, and a warning that between merging this PR and its first promotion run, production receives no deployments (expected).
|
|
||||||
|
|
||||||
- [ ] **Step 2: Validate after merge (human-in-the-loop):**
|
|
||||||
1. Merge this PR → confirm the `Stage (build -> staging -> E2E gate)` run goes green and **no** production deployment happens (prod image digest unchanged: `docker buildx imagetools inspect <image>:prod` before/after, or check the Portainer prod stack's last-update time).
|
|
||||||
2. Test something trivial on staging.
|
|
||||||
3. Run "Promote to Production (manual)" with the SHA empty → confirm e2e verification passes, retag happens, prod becomes healthy.
|
|
||||||
4. Negative test: run the promote workflow with a garbage SHA (e.g. `deadbeef1`) → confirm it fails at resolve/verify without touching `:prod`.
|
|
||||||
|
|
||||||
- [ ] **Step 3: Update the ledger/memory** with the new deploy model so future sessions stop assuming auto-promotion.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Out of scope / future options
|
|
||||||
|
|
||||||
- Notifications when staging is ready for testing (Gitea can email on workflow completion; a webhook to ntfy/Matrix could be added later).
|
|
||||||
- Restricting who can run the promote workflow: Gitea 1.26 runs `workflow_dispatch` with the permissions of the dispatching user; for a solo repo this is already effectively restricted.
|
|
||||||
- The tag-driven fallback (`on: push: tags: promote-*`) if `workflow_dispatch` ever proves unreliable on the runner.
|
|
||||||
@@ -1,215 +0,0 @@
|
|||||||
# Exam Results Taxonomy — Phase Grouping and Sixth-Form Separation
|
|
||||||
|
|
||||||
**Date:** 2026-07-07
|
|
||||||
**Status:** Approved design (taxonomy/analysis only — no implementation in this doc's scope)
|
|
||||||
|
|
||||||
## Purpose
|
|
||||||
|
|
||||||
Classify every exam-result metric SchoolCompare displays today into four phase
|
|
||||||
groups — **Primary**, **Secondary**, **Sixth form**, **Other** — and define an
|
|
||||||
authoritative rule for separating schools that have a sixth form from those
|
|
||||||
that don't. This document is the reference for:
|
|
||||||
|
|
||||||
1. How the UI should group results sections and rankings by phase.
|
|
||||||
2. The future KS5 (A-level) ingestion work — the Sixth form group lists the
|
|
||||||
concrete DfE metrics as placeholders with source columns.
|
|
||||||
3. Replacing the fragile `age_range contains "18"` heuristic with the GIAS
|
|
||||||
`OfficialSixthForm` flag.
|
|
||||||
|
|
||||||
## 1. Grouping principle
|
|
||||||
|
|
||||||
Metrics are grouped by **the key stage of the assessment**, not by the phase
|
|
||||||
of the school displaying them. An all-through school (4–18) shows metrics in
|
|
||||||
all three exam groups; a pure primary shows only the Primary group.
|
|
||||||
|
|
||||||
| Group | Assessments | Key stage | Taken at age | Data status |
|
|
||||||
|---|---|---|---|---|
|
|
||||||
| **Primary** | KS2 SATs (reading, writing TA, maths, GPS, science TA) | KS2 | 10–11 (Year 6) | ✅ Live — `marts.fact_ks2_performance` |
|
|
||||||
| **Secondary** | GCSEs, Attainment 8 / Progress 8, EBacc | KS4 | 15–16 (Year 11) | ✅ Live — `marts.fact_ks4_performance` |
|
|
||||||
| **Sixth form** | A levels, applied general, tech levels | KS5 (16–18) | 17–18 (Year 12–13) | ⏳ Not ingested — placeholders in §4 |
|
|
||||||
| **Other** | Non-exam context displayed alongside results | n/a | n/a | ✅ Live — various marts |
|
|
||||||
|
|
||||||
Not covered (not displayed today, candidates for future "Other"/Primary):
|
|
||||||
EYFS Good Level of Development, Year 1 Phonics check, Year 4 Multiplication
|
|
||||||
Tables Check, KS1 assessments (no longer published at school level by DfE).
|
|
||||||
|
|
||||||
## 2. Metric-by-metric mapping (current site)
|
|
||||||
|
|
||||||
Every key in `backend/schemas.py` `METRIC_DEFINITIONS` — the single source of
|
|
||||||
truth for what the site displays — mapped to its phase group. `category` is
|
|
||||||
the existing schema category; source columns are the DfE names used at
|
|
||||||
ingestion (legacy performance-tables CSV for KS2, EES for KS4).
|
|
||||||
|
|
||||||
### Primary (KS2 SATs)
|
|
||||||
|
|
||||||
| Metric key | Category | DfE source column |
|
|
||||||
|---|---|---|
|
|
||||||
| `rwm_expected_pct` | expected | `PTRWM_EXP` |
|
|
||||||
| `reading_expected_pct` | expected | `PTREAD_EXP` |
|
|
||||||
| `writing_expected_pct` | expected | `PTWRITTA_EXP` |
|
|
||||||
| `maths_expected_pct` | expected | `PTMAT_EXP` |
|
|
||||||
| `gps_expected_pct` | expected | `PTGPS_EXP` |
|
|
||||||
| `science_expected_pct` | expected | `PTSCITA_EXP` |
|
|
||||||
| `rwm_high_pct` | higher | `PTRWM_HIGH` |
|
|
||||||
| `reading_high_pct` | higher | `PTREAD_HIGH` |
|
|
||||||
| `writing_high_pct` | higher | `PTWRITTA_HIGH` |
|
|
||||||
| `maths_high_pct` | higher | `PTMAT_HIGH` |
|
|
||||||
| `gps_high_pct` | higher | `PTGPS_HIGH` |
|
|
||||||
| `reading_progress` | progress | `READPROG` |
|
|
||||||
| `writing_progress` | progress | `WRITPROG` |
|
|
||||||
| `maths_progress` | progress | `MATPROG` |
|
|
||||||
| `reading_avg_score` | average | `READ_AVERAGE` |
|
|
||||||
| `maths_avg_score` | average | `MAT_AVERAGE` |
|
|
||||||
| `gps_avg_score` | average | `GPS_AVERAGE` |
|
|
||||||
| `rwm_expected_boys_pct` | gender | `PTRWM_EXP_B` |
|
|
||||||
| `rwm_expected_girls_pct` | gender | `PTRWM_EXP_G` |
|
|
||||||
| `rwm_high_boys_pct` | gender | `PTRWM_HIGH_B` |
|
|
||||||
| `rwm_high_girls_pct` | gender | `PTRWM_HIGH_G` |
|
|
||||||
| `rwm_expected_disadvantaged_pct` | equity | `PTRWM_EXP_FSM6CLA1A` |
|
|
||||||
| `rwm_expected_non_disadvantaged_pct` | equity | `PTRWM_EXP_NotFSM6CLA1A` |
|
|
||||||
| `disadvantaged_gap` | equity | `DIFFN_RWM_EXP` |
|
|
||||||
| `reading_absence_pct` | absence | `PTREAD_AT` |
|
|
||||||
| `gps_absence_pct` | absence | `PTGPS_AT` |
|
|
||||||
| `maths_absence_pct` | absence | `PTMAT_AT` |
|
|
||||||
| `writing_absence_pct` | absence | `PTWRITTA_AD` |
|
|
||||||
| `science_absence_pct` | absence | `PTSCITA_AD` |
|
|
||||||
| `rwm_expected_3yr_pct` | trends | `PTRWM_EXP_3YR` |
|
|
||||||
| `reading_avg_3yr` | trends | `READ_AVERAGE_3YR` |
|
|
||||||
| `maths_avg_3yr` | trends | `MAT_AVERAGE_3YR` |
|
|
||||||
|
|
||||||
The absence metrics measure absence *from KS2 tests*, so they belong to
|
|
||||||
Primary even though they are not attainment scores. National comparators for
|
|
||||||
this group come from `marts.fact_ks2_national_averages`.
|
|
||||||
|
|
||||||
### Secondary (KS4 / GCSE)
|
|
||||||
|
|
||||||
| Metric key | Category | EES source column |
|
|
||||||
|---|---|---|
|
|
||||||
| `attainment_8_score` | gcse | `attainment8_average` |
|
|
||||||
| `progress_8_score` | gcse | `progress8_average` |
|
|
||||||
| `english_maths_standard_pass_pct` | gcse | `engmath_94_percent` |
|
|
||||||
| `english_maths_strong_pass_pct` | gcse | `engmath_95_percent` |
|
|
||||||
| `ebacc_entry_pct` | gcse | `ebacc_entering_percent` |
|
|
||||||
| `ebacc_standard_pass_pct` | gcse | `ebacc_94_percent` |
|
|
||||||
| `ebacc_strong_pass_pct` | gcse | `ebacc_95_percent` |
|
|
||||||
| `ebacc_avg_score` | gcse | `ebacc_aps_average` |
|
|
||||||
| `gcse_grade_91_pct` | gcse | `gcse_91_percent` |
|
|
||||||
|
|
||||||
Also stored in `marts.fact_ks4_performance` (and `fact_performance`) but not
|
|
||||||
yet in `METRIC_DEFINITIONS` — Secondary group members when surfaced:
|
|
||||||
`progress_8_lower_ci`, `progress_8_upper_ci`, `progress_8_english`,
|
|
||||||
`progress_8_maths`, `progress_8_ebacc`, `progress_8_open`,
|
|
||||||
`prior_attainment_avg` (KS2 baseline of the GCSE cohort), `sen_pct`.
|
|
||||||
|
|
||||||
### Sixth form (KS5)
|
|
||||||
|
|
||||||
No metrics today. The secondary school detail view renders a static note
|
|
||||||
("Post-16 destination data coming soon") when the school has a sixth form.
|
|
||||||
Placeholders for ingestion are specified in §4.
|
|
||||||
|
|
||||||
### Other (non-exam context)
|
|
||||||
|
|
||||||
Displayed alongside results but not tied to any assessment:
|
|
||||||
|
|
||||||
| Metric key / surface | Category | Source |
|
|
||||||
|---|---|---|
|
|
||||||
| `disadvantaged_pct` | context | KS2 CSV `PTFSM6CLA1A` |
|
|
||||||
| `eal_pct` | context | KS2 CSV `PTEALGRP2` |
|
|
||||||
| `sen_support_pct` | context | KS2 CSV `PSENELK` (KS4 fallback `sen_no_ehcp_pupil_percent`) |
|
|
||||||
| `stability_pct` | context | KS2 CSV `PTMOBN` |
|
|
||||||
| Ofsted grades incl. `sixth_form_provision` / `rc_sixth_form` | — | `marts.fact_ofsted_inspection` |
|
|
||||||
| Admissions (offers, oversubscription) | — | `marts.fact_admissions` |
|
|
||||||
| Finance (per-pupil spend, cost shares) | — | `marts.fact_finance` |
|
|
||||||
| Deprivation (IDACI) | — | `marts.fact_deprivation` |
|
|
||||||
| Pupil characteristics (census) | — | `marts.fact_pupil_characteristics` |
|
|
||||||
|
|
||||||
Note: the context metrics are cohort characteristics of the KS2 cohort at
|
|
||||||
source, but they are presented (and should stay presented) as school-level
|
|
||||||
context, so they group as Other, not Primary.
|
|
||||||
|
|
||||||
## 3. Sixth-form separation
|
|
||||||
|
|
||||||
### Definition (authoritative)
|
|
||||||
|
|
||||||
> A school **has a sixth form** iff GIAS `OfficialSixthForm (name)` =
|
|
||||||
> `"Has a sixth form"` for its URN.
|
|
||||||
|
|
||||||
GIAS values are `Has a sixth form`, `Does not have a sixth form`, and
|
|
||||||
`Not applicable` / blank. `Not applicable` (nurseries, primaries, PRUs) maps
|
|
||||||
to **false**. This field is the DfE's registry flag, updated continuously,
|
|
||||||
and is the only source that correctly classifies:
|
|
||||||
|
|
||||||
- 16–19 sixth-form colleges and UTCs (age ranges like `14-19`, `16-19` that
|
|
||||||
the current substring heuristic misclassifies as *no* sixth form);
|
|
||||||
- schools whose statutory age range extends to 18 on paper but which have no
|
|
||||||
registered post-16 provision.
|
|
||||||
|
|
||||||
### Pipeline change (implemented 2026-07-07)
|
|
||||||
|
|
||||||
1. `stg_gias_establishments.sql`: add
|
|
||||||
`"OfficialSixthForm (name)" as official_sixth_form`.
|
|
||||||
2. `dim_school.sql` (+ `models.py` `DimSchool`, `_marts_schema.yml`): add
|
|
||||||
`has_sixth_form boolean` = `official_sixth_form = 'Has a sixth form'`.
|
|
||||||
3. Expose `has_sixth_form` on the school API payloads.
|
|
||||||
|
|
||||||
Implemented in `feat/gias-sixth-form-flag` — see
|
|
||||||
`docs/superpowers/plans/2026-07-07-gias-sixth-form-flag.md`.
|
|
||||||
|
|
||||||
### Current heuristic — audit of `age_range` ~ "18" sites
|
|
||||||
|
|
||||||
All must migrate to the `has_sixth_form` flag once exposed:
|
|
||||||
|
|
||||||
| Site | Current behaviour |
|
|
||||||
|---|---|
|
|
||||||
| `backend/app.py:419-422` | `/api/schools?has_sixth_form=yes\|no` filters on `age_range.str.contains("18")` |
|
|
||||||
| `nextjs-app/components/SecondarySchoolDetailView.tsx:101` | "Sixth form" badge + coming-soon note from `age_range?.includes('18')` |
|
|
||||||
| `nextjs-app/components/FilterBar.tsx:370-372` | Filter labels hard-code "(11-18)" / "(11-16)" — labels should drop the age-range parenthetical since sixth form ≠ age range |
|
|
||||||
|
|
||||||
Fallback rule: if GIAS is blank for a URN (rare; new establishments), fall
|
|
||||||
back to the age-range heuristic and log the URN.
|
|
||||||
|
|
||||||
### UI separation rules
|
|
||||||
|
|
||||||
- **School page**: schools with `has_sixth_form = true` show a Sixth form
|
|
||||||
results section (placeholder until KS5 data lands); schools without never
|
|
||||||
show it. Badge on the header as today, but driven by the flag.
|
|
||||||
- **Search/rankings filter**: "With sixth form" / "Without sixth form" uses
|
|
||||||
the flag; applies to secondary and all-through phases.
|
|
||||||
- **Comparison**: when comparing a with-sixth-form school against one
|
|
||||||
without, the Sixth form group renders "No sixth form" for the latter
|
|
||||||
rather than blank cells, making the structural difference explicit.
|
|
||||||
|
|
||||||
## 4. Sixth form placeholders — future KS5 ingestion spec
|
|
||||||
|
|
||||||
Source: DfE "A level and other 16 to 18 results" (EES, preferred — matches
|
|
||||||
the KS4 EES tap) or legacy performance-tables `england_ks5final.csv`.
|
|
||||||
Column names below are from the legacy KS5 CSV; verify against the EES
|
|
||||||
release chosen at ingestion time.
|
|
||||||
|
|
||||||
| Proposed metric key | Name | Legacy source column | Type |
|
|
||||||
|---|---|---|---|
|
|
||||||
| `alevel_aps_per_entry` | A level average points per entry | `TALLPPE_ALEV_1618` | score |
|
|
||||||
| `alevel_avg_grade` | A level average grade (e.g. B-) | `TALLPPEGRD_ALEV_1618` | grade |
|
|
||||||
| `academic_aps_per_entry` | Academic qualifications APS per entry | `TALLPPE_ACAD_1618` | score |
|
|
||||||
| `applied_general_aps_per_entry` | Applied general APS per entry | `TALLPPE_AGEN_1618` | score |
|
|
||||||
| `tech_level_aps_per_entry` | Tech level APS per entry | `TALLPPE_TLEV_1618` | score |
|
|
||||||
| `english_progress_1618` | English progress (16–18, unfinished GCSE 4+) | `PROGENG_1618` | score |
|
|
||||||
| `maths_progress_1618` | Maths progress (16–18) | `PROGMAT_1618` | score |
|
|
||||||
| `ks5_cohort_size` | Students at end of 16–18 study | `TALLPUP_1618` | count |
|
|
||||||
| `alevel_3plus_aab_pct` | % achieving AAB+ in ≥2 facilitating subjects | `TAAB2FAC_1618` | percentage |
|
|
||||||
| `ks5_retention_pct` | Retention (completed main programme) | study-programme retention measure | percentage |
|
|
||||||
| `ks5_destinations_pct` | Sustained education/employment destination | 16–18 destination measures dataset | percentage |
|
|
||||||
|
|
||||||
Proposed landing shape mirrors KS4: `stg_ees_ks5.sql` →
|
|
||||||
`int_ks5_with_lineage.sql` → `marts.fact_ks5_performance` (one row per URN
|
|
||||||
per year), joined into `fact_performance`, with a `category: "sixth_form"`
|
|
||||||
(or `"alevel"`) block added to `METRIC_DEFINITIONS`.
|
|
||||||
|
|
||||||
## 5. Out of scope
|
|
||||||
|
|
||||||
- Any implementation (pipeline, API, or UI changes) — this is the taxonomy
|
|
||||||
reference; implementation work items are §3 "Pipeline change", the
|
|
||||||
heuristic migration audit, and §4 ingestion, each to be planned separately.
|
|
||||||
- Middle schools (deemed secondary/primary): they follow the assessment-based
|
|
||||||
grouping automatically — no special casing.
|
|
||||||
- Independent schools: no DfE performance data published; unaffected.
|
|
||||||
@@ -1,182 +0,0 @@
|
|||||||
# GIAS Code Dictionaries — Codes in Marts, Names in Code
|
|
||||||
|
|
||||||
**Date:** 2026-07-09
|
|
||||||
**Status:** Implemented 2026-07-09 — see docs/superpowers/plans/2026-07-09-gias-code-dictionaries.md
|
|
||||||
|
|
||||||
## Goal
|
|
||||||
|
|
||||||
Six GIAS classification fields are stored in the marts as repeated name
|
|
||||||
strings. Replace them with the official DfE integer codes and translate
|
|
||||||
code → name in application code. After this change the marts carry only
|
|
||||||
codes for:
|
|
||||||
|
|
||||||
| GIAS field | Today (marts, string) | After (marts, int) |
|
|
||||||
|---|---|---|
|
|
||||||
| `TypeOfEstablishment (name)` | `dim_school.school_type` | `school_type_code` |
|
|
||||||
| `EstablishmentStatus (name)` | `dim_school.status` | `status_code` |
|
|
||||||
| `PhaseOfEducation (name)` | `dim_school.phase` | `phase_code` |
|
|
||||||
| `OfficialSixthForm (name)` | (already reduced to `has_sixth_form` bool) | `official_sixth_form_code` in staging only; mart keeps the bool |
|
|
||||||
| `ReligiousCharacter (name)` | `dim_school.religious_character` | `religious_character_code` |
|
|
||||||
| `AdmissionsPolicy (name)` | `dim_school.admissions_policy` | `admissions_policy_code` |
|
|
||||||
|
|
||||||
Motivation: smaller marts and stable enum values for filtering. (Honest
|
|
||||||
sizing note: at ~25k open schools the raw performance win is modest; the
|
|
||||||
durable benefits are storage, DfE-governed vocabulary, and filter values
|
|
||||||
that can't drift with GIAS renames.)
|
|
||||||
|
|
||||||
## Decisions (made during brainstorming)
|
|
||||||
|
|
||||||
1. **GIAS native codes**, not custom enums. The GIAS bulk CSV publishes an
|
|
||||||
official `X (code)` column beside every `X (name)` column. We ingest the
|
|
||||||
DfE's own codes; no invented mapping to maintain.
|
|
||||||
2. **Translation lives in the backend at the API boundary.** The API keeps
|
|
||||||
serving today's name strings; the frontend, e2e journeys, and API
|
|
||||||
consumers are untouched.
|
|
||||||
|
|
||||||
## Design
|
|
||||||
|
|
||||||
### 1. Tap (Singer schema)
|
|
||||||
|
|
||||||
Add the six `(code)` columns to `GIASEstablishmentsStream.schema` in
|
|
||||||
`pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py`:
|
|
||||||
|
|
||||||
```
|
|
||||||
"TypeOfEstablishment (code)", "EstablishmentStatus (code)",
|
|
||||||
"PhaseOfEducation (code)", "OfficialSixthForm (code)",
|
|
||||||
"ReligiousCharacter (code)", "AdmissionsPolicy (code)"
|
|
||||||
```
|
|
||||||
|
|
||||||
The `(name)` columns **stay declared** — raw keeps both so we can detect
|
|
||||||
dictionary drift (§4) and regenerate dictionaries from live data.
|
|
||||||
|
|
||||||
### 2. Staging (`stg_gias_establishments.sql`)
|
|
||||||
|
|
||||||
- Add int casts: `school_type_code`, `status_code`, `phase_code`,
|
|
||||||
`official_sixth_form_code`, `religious_character_code`,
|
|
||||||
`admissions_policy_code` (all `cast(nullif(trim(...), '') as integer)`).
|
|
||||||
- Remove the corresponding name columns from the staging select
|
|
||||||
(`school_type`, `status`, `phase`, `official_sixth_form`,
|
|
||||||
`religious_character`, `admissions_policy`). Names live only in raw.
|
|
||||||
|
|
||||||
### 3. Marts
|
|
||||||
|
|
||||||
**`dim_school`** stores codes only:
|
|
||||||
|
|
||||||
- `school_type_code`, `status_code`, `phase_code`,
|
|
||||||
`religious_character_code`, `admissions_policy_code` replace their
|
|
||||||
string columns.
|
|
||||||
- Status filter becomes `where status_code in (<open>, <proposed-to-close>)`.
|
|
||||||
The numeric values are read from live raw data at implementation time
|
|
||||||
(`select distinct "EstablishmentStatus (code)", "EstablishmentStatus (name)"`),
|
|
||||||
never assumed from memory. Same filter in `dim_location`.
|
|
||||||
- `has_sixth_form` derives from `official_sixth_form_code`
|
|
||||||
(`<has-code>` → true, `<does-not>/<not-applicable>` → false, null →
|
|
||||||
`statutory_high_age >= 18` fallback). The `lower(trim(...))` string guard
|
|
||||||
becomes obsolete and is removed.
|
|
||||||
- `phase_code` derivation keeps today's cascade but emits codes:
|
|
||||||
1. GIAS `phase_code` when it is a real value (not the not-applicable code);
|
|
||||||
2. statutory-age inference emits the matching GIAS code
|
|
||||||
(Primary / Secondary / All-through — numeric values confirmed from
|
|
||||||
live data at implementation);
|
|
||||||
3. school-name heuristics (unchanged — they match `school_name`, which is
|
|
||||||
not one of the six fields) emit the same codes;
|
|
||||||
4. else null.
|
|
||||||
- dbt schema tests: `accepted_values` (severity **warn**) on every code
|
|
||||||
column, values taken from the dictionary; `not_null` warn on `phase_code`
|
|
||||||
(mirrors today's phase test); `has_sixth_form` tests unchanged.
|
|
||||||
|
|
||||||
**`dim_location`**: only the status filter changes (must stay byte-identical
|
|
||||||
to `dim_school`'s — the API inner-joins the two).
|
|
||||||
|
|
||||||
### 4. Dictionaries
|
|
||||||
|
|
||||||
**Canonical module: `backend/gias_codes.py`**
|
|
||||||
|
|
||||||
```python
|
|
||||||
ESTABLISHMENT_STATUS: dict[int, str]
|
|
||||||
SCHOOL_TYPE: dict[int, str]
|
|
||||||
PHASE_OF_EDUCATION: dict[int, str]
|
|
||||||
OFFICIAL_SIXTH_FORM: dict[int, str]
|
|
||||||
RELIGIOUS_CHARACTER: dict[int, str]
|
|
||||||
ADMISSIONS_POLICY: dict[int, str]
|
|
||||||
|
|
||||||
def translate(code: int | None, mapping: dict[int, str]) -> str | None:
|
|
||||||
"""None -> None; unknown code -> 'Unknown (<code>)' + warning log."""
|
|
||||||
```
|
|
||||||
|
|
||||||
- Contents are generated from live raw data
|
|
||||||
(`SELECT DISTINCT code, name FROM raw.gias_establishments ...` per field)
|
|
||||||
and sanity-checked against the DfE GIAS registers. Names must be
|
|
||||||
byte-identical to what the API serves today.
|
|
||||||
- Unknown codes never blank the UI: `translate` returns `"Unknown (<code>)"`
|
|
||||||
and logs, so a new DfE value degrades gracefully.
|
|
||||||
|
|
||||||
**Pipeline copy: `pipeline/scripts/gias_codes.py`**
|
|
||||||
|
|
||||||
The app and pipeline Docker images have disjoint build contexts
|
|
||||||
(`Dockerfile` copies `backend/`; `pipeline/Dockerfile` copies `pipeline/`),
|
|
||||||
so the Typesense sync cannot import the backend module. It gets a
|
|
||||||
byte-identical copy, and a backend unit test asserts
|
|
||||||
`backend/gias_codes.py` and `pipeline/scripts/gias_codes.py` have identical
|
|
||||||
content — drift fails CI. (Deliberately chosen over codegen: six dicts do
|
|
||||||
not justify build machinery.)
|
|
||||||
|
|
||||||
**Seed for drift detection: `pipeline/transform/seeds/gias_code_names.csv`**
|
|
||||||
|
|
||||||
Columns `field,code,name` mirroring the dictionary. A dbt test (severity
|
|
||||||
warn) compares live raw `(code, name)` pairs against the seed; when DfE adds
|
|
||||||
or renames a value the nightly run warns, prompting a dictionary + seed
|
|
||||||
update in one PR.
|
|
||||||
|
|
||||||
### 5. Backend translation (API contract unchanged)
|
|
||||||
|
|
||||||
- `_MAIN_QUERY` selects the code columns instead of the name columns.
|
|
||||||
- `load_school_data_as_dataframe()` translates immediately after
|
|
||||||
`pd.read_sql`, writing today's column names:
|
|
||||||
|
|
||||||
```python
|
|
||||||
df["phase"] = df["phase_code"].map(...)
|
|
||||||
df["school_type"] = df["school_type_code"].map(...) # then normalize_school_type as today
|
|
||||||
df["status"] = df["status_code"].map(...)
|
|
||||||
df["religious_denomination"] = df["religious_character_code"].map(...)
|
|
||||||
df["admissions_policy"] = df["admissions_policy_code"].map(...)
|
|
||||||
```
|
|
||||||
|
|
||||||
Everything downstream — `PHASE_GROUPS`, filters, payload builders,
|
|
||||||
`/api/filters`, frontend, e2e — sees exactly today's strings. No frontend
|
|
||||||
changes.
|
|
||||||
|
|
||||||
- `backend/models.py` `DimSchool`: string columns replaced by
|
|
||||||
`*_code = Column(Integer)`.
|
|
||||||
|
|
||||||
### 6. Typesense sync
|
|
||||||
|
|
||||||
`pipeline/scripts/sync_typesense.py` selects `phase`, `school_type`,
|
|
||||||
`religious_character` today. It switches to the code columns and translates
|
|
||||||
via `pipeline/scripts/gias_codes.py` before indexing, so facet values in
|
|
||||||
search are unchanged.
|
|
||||||
|
|
||||||
### 7. Rollout
|
|
||||||
|
|
||||||
- No DB migration: marts are full-rebuild tables.
|
|
||||||
- Deploy window: until the first post-merge pipeline run, the old marts
|
|
||||||
still carry string columns while the new backend queries code columns, so
|
|
||||||
the backend's query fails and it serves empty data (the one-column retry
|
|
||||||
built for `has_sixth_form` doesn't generalise to six columns, and a full
|
|
||||||
old-schema fallback query isn't worth it). **Decision: accept the window
|
|
||||||
and close it operationally — the runbook is merge → deploy → trigger
|
|
||||||
`school_data_daily` immediately.** The DAG's final step already calls
|
|
||||||
`/api/admin/reload`, so the backend recovers without a restart.
|
|
||||||
- Tests: backend unit tests for `translate()` (known / unknown / None),
|
|
||||||
payload tests asserting names still served, the file-parity test, dbt
|
|
||||||
schema/seed tests. Frontend: no changes; existing Jest suite is the
|
|
||||||
regression net.
|
|
||||||
|
|
||||||
## Out of scope
|
|
||||||
|
|
||||||
- Recoding other string columns (`gender`, `urban_rural`,
|
|
||||||
`nursery_provision`, `local_authority_name` …) — same pattern can follow
|
|
||||||
later if this proves out.
|
|
||||||
- Collapsing academy subtypes (today's `normalize_school_type`) — kept
|
|
||||||
as-is, applied after translation.
|
|
||||||
- Serving codes through the API — the contract deliberately keeps names.
|
|
||||||
@@ -1,177 +0,0 @@
|
|||||||
# Compare Screen Redesign — Expert Data Review
|
|
||||||
|
|
||||||
**Date:** 2026-07-11
|
|
||||||
**Reviewer:** subagent briefed as an English education-standards / DfE-Ofsted data expert
|
|
||||||
**Subject:** desktop + mobile compare mockups and the redesign spec
|
|
||||||
(`2026-07-11-compare-screen-redesign-design.md`)
|
|
||||||
**Status:** first-pass must-fixes applied 2026-07-12; second-pass
|
|
||||||
findings (below) applied 2026-07-12 — mockups + spec §4/§8 updated
|
|
||||||
|
|
||||||
## Must-fix
|
|
||||||
|
|
||||||
1. **COVID gap is wrong and drops a real results year.** KS2 tests were
|
|
||||||
cancelled 2019/20 and 2020/21 only; they resumed in 2021/22 with
|
|
||||||
published school-level results (England RWM ≈ 59%). The mockup charts
|
|
||||||
omit 2021/22 entirely and the tooltip claims no tests were held
|
|
||||||
2019/20–2021/22. Fix: add 2021/22 to axis and all series; shrink the
|
|
||||||
gap band; optionally annotate 2021/22 with DfE's post-pandemic
|
|
||||||
comparability caution.
|
|
||||||
2. **Report-card at-a-glance summary miscounts areas.** Detail list has
|
|
||||||
4 Strong / 2 Expected / 1 Attention needed + Safeguarding met, but
|
|
||||||
the summary says "3 areas Expected standard" — it counts safeguarding
|
|
||||||
as a graded area. Safeguarding is a separate binary judgement and
|
|
||||||
must be excluded from rating counts.
|
|
||||||
3. **"Where the offers went" derivation is unsound.** Places − 1st-pref
|
|
||||||
offers ≠ "second or third choices": the residual can include 4th–6th
|
|
||||||
preference offers (pan-London scheme) and LA-allocated children who
|
|
||||||
didn't choose the school; and offers don't necessarily equal PAN.
|
|
||||||
Use the real 2nd/3rd-preference fields being promoted from
|
|
||||||
`raw.ees_admissions`; until then drop the row.
|
|
||||||
4. **Ofsted timeline in the copy is wrong.** Overall grades were
|
|
||||||
abolished September 2024, not November 2025; Sept 2024–Nov 2025
|
|
||||||
inspections kept the four key judgements without an overall grade
|
|
||||||
(ungraded inspections carried grades forward). Neither mockup shows
|
|
||||||
the interim regime, which will dominate real comparisons. Fix copy
|
|
||||||
and add an interim example.
|
|
||||||
5. **Barclay's "published an overall grade only — no area-by-area
|
|
||||||
detail" misdescribes inspections.** No inspection type does that; a
|
|
||||||
2021 graded inspection necessarily had subgrades — the gap is in our
|
|
||||||
dataset. If it was an ungraded (s8) inspection, "Outstanding" is a
|
|
||||||
carried-forward grade and should say so. Fix: "We don't hold
|
|
||||||
area-by-area detail for this inspection", and distinguish graded vs
|
|
||||||
ungraded in the data model.
|
|
||||||
|
|
||||||
## Should-fix
|
|
||||||
|
|
||||||
6. Writing is teacher assessment, not a test — "national tests and
|
|
||||||
teacher assessments"; note TA caveat on the Writing strip.
|
|
||||||
7. Verify renewed-framework wording against Ofsted's final toolkit:
|
|
||||||
likely "Needs attention" (not "Attention needed") and "Personal
|
|
||||||
development and well-being" (which otherwise collides with the
|
|
||||||
identically-named legacy judgement). Pin every label to the
|
|
||||||
published toolkit.
|
|
||||||
8. "Expected standard" now means two things on one page (Ofsted area
|
|
||||||
rating vs KS2 measure) — disambiguate in tooltips.
|
|
||||||
9. Disadvantaged row: DfE definition includes looked-after / previously
|
|
||||||
looked-after children, not just FSM6; benchmark labels inconsistent
|
|
||||||
across desktop/mobile; subgroup percentages need cohort sizes or a
|
|
||||||
volatility threshold before chips are attached.
|
|
||||||
10. "Trend, last 7 years" spans ten years; sparklines render the COVID
|
|
||||||
gap as equal spacing (the exact defect the audit criticises) and
|
|
||||||
"Improved: 52% → 87%" endpoint-cherry-picks a volatile series.
|
|
||||||
11. At-a-glance "Getting a place" uses different metrics per school
|
|
||||||
(Barclay is also oversubscribed on total preferences but shows a
|
|
||||||
green chip). Standardise on first-preference success %. Explain the
|
|
||||||
equal-preference rule; condition "living close by matters" on the
|
|
||||||
school's actual oversubscription criteria.
|
|
||||||
12. "457 applications for 180 places" = total preferences at any rank,
|
|
||||||
not head-to-head applicants; lead with first preferences vs places.
|
|
||||||
Add offers-vs-final-intake (waiting lists/appeals) caveat.
|
|
||||||
13. Elmhurst's subgrade list is likely missing Early years provision
|
|
||||||
(school has a nursery) — possible pipeline gap.
|
|
||||||
14. "Ofsted rating" label is obsolete post-Sept-2024 — use "Latest
|
|
||||||
Ofsted inspection"; check whether Oct 2021 is the latest inspection
|
|
||||||
or merely the latest graded one.
|
|
||||||
15. SEN: "EHCP plans" is redundant; 28% SEN support often indicates
|
|
||||||
resourced provision — add a note; England SEN-support ≈ 14%, not 13%.
|
|
||||||
|
|
||||||
## Nice-to-have
|
|
||||||
|
|
||||||
16. Consistent labelling of official DfE vs dataset-computed benchmarks
|
|
||||||
(and medians shouldn't be called averages inconsistently).
|
|
||||||
17. England 2015/16 RWM (53%) exists in DfE publications — the null is
|
|
||||||
a dataset gap; source it or the England line looks broken.
|
|
||||||
18. "1 in 4 first choices missed out" — actually more than 1 in 4.
|
|
||||||
19. "1,273 of 1,260 places (full)" is over capacity; capacity figures
|
|
||||||
are often stale — say "at or above capacity".
|
|
||||||
20. State the actual suppression rule (DfE: ≤5 pupils suppressed,
|
|
||||||
small numbers rounded) instead of "a handful".
|
|
||||||
21. Spec §4.3 progress chips can't exist for displayed years: KS2
|
|
||||||
progress ended with 2022/23 (no KS1 baseline) and returns
|
|
||||||
~2027/28 with the reception baseline. Make explicit in the spec.
|
|
||||||
IDACI (spec §4.5) is absent from mockups; if shipped, caveat it
|
|
||||||
describes pupils' neighbourhoods, not the school.
|
|
||||||
22. Tooltips should give the official term "first preference" alongside
|
|
||||||
the plain-English "first choice".
|
|
||||||
|
|
||||||
## Overall assessment (verbatim gist)
|
|
||||||
|
|
||||||
The bones are genuinely good by education-data standards —
|
|
||||||
England-average anchoring, explicit non-comparability messaging across
|
|
||||||
Ofsted regimes, refusal to synthesise an overall grade, time-true
|
|
||||||
x-axis, neutral FSM/EAL framing — better than most commercial
|
|
||||||
school-comparison sites. But items 1–5 are outright factual errors or
|
|
||||||
misdescriptions that a well-informed parent or Ofsted would catch;
|
|
||||||
the admissions section needs the most conceptual work (equal
|
|
||||||
preference, preferences-vs-applicants, offers-vs-intake). Fix 1–5
|
|
||||||
before user testing; the rest fold into the planned PRs.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
# Second-pass review (2026-07-12)
|
|
||||||
|
|
||||||
Same reviewer, after the must-fixes and the new three-tier metric
|
|
||||||
exposure model were applied.
|
|
||||||
|
|
||||||
## Verification of first-pass must-fixes
|
|
||||||
|
|
||||||
- **1 (COVID/2021/22): resolved.** Time-true axis, band covers only the
|
|
||||||
cancelled years, England 58.7% consistent with official figures,
|
|
||||||
dataset gaps break lines honestly; reading/maths England series all
|
|
||||||
match published figures; RWM ≤ min(subject) checks pass.
|
|
||||||
- **2 (report-card count): resolved** — safeguarding excluded, spec §8.2.
|
|
||||||
- **3 (offers derivation): resolved** — row removed, spec §8.3 bans it.
|
|
||||||
- **4 (Ofsted timeline): resolved on desktop; mobile omits the interim
|
|
||||||
regime clause** (see finding 6).
|
|
||||||
- **5 (Barclay explanation): resolved.**
|
|
||||||
|
|
||||||
## New findings
|
|
||||||
|
|
||||||
1. **Should-fix — scaled-score strip domain contradicts caption.**
|
|
||||||
Caption says "scaled scores run 80–120", strips render 100–120;
|
|
||||||
truncated domain exaggerates small gaps and below-100 averages
|
|
||||||
would fall off the edge. Render 80–120, or caption the 100–120
|
|
||||||
window honestly and define below-100 behaviour.
|
|
||||||
2. **Should-fix — scaled-score England ticks (106/105/105) unsourced.**
|
|
||||||
Plausible but hand-entered; verify against DfE 2024/25 tables and
|
|
||||||
add loading official England scaled scores to the pipeline list
|
|
||||||
(absent from §8.1/§8.6).
|
|
||||||
3. **Should-fix — "Writing" listed under "Higher standard" in the
|
|
||||||
picker.** Writing TA outcome is "greater depth" (GDS), never
|
|
||||||
"higher standard". Label "Writing — greater depth (teacher
|
|
||||||
assessment)"; tooltip the combined higher-standard composition.
|
|
||||||
4. Nice — "grammar & punctuation" summary line drops "spelling" (GPS).
|
|
||||||
5. Nice — science is teacher-assessed (no KS2 test since 2009) and
|
|
||||||
coarse; tooltip it like writing; reconsider its tier-2 slot.
|
|
||||||
6. **Should-fix — mobile Ofsted copy skips the interim regime**
|
|
||||||
(Sept 2024–Nov 2025) that desktop explains. One clause fixes it.
|
|
||||||
7. **Should-fix — benchmark provenance still inconsistent** (EAL
|
|
||||||
tooltip unsourced; FSM/disadvantaged chips vs tooltips use three
|
|
||||||
vocabularies; header note says all England averages are official).
|
|
||||||
Adopt one house style: official = "England average", computed =
|
|
||||||
"benchmark / typical state school (our dataset)". Also tighten EAL
|
|
||||||
definition to census wording ("first language known or believed to
|
|
||||||
be other than English").
|
|
||||||
8. Nice — "community primaries" distance note attached to an academy
|
|
||||||
(Elmhurst); say "non-faith primaries" or condition on policy field.
|
|
||||||
9. Nice — "Improving since 2022" → "since 2022/23".
|
|
||||||
10. Nice — England chart tooltips show decimals; §7 mandates whole
|
|
||||||
percents.
|
|
||||||
|
|
||||||
## Residual gaps not covered by spec §8
|
|
||||||
|
|
||||||
11. Spec promises IDACI-in-words, Attendance section, and tier-2
|
|
||||||
gender/absence that the mockups never show — mark post-v1 or
|
|
||||||
demonstrate, so implementation scope is unambiguous.
|
|
||||||
12. Add official England scaled-score averages to the pipeline task
|
|
||||||
list.
|
|
||||||
13. Add the writing/greater-depth terminology rule to §8.7.
|
|
||||||
|
|
||||||
## Verdict
|
|
||||||
|
|
||||||
All must-fixes genuinely resolved; the tier model is conceptually
|
|
||||||
sound ("no measure is lost", honest dataset-gap breaks, grouped
|
|
||||||
picker). Remaining issues are contained: one internal contradiction
|
|
||||||
(80–120 vs 100–120), one provenance inconsistency, one terminology
|
|
||||||
error (writing/GDS). With findings 1–3 and 6–7 addressed, the data
|
|
||||||
framing is fit to put in front of parents.
|
|
||||||
@@ -1,324 +0,0 @@
|
|||||||
# Compare Screen Redesign — Audit & Design
|
|
||||||
|
|
||||||
**Date:** 2026-07-11
|
|
||||||
**Status:** Draft — awaiting review
|
|
||||||
**Scope:** `/compare` page (nextjs-app), `/api/compare` endpoint (backend)
|
|
||||||
|
|
||||||
## 1. Audit of the current screen
|
|
||||||
|
|
||||||
The current compare page (`nextjs-app/components/ComparisonView.tsx`) is a
|
|
||||||
single-metric analyst tool: a `<select>` with ~40 KS2/GCSE metrics, one
|
|
||||||
line chart over time, and a year-by-year table — all for the one selected
|
|
||||||
metric. Observed on production with 3 primary schools:
|
|
||||||
|
|
||||||
**What works**
|
|
||||||
|
|
||||||
- URL-shareable state (`?urns=…&metric=…`), native share sheet.
|
|
||||||
- Phase tabs (primary/secondary) with sensible auto-detection.
|
|
||||||
- Colour-coded school cards tied to chart series.
|
|
||||||
- Metric descriptions from `/api/metrics` (single source of truth).
|
|
||||||
|
|
||||||
**What doesn't**
|
|
||||||
|
|
||||||
1. **Performance-only.** The database already holds Ofsted inspections,
|
|
||||||
admissions/oversubscription history, pupil characteristics (FSM/EAL),
|
|
||||||
SEN, deprivation (IDACI), finance, capacity, faith, gender, trust —
|
|
||||||
none of it reaches the compare screen. `/api/compare` returns only
|
|
||||||
`yearly_data` + minimal `school_info`, while `/api/schools/{urn}`
|
|
||||||
already returns all supplementary blocks.
|
|
||||||
2. **One metric at a time.** A parent must know which of ~40 metrics
|
|
||||||
matters, select each in turn, and hold results in their head. There is
|
|
||||||
no side-by-side overview and no way to see two dimensions at once.
|
|
||||||
3. **No benchmarks.** Numbers float without anchors: is 79% RWM good?
|
|
||||||
The DB has official national averages (`fact_ks2_national_averages`)
|
|
||||||
but the page never shows them.
|
|
||||||
4. **Domain jargon untranslated.** "GPS Expected %", "Progress scores",
|
|
||||||
"RWM Combined" assume DfE literacy. The only plain-English help is one
|
|
||||||
note for progress scores.
|
|
||||||
5. **Raw numbers, no judgement support.** 87.0% vs 92.0% vs 79.0% — the
|
|
||||||
page never says "all three are well above the England average of 62%",
|
|
||||||
which is the fact a parent actually needs.
|
|
||||||
6. **Bugs/paper cuts observed:** the third school's series did not render
|
|
||||||
on the production chart despite table data (worth a separate fix);
|
|
||||||
the COVID gap (2018/19 → 2022/23) renders as equal spacing with no
|
|
||||||
annotation; table shows "87.0%" precision that implies false accuracy.
|
|
||||||
|
|
||||||
## 2. Data inventory (available vs shown)
|
|
||||||
|
|
||||||
| Domain | Source table | On detail page | On compare |
|
|
||||||
|---|---|---|---|
|
|
||||||
| KS2 attainment/progress | fact_ks2_performance | yes | **yes** (only thing shown) |
|
|
||||||
| National averages | fact_ks2_national_averages | partial | no |
|
|
||||||
| Ofsted (latest + subgrades + report-card fields) | fact_ofsted_inspection, dim_school | yes | no |
|
|
||||||
| Admissions & oversubscription (multi-year) | fact_admissions | yes | no |
|
|
||||||
| Pupil characteristics (FSM, EAL, gender split) | fact_pupil_characteristics | yes | no |
|
|
||||||
| Context (SEN, disadvantaged, stability, absence) | fact_ks2_performance | via metric picker | buried in picker |
|
|
||||||
| Deprivation (IDACI) | fact_deprivation | yes | no |
|
|
||||||
| Finance (per-pupil spend) | fact_finance | yes | no |
|
|
||||||
| School facts (capacity, faith, ages, trust, nursery, gender) | dim_school | yes | no |
|
|
||||||
| Location/distance | dim_location | map | no |
|
|
||||||
|
|
||||||
## 3. Design goals
|
|
||||||
|
|
||||||
1. **Answer parent questions, in order:** Is it a good school (Ofsted)?
|
|
||||||
Do children do well there (academics vs England)? Will my child get a
|
|
||||||
place (admissions)? What is the school like (size, community, faith)?
|
|
||||||
2. **Every number gets an anchor** — the England average, rendered as a
|
|
||||||
consistent visual tick, plus a plain-English chip
|
|
||||||
(Above / Close to / Below England average).
|
|
||||||
3. **Plain English first, jargon on demand.** Labels are questions or
|
|
||||||
sentences ("Children reaching the expected standard in reading,
|
|
||||||
writing and maths"), codes/acronyms live in tooltips.
|
|
||||||
4. **Scan whole-picture first, drill down second.** The single-metric
|
|
||||||
trend explorer survives, demoted to an "Explore trends" section at the
|
|
||||||
bottom rather than being the entire page.
|
|
||||||
|
|
||||||
## 4. Proposed structure
|
|
||||||
|
|
||||||
Columns = schools (max 4 visible on desktop, horizontal scroll beyond),
|
|
||||||
rows = dimensions. Sticky compact school header keeps column identity
|
|
||||||
while scrolling. Sections, in order:
|
|
||||||
|
|
||||||
1. **At a glance** — verdict row per school: Ofsted badge, headline
|
|
||||||
attainment vs England (dot strip + chip), oversubscription chip,
|
|
||||||
size, distance (when a location is set).
|
|
||||||
2. **Ofsted inspection** — must handle all three inspection regimes,
|
|
||||||
which will coexist in comparisons for years:
|
|
||||||
- **Legacy graded (pre-Sept 2024):** overall grade badge
|
|
||||||
(Outstanding/Good/Requires improvement/Inadequate). Subgrades,
|
|
||||||
where published, are rendered in the **same area-by-rating chip
|
|
||||||
list UX as report cards** (one row per judgement area, rating as
|
|
||||||
a chip) — one visual grammar for inspection detail across both
|
|
||||||
regimes. Where our dataset has no subgrades for an inspection,
|
|
||||||
say so honestly ("We don't hold area-by-area detail for this
|
|
||||||
inspection") and point to the school's Ofsted page — never claim
|
|
||||||
the inspection itself published no detail (graded inspections
|
|
||||||
always have subgrades; if it was ungraded, the grade is
|
|
||||||
carried forward and must be labelled as such).
|
|
||||||
- **Interim ungraded (Sept 2024 – Nov 2025):** parsed outcome
|
|
||||||
("remains Good") shown as the effective grade, marked as such.
|
|
||||||
- **Renewed framework report card (from Nov 2025):** no overall
|
|
||||||
grade exists. Render the report card as an area-by-rating list
|
|
||||||
using Ofsted's 5-point scale (Exceptional / Strong standard /
|
|
||||||
Expected standard / Attention needed / Urgent improvement) across
|
|
||||||
the evaluation areas we model (`rc_inclusion`,
|
|
||||||
`rc_curriculum_teaching`, `rc_achievement`,
|
|
||||||
`rc_attendance_behaviour`, `rc_personal_development`,
|
|
||||||
`rc_leadership_governance`, `rc_early_years`, `rc_sixth_form`)
|
|
||||||
plus the separate safeguarding met/not-met flag. **At-a-glance
|
|
||||||
summary rule:** never an unlabelled colour strip — summarise by
|
|
||||||
counting areas per rating, best first ("5 areas Strong standard ·
|
|
||||||
3 areas Expected standard"), and always name any area rated
|
|
||||||
Attention needed or Urgent improvement explicitly (never fold
|
|
||||||
problems into a count), plus "Safeguarding not met" whenever that
|
|
||||||
flag is false. When everything is Expected standard or better,
|
|
||||||
add the reassurance line "No areas need attention".
|
|
||||||
When a comparison mixes regimes, show a one-line comparability note
|
|
||||||
("Ofsted changed how it reports in Nov 2025 — a report card and an
|
|
||||||
older overall grade aren't directly comparable"). Never derive a
|
|
||||||
fake overall grade from report-card areas.
|
|
||||||
3. **Academics (KS2)** — one dot-strip row per headline measure (RWM
|
|
||||||
expected, RWM higher, reading/writing/maths expected), each with the
|
|
||||||
England-average tick and per-school dots; copy must say "tests and
|
|
||||||
teacher assessments" (writing is TA, not a test). Progress scores
|
|
||||||
translated to Above/Average/Below chips (CI-based) — **but note KS2
|
|
||||||
progress measures ended with 2022/23** (no KS1 baseline afterwards)
|
|
||||||
and return only when the reception-baseline cohort reaches Y6
|
|
||||||
(~2027/28), so progress chips apply to historical years in the
|
|
||||||
trends explorer, not the headline view. Sparkline per school over
|
|
||||||
the full published period, with an honest gap for the cancelled
|
|
||||||
test years (2019/20–2020/21). Disadvantaged-pupils row under an
|
|
||||||
"Equity" subheading, always with cohort size shown and DfE's full
|
|
||||||
definition (FSM6 **or** looked-after/previously looked-after).
|
|
||||||
4. **Getting a place** — oversubscription ratio as plain sentence
|
|
||||||
("184 applications for 80 places"), first-preference success %, trend
|
|
||||||
vs last year, admissions policy.
|
|
||||||
5. **Who goes there** — pupils on roll (vs capacity), boys/girls, FSM %,
|
|
||||||
EAL %, SEN support %, faith, ages, nursery, trust. *Post-v1:* IDACI
|
|
||||||
decile in words (needs a coverage check of `fact_deprivation` and
|
|
||||||
the neighbourhood-not-school caveat, §8.7).
|
|
||||||
6. **Attendance** — *post-v1.* The KS2 test-day absence fields are the
|
|
||||||
only per-school absence data we hold; they're near-zero for most
|
|
||||||
schools and easy to misread as general attendance. Ship only if a
|
|
||||||
general-absence source lands.
|
|
||||||
7. **Explore trends** (existing feature, collapsed) — metric picker +
|
|
||||||
multi-year line chart + table, with an added England-average
|
|
||||||
reference line and a COVID-gap annotation.
|
|
||||||
|
|
||||||
**Metric exposure model (three tiers).** No measure from the current
|
|
||||||
page is lost; they surface at three levels of prominence:
|
|
||||||
- **Tier 1 — headline strips (always visible):** RWM expected,
|
|
||||||
reading/writing/maths expected, RWM higher standard.
|
|
||||||
- **Tier 2 — "More measures" expansion inside Academics:** GPS and
|
|
||||||
science expected % (science labelled teacher-assessed), average
|
|
||||||
scaled scores (reading/maths/GPS, same dot-strip grammar showing
|
|
||||||
the 100–120 window of the 80–120 scale, widening below 100, with
|
|
||||||
the England tick) — one tap/click away, same visual language.
|
|
||||||
*Post-v1:* gender split and absence (see §4.6).
|
|
||||||
- **Tier 3 — Explore trends:** the full grouped catalogue (the
|
|
||||||
current page's ~40 metrics, including equity and school-context
|
|
||||||
measures, and the GCSE set for secondary phase) drives the
|
|
||||||
year-by-year chart and table via the grouped metric picker.
|
|
||||||
The tier assignment is a content decision per phase (secondary:
|
|
||||||
Attainment 8, Progress 8 banding, grade 5+ English & maths as tier 1;
|
|
||||||
EBacc and subject entries as tier 2).
|
|
||||||
|
|
||||||
Finance (per-pupil spend) is deliberately deferred: low parent value,
|
|
||||||
risk of misreading. Revisit later.
|
|
||||||
|
|
||||||
**Mobile (design target — mobile first):** the desktop grid is the
|
|
||||||
adaptation, not the other way round. On mobile the layout goes
|
|
||||||
*measure-first*: each row is one measure with all schools listed under
|
|
||||||
it (colour dot + short name + value + chip), so comparison never
|
|
||||||
requires horizontal swiping between school cards. A sticky horizontal
|
|
||||||
school-chip bar keeps identity and add/remove available while
|
|
||||||
scrolling. Dot strips already read measure-first and carry over
|
|
||||||
unchanged. The trend chart scrolls horizontally inside its container.
|
|
||||||
|
|
||||||
## 5. Data strategy — existing dataset only
|
|
||||||
|
|
||||||
Constraint (agreed 2026-07-11): use only data already in marts plus
|
|
||||||
fields already present in the `raw` schema extracts we pull today.
|
|
||||||
No new external sources.
|
|
||||||
|
|
||||||
**Gaps in the mockup, resolved within this constraint:**
|
|
||||||
|
|
||||||
| Mockup element | Resolution |
|
|
||||||
|---|---|
|
|
||||||
| England average for disadvantaged pupils | Compute from our own data: `stg_ees_ks2` already pivots the Disadvantaged breakdown per school; aggregate it (weighted by eligible pupils) into `fact_ks2_national_averages` or compute in the API. Label it "England average (state schools)". |
|
|
||||||
| England context for FSM / EAL / SEN chips | Compute dataset-wide medians per phase, same pattern as `/api/national-averages` does for KS4. |
|
|
||||||
| "Much larger than average" size label | Dataset median pupils-on-roll per phase. |
|
|
||||||
| Ofsted link | We don't have deep links to the latest report, so always link to the school's Ofsted provider page, `https://reports.ofsted.gov.uk/provider/21/{urn}`, derived from URN (label it "the school's Ofsted page", not "the report"). |
|
|
||||||
|
|
||||||
**Raw fields we already pull but don't store — promote to marts (one
|
|
||||||
dbt/pipeline PR, no tap changes):**
|
|
||||||
|
|
||||||
- `raw.ees_admissions`: 2nd/3rd preference applications and offers,
|
|
||||||
total-preference counts, cross-LA applications and offers → richer
|
|
||||||
"Getting a place" (e.g. "offers reached 2nd-choice families",
|
|
||||||
competition from outside the borough).
|
|
||||||
- `raw.ees_ks2_attainment`: progress-measure confidence intervals and
|
|
||||||
"working towards" % → lets the Above/Average/Below progress chips be
|
|
||||||
statistically honest (band by CI overlap with 0, mirroring DfE
|
|
||||||
methodology) instead of thresholding the point estimate.
|
|
||||||
- `raw.ees_ks4_performance` / `ees_ks4_info`: `progress8_banding`
|
|
||||||
(DfE's own plain-English "well above average … well below average"
|
|
||||||
label — exactly the chip we want for secondary), EBacc entry/APS,
|
|
||||||
grade-5+ English & maths, `attainment8_diffn`/`progress8_diffn`
|
|
||||||
(disadvantage gaps) → the secondary-phase version of the Academics
|
|
||||||
section.
|
|
||||||
- `raw.ees_census`: young-carer % and the ethnicity breakdown →
|
|
||||||
optional "Who goes there" enrichment; hold for a later iteration
|
|
||||||
(presentation needs care), but the data requires no new extract.
|
|
||||||
- `raw.ofsted_inspections` / tap-uk-ofsted: the `rc_*` report-card
|
|
||||||
columns exist in staging/marts but are stubbed `null` — the tap has a
|
|
||||||
TODO to map the report-card column names from the Ofsted MI file
|
|
||||||
(same monthly extract we already download; inspections from Nov 2025
|
|
||||||
onward carry them). This is the one promotion that needs a small tap
|
|
||||||
schema addition, and it's a prerequisite for the new-framework Ofsted
|
|
||||||
display above.
|
|
||||||
|
|
||||||
Explicitly out (not in any current extract): school-level phonics,
|
|
||||||
workforce/teacher data, per-school attendance beyond the KS2 test-day
|
|
||||||
absence fields, Ofsted report-card documents themselves.
|
|
||||||
|
|
||||||
## 6. API changes
|
|
||||||
|
|
||||||
Extend `GET /api/compare` response per URN with the same supplementary
|
|
||||||
blocks the detail endpoint already builds (`get_supplementary_data`):
|
|
||||||
`ofsted`, `census`, `admissions` (+ `admissions_history`), `deprivation`,
|
|
||||||
plus a top-level `national_averages` block for the latest year. Reuse the
|
|
||||||
existing function; no new tables. Response stays backward-compatible
|
|
||||||
(additive fields only). Add derived helper fields server-side or compute
|
|
||||||
chips client-side from `national_averages` (client-side preferred — no
|
|
||||||
schema churn).
|
|
||||||
|
|
||||||
## 7. Accessibility & comprehension devices
|
|
||||||
|
|
||||||
- Verdict chips are text + colour + position (never colour alone).
|
|
||||||
- Every acronym has a tooltip using existing `MetricTooltip`.
|
|
||||||
- "How to read this" one-liner at the top of each section.
|
|
||||||
- Chart palette: coral `#e07256`, teal `#00949b`, purple `#8664c9`
|
|
||||||
(validated: lightness band, chroma, CVD separation, contrast — the
|
|
||||||
current `--chart-2/-4` tokens fail chroma/contrast checks and should
|
|
||||||
be nudged to these).
|
|
||||||
- Numbers rounded to whole percents; England tick labelled on first use.
|
|
||||||
|
|
||||||
## 8. Expert-review requirements
|
|
||||||
|
|
||||||
An adversarial review by an education-data expert (full findings in
|
|
||||||
`2026-07-11-compare-screen-expert-review.md`) was applied to the
|
|
||||||
mockups on 2026-07-12. The following are binding requirements for
|
|
||||||
implementation, beyond what the mockups can show:
|
|
||||||
|
|
||||||
1. **Chart truthfulness:** KS2 tests were cancelled 2019/20–2020/21
|
|
||||||
only. **2021/22 school-level figures are a permanent source gap** —
|
|
||||||
DfE stated it would not publish KS2 2021/22 in performance tables
|
|
||||||
(verified 2026-07-12 against EES, the CSP download service, and
|
|
||||||
DfE release notes; see `# TASK 6 VERIFICATION` in
|
|
||||||
`pipeline/scripts/diagnose_compare_gaps.py`). The chart's England-
|
|
||||||
only 2021/22 point with broken school lines is therefore the
|
|
||||||
correct permanent rendering; copy should say "DfE didn't publish
|
|
||||||
school-level figures for 2021/22", not "not in our dataset yet".
|
|
||||||
The 2015/16 national figure and the GPS/science/scaled-score
|
|
||||||
England averages ARE loadable (mapping already correct; refreshed
|
|
||||||
raw extract backfills them). Never render missing years as if time
|
|
||||||
were continuous.
|
|
||||||
2. **Report-card summaries** count graded areas only — safeguarding is
|
|
||||||
a separate binary flag, never included in rating counts.
|
|
||||||
3. **Admissions:** use the real preference-breakdown fields from
|
|
||||||
`raw.ees_admissions`; never derive "lower-preference offers" as
|
|
||||||
places − first-preference offers. Frame total applications as
|
|
||||||
"named on N forms" (any rank), lead with first-preference success,
|
|
||||||
and standardise at-a-glance chips on that one metric. Explain the
|
|
||||||
equal-preference rule; caveat offers vs final intake (waiting
|
|
||||||
lists/appeals); condition "distance decides" on the school's actual
|
|
||||||
oversubscription criteria where we have the admissions-policy field.
|
|
||||||
4. **Ofsted:** overall grades ended September 2024 (report cards from
|
|
||||||
November 2025); the interim regime must be renderable. Distinguish
|
|
||||||
graded (s5) vs ungraded (s8) inspections and surface carried-forward
|
|
||||||
grades as such; "we don't hold the detail" is a statement about our
|
|
||||||
dataset, never about the inspection. Verify every scale/area label
|
|
||||||
against Ofsted's final published toolkit before launch (e.g. "Needs
|
|
||||||
attention" vs "Attention needed"; "Personal development and
|
|
||||||
well-being" vs the identically-named legacy judgement). Check
|
|
||||||
whether a school's latest inspection is merely its latest *graded*
|
|
||||||
one. Confirm Early years provision subgrades flow through the
|
|
||||||
pipeline for schools with nurseries.
|
|
||||||
5. **Subgroup honesty:** disadvantaged-pupil percentages carry cohort
|
|
||||||
sizes and follow the DfE suppression rule (≤5 pupils suppressed);
|
|
||||||
state the rule verbatim in the footer.
|
|
||||||
6. **Benchmark provenance:** official DfE figures and
|
|
||||||
dataset-computed benchmarks must be labelled distinctly and
|
|
||||||
consistently everywhere (a computed median is a "benchmark",
|
|
||||||
not an "England average").
|
|
||||||
7. **Copy details:** "Latest Ofsted inspection" (not "Ofsted rating");
|
|
||||||
"EHC plans"; SEN-support benchmark ≈14%; high SEN share may
|
|
||||||
indicate resourced provision (say so neutrally); "at or above
|
|
||||||
capacity" rather than "full" (capacity data is often stale);
|
|
||||||
disambiguate Ofsted's "Expected standard" from the KS2 measure;
|
|
||||||
give official terms ("first preference") alongside plain English.
|
|
||||||
Writing has no "higher standard" — its TA outcome is "greater
|
|
||||||
depth (GDS)"; never list writing under a higher-standard group.
|
|
||||||
Science and writing are teacher-assessed and must be labelled as
|
|
||||||
such (no KS2 science test since 2009). House style for benchmark
|
|
||||||
provenance: official DfE figures say "England average"; computed
|
|
||||||
figures say "state-school average (computed from our dataset)" —
|
|
||||||
applied to every chip, tooltip, header note and section intro.
|
|
||||||
EAL uses the census wording: first language known or believed to
|
|
||||||
be other than English. If IDACI ships, caveat that it describes
|
|
||||||
pupils' home neighbourhoods, not the school.
|
|
||||||
|
|
||||||
## 9. Rollout
|
|
||||||
|
|
||||||
1. **PR 1 (backend):** extend `/api/compare` + tests.
|
|
||||||
2. **PR 2 (frontend):** new compare layout behind the existing route;
|
|
||||||
e2e journey updated in the same PR (promotion gate).
|
|
||||||
3. **Fix separately:** missing third series on the current chart.
|
|
||||||
|
|
||||||
## 10. Open questions for review
|
|
||||||
|
|
||||||
- Max schools: keep 10 in API but cap visible columns at 4 with scroll?
|
|
||||||
- Should distance-from-home appear when the user searched by postcode
|
|
||||||
(data exists via `dim_location`)?
|
|
||||||
- Keep finance out of v1? (Recommended: yes, out.)
|
|
||||||
@@ -25,16 +25,6 @@ test('home page loads with hero search', async ({ page }) => {
|
|||||||
await expect(page.getByPlaceholder('School name or postcode').first()).toBeVisible();
|
await expect(page.getByPlaceholder('School name or postcode').first()).toBeVisible();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('home hero offers a "use my location" shortcut beside the search box', async ({ page }) => {
|
|
||||||
await page.goto('/');
|
|
||||||
// The geolocation shortcut lives inside the hero search card, right under the
|
|
||||||
// search input — not in a separate strip further down the page.
|
|
||||||
const searchInput = page.getByPlaceholder('School name or postcode').first();
|
|
||||||
await expect(searchInput).toBeVisible();
|
|
||||||
const nearMe = page.getByRole('button', { name: /use my location/i });
|
|
||||||
await expect(nearMe).toBeVisible();
|
|
||||||
});
|
|
||||||
|
|
||||||
test('searching by name returns school results', async ({ page }) => {
|
test('searching by name returns school results', async ({ page }) => {
|
||||||
await searchByName(page, 'primary');
|
await searchByName(page, 'primary');
|
||||||
await expect(schoolLinks(page).first()).toBeVisible({ timeout: 15_000 });
|
await expect(schoolLinks(page).first()).toBeVisible({ timeout: 15_000 });
|
||||||
@@ -60,34 +50,6 @@ test('school detail page renders name and performance data', async ({ page }) =>
|
|||||||
await expect(page.locator('canvas:visible').first()).toBeVisible({ timeout: 15_000 });
|
await expect(page.locator('canvas:visible').first()).toBeVisible({ timeout: 15_000 });
|
||||||
});
|
});
|
||||||
|
|
||||||
test('school with no performance data still gets a working detail page', async ({ page }) => {
|
|
||||||
// Schools without KS2/KS4 results (special post-16 institutions, sixth-form
|
|
||||||
// centres, PRUs) used to 500 in the API — NaN GIAS fields broke JSON
|
|
||||||
// serialization — which the frontend rendered as a 404 on every such SEO
|
|
||||||
// landing page. Find one via the search API (year === null marks "no
|
|
||||||
// performance rows") and assert its page renders.
|
|
||||||
const candidates: number[] = [];
|
|
||||||
for (const q of ['post 16', 'specialist college', 'sixth form']) {
|
|
||||||
const resp = await page.request.get(
|
|
||||||
`/api/schools?search=${encodeURIComponent(q)}&per_page=20`
|
|
||||||
);
|
|
||||||
if (!resp.ok()) continue;
|
|
||||||
const body = await resp.json();
|
|
||||||
for (const s of body.schools ?? []) {
|
|
||||||
if (s.year === null && s.urn) candidates.push(s.urn);
|
|
||||||
}
|
|
||||||
if (candidates.length) break;
|
|
||||||
}
|
|
||||||
test.skip(candidates.length === 0, 'no results-less school in this dataset');
|
|
||||||
|
|
||||||
const detail = await page.request.get(`/api/schools/${candidates[0]}`);
|
|
||||||
expect(detail.status(), 'detail API must not 500 for a results-less school').toBe(200);
|
|
||||||
|
|
||||||
await page.goto(`/school/${candidates[0]}`);
|
|
||||||
await page.waitForURL(/\/school\/\d+-/); // redirected to canonical slug
|
|
||||||
await expect(page.locator('h1').first()).toBeVisible();
|
|
||||||
});
|
|
||||||
|
|
||||||
test('school hero map opens fullscreen on mobile without the Fullscreen API', async ({ page }) => {
|
test('school hero map opens fullscreen on mobile without the Fullscreen API', async ({ page }) => {
|
||||||
// iOS Safari has no Element.requestFullscreen; the map must fall back to a
|
// iOS Safari has no Element.requestFullscreen; the map must fall back to a
|
||||||
// CSS overlay. Simulate that by removing the API before any page script runs.
|
// CSS overlay. Simulate that by removing the API before any page script runs.
|
||||||
@@ -113,31 +75,6 @@ test('school hero map opens fullscreen on mobile without the Fullscreen API', as
|
|||||||
await expect(openMap).toBeVisible();
|
await expect(openMap).toBeVisible();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('results map fullscreen falls back to an overlay on iOS', async ({ page }) => {
|
|
||||||
// Same iOS gap as the hero map: no Element.requestFullscreen, so the results
|
|
||||||
// map's fullscreen button must fall back to a CSS overlay.
|
|
||||||
await page.setViewportSize({ width: 390, height: 844 });
|
|
||||||
await page.addInitScript(() => {
|
|
||||||
// @ts-expect-error deliberate API removal
|
|
||||||
delete Element.prototype.requestFullscreen;
|
|
||||||
});
|
|
||||||
|
|
||||||
await searchByName(page, 'B1 1BB');
|
|
||||||
await expect(schoolLinks(page).first()).toBeVisible({ timeout: 15_000 });
|
|
||||||
|
|
||||||
// Switch to the map view, then open the map fullscreen.
|
|
||||||
await page.getByRole('button', { name: 'Map', exact: true }).click();
|
|
||||||
const openFs = page.getByRole('button', { name: 'View map fullscreen' });
|
|
||||||
await expect(openFs).toBeVisible({ timeout: 15_000 });
|
|
||||||
await openFs.click();
|
|
||||||
|
|
||||||
// The button flips to its exit state once the overlay is up.
|
|
||||||
const exitFs = page.getByRole('button', { name: 'Exit fullscreen' });
|
|
||||||
await expect(exitFs).toBeVisible();
|
|
||||||
await exitFs.click();
|
|
||||||
await expect(openFs).toBeVisible();
|
|
||||||
});
|
|
||||||
|
|
||||||
test('comparing two schools shows both side by side', async ({ page }) => {
|
test('comparing two schools shows both side by side', async ({ page }) => {
|
||||||
// Collect two school URNs from search results, then load the share URL
|
// Collect two school URNs from search results, then load the share URL
|
||||||
await searchByName(page, 'primary');
|
await searchByName(page, 'primary');
|
||||||
@@ -154,37 +91,6 @@ test('comparing two schools shows both side by side', async ({ page }) => {
|
|||||||
await expect(page.locator(`a[href*="${urns[1]}"]`).first()).toBeVisible();
|
await expect(page.locator(`a[href*="${urns[1]}"]`).first()).toBeVisible();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('compare chart on mobile shows school chips with tap-to-focus', async ({ page }) => {
|
|
||||||
await page.setViewportSize({ width: 390, height: 844 });
|
|
||||||
|
|
||||||
await searchByName(page, 'primary');
|
|
||||||
await expect(schoolLinks(page).first()).toBeVisible({ timeout: 15_000 });
|
|
||||||
const hrefs = await schoolLinks(page).evaluateAll((links) =>
|
|
||||||
links.map((l) => (l as HTMLAnchorElement).getAttribute('href') || '')
|
|
||||||
);
|
|
||||||
const urns = [...new Set(hrefs.map((h) => h.match(/\/school\/(\d+)/)?.[1]).filter(Boolean))];
|
|
||||||
// Compare three schools, not two: a "primary" search can return all-through
|
|
||||||
// schools that classify as secondary, and the chips only appear for the
|
|
||||||
// active phase. With three schools across two phases, the auto-selected
|
|
||||||
// majority phase always holds ≥2, so the chip legend is guaranteed to render.
|
|
||||||
expect(urns.length).toBeGreaterThanOrEqual(3);
|
|
||||||
|
|
||||||
await page.goto(`/compare?urns=${urns[0]},${urns[1]},${urns[2]}`);
|
|
||||||
await expect(page.locator('canvas:visible').first()).toBeVisible({ timeout: 15_000 });
|
|
||||||
|
|
||||||
// The mobile chart legend renders one chip per school in the active phase.
|
|
||||||
const chipGroup = page.getByRole('group', { name: /highlight a school/i });
|
|
||||||
const chips = chipGroup.getByRole('button');
|
|
||||||
await expect(chips.first()).toBeVisible({ timeout: 15_000 });
|
|
||||||
expect(await chips.count()).toBeGreaterThanOrEqual(2);
|
|
||||||
|
|
||||||
// Tapping a chip focuses that school's line; tapping again releases it.
|
|
||||||
await chips.first().click();
|
|
||||||
await expect(chips.first()).toHaveAttribute('aria-pressed', 'true');
|
|
||||||
await chips.first().click();
|
|
||||||
await expect(chips.first()).toHaveAttribute('aria-pressed', 'false');
|
|
||||||
});
|
|
||||||
|
|
||||||
test('rankings page loads a populated table', async ({ page }) => {
|
test('rankings page loads a populated table', async ({ page }) => {
|
||||||
await page.goto('/rankings');
|
await page.goto('/rankings');
|
||||||
await expect(page.getByRole('heading', { name: /rankings/i }).first()).toBeVisible();
|
await expect(page.getByRole('heading', { name: /rankings/i }).first()).toBeVisible();
|
||||||
|
|||||||
@@ -22,9 +22,7 @@ COPY . .
|
|||||||
ENV NEXT_TELEMETRY_DISABLED=1
|
ENV NEXT_TELEMETRY_DISABLED=1
|
||||||
ENV NODE_ENV=production
|
ENV NODE_ENV=production
|
||||||
|
|
||||||
# Default backend URL for any server-side fetch during `next build`. The
|
# Build argument for FastAPI URL (used by Next.js rewrites at build time)
|
||||||
# runtime /api proxy reads FASTAPI_URL per request (see app/api/[...path]),
|
|
||||||
# so the deployed container's env is what actually routes traffic.
|
|
||||||
ARG FASTAPI_URL=http://backend:80/api
|
ARG FASTAPI_URL=http://backend:80/api
|
||||||
ENV FASTAPI_URL=${FASTAPI_URL}
|
ENV FASTAPI_URL=${FASTAPI_URL}
|
||||||
|
|
||||||
|
|||||||
@@ -1,67 +0,0 @@
|
|||||||
/**
|
|
||||||
* SecondarySchoolRow — sixth-form tag must come from the GIAS
|
|
||||||
* has_sixth_form flag, not the age_range-contains-"18" heuristic.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import '@testing-library/jest-dom';
|
|
||||||
import { render, screen } from '@testing-library/react';
|
|
||||||
import { SecondarySchoolRow } from '@/components/SecondarySchoolRow';
|
|
||||||
import type { School } from '@/lib/types';
|
|
||||||
|
|
||||||
const base = {
|
|
||||||
urn: 100002,
|
|
||||||
school_name: 'Beta Sixth Form College',
|
|
||||||
local_authority: 'Testshire',
|
|
||||||
school_type: 'Academy',
|
|
||||||
phase: 'Secondary',
|
|
||||||
gender: 'Mixed',
|
|
||||||
attainment_8_score: 50.0,
|
|
||||||
} as unknown as School;
|
|
||||||
|
|
||||||
describe('SecondarySchoolRow sixth-form tag', () => {
|
|
||||||
it('shows the tag for a 16-19 college with the GIAS flag set', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow
|
|
||||||
school={{ ...base, age_range: '16-19', has_sixth_form: true }}
|
|
||||||
/>,
|
|
||||||
);
|
|
||||||
expect(screen.getByText('Sixth form')).toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag for an 11-18 school without a registered sixth form', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow
|
|
||||||
school={{ ...base, age_range: '11-18', has_sixth_form: false }}
|
|
||||||
/>,
|
|
||||||
);
|
|
||||||
expect(screen.queryByText('Sixth form')).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag when the flag is missing (pipeline not yet re-run)', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow school={{ ...base, age_range: '11-18' }} />,
|
|
||||||
);
|
|
||||||
expect(screen.queryByText('Sixth form')).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('SecondarySchoolRow proposed-to-close tag', () => {
|
|
||||||
it('shows the tag when GIAS status is "Open, but proposed to close"', () => {
|
|
||||||
render(
|
|
||||||
<SecondarySchoolRow
|
|
||||||
school={{ ...base, status: 'Open, but proposed to close' }}
|
|
||||||
/>,
|
|
||||||
);
|
|
||||||
expect(screen.getByText(/Proposed to close/)).toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag for a plain open school', () => {
|
|
||||||
render(<SecondarySchoolRow school={{ ...base, status: 'Open' }} />);
|
|
||||||
expect(screen.queryByText(/Proposed to close/)).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('hides the tag when status is missing', () => {
|
|
||||||
render(<SecondarySchoolRow school={base} />);
|
|
||||||
expect(screen.queryByText(/Proposed to close/)).not.toBeInTheDocument();
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -9,8 +9,6 @@ import {
|
|||||||
isValidPostcode,
|
isValidPostcode,
|
||||||
debounce,
|
debounce,
|
||||||
buildOfstedListBadge,
|
buildOfstedListBadge,
|
||||||
metricKind,
|
|
||||||
computeYBounds,
|
|
||||||
} from '@/lib/utils';
|
} from '@/lib/utils';
|
||||||
|
|
||||||
describe('formatPercentage', () => {
|
describe('formatPercentage', () => {
|
||||||
@@ -161,65 +159,3 @@ describe('buildOfstedListBadge', () => {
|
|||||||
expect(badge.cssClass).toBe('ofstedPending');
|
expect(badge.cssClass).toBe('ofstedPending');
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe('metricKind', () => {
|
|
||||||
it('classifies metrics by key', () => {
|
|
||||||
expect(metricKind('rwm_expected_pct')).toBe('percentage');
|
|
||||||
expect(metricKind('absence_rate')).toBe('percentage');
|
|
||||||
expect(metricKind('reading_progress')).toBe('progress');
|
|
||||||
expect(metricKind('progress_8_score')).toBe('progress');
|
|
||||||
expect(metricKind('attainment_8_score')).toBe('score');
|
|
||||||
expect(metricKind('reading_avg_score')).toBe('score');
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('computeYBounds', () => {
|
|
||||||
it('tightens clustered percentages instead of framing 0-100', () => {
|
|
||||||
const b = computeYBounds([86, 86, 86, 80, 96], 'percentage');
|
|
||||||
expect(b.min).toBeGreaterThanOrEqual(0);
|
|
||||||
expect(b.max).toBeLessThanOrEqual(100);
|
|
||||||
expect(b.min).toBeGreaterThan(50);
|
|
||||||
expect(b.max! - b.min!).toBeGreaterThanOrEqual(10);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('never widens percentages beyond 0-100 for non-negative data', () => {
|
|
||||||
const b = computeYBounds([2, 5, 98], 'percentage');
|
|
||||||
expect(b.min).toBe(0);
|
|
||||||
expect(b.max).toBe(100);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('does not clamp to zero when pct-named trend data is negative', () => {
|
|
||||||
const b = computeYBounds([-12, -3, 4], 'percentage');
|
|
||||||
expect(b.min).toBeLessThan(-12);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('keeps progress bounds symmetric around zero', () => {
|
|
||||||
const b = computeYBounds([-1.2, 0.4, 2.1], 'progress');
|
|
||||||
expect(b.min).toBe(-b.max!);
|
|
||||||
expect(b.min).toBeLessThanOrEqual(-1.2);
|
|
||||||
expect(b.max).toBeGreaterThanOrEqual(2.1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('fits score metrics without a fixed frame', () => {
|
|
||||||
const b = computeYBounds([42.3, 48.9, 51.2], 'score');
|
|
||||||
expect(b.min).toBeGreaterThanOrEqual(0);
|
|
||||||
expect(b.min).toBeLessThanOrEqual(42.3);
|
|
||||||
expect(b.max).toBeGreaterThanOrEqual(51.2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('returns empty bounds when there is no numeric data', () => {
|
|
||||||
expect(computeYBounds([null, undefined, NaN], 'percentage')).toEqual({});
|
|
||||||
expect(computeYBounds([], 'progress')).toEqual({});
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('isProposedToClose', () => {
|
|
||||||
const { isProposedToClose } = require('@/lib/utils');
|
|
||||||
|
|
||||||
it('is true only for the exact GIAS proposed-to-close status', () => {
|
|
||||||
expect(isProposedToClose({ status: 'Open, but proposed to close' })).toBe(true);
|
|
||||||
expect(isProposedToClose({ status: 'Open' })).toBe(false);
|
|
||||||
expect(isProposedToClose({ status: null })).toBe(false);
|
|
||||||
expect(isProposedToClose({})).toBe(false);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|||||||
@@ -1,75 +0,0 @@
|
|||||||
/**
|
|
||||||
* Runtime proxy for /api/* → the FastAPI backend.
|
|
||||||
*
|
|
||||||
* This replaces the old next.config.js `rewrites()` proxy, whose destination
|
|
||||||
* was baked into the build (routes-manifest.json) from FASTAPI_URL at build
|
|
||||||
* time. Because one frontend image is promoted staging→prod, a baked hostname
|
|
||||||
* forced every environment to name the backend identically; a mismatch (e.g.
|
|
||||||
* a `backend_stg` service) produced `getaddrinfo ENOTFOUND backend`.
|
|
||||||
*
|
|
||||||
* A route handler reads process.env.FASTAPI_URL on each request, so the same
|
|
||||||
* image adapts to whatever the backend is called in each environment.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import { type NextRequest, NextResponse } from 'next/server';
|
|
||||||
|
|
||||||
export const dynamic = 'force-dynamic';
|
|
||||||
export const runtime = 'nodejs';
|
|
||||||
|
|
||||||
// FASTAPI_URL already includes the `/api` suffix (e.g. http://backend:80/api).
|
|
||||||
function backendBase(): string {
|
|
||||||
return process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
|
||||||
}
|
|
||||||
|
|
||||||
// Hop-by-hop / length headers must not be copied across a proxy — undici has
|
|
||||||
// already decoded the body, so a stale content-encoding/length corrupts it.
|
|
||||||
const STRIPPED_RESPONSE_HEADERS = ['content-encoding', 'content-length', 'transfer-encoding', 'connection'];
|
|
||||||
const METHODS_WITH_BODY = new Set(['POST', 'PUT', 'PATCH', 'DELETE']);
|
|
||||||
|
|
||||||
async function handler(req: NextRequest, ctx: { params: Promise<{ path: string[] }> }) {
|
|
||||||
const { path } = await ctx.params;
|
|
||||||
const target = `${backendBase()}/${path.join('/')}${req.nextUrl.search}`;
|
|
||||||
|
|
||||||
const headers = new Headers(req.headers);
|
|
||||||
headers.delete('host');
|
|
||||||
headers.delete('connection');
|
|
||||||
|
|
||||||
const init: RequestInit & { duplex?: 'half' } = {
|
|
||||||
method: req.method,
|
|
||||||
headers,
|
|
||||||
redirect: 'manual',
|
|
||||||
cache: 'no-store',
|
|
||||||
};
|
|
||||||
if (METHODS_WITH_BODY.has(req.method)) {
|
|
||||||
init.body = req.body;
|
|
||||||
init.duplex = 'half';
|
|
||||||
}
|
|
||||||
|
|
||||||
let upstream: Response;
|
|
||||||
try {
|
|
||||||
upstream = await fetch(target, init);
|
|
||||||
} catch (err) {
|
|
||||||
// e.g. DNS failure or connection refused — surface a clean 502 instead of
|
|
||||||
// an opaque proxy crash so callers can degrade gracefully.
|
|
||||||
return NextResponse.json({ detail: 'Upstream request failed' }, { status: 502 });
|
|
||||||
}
|
|
||||||
|
|
||||||
const responseHeaders = new Headers(upstream.headers);
|
|
||||||
for (const h of STRIPPED_RESPONSE_HEADERS) responseHeaders.delete(h);
|
|
||||||
|
|
||||||
return new NextResponse(upstream.body, {
|
|
||||||
status: upstream.status,
|
|
||||||
statusText: upstream.statusText,
|
|
||||||
headers: responseHeaders,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
export {
|
|
||||||
handler as GET,
|
|
||||||
handler as HEAD,
|
|
||||||
handler as POST,
|
|
||||||
handler as PUT,
|
|
||||||
handler as PATCH,
|
|
||||||
handler as DELETE,
|
|
||||||
handler as OPTIONS,
|
|
||||||
};
|
|
||||||
@@ -133,7 +133,7 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
|
|||||||
notFound();
|
notFound();
|
||||||
}
|
}
|
||||||
|
|
||||||
const { school_info, yearly_data, absence_data, ofsted, census, admissions, admissions_history, sen_detail, phonics, deprivation, finance } = data;
|
const { school_info, yearly_data, absence_data, ofsted, parent_view, census, admissions, admissions_history, sen_detail, phonics, deprivation, finance } = data;
|
||||||
|
|
||||||
// Redirect bare URN to canonical slug URL
|
// Redirect bare URN to canonical slug URL
|
||||||
const canonicalSlug = schoolUrl(urn, school_info.school_name).replace('/school/', '');
|
const canonicalSlug = schoolUrl(urn, school_info.school_name).replace('/school/', '');
|
||||||
@@ -189,6 +189,7 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
|
|||||||
yearlyData={yearly_data}
|
yearlyData={yearly_data}
|
||||||
absenceData={absence_data}
|
absenceData={absence_data}
|
||||||
ofsted={ofsted ?? null}
|
ofsted={ofsted ?? null}
|
||||||
|
parentView={parent_view ?? null}
|
||||||
census={census ?? null}
|
census={census ?? null}
|
||||||
admissions={admissions ?? null}
|
admissions={admissions ?? null}
|
||||||
senDetail={sen_detail ?? null}
|
senDetail={sen_detail ?? null}
|
||||||
@@ -202,6 +203,7 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
|
|||||||
yearlyData={yearly_data}
|
yearlyData={yearly_data}
|
||||||
absenceData={absence_data}
|
absenceData={absence_data}
|
||||||
ofsted={ofsted ?? null}
|
ofsted={ofsted ?? null}
|
||||||
|
parentView={parent_view ?? null}
|
||||||
census={census ?? null}
|
census={census ?? null}
|
||||||
admissions={admissions ?? null}
|
admissions={admissions ?? null}
|
||||||
admissionsHistory={admissions_history ?? []}
|
admissionsHistory={admissions_history ?? []}
|
||||||
|
|||||||
@@ -1,32 +0,0 @@
|
|||||||
/**
|
|
||||||
* Runtime proxy for /sitemap.xml → the FastAPI backend's generated sitemap.
|
|
||||||
*
|
|
||||||
* Like the /api/* proxy, this reads FASTAPI_URL at request time rather than
|
|
||||||
* baking the backend host into the build, so one image works in every
|
|
||||||
* environment. robots.ts points crawlers here.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import { NextResponse } from 'next/server';
|
|
||||||
|
|
||||||
export const dynamic = 'force-dynamic';
|
|
||||||
export const runtime = 'nodejs';
|
|
||||||
|
|
||||||
function backendOrigin(): string {
|
|
||||||
const base = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL || 'http://localhost:8000/api';
|
|
||||||
return base.replace(/\/api$/, '');
|
|
||||||
}
|
|
||||||
|
|
||||||
export async function GET() {
|
|
||||||
let upstream: Response;
|
|
||||||
try {
|
|
||||||
upstream = await fetch(`${backendOrigin()}/sitemap.xml`, { cache: 'no-store' });
|
|
||||||
} catch {
|
|
||||||
return new NextResponse('Sitemap temporarily unavailable', { status: 502 });
|
|
||||||
}
|
|
||||||
|
|
||||||
const body = await upstream.text();
|
|
||||||
return new NextResponse(body, {
|
|
||||||
status: upstream.status,
|
|
||||||
headers: { 'content-type': upstream.headers.get('content-type') || 'application/xml' },
|
|
||||||
});
|
|
||||||
}
|
|
||||||
@@ -1,67 +0,0 @@
|
|||||||
/* Chart wrapper: chips (mobile) above, canvas filling the rest of the
|
|
||||||
parent .chartContainer, whose fixed height drives Chart.js sizing via
|
|
||||||
maintainAspectRatio: false. */
|
|
||||||
.wrapper {
|
|
||||||
display: flex;
|
|
||||||
flex-direction: column;
|
|
||||||
height: 100%;
|
|
||||||
}
|
|
||||||
|
|
||||||
.canvasBox {
|
|
||||||
position: relative;
|
|
||||||
flex: 1 1 auto;
|
|
||||||
min-height: 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* School chips: mobile-only legend + tap-to-focus control. Desktop keeps
|
|
||||||
Chart.js's built-in legend (with per-school point shapes). */
|
|
||||||
.chips {
|
|
||||||
display: none;
|
|
||||||
}
|
|
||||||
|
|
||||||
@media (max-width: 640px) {
|
|
||||||
.chips {
|
|
||||||
/* Two chips per row so long school names don't crowd into a single
|
|
||||||
line; each chip fills its column and truncates with an ellipsis. */
|
|
||||||
display: grid;
|
|
||||||
grid-template-columns: 1fr 1fr;
|
|
||||||
gap: 6px;
|
|
||||||
padding-bottom: 8px;
|
|
||||||
}
|
|
||||||
|
|
||||||
.chip {
|
|
||||||
display: inline-flex;
|
|
||||||
align-items: center;
|
|
||||||
gap: 6px;
|
|
||||||
min-height: 40px;
|
|
||||||
min-width: 0;
|
|
||||||
padding: 4px 10px;
|
|
||||||
border: 1px solid rgba(0, 0, 0, .12);
|
|
||||||
border-radius: 999px;
|
|
||||||
background: transparent;
|
|
||||||
cursor: pointer;
|
|
||||||
font-size: 12px;
|
|
||||||
font-weight: 600;
|
|
||||||
}
|
|
||||||
|
|
||||||
.chip[aria-pressed="true"] {
|
|
||||||
background: rgba(0, 0, 0, .06);
|
|
||||||
border-color: rgba(0, 0, 0, .35);
|
|
||||||
}
|
|
||||||
|
|
||||||
.chipDot {
|
|
||||||
flex: 0 0 auto;
|
|
||||||
width: 10px;
|
|
||||||
height: 10px;
|
|
||||||
border-radius: 50%;
|
|
||||||
}
|
|
||||||
|
|
||||||
.chipName {
|
|
||||||
overflow: hidden;
|
|
||||||
text-overflow: ellipsis;
|
|
||||||
white-space: nowrap;
|
|
||||||
/* min-width:0 lets the name shrink inside the grid cell so the
|
|
||||||
ellipsis kicks in instead of overflowing. */
|
|
||||||
min-width: 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,82 +1,47 @@
|
|||||||
/**
|
/**
|
||||||
* ComparisonChart Component
|
* ComparisonChart Component
|
||||||
* Multi-school comparison chart using Chart.js.
|
* Multi-school comparison chart using Chart.js
|
||||||
*
|
|
||||||
* Desktop: built-in legend (point-style markers double as per-school shapes).
|
|
||||||
* Mobile (≤640px): the in-chart legend and axis titles are dropped in favour
|
|
||||||
* of a chip row above the canvas; tapping a chip highlights that school's
|
|
||||||
* line and dims the rest. The y-axis auto-fits the data on all viewports so
|
|
||||||
* clustered schools stay distinguishable.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import { useEffect, useState } from 'react';
|
|
||||||
import { Line } from 'react-chartjs-2';
|
import { Line } from 'react-chartjs-2';
|
||||||
import { ChartOptions, ChartDataset, PointStyle } from 'chart.js';
|
import { ChartOptions } from 'chart.js';
|
||||||
import '@/lib/chartSetup';
|
import '@/lib/chartSetup';
|
||||||
import type { ComparisonData } from '@/lib/types';
|
import type { ComparisonData } from '@/lib/types';
|
||||||
import {
|
import { CHART_COLORS, formatAcademicYear } from '@/lib/utils';
|
||||||
CHART_COLORS,
|
|
||||||
CHART_TEXT_COLORS,
|
|
||||||
computeYBounds,
|
|
||||||
formatAcademicYear,
|
|
||||||
metricKind,
|
|
||||||
rgbToRgba,
|
|
||||||
} from '@/lib/utils';
|
|
||||||
import { useIsMobile } from '@/hooks/useIsMobile';
|
|
||||||
import { track } from '@/lib/analytics';
|
|
||||||
import styles from './ComparisonChart.module.css';
|
|
||||||
|
|
||||||
interface ComparisonChartProps {
|
interface ComparisonChartProps {
|
||||||
comparisonData: Record<string, ComparisonData>;
|
comparisonData: Record<string, ComparisonData>;
|
||||||
/** Ordered as displayed in the school cards, so colours match by index. */
|
|
||||||
schools: Array<{ urn: number; school_name: string }>;
|
|
||||||
metric: string;
|
metric: string;
|
||||||
metricLabel: string;
|
metricLabel: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
// One shape per basket slot (MAX_SCHOOLS = 5) — secondary encoding so
|
export function ComparisonChart({ comparisonData, metric, metricLabel }: ComparisonChartProps) {
|
||||||
// converging lines stay tellable apart without relying on hue alone.
|
// Get all schools and their data
|
||||||
const POINT_STYLES: PointStyle[] = ['circle', 'triangle', 'rect', 'rectRot', 'star'];
|
const schools = Object.entries(comparisonData);
|
||||||
|
|
||||||
export function ComparisonChart({ comparisonData, schools, metric, metricLabel }: ComparisonChartProps) {
|
|
||||||
const isMobile = useIsMobile();
|
|
||||||
const [focusedUrn, setFocusedUrn] = useState<number | null>(null);
|
|
||||||
|
|
||||||
// A focused school that leaves the basket must not linger.
|
|
||||||
const urnKey = schools.map((s) => s.urn).join(',');
|
|
||||||
useEffect(() => {
|
|
||||||
setFocusedUrn(null);
|
|
||||||
}, [urnKey]);
|
|
||||||
|
|
||||||
if (schools.length === 0) {
|
if (schools.length === 0) {
|
||||||
return <div>No data available</div>;
|
return <div>No data available</div>;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Union of years across all schools — coverage differs between them.
|
// Get years from first school (assuming all schools have same years)
|
||||||
const years = [
|
const years = schools[0][1].yearly_data.map((d) => d.year).sort((a, b) => a - b);
|
||||||
...new Set(schools.flatMap((s) => comparisonData[String(s.urn)]?.yearly_data.map((d) => d.year) ?? [])),
|
|
||||||
].sort((a, b) => a - b);
|
|
||||||
|
|
||||||
const datasets: ChartDataset<'line'>[] = schools.map((school, index) => {
|
// Create datasets for each school
|
||||||
const data = comparisonData[String(school.urn)];
|
const datasets = schools.map(([urn, data], index) => {
|
||||||
|
const schoolInfo = data.school_info;
|
||||||
const color = CHART_COLORS[index % CHART_COLORS.length];
|
const color = CHART_COLORS[index % CHART_COLORS.length];
|
||||||
const dimmed = focusedUrn !== null && focusedUrn !== school.urn;
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
label: school.school_name,
|
label: schoolInfo.school_name,
|
||||||
data: years.map((year) => {
|
data: years.map((year) => {
|
||||||
const yearData = data?.yearly_data.find((d) => d.year === year);
|
const yearData = data.yearly_data.find((d) => d.year === year);
|
||||||
if (!yearData) return null;
|
if (!yearData) return null;
|
||||||
return yearData[metric as keyof typeof yearData] as number | null;
|
return yearData[metric as keyof typeof yearData] as number | null;
|
||||||
}),
|
}),
|
||||||
borderColor: dimmed ? rgbToRgba(color, 0.2) : color,
|
borderColor: color,
|
||||||
backgroundColor: dimmed ? 'transparent' : rgbToRgba(color, 0.1),
|
backgroundColor: color.replace('rgb', 'rgba').replace(')', ', 0.1)'),
|
||||||
borderWidth: focusedUrn === school.urn ? 3 : dimmed ? 1.5 : 2,
|
|
||||||
pointStyle: POINT_STYLES[index % POINT_STYLES.length],
|
|
||||||
pointRadius: dimmed ? 2 : isMobile ? 3 : 4,
|
|
||||||
pointHoverRadius: isMobile ? 5 : 6,
|
|
||||||
tension: 0.3,
|
tension: 0.3,
|
||||||
spanGaps: true,
|
spanGaps: true,
|
||||||
};
|
};
|
||||||
@@ -87,11 +52,9 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
datasets,
|
datasets,
|
||||||
};
|
};
|
||||||
|
|
||||||
const kind = metricKind(metric);
|
// Determine if metric is a progress score or percentage
|
||||||
const yBounds = computeYBounds(
|
const isProgressScore = metric.includes('progress');
|
||||||
datasets.flatMap((ds) => ds.data as Array<number | null>),
|
const isPercentage = metric.includes('pct') || metric.includes('rate');
|
||||||
kind,
|
|
||||||
);
|
|
||||||
|
|
||||||
const options: ChartOptions<'line'> = {
|
const options: ChartOptions<'line'> = {
|
||||||
responsive: true,
|
responsive: true,
|
||||||
@@ -102,7 +65,6 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
},
|
},
|
||||||
plugins: {
|
plugins: {
|
||||||
legend: {
|
legend: {
|
||||||
display: !isMobile,
|
|
||||||
position: 'top' as const,
|
position: 'top' as const,
|
||||||
labels: {
|
labels: {
|
||||||
usePointStyle: true,
|
usePointStyle: true,
|
||||||
@@ -112,22 +74,26 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
},
|
},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
// No in-chart title: the section heading and metric selector above the
|
|
||||||
// chart already state the metric.
|
|
||||||
title: {
|
title: {
|
||||||
display: false,
|
display: true,
|
||||||
|
text: `${metricLabel} - Comparison`,
|
||||||
|
font: {
|
||||||
|
size: 16,
|
||||||
|
weight: 'bold',
|
||||||
|
},
|
||||||
|
padding: {
|
||||||
|
bottom: 20,
|
||||||
|
},
|
||||||
},
|
},
|
||||||
tooltip: {
|
tooltip: {
|
||||||
backgroundColor: 'rgba(0, 0, 0, 0.8)',
|
backgroundColor: 'rgba(0, 0, 0, 0.8)',
|
||||||
padding: isMobile ? 10 : 12,
|
padding: 12,
|
||||||
titleFont: {
|
titleFont: {
|
||||||
size: isMobile ? 12 : 14,
|
size: 14,
|
||||||
},
|
},
|
||||||
bodyFont: {
|
bodyFont: {
|
||||||
size: isMobile ? 11 : 13,
|
size: 13,
|
||||||
},
|
},
|
||||||
usePointStyle: true,
|
|
||||||
itemSort: (a, b) => (b.parsed.y ?? -Infinity) - (a.parsed.y ?? -Infinity),
|
|
||||||
callbacks: {
|
callbacks: {
|
||||||
label: function (context) {
|
label: function (context) {
|
||||||
let label = context.dataset.label || '';
|
let label = context.dataset.label || '';
|
||||||
@@ -135,7 +101,13 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
label += ': ';
|
label += ': ';
|
||||||
}
|
}
|
||||||
if (context.parsed.y !== null) {
|
if (context.parsed.y !== null) {
|
||||||
label += context.parsed.y.toFixed(1) + (kind === 'percentage' ? '%' : '');
|
if (isProgressScore) {
|
||||||
|
label += context.parsed.y.toFixed(1);
|
||||||
|
} else if (isPercentage) {
|
||||||
|
label += context.parsed.y.toFixed(1) + '%';
|
||||||
|
} else {
|
||||||
|
label += context.parsed.y.toFixed(1);
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
label += 'N/A';
|
label += 'N/A';
|
||||||
}
|
}
|
||||||
@@ -149,18 +121,17 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
type: 'linear' as const,
|
type: 'linear' as const,
|
||||||
display: true,
|
display: true,
|
||||||
title: {
|
title: {
|
||||||
display: !isMobile,
|
display: true,
|
||||||
text: kind === 'percentage' ? 'Percentage (%)' : kind === 'progress' ? 'Progress Score' : 'Value',
|
text: isPercentage ? 'Percentage (%)' : isProgressScore ? 'Progress Score' : 'Value',
|
||||||
font: {
|
font: {
|
||||||
size: 12,
|
size: 12,
|
||||||
weight: 'bold',
|
weight: 'bold',
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
...yBounds,
|
...(isPercentage && {
|
||||||
ticks: {
|
min: 0,
|
||||||
font: { size: isMobile ? 10 : 12 },
|
max: 100,
|
||||||
...(isMobile && { maxTicksLimit: 5 }),
|
}),
|
||||||
},
|
|
||||||
grid: {
|
grid: {
|
||||||
color: 'rgba(0, 0, 0, 0.05)',
|
color: 'rgba(0, 0, 0, 0.05)',
|
||||||
},
|
},
|
||||||
@@ -170,58 +141,16 @@ export function ComparisonChart({ comparisonData, schools, metric, metricLabel }
|
|||||||
display: false,
|
display: false,
|
||||||
},
|
},
|
||||||
title: {
|
title: {
|
||||||
display: !isMobile,
|
display: true,
|
||||||
text: 'Year',
|
text: 'Year',
|
||||||
font: {
|
font: {
|
||||||
size: 12,
|
size: 12,
|
||||||
weight: 'bold',
|
weight: 'bold',
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
ticks: {
|
|
||||||
font: { size: isMobile ? 10 : 12 },
|
|
||||||
...(isMobile && { maxRotation: 0, autoSkip: true, maxTicksLimit: 4 }),
|
|
||||||
},
|
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
|
|
||||||
const toggleFocus = (urn: number) => {
|
return <Line data={chartData} options={options} />;
|
||||||
const next = focusedUrn === urn ? null : urn;
|
|
||||||
setFocusedUrn(next);
|
|
||||||
if (next !== null) track('compare_focus_school', { urn: next });
|
|
||||||
};
|
|
||||||
|
|
||||||
return (
|
|
||||||
<div className={styles.wrapper}>
|
|
||||||
{/* Mobile legend + focus control; a single series needs no legend. */}
|
|
||||||
{schools.length > 1 && (
|
|
||||||
<div className={styles.chips} role="group" aria-label="Highlight a school on the chart">
|
|
||||||
{schools.map((school, index) => (
|
|
||||||
<button
|
|
||||||
key={school.urn}
|
|
||||||
type="button"
|
|
||||||
className={styles.chip}
|
|
||||||
aria-pressed={focusedUrn === school.urn}
|
|
||||||
onClick={() => toggleFocus(school.urn)}
|
|
||||||
>
|
|
||||||
<span
|
|
||||||
className={styles.chipDot}
|
|
||||||
style={{ background: CHART_COLORS[index % CHART_COLORS.length] }}
|
|
||||||
aria-hidden="true"
|
|
||||||
/>
|
|
||||||
<span
|
|
||||||
className={styles.chipName}
|
|
||||||
style={{ color: CHART_TEXT_COLORS[index % CHART_TEXT_COLORS.length] }}
|
|
||||||
>
|
|
||||||
{school.school_name}
|
|
||||||
</span>
|
|
||||||
</button>
|
|
||||||
))}
|
|
||||||
</div>
|
|
||||||
)}
|
|
||||||
<div className={styles.canvasBox}>
|
|
||||||
<Line data={chartData} options={options} aria-label={`${metricLabel} comparison chart`} />
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -454,10 +454,7 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
.chartContainer {
|
.chartContainer {
|
||||||
/* Taller than desktop's proportion would suggest: the chip legend row
|
height: 300px;
|
||||||
sits inside, and the in-chart title/legend/axis titles are gone, so
|
|
||||||
nearly all of this is plot area. */
|
|
||||||
height: 340px;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
.comparisonTable {
|
.comparisonTable {
|
||||||
|
|||||||
@@ -111,10 +111,8 @@ export function ComparisonView({
|
|||||||
setComparisonData(data.comparison);
|
setComparisonData(data.comparison);
|
||||||
})
|
})
|
||||||
.catch((err) => {
|
.catch((err) => {
|
||||||
// Keep whatever we already have (SSR data or a previous fetch) rather
|
|
||||||
// than blanking the chart — a transient refetch failure shouldn't
|
|
||||||
// destroy a working comparison the user is looking at.
|
|
||||||
console.error('Failed to fetch comparison:', err);
|
console.error('Failed to fetch comparison:', err);
|
||||||
|
setComparisonData(null);
|
||||||
});
|
});
|
||||||
} else {
|
} else {
|
||||||
setComparisonData(null);
|
setComparisonData(null);
|
||||||
@@ -431,7 +429,6 @@ export function ComparisonView({
|
|||||||
<div className={styles.chartContainer}>
|
<div className={styles.chartContainer}>
|
||||||
<ComparisonChart
|
<ComparisonChart
|
||||||
comparisonData={activeComparisonData}
|
comparisonData={activeComparisonData}
|
||||||
schools={activeSchools}
|
|
||||||
metric={selectedMetric}
|
metric={selectedMetric}
|
||||||
metricLabel={metricLabel}
|
metricLabel={metricLabel}
|
||||||
/>
|
/>
|
||||||
|
|||||||
@@ -36,91 +36,6 @@
|
|||||||
margin-bottom: 0;
|
margin-bottom: 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
.searchHint {
|
|
||||||
margin: 0.875rem 0 0;
|
|
||||||
font-size: 0.95rem;
|
|
||||||
color: var(--text-secondary, #5a554d);
|
|
||||||
text-align: center;
|
|
||||||
}
|
|
||||||
|
|
||||||
.searchHint strong {
|
|
||||||
color: var(--text-primary, #1a1612);
|
|
||||||
font-weight: 600;
|
|
||||||
}
|
|
||||||
|
|
||||||
@media (max-width: 600px) {
|
|
||||||
.searchHint {
|
|
||||||
font-size: 0.85rem;
|
|
||||||
text-align: left;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
.nearMeRow {
|
|
||||||
display: flex;
|
|
||||||
flex-direction: column;
|
|
||||||
align-items: center;
|
|
||||||
gap: 0.5rem;
|
|
||||||
margin-top: 0.75rem;
|
|
||||||
}
|
|
||||||
|
|
||||||
.nearMeBtn {
|
|
||||||
display: inline-flex;
|
|
||||||
align-items: center;
|
|
||||||
gap: 0.5rem;
|
|
||||||
padding: 0.625rem 1.375rem;
|
|
||||||
background: var(--accent-teal, #2d7d7d);
|
|
||||||
color: #fff;
|
|
||||||
border: none;
|
|
||||||
border-radius: 999px;
|
|
||||||
font-size: 0.9375rem;
|
|
||||||
font-weight: 600;
|
|
||||||
cursor: pointer;
|
|
||||||
transition: background 0.2s ease, transform 0.15s ease;
|
|
||||||
font-family: inherit;
|
|
||||||
}
|
|
||||||
|
|
||||||
.nearMeBtn:hover:not(:disabled) {
|
|
||||||
background: #235f5f;
|
|
||||||
transform: translateY(-1px);
|
|
||||||
}
|
|
||||||
|
|
||||||
.nearMeBtn:disabled {
|
|
||||||
opacity: 0.7;
|
|
||||||
cursor: not-allowed;
|
|
||||||
}
|
|
||||||
|
|
||||||
.nearMeSpinner {
|
|
||||||
display: inline-block;
|
|
||||||
width: 14px;
|
|
||||||
height: 14px;
|
|
||||||
border: 2px solid rgba(255, 255, 255, 0.35);
|
|
||||||
border-top-color: #fff;
|
|
||||||
border-radius: 50%;
|
|
||||||
animation: nearMeSpin 0.7s linear infinite;
|
|
||||||
flex-shrink: 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
@keyframes nearMeSpin {
|
|
||||||
to {
|
|
||||||
transform: rotate(360deg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
.geoError {
|
|
||||||
font-size: 0.8125rem;
|
|
||||||
color: var(--accent-coral-dark, #b04a2e);
|
|
||||||
margin: 0;
|
|
||||||
max-width: 340px;
|
|
||||||
text-align: center;
|
|
||||||
}
|
|
||||||
|
|
||||||
@media (max-width: 600px) {
|
|
||||||
.nearMeBtn {
|
|
||||||
width: 100%;
|
|
||||||
justify-content: center;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
.searchSection {
|
.searchSection {
|
||||||
margin-bottom: 0;
|
margin-bottom: 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -11,21 +11,9 @@ interface FilterBarProps {
|
|||||||
filters: Filters;
|
filters: Filters;
|
||||||
isHero?: boolean;
|
isHero?: boolean;
|
||||||
resultFilters?: ResultFilters;
|
resultFilters?: ResultFilters;
|
||||||
// Geolocation "use my location" affordance, shown beside the hero search box.
|
|
||||||
// The state and handler live in HomeView (which owns the geolocation flow).
|
|
||||||
onNearMe?: () => void;
|
|
||||||
geoState?: "idle" | "requesting" | "error";
|
|
||||||
geoError?: string | null;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export function FilterBar({
|
export function FilterBar({ filters, isHero, resultFilters }: FilterBarProps) {
|
||||||
filters,
|
|
||||||
isHero,
|
|
||||||
resultFilters,
|
|
||||||
onNearMe,
|
|
||||||
geoState = "idle",
|
|
||||||
geoError,
|
|
||||||
}: FilterBarProps) {
|
|
||||||
const router = useRouter();
|
const router = useRouter();
|
||||||
const pathname = usePathname();
|
const pathname = usePathname();
|
||||||
const searchParams = useSearchParams();
|
const searchParams = useSearchParams();
|
||||||
@@ -194,52 +182,6 @@ export function FilterBar({
|
|||||||
{isPending ? <div className={styles.spinner}></div> : "Search"}
|
{isPending ? <div className={styles.spinner}></div> : "Search"}
|
||||||
</button>
|
</button>
|
||||||
</div>
|
</div>
|
||||||
{isHero && (
|
|
||||||
<>
|
|
||||||
<p className={styles.searchHint}>
|
|
||||||
Search by <strong>school name</strong> — or use your{" "}
|
|
||||||
<strong>postcode</strong> for the nearest schools.
|
|
||||||
</p>
|
|
||||||
{onNearMe && (
|
|
||||||
<div className={styles.nearMeRow}>
|
|
||||||
<button
|
|
||||||
type="button"
|
|
||||||
className={styles.nearMeBtn}
|
|
||||||
onClick={onNearMe}
|
|
||||||
disabled={geoState === "requesting"}
|
|
||||||
>
|
|
||||||
{geoState === "requesting" ? (
|
|
||||||
<>
|
|
||||||
<span className={styles.nearMeSpinner} aria-hidden="true" />
|
|
||||||
Locating you…
|
|
||||||
</>
|
|
||||||
) : (
|
|
||||||
<>
|
|
||||||
<svg
|
|
||||||
width="15"
|
|
||||||
height="15"
|
|
||||||
viewBox="0 0 24 24"
|
|
||||||
fill="none"
|
|
||||||
stroke="currentColor"
|
|
||||||
strokeWidth="2.5"
|
|
||||||
aria-hidden="true"
|
|
||||||
>
|
|
||||||
<path d="M12 2a7 7 0 0 1 7 7c0 5.25-7 13-7 13S5 14.25 5 9a7 7 0 0 1 7-7z" />
|
|
||||||
<circle cx="12" cy="9" r="2.5" />
|
|
||||||
</svg>
|
|
||||||
Use my location
|
|
||||||
</>
|
|
||||||
)}
|
|
||||||
</button>
|
|
||||||
{geoError && (
|
|
||||||
<p className={styles.geoError} role="alert">
|
|
||||||
{geoError}
|
|
||||||
</p>
|
|
||||||
)}
|
|
||||||
</div>
|
|
||||||
)}
|
|
||||||
</>
|
|
||||||
)}
|
|
||||||
</form>
|
</form>
|
||||||
|
|
||||||
{!isHero && (
|
{!isHero && (
|
||||||
@@ -368,8 +310,8 @@ export function FilterBar({
|
|||||||
disabled={isPending}
|
disabled={isPending}
|
||||||
>
|
>
|
||||||
<option value="">With or without sixth form</option>
|
<option value="">With or without sixth form</option>
|
||||||
<option value="yes">With sixth form</option>
|
<option value="yes">With sixth form (11-18)</option>
|
||||||
<option value="no">Without sixth form</option>
|
<option value="no">Without sixth form (11-16)</option>
|
||||||
</select>
|
</select>
|
||||||
|
|
||||||
{admissionsPolicyOptions.length > 0 && (
|
{admissionsPolicyOptions.length > 0 && (
|
||||||
|
|||||||
@@ -369,16 +369,6 @@
|
|||||||
|
|
||||||
.viewToggle {
|
.viewToggle {
|
||||||
justify-content: center;
|
justify-content: center;
|
||||||
flex-shrink: 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* The sort <select> sizes to its widest option ("Highest Reading, Writing
|
|
||||||
& Maths %"), which overflows a phone viewport — beside the view toggle it
|
|
||||||
ran off the right edge. Let it flex into the remaining space and shrink;
|
|
||||||
the selected label truncates instead of pushing past the screen. */
|
|
||||||
.sortSelect {
|
|
||||||
flex: 1;
|
|
||||||
min-width: 0;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
.mapViewContainer {
|
.mapViewContainer {
|
||||||
@@ -506,6 +496,68 @@
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
.discoverySection {
|
||||||
|
padding: 0.5rem 0 0.5rem;
|
||||||
|
text-align: center;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nearMeRow {
|
||||||
|
display: flex;
|
||||||
|
flex-direction: column;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.5rem;
|
||||||
|
margin-bottom: 1.25rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nearMeBtn {
|
||||||
|
display: inline-flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.5rem;
|
||||||
|
padding: 0.625rem 1.375rem;
|
||||||
|
background: var(--accent-teal, #2d7d7d);
|
||||||
|
color: #fff;
|
||||||
|
border: none;
|
||||||
|
border-radius: 999px;
|
||||||
|
font-size: 0.9375rem;
|
||||||
|
font-weight: 600;
|
||||||
|
cursor: pointer;
|
||||||
|
transition: background 0.2s ease, transform 0.15s ease;
|
||||||
|
font-family: inherit;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nearMeBtn:hover:not(:disabled) {
|
||||||
|
background: #235f5f;
|
||||||
|
transform: translateY(-1px);
|
||||||
|
}
|
||||||
|
|
||||||
|
.nearMeBtn:disabled {
|
||||||
|
opacity: 0.7;
|
||||||
|
cursor: not-allowed;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nearMeBtnSpinner {
|
||||||
|
display: inline-block;
|
||||||
|
width: 14px;
|
||||||
|
height: 14px;
|
||||||
|
border: 2px solid rgba(255, 255, 255, 0.35);
|
||||||
|
border-top-color: #fff;
|
||||||
|
border-radius: 50%;
|
||||||
|
animation: nearMeSpin 0.7s linear infinite;
|
||||||
|
flex-shrink: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
@keyframes nearMeSpin {
|
||||||
|
to { transform: rotate(360deg); }
|
||||||
|
}
|
||||||
|
|
||||||
|
.geoError {
|
||||||
|
font-size: 0.8125rem;
|
||||||
|
color: var(--accent-coral-dark, #b04a2e);
|
||||||
|
margin: 0;
|
||||||
|
max-width: 340px;
|
||||||
|
text-align: center;
|
||||||
|
}
|
||||||
|
|
||||||
.quickSearches {
|
.quickSearches {
|
||||||
display: flex;
|
display: flex;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
|
|||||||
@@ -284,11 +284,37 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
|
|||||||
filters={filters}
|
filters={filters}
|
||||||
isHero={!isSearchActive}
|
isHero={!isSearchActive}
|
||||||
resultFilters={initialSchools.result_filters}
|
resultFilters={initialSchools.result_filters}
|
||||||
onNearMe={handleNearMe}
|
|
||||||
geoState={geoState}
|
|
||||||
geoError={geoError}
|
|
||||||
/>
|
/>
|
||||||
|
|
||||||
|
{/* Discovery section shown on landing page before any search */}
|
||||||
|
{!isSearchActive && initialSchools.schools.length === 0 && (
|
||||||
|
<div className={styles.discoverySection}>
|
||||||
|
<div className={styles.nearMeRow}>
|
||||||
|
<button
|
||||||
|
className={styles.nearMeBtn}
|
||||||
|
onClick={handleNearMe}
|
||||||
|
disabled={geoState === 'requesting'}
|
||||||
|
>
|
||||||
|
{geoState === 'requesting' ? (
|
||||||
|
<>
|
||||||
|
<span className={styles.nearMeBtnSpinner} aria-hidden="true" />
|
||||||
|
Locating you…
|
||||||
|
</>
|
||||||
|
) : (
|
||||||
|
<>
|
||||||
|
<svg width="15" height="15" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth="2.5" aria-hidden="true">
|
||||||
|
<path d="M12 2a7 7 0 0 1 7 7c0 5.25-7 13-7 13S5 14.25 5 9a7 7 0 0 1 7-7z"/>
|
||||||
|
<circle cx="12" cy="9" r="2.5"/>
|
||||||
|
</svg>
|
||||||
|
Schools near me
|
||||||
|
</>
|
||||||
|
)}
|
||||||
|
</button>
|
||||||
|
{geoError && <p className={styles.geoError} role="alert">{geoError}</p>}
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
{/* Admissions countdown strip — only on landing page */}
|
{/* Admissions countdown strip — only on landing page */}
|
||||||
{!isSearchActive && (
|
{!isSearchActive && (
|
||||||
<section className={styles.admissionsStrip}>
|
<section className={styles.admissionsStrip}>
|
||||||
|
|||||||
@@ -10,13 +10,12 @@
|
|||||||
|
|
||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import { useMemo, useState } from 'react';
|
import { useEffect, useMemo, useState } from 'react';
|
||||||
import { Line } from 'react-chartjs-2';
|
import { Line } from 'react-chartjs-2';
|
||||||
import { ChartOptions, ChartDataset } from 'chart.js';
|
import { ChartOptions, ChartDataset } from 'chart.js';
|
||||||
import '@/lib/chartSetup';
|
import '@/lib/chartSetup';
|
||||||
import type { SchoolResult } from '@/lib/types';
|
import type { SchoolResult } from '@/lib/types';
|
||||||
import { formatAcademicYear } from '@/lib/utils';
|
import { formatAcademicYear } from '@/lib/utils';
|
||||||
import { useIsMobile } from '@/hooks/useIsMobile';
|
|
||||||
import { track } from '@/lib/analytics';
|
import { track } from '@/lib/analytics';
|
||||||
import styles from './PerformanceChart.module.css';
|
import styles from './PerformanceChart.module.css';
|
||||||
|
|
||||||
@@ -69,7 +68,16 @@ export function PerformanceChart({
|
|||||||
const sortedData = [...data].sort((a, b) => a.year - b.year);
|
const sortedData = [...data].sort((a, b) => a.year - b.year);
|
||||||
const years = sortedData.map(d => formatAcademicYear(d.year));
|
const years = sortedData.map(d => formatAcademicYear(d.year));
|
||||||
|
|
||||||
const isMobile = useIsMobile();
|
// ── Mobile detection ─────────────────────────────────────────────────
|
||||||
|
// Hydration-safe: SSR renders desktop; client flips to mobile after mount.
|
||||||
|
const [isMobile, setIsMobile] = useState(false);
|
||||||
|
useEffect(() => {
|
||||||
|
const mq = window.matchMedia('(max-width: 640px)');
|
||||||
|
const update = () => setIsMobile(mq.matches);
|
||||||
|
update();
|
||||||
|
mq.addEventListener('change', update);
|
||||||
|
return () => mq.removeEventListener('change', update);
|
||||||
|
}, []);
|
||||||
|
|
||||||
// ── Build per-year national averages ─────────────────────────────────
|
// ── Build per-year national averages ─────────────────────────────────
|
||||||
const natRefRwm: (number | null)[] = sortedData.map(d => {
|
const natRefRwm: (number | null)[] = sortedData.map(d => {
|
||||||
|
|||||||
@@ -713,6 +713,18 @@
|
|||||||
margin: -0.5rem 0 1rem;
|
margin: -0.5rem 0 1rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Response count badge */
|
||||||
|
.responseBadge {
|
||||||
|
font-size: 0.75rem;
|
||||||
|
font-weight: 500;
|
||||||
|
font-family: var(--font-dm-sans), sans-serif;
|
||||||
|
color: var(--text-muted, #8a847a);
|
||||||
|
background: var(--bg-secondary, #f3ede4);
|
||||||
|
padding: 0.1rem 0.5rem;
|
||||||
|
border-radius: 999px;
|
||||||
|
margin-left: auto;
|
||||||
|
}
|
||||||
|
|
||||||
.subSectionTitle {
|
.subSectionTitle {
|
||||||
font-size: 0.875rem;
|
font-size: 0.875rem;
|
||||||
font-weight: 600;
|
font-weight: 600;
|
||||||
@@ -720,6 +732,18 @@
|
|||||||
margin: 1.25rem 0 0.75rem;
|
margin: 1.25rem 0 0.75rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Parent recommendation line in Ofsted section */
|
||||||
|
.parentRecommendLine {
|
||||||
|
font-size: 0.85rem;
|
||||||
|
color: var(--text-secondary, #5c564d);
|
||||||
|
margin: 0.5rem 0 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentRecommendLine strong {
|
||||||
|
color: var(--accent-teal, #2d7d7d);
|
||||||
|
font-weight: 700;
|
||||||
|
}
|
||||||
|
|
||||||
/* Metrics Grid & Cards */
|
/* Metrics Grid & Cards */
|
||||||
.metricsGrid {
|
.metricsGrid {
|
||||||
display: grid;
|
display: grid;
|
||||||
@@ -1070,6 +1094,49 @@
|
|||||||
text-decoration: underline;
|
text-decoration: underline;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Parent View */
|
||||||
|
.parentViewGrid {
|
||||||
|
display: flex;
|
||||||
|
flex-direction: column;
|
||||||
|
gap: 0.5rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewRow {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.75rem;
|
||||||
|
font-size: 0.875rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewLabel {
|
||||||
|
flex: 0 0 18rem;
|
||||||
|
color: var(--text-secondary, #5c564d);
|
||||||
|
font-size: 0.8125rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewBar {
|
||||||
|
flex: 1;
|
||||||
|
height: 0.5rem;
|
||||||
|
background: var(--bg-secondary, #f3ede4);
|
||||||
|
border-radius: 4px;
|
||||||
|
overflow: hidden;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewFill {
|
||||||
|
height: 100%;
|
||||||
|
background: var(--accent-teal, #2d7d7d);
|
||||||
|
border-radius: 4px;
|
||||||
|
transition: width 0.4s ease;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewPct {
|
||||||
|
flex: 0 0 2.75rem;
|
||||||
|
text-align: right;
|
||||||
|
font-size: 0.8125rem;
|
||||||
|
font-weight: 600;
|
||||||
|
color: var(--text-primary, #1a1612);
|
||||||
|
}
|
||||||
|
|
||||||
/* Admissions badge — uses unified status colours */
|
/* Admissions badge — uses unified status colours */
|
||||||
.admissionsBadge {
|
.admissionsBadge {
|
||||||
display: inline-flex;
|
display: inline-flex;
|
||||||
@@ -1202,6 +1269,25 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
@media (max-width: 480px) {
|
@media (max-width: 480px) {
|
||||||
|
.parentViewRow {
|
||||||
|
flex-direction: column;
|
||||||
|
align-items: flex-start;
|
||||||
|
gap: 0.25rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewLabel {
|
||||||
|
flex: none;
|
||||||
|
max-width: 100%;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewBar {
|
||||||
|
width: 100%;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewPct {
|
||||||
|
flex: none;
|
||||||
|
}
|
||||||
|
|
||||||
.card {
|
.card {
|
||||||
padding: 1rem;
|
padding: 1rem;
|
||||||
}
|
}
|
||||||
@@ -1543,18 +1629,3 @@
|
|||||||
.historyDisclosure[open] > .historyToggle::before {
|
.historyDisclosure[open] > .historyToggle::before {
|
||||||
transform: rotate(90deg);
|
transform: rotate(90deg);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* GIAS "Open, but proposed to close" notice strip */
|
|
||||||
.closingStrip {
|
|
||||||
background: #fdf6e3;
|
|
||||||
border-left: 4px solid #e2c96f;
|
|
||||||
border-radius: 0 6px 6px 0;
|
|
||||||
padding: 0.55rem 0.9rem;
|
|
||||||
margin: 0.5rem 0;
|
|
||||||
font-size: 0.88rem;
|
|
||||||
color: #6e5a00;
|
|
||||||
max-width: 68ch;
|
|
||||||
}
|
|
||||||
.closingStrip strong {
|
|
||||||
color: #8a6200;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -13,12 +13,12 @@ import { SchoolHeroMap, type SchoolHeroMapHandle } from './SchoolHeroMap';
|
|||||||
import { MetricTooltip } from './MetricTooltip';
|
import { MetricTooltip } from './MetricTooltip';
|
||||||
import type {
|
import type {
|
||||||
School, SchoolResult, AbsenceData,
|
School, SchoolResult, AbsenceData,
|
||||||
OfstedInspection, SchoolCensus,
|
OfstedInspection, OfstedParentView, SchoolCensus,
|
||||||
SchoolAdmissions, SenDetail, Phonics,
|
SchoolAdmissions, SenDetail, Phonics,
|
||||||
SchoolDeprivation, SchoolFinance, NationalAverages,
|
SchoolDeprivation, SchoolFinance, NationalAverages,
|
||||||
} from '@/lib/types';
|
} from '@/lib/types';
|
||||||
import {
|
import {
|
||||||
formatPercentage, formatProgress, formatAcademicYear, isProposedToClose,
|
formatPercentage, formatProgress, formatAcademicYear,
|
||||||
} from '@/lib/utils';
|
} from '@/lib/utils';
|
||||||
import { DeltaChip } from './DeltaChip';
|
import { DeltaChip } from './DeltaChip';
|
||||||
|
|
||||||
@@ -63,6 +63,7 @@ interface SchoolDetailViewProps {
|
|||||||
yearlyData: SchoolResult[];
|
yearlyData: SchoolResult[];
|
||||||
absenceData: AbsenceData | null;
|
absenceData: AbsenceData | null;
|
||||||
ofsted: OfstedInspection | null;
|
ofsted: OfstedInspection | null;
|
||||||
|
parentView: OfstedParentView | null;
|
||||||
census: SchoolCensus | null;
|
census: SchoolCensus | null;
|
||||||
admissions: SchoolAdmissions | null;
|
admissions: SchoolAdmissions | null;
|
||||||
admissionsHistory: SchoolAdmissions[];
|
admissionsHistory: SchoolAdmissions[];
|
||||||
@@ -74,7 +75,7 @@ interface SchoolDetailViewProps {
|
|||||||
|
|
||||||
export function SchoolDetailView({
|
export function SchoolDetailView({
|
||||||
schoolInfo, yearlyData, absenceData,
|
schoolInfo, yearlyData, absenceData,
|
||||||
ofsted, census, admissions, admissionsHistory, senDetail, phonics, deprivation, finance,
|
ofsted, parentView, census, admissions, admissionsHistory, senDetail, phonics, deprivation, finance,
|
||||||
}: SchoolDetailViewProps) {
|
}: SchoolDetailViewProps) {
|
||||||
const router = useRouter();
|
const router = useRouter();
|
||||||
const { addSchool, removeSchool, isSelected } = useComparison();
|
const { addSchool, removeSchool, isSelected } = useComparison();
|
||||||
@@ -233,6 +234,8 @@ export function SchoolDetailView({
|
|||||||
if (hasInclusionData) navItems.push({ id: 'inclusion', label: 'Pupils' });
|
if (hasInclusionData) navItems.push({ id: 'inclusion', label: 'Pupils' });
|
||||||
if (yearlyData.length > 0) navItems.push({ id: 'history', label: 'History' });
|
if (yearlyData.length > 0) navItems.push({ id: 'history', label: 'History' });
|
||||||
if (hasPhonics && isPrimary) navItems.push({ id: 'phonics', label: 'Phonics' });
|
if (hasPhonics && isPrimary) navItems.push({ id: 'phonics', label: 'Phonics' });
|
||||||
|
if (parentView && parentView.total_responses != null && parentView.total_responses > 0)
|
||||||
|
navItems.push({ id: 'parents', label: 'Parents' });
|
||||||
if (hasSchoolLife) navItems.push({ id: 'school-life', label: 'School Life' });
|
if (hasSchoolLife) navItems.push({ id: 'school-life', label: 'School Life' });
|
||||||
if (hasDeprivation) navItems.push({ id: 'local-area', label: 'Local Area' });
|
if (hasDeprivation) navItems.push({ id: 'local-area', label: 'Local Area' });
|
||||||
if (hasFinance) navItems.push({ id: 'finances', label: 'Finances' });
|
if (hasFinance) navItems.push({ id: 'finances', label: 'Finances' });
|
||||||
@@ -313,12 +316,6 @@ export function SchoolDetailView({
|
|||||||
<span className={styles.metaItem}>{schoolInfo.gender}'s school</span>
|
<span className={styles.metaItem}>{schoolInfo.gender}'s school</span>
|
||||||
)}
|
)}
|
||||||
</div>
|
</div>
|
||||||
{isProposedToClose(schoolInfo) && (
|
|
||||||
<div className={styles.closingStrip} role="note">
|
|
||||||
<strong>⚠ Proposed to close</strong> — this school is proposed for closure,
|
|
||||||
check with the local authority before applying.
|
|
||||||
</div>
|
|
||||||
)}
|
|
||||||
{schoolInfo.address && (
|
{schoolInfo.address && (
|
||||||
<p className={styles.address}>
|
<p className={styles.address}>
|
||||||
{schoolInfo.address}{schoolInfo.postcode && `, ${schoolInfo.postcode}`}
|
{schoolInfo.address}{schoolInfo.postcode && `, ${schoolInfo.postcode}`}
|
||||||
@@ -552,6 +549,11 @@ export function SchoolDetailView({
|
|||||||
) : null;
|
) : null;
|
||||||
})}
|
})}
|
||||||
</div>
|
</div>
|
||||||
|
{parentView?.q_recommend_pct != null && parentView.total_responses != null && parentView.total_responses > 0 && (
|
||||||
|
<p className={styles.parentRecommendLine}>
|
||||||
|
<strong>{Math.round(parentView.q_recommend_pct)}%</strong> of parents would recommend this school ({parentView.total_responses.toLocaleString()} responses)
|
||||||
|
</p>
|
||||||
|
)}
|
||||||
</>
|
</>
|
||||||
) : (
|
) : (
|
||||||
/* ── Old OEIF layout ── */
|
/* ── Old OEIF layout ── */
|
||||||
@@ -570,6 +572,11 @@ export function SchoolDetailView({
|
|||||||
<p className={styles.ofstedDisclaimer}>
|
<p className={styles.ofstedDisclaimer}>
|
||||||
From September 2024, Ofsted no longer makes an overall effectiveness judgement in inspections of state-funded schools.
|
From September 2024, Ofsted no longer makes an overall effectiveness judgement in inspections of state-funded schools.
|
||||||
</p>
|
</p>
|
||||||
|
{parentView?.q_recommend_pct != null && parentView.total_responses != null && parentView.total_responses > 0 && (
|
||||||
|
<p className={styles.parentRecommendLine}>
|
||||||
|
<strong>{Math.round(parentView.q_recommend_pct)}%</strong> of parents would recommend this school ({parentView.total_responses.toLocaleString()} responses)
|
||||||
|
</p>
|
||||||
|
)}
|
||||||
{oeifAllSameGrade ? (
|
{oeifAllSameGrade ? (
|
||||||
<p className={styles.ofstedAllSame}>
|
<p className={styles.ofstedAllSame}>
|
||||||
Rated <strong>{OFSTED_LABELS[ofsted.overall_effectiveness!]}</strong> across all inspected areas — Quality of Teaching, Behaviour, Pupils' Development and Leadership.
|
Rated <strong>{OFSTED_LABELS[ofsted.overall_effectiveness!]}</strong> across all inspected areas — Quality of Teaching, Behaviour, Pupils' Development and Leadership.
|
||||||
@@ -1122,6 +1129,42 @@ export function SchoolDetailView({
|
|||||||
</section>
|
</section>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
{/* What Parents Say */}
|
||||||
|
{parentView && parentView.total_responses != null && parentView.total_responses > 0 && (
|
||||||
|
<section id="parents" className={styles.card}>
|
||||||
|
<h2 className={styles.sectionTitle}>
|
||||||
|
What Parents Say
|
||||||
|
<span className={styles.responseBadge}>
|
||||||
|
{parentView.total_responses.toLocaleString()} responses
|
||||||
|
</span>
|
||||||
|
</h2>
|
||||||
|
<p className={styles.sectionSubtitle}>
|
||||||
|
From the Ofsted Parent View survey — parents share their experience of this school.
|
||||||
|
</p>
|
||||||
|
<div className={styles.parentViewGrid}>
|
||||||
|
{[
|
||||||
|
{ label: 'Would recommend this school', pct: parentView.q_recommend_pct },
|
||||||
|
{ label: 'My child is happy here', pct: parentView.q_happy_pct },
|
||||||
|
{ label: 'My child feels safe here', pct: parentView.q_safe_pct },
|
||||||
|
{ label: 'Teaching is good', pct: parentView.q_teaching_pct },
|
||||||
|
{ label: 'My child makes good progress', pct: parentView.q_progress_pct },
|
||||||
|
{ label: 'School looks after pupils\' wellbeing', pct: parentView.q_wellbeing_pct },
|
||||||
|
{ label: 'Behaviour is well managed', pct: parentView.q_behaviour_pct },
|
||||||
|
{ label: 'School deals well with bullying', pct: parentView.q_bullying_pct },
|
||||||
|
{ label: 'Communicates well with parents', pct: parentView.q_communication_pct },
|
||||||
|
].filter(q => q.pct != null).map(({ label, pct }) => (
|
||||||
|
<div key={label} className={styles.parentViewRow}>
|
||||||
|
<span className={styles.parentViewLabel}>{label}</span>
|
||||||
|
<div className={styles.parentViewBar}>
|
||||||
|
<div className={styles.parentViewFill} style={{ width: `${pct}%` }} />
|
||||||
|
</div>
|
||||||
|
<span className={styles.parentViewPct}>{Math.round(pct!)}%</span>
|
||||||
|
</div>
|
||||||
|
))}
|
||||||
|
</div>
|
||||||
|
</section>
|
||||||
|
)}
|
||||||
|
|
||||||
{/* School Life */}
|
{/* School Life */}
|
||||||
{hasSchoolLife && (
|
{hasSchoolLife && (
|
||||||
<section id="school-life" className={styles.card}>
|
<section id="school-life" className={styles.card}>
|
||||||
|
|||||||
@@ -10,15 +10,6 @@
|
|||||||
height: 100dvh;
|
height: 100dvh;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Fallback fullscreen (iOS Safari — no Element.requestFullscreen): the API
|
|
||||||
can't promote the element, so pin it over the page ourselves. Above the
|
|
||||||
comparison toast (3000) and the bottom nav; below modals (9999+). */
|
|
||||||
.mapWrapper.fsFallback {
|
|
||||||
position: fixed;
|
|
||||||
inset: 0;
|
|
||||||
z-index: 5000;
|
|
||||||
}
|
|
||||||
|
|
||||||
.fullscreenBtn {
|
.fullscreenBtn {
|
||||||
position: absolute;
|
position: absolute;
|
||||||
top: 0.625rem;
|
top: 0.625rem;
|
||||||
|
|||||||
@@ -33,52 +33,22 @@ interface SchoolMapProps {
|
|||||||
|
|
||||||
export function SchoolMap({ schools, center, zoom = 13, referencePoint, onMarkerClick, nationalAvgRwm, laAverages }: SchoolMapProps) {
|
export function SchoolMap({ schools, center, zoom = 13, referencePoint, onMarkerClick, nationalAvgRwm, laAverages }: SchoolMapProps) {
|
||||||
const wrapperRef = useRef<HTMLDivElement>(null);
|
const wrapperRef = useRef<HTMLDivElement>(null);
|
||||||
const [nativeFullscreen, setNativeFullscreen] = useState(false);
|
const [isFullscreen, setIsFullscreen] = useState(false);
|
||||||
// iOS Safari has no Element.requestFullscreen — fall back to a fixed-position
|
|
||||||
// overlay driven by state instead of the Fullscreen API.
|
|
||||||
const [fallbackFullscreen, setFallbackFullscreen] = useState(false);
|
|
||||||
const isFullscreen = nativeFullscreen || fallbackFullscreen;
|
|
||||||
|
|
||||||
// Sync state with browser fullscreen events (e.g. Escape key)
|
// Sync state with browser fullscreen events (e.g. Escape key)
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
const onFsChange = () => setNativeFullscreen(!!document.fullscreenElement);
|
const onFsChange = () => setIsFullscreen(!!document.fullscreenElement);
|
||||||
document.addEventListener('fullscreenchange', onFsChange);
|
document.addEventListener('fullscreenchange', onFsChange);
|
||||||
return () => document.removeEventListener('fullscreenchange', onFsChange);
|
return () => document.removeEventListener('fullscreenchange', onFsChange);
|
||||||
}, []);
|
}, []);
|
||||||
|
|
||||||
// Lock body scroll while the fallback overlay is up.
|
|
||||||
useEffect(() => {
|
|
||||||
if (!fallbackFullscreen) return;
|
|
||||||
const prev = document.body.style.overflow;
|
|
||||||
document.body.style.overflow = 'hidden';
|
|
||||||
return () => { document.body.style.overflow = prev; };
|
|
||||||
}, [fallbackFullscreen]);
|
|
||||||
|
|
||||||
// Leaflet re-measures on window resize (trackResize). Native fullscreen fires
|
|
||||||
// one; the CSS fallback overlay changes size without a resize event, so nudge
|
|
||||||
// Leaflet after the layout settles or the map fills only part of the screen.
|
|
||||||
useEffect(() => {
|
|
||||||
const id = requestAnimationFrame(() => window.dispatchEvent(new Event('resize')));
|
|
||||||
return () => cancelAnimationFrame(id);
|
|
||||||
}, [isFullscreen]);
|
|
||||||
|
|
||||||
const toggleFullscreen = useCallback(() => {
|
const toggleFullscreen = useCallback(() => {
|
||||||
if (document.fullscreenElement) {
|
if (!document.fullscreenElement) {
|
||||||
document.exitFullscreen().catch(() => {});
|
wrapperRef.current?.requestFullscreen();
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (fallbackFullscreen) {
|
|
||||||
setFallbackFullscreen(false);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
const el = wrapperRef.current;
|
|
||||||
if (!el) return;
|
|
||||||
if (el.requestFullscreen) {
|
|
||||||
el.requestFullscreen().catch(() => setFallbackFullscreen(true));
|
|
||||||
} else {
|
} else {
|
||||||
setFallbackFullscreen(true);
|
document.exitFullscreen();
|
||||||
}
|
}
|
||||||
}, [fallbackFullscreen]);
|
}, []);
|
||||||
|
|
||||||
// Calculate center if not provided
|
// Calculate center if not provided
|
||||||
const mapCenter: [number, number] = center || (() => {
|
const mapCenter: [number, number] = center || (() => {
|
||||||
@@ -94,7 +64,7 @@ export function SchoolMap({ schools, center, zoom = 13, referencePoint, onMarker
|
|||||||
})();
|
})();
|
||||||
|
|
||||||
return (
|
return (
|
||||||
<div ref={wrapperRef} className={`${styles.mapWrapper} ${isFullscreen ? styles.fullscreen : ''} ${fallbackFullscreen ? styles.fsFallback : ''}`}>
|
<div ref={wrapperRef} className={`${styles.mapWrapper} ${isFullscreen ? styles.fullscreen : ''}`}>
|
||||||
<button
|
<button
|
||||||
className={styles.fullscreenBtn}
|
className={styles.fullscreenBtn}
|
||||||
onClick={toggleFullscreen}
|
onClick={toggleFullscreen}
|
||||||
|
|||||||
@@ -254,10 +254,3 @@
|
|||||||
justify-content: center;
|
justify-content: center;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/* GIAS "Open, but proposed to close" marker */
|
|
||||||
.attrClosing {
|
|
||||||
background: #fdf6e3;
|
|
||||||
color: #8a6200;
|
|
||||||
border: 1px solid #e2c96f;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import type { School } from '@/lib/types';
|
import type { School } from '@/lib/types';
|
||||||
import { formatPercentage, calculateTrend, getPhaseStyle, schoolUrl, buildOfstedListBadge, formatAgeRange, isProposedToClose } from '@/lib/utils';
|
import { formatPercentage, calculateTrend, getPhaseStyle, schoolUrl, buildOfstedListBadge, formatAgeRange } from '@/lib/utils';
|
||||||
import styles from './SchoolRow.module.css';
|
import styles from './SchoolRow.module.css';
|
||||||
|
|
||||||
interface SchoolRowProps {
|
interface SchoolRowProps {
|
||||||
@@ -78,9 +78,6 @@ export function SchoolRow({
|
|||||||
{school.age_range && <span className={styles.attr}>{formatAgeRange(school.age_range)}</span>}
|
{school.age_range && <span className={styles.attr}>{formatAgeRange(school.age_range)}</span>}
|
||||||
{showDenomination && <span className={styles.attr}>{school.religious_denomination}</span>}
|
{showDenomination && <span className={styles.attr}>{school.religious_denomination}</span>}
|
||||||
{showGender && <span className={styles.attr}>{school.gender}</span>}
|
{showGender && <span className={styles.attr}>{school.gender}</span>}
|
||||||
{isProposedToClose(school) && (
|
|
||||||
<span className={`${styles.attr} ${styles.attrClosing}`}>⚠ Proposed to close</span>
|
|
||||||
)}
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
{/* Line 3: Key stats */}
|
{/* Line 3: Key stats */}
|
||||||
|
|||||||
@@ -383,6 +383,17 @@
|
|||||||
margin: 1.25rem 0 0.75rem;
|
margin: 1.25rem 0 0.75rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
.responseBadge {
|
||||||
|
font-size: 0.75rem;
|
||||||
|
font-weight: 500;
|
||||||
|
font-family: var(--font-dm-sans), sans-serif;
|
||||||
|
color: var(--text-muted, #8a847a);
|
||||||
|
background: var(--bg-secondary, #f3ede4);
|
||||||
|
padding: 0.1rem 0.5rem;
|
||||||
|
border-radius: 999px;
|
||||||
|
margin-left: auto;
|
||||||
|
}
|
||||||
|
|
||||||
/* ── Progress 8 suspension banner ───────────────────── */
|
/* ── Progress 8 suspension banner ───────────────────── */
|
||||||
.p8Banner {
|
.p8Banner {
|
||||||
background: rgba(180, 120, 0, 0.1);
|
background: rgba(180, 120, 0, 0.1);
|
||||||
@@ -653,6 +664,60 @@
|
|||||||
text-decoration: underline;
|
text-decoration: underline;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* ── Parent View ─────────────────────────────────────── */
|
||||||
|
.parentRecommendLine {
|
||||||
|
font-size: 0.85rem;
|
||||||
|
color: var(--text-secondary, #5c564d);
|
||||||
|
margin: 0.5rem 0 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentRecommendLine strong {
|
||||||
|
color: var(--accent-teal, #2d7d7d);
|
||||||
|
font-weight: 700;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewGrid {
|
||||||
|
display: flex;
|
||||||
|
flex-direction: column;
|
||||||
|
gap: 0.5rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewRow {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.75rem;
|
||||||
|
font-size: 0.875rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewLabel {
|
||||||
|
flex: 0 0 18rem;
|
||||||
|
color: var(--text-secondary, #5c564d);
|
||||||
|
font-size: 0.8125rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewBar {
|
||||||
|
flex: 1;
|
||||||
|
height: 0.5rem;
|
||||||
|
background: var(--bg-secondary, #f3ede4);
|
||||||
|
border-radius: 4px;
|
||||||
|
overflow: hidden;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewFill {
|
||||||
|
height: 100%;
|
||||||
|
background: var(--accent-teal, #2d7d7d);
|
||||||
|
border-radius: 4px;
|
||||||
|
transition: width 0.4s ease;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewPct {
|
||||||
|
flex: 0 0 2.75rem;
|
||||||
|
text-align: right;
|
||||||
|
font-size: 0.8125rem;
|
||||||
|
font-weight: 600;
|
||||||
|
color: var(--text-primary, #1a1612);
|
||||||
|
}
|
||||||
|
|
||||||
/* ── Admissions ──────────────────────────────────────── */
|
/* ── Admissions ──────────────────────────────────────── */
|
||||||
.admissionsTypeBadge {
|
.admissionsTypeBadge {
|
||||||
border-radius: 6px;
|
border-radius: 6px;
|
||||||
@@ -1070,6 +1135,10 @@
|
|||||||
font-size: 1rem;
|
font-size: 1rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
.parentViewLabel {
|
||||||
|
flex-basis: 10rem;
|
||||||
|
}
|
||||||
|
|
||||||
.ofstedReportLink {
|
.ofstedReportLink {
|
||||||
margin-left: 0;
|
margin-left: 0;
|
||||||
display: block;
|
display: block;
|
||||||
@@ -1082,6 +1151,25 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
@media (max-width: 480px) {
|
@media (max-width: 480px) {
|
||||||
|
.parentViewRow {
|
||||||
|
flex-direction: column;
|
||||||
|
align-items: flex-start;
|
||||||
|
gap: 0.25rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewLabel {
|
||||||
|
flex: none;
|
||||||
|
max-width: 100%;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewBar {
|
||||||
|
width: 100%;
|
||||||
|
}
|
||||||
|
|
||||||
|
.parentViewPct {
|
||||||
|
flex: none;
|
||||||
|
}
|
||||||
|
|
||||||
.metricsGrid {
|
.metricsGrid {
|
||||||
grid-template-columns: 1fr 1fr;
|
grid-template-columns: 1fr 1fr;
|
||||||
gap: 0.5rem;
|
gap: 0.5rem;
|
||||||
@@ -1099,18 +1187,3 @@
|
|||||||
padding: 0.75rem;
|
padding: 0.75rem;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/* GIAS "Open, but proposed to close" notice strip */
|
|
||||||
.closingStrip {
|
|
||||||
background: #fdf6e3;
|
|
||||||
border-left: 4px solid #e2c96f;
|
|
||||||
border-radius: 0 6px 6px 0;
|
|
||||||
padding: 0.55rem 0.9rem;
|
|
||||||
margin: 0.5rem 0;
|
|
||||||
font-size: 0.88rem;
|
|
||||||
color: #6e5a00;
|
|
||||||
max-width: 68ch;
|
|
||||||
}
|
|
||||||
.closingStrip strong {
|
|
||||||
color: #8a6200;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -19,11 +19,11 @@ const PerformanceChart = dynamic(
|
|||||||
);
|
);
|
||||||
import type {
|
import type {
|
||||||
School, SchoolResult, AbsenceData,
|
School, SchoolResult, AbsenceData,
|
||||||
OfstedInspection, SchoolCensus,
|
OfstedInspection, OfstedParentView, SchoolCensus,
|
||||||
SchoolAdmissions, SenDetail, Phonics,
|
SchoolAdmissions, SenDetail, Phonics,
|
||||||
SchoolDeprivation, SchoolFinance, NationalAverages,
|
SchoolDeprivation, SchoolFinance, NationalAverages,
|
||||||
} from '@/lib/types';
|
} from '@/lib/types';
|
||||||
import { formatPercentage, formatProgress, formatAcademicYear, formatAgeRange, isProposedToClose } from '@/lib/utils';
|
import { formatPercentage, formatProgress, formatAcademicYear, formatAgeRange } from '@/lib/utils';
|
||||||
import { DeltaChip } from './DeltaChip';
|
import { DeltaChip } from './DeltaChip';
|
||||||
import { track, getNavigationSource } from '@/lib/analytics';
|
import { track, getNavigationSource } from '@/lib/analytics';
|
||||||
import styles from './SecondarySchoolDetailView.module.css';
|
import styles from './SecondarySchoolDetailView.module.css';
|
||||||
@@ -65,6 +65,7 @@ interface SecondarySchoolDetailViewProps {
|
|||||||
yearlyData: SchoolResult[];
|
yearlyData: SchoolResult[];
|
||||||
absenceData: AbsenceData | null;
|
absenceData: AbsenceData | null;
|
||||||
ofsted: OfstedInspection | null;
|
ofsted: OfstedInspection | null;
|
||||||
|
parentView: OfstedParentView | null;
|
||||||
census: SchoolCensus | null;
|
census: SchoolCensus | null;
|
||||||
admissions: SchoolAdmissions | null;
|
admissions: SchoolAdmissions | null;
|
||||||
senDetail: SenDetail | null;
|
senDetail: SenDetail | null;
|
||||||
@@ -75,7 +76,7 @@ interface SecondarySchoolDetailViewProps {
|
|||||||
|
|
||||||
export function SecondarySchoolDetailView({
|
export function SecondarySchoolDetailView({
|
||||||
schoolInfo, yearlyData,
|
schoolInfo, yearlyData,
|
||||||
ofsted, census, admissions, senDetail, deprivation, finance, absenceData,
|
ofsted, parentView, census, admissions, senDetail, deprivation, finance, absenceData,
|
||||||
}: SecondarySchoolDetailViewProps) {
|
}: SecondarySchoolDetailViewProps) {
|
||||||
const router = useRouter();
|
const router = useRouter();
|
||||||
// Hero map — the "View on map" link opens its fullscreen view.
|
// Hero map — the "View on map" link opens its fullscreen view.
|
||||||
@@ -98,9 +99,9 @@ export function SecondarySchoolDetailView({
|
|||||||
|
|
||||||
const secondaryAvg = nationalAvg?.secondary ?? {};
|
const secondaryAvg = nationalAvg?.secondary ?? {};
|
||||||
|
|
||||||
// GIAS OfficialSixthForm flag; missing (pipeline not yet re-run) => false.
|
const hasSixthForm = schoolInfo.age_range?.includes('18') ?? false;
|
||||||
const hasSixthForm = schoolInfo.has_sixth_form ?? false;
|
|
||||||
const hasFinance = finance != null && finance.per_pupil_spend != null;
|
const hasFinance = finance != null && finance.per_pupil_spend != null;
|
||||||
|
const hasParents = parentView != null && parentView.total_responses != null && parentView.total_responses > 0;
|
||||||
const hasDeprivation = deprivation != null && deprivation.idaci_decile != null;
|
const hasDeprivation = deprivation != null && deprivation.idaci_decile != null;
|
||||||
const hasLocation = schoolInfo.latitude != null && schoolInfo.longitude != null;
|
const hasLocation = schoolInfo.latitude != null && schoolInfo.longitude != null;
|
||||||
const hasWellbeing = (latestResults?.sen_support_pct != null || latestResults?.sen_ehcp_pct != null) || hasDeprivation;
|
const hasWellbeing = (latestResults?.sen_support_pct != null || latestResults?.sen_ehcp_pct != null) || hasDeprivation;
|
||||||
@@ -158,6 +159,7 @@ export function SecondarySchoolDetailView({
|
|||||||
if (hasResults) navItems.push({ id: 'gcse', label: 'GCSEs' });
|
if (hasResults) navItems.push({ id: 'gcse', label: 'GCSEs' });
|
||||||
if (admissions) navItems.push({ id: 'admissions', label: 'Admissions' });
|
if (admissions) navItems.push({ id: 'admissions', label: 'Admissions' });
|
||||||
if (yearlyData.length > 1) navItems.push({ id: 'history', label: 'History' });
|
if (yearlyData.length > 1) navItems.push({ id: 'history', label: 'History' });
|
||||||
|
if (hasParents) navItems.push({ id: 'parents', label: 'Parents' });
|
||||||
if (hasWellbeing) navItems.push({ id: 'wellbeing', label: 'Wellbeing' });
|
if (hasWellbeing) navItems.push({ id: 'wellbeing', label: 'Wellbeing' });
|
||||||
if (hasFinance) navItems.push({ id: 'finances', label: 'Finances' });
|
if (hasFinance) navItems.push({ id: 'finances', label: 'Finances' });
|
||||||
|
|
||||||
@@ -237,12 +239,6 @@ export function SecondarySchoolDetailView({
|
|||||||
</span>
|
</span>
|
||||||
)}
|
)}
|
||||||
</div>
|
</div>
|
||||||
{isProposedToClose(schoolInfo) && (
|
|
||||||
<div className={styles.closingStrip} role="note">
|
|
||||||
<strong>⚠ Proposed to close</strong> — this school is proposed for closure,
|
|
||||||
check with the local authority before applying.
|
|
||||||
</div>
|
|
||||||
)}
|
|
||||||
{schoolInfo.address && (
|
{schoolInfo.address && (
|
||||||
<p className={styles.address}>
|
<p className={styles.address}>
|
||||||
{schoolInfo.address}{schoolInfo.postcode && `, ${schoolInfo.postcode}`}
|
{schoolInfo.address}{schoolInfo.postcode && `, ${schoolInfo.postcode}`}
|
||||||
@@ -439,6 +435,11 @@ export function SecondarySchoolDetailView({
|
|||||||
</div>
|
</div>
|
||||||
</>
|
</>
|
||||||
)}
|
)}
|
||||||
|
{hasParents && (
|
||||||
|
<p className={styles.parentRecommendLine}>
|
||||||
|
<strong>{Math.round(parentView!.q_recommend_pct!)}%</strong> of parents would recommend this school ({parentView!.total_responses!.toLocaleString()} responses)
|
||||||
|
</p>
|
||||||
|
)}
|
||||||
</section>
|
</section>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
@@ -774,6 +775,42 @@ export function SecondarySchoolDetailView({
|
|||||||
</details>
|
</details>
|
||||||
</section>
|
</section>
|
||||||
)}
|
)}
|
||||||
|
{/* ── Parent View ────────────────────────────────── */}
|
||||||
|
{hasParents && parentView && (
|
||||||
|
<section id="parents" className={styles.card}>
|
||||||
|
<h2 className={styles.sectionTitle}>
|
||||||
|
What Parents Say
|
||||||
|
<span className={styles.responseBadge}>
|
||||||
|
{parentView.total_responses!.toLocaleString()} responses
|
||||||
|
</span>
|
||||||
|
</h2>
|
||||||
|
<p className={styles.sectionSubtitle}>
|
||||||
|
From the Ofsted Parent View survey — parents share their experience of this school.
|
||||||
|
</p>
|
||||||
|
<div className={styles.parentViewGrid}>
|
||||||
|
{[
|
||||||
|
{ label: 'Would recommend this school', pct: parentView.q_recommend_pct },
|
||||||
|
{ label: 'My child is happy here', pct: parentView.q_happy_pct },
|
||||||
|
{ label: 'My child feels safe here', pct: parentView.q_safe_pct },
|
||||||
|
{ label: 'Teaching is good', pct: parentView.q_teaching_pct },
|
||||||
|
{ label: 'My child makes good progress', pct: parentView.q_progress_pct },
|
||||||
|
{ label: 'School looks after pupils\' wellbeing', pct: parentView.q_wellbeing_pct },
|
||||||
|
{ label: 'Behaviour is well managed', pct: parentView.q_behaviour_pct },
|
||||||
|
{ label: 'School deals well with bullying', pct: parentView.q_bullying_pct },
|
||||||
|
{ label: 'Communicates well with parents', pct: parentView.q_communication_pct },
|
||||||
|
].filter(q => q.pct != null).map(({ label, pct }) => (
|
||||||
|
<div key={label} className={styles.parentViewRow}>
|
||||||
|
<span className={styles.parentViewLabel}>{label}</span>
|
||||||
|
<div className={styles.parentViewBar}>
|
||||||
|
<div className={styles.parentViewFill} style={{ width: `${pct}%` }} />
|
||||||
|
</div>
|
||||||
|
<span className={styles.parentViewPct}>{Math.round(pct!)}%</span>
|
||||||
|
</div>
|
||||||
|
))}
|
||||||
|
</div>
|
||||||
|
</section>
|
||||||
|
)}
|
||||||
|
|
||||||
{/* ── Wellbeing ──────────────────────────────────── */}
|
{/* ── Wellbeing ──────────────────────────────────── */}
|
||||||
{hasWellbeing && (
|
{hasWellbeing && (
|
||||||
<section id="wellbeing" className={styles.card}>
|
<section id="wellbeing" className={styles.card}>
|
||||||
|
|||||||
@@ -266,9 +266,3 @@
|
|||||||
justify-content: center;
|
justify-content: center;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
.closingTag {
|
|
||||||
background: #fdf6e3;
|
|
||||||
color: #8a6200;
|
|
||||||
border: 1px solid #e2c96f;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@
|
|||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import type { School } from '@/lib/types';
|
import type { School } from '@/lib/types';
|
||||||
import { buildOfstedListBadge, getPhaseStyle, schoolUrl, formatAgeRange, isProposedToClose } from '@/lib/utils';
|
import { buildOfstedListBadge, getPhaseStyle, schoolUrl, formatAgeRange } from '@/lib/utils';
|
||||||
import styles from './SecondarySchoolRow.module.css';
|
import styles from './SecondarySchoolRow.module.css';
|
||||||
|
|
||||||
function detectAdmissionsTag(school: School): string | null {
|
function detectAdmissionsTag(school: School): string | null {
|
||||||
@@ -23,8 +23,7 @@ function detectAdmissionsTag(school: School): string | null {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function hasSixthForm(school: School): boolean {
|
function hasSixthForm(school: School): boolean {
|
||||||
// GIAS OfficialSixthForm flag; missing (pipeline not yet re-run) => false.
|
return school.age_range?.includes('18') ?? false;
|
||||||
return school.has_sixth_form ?? false;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
interface SecondarySchoolRowProps {
|
interface SecondarySchoolRowProps {
|
||||||
@@ -97,9 +96,6 @@ export function SecondarySchoolRow({
|
|||||||
{admissionsTag}
|
{admissionsTag}
|
||||||
</span>
|
</span>
|
||||||
)}
|
)}
|
||||||
{isProposedToClose(school) && (
|
|
||||||
<span className={`${styles.provisionTag} ${styles.closingTag}`}>⚠ Proposed to close</span>
|
|
||||||
)}
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
{/* Line 3: KS4 stats */}
|
{/* Line 3: KS4 stats */}
|
||||||
|
|||||||
@@ -1,23 +0,0 @@
|
|||||||
/**
|
|
||||||
* Viewport hook shared by the chart components.
|
|
||||||
* Hydration-safe: SSR and the first client render report desktop; the
|
|
||||||
* media-query subscription flips the value after mount.
|
|
||||||
*/
|
|
||||||
|
|
||||||
'use client';
|
|
||||||
|
|
||||||
import { useEffect, useState } from 'react';
|
|
||||||
|
|
||||||
export function useIsMobile(maxWidth = 640): boolean {
|
|
||||||
const [isMobile, setIsMobile] = useState(false);
|
|
||||||
|
|
||||||
useEffect(() => {
|
|
||||||
const mq = window.matchMedia(`(max-width: ${maxWidth}px)`);
|
|
||||||
const update = () => setIsMobile(mq.matches);
|
|
||||||
update();
|
|
||||||
mq.addEventListener('change', update);
|
|
||||||
return () => mq.removeEventListener('change', update);
|
|
||||||
}, [maxWidth]);
|
|
||||||
|
|
||||||
return isMobile;
|
|
||||||
}
|
|
||||||
@@ -29,7 +29,6 @@ export type EventName =
|
|||||||
| 'compare_viewed'
|
| 'compare_viewed'
|
||||||
| 'compare_metric_changed'
|
| 'compare_metric_changed'
|
||||||
| 'compare_shared'
|
| 'compare_shared'
|
||||||
| 'compare_focus_school'
|
|
||||||
// Operational
|
// Operational
|
||||||
| 'api_error'
|
| 'api_error'
|
||||||
| 'results_load_more';
|
| 'results_load_more';
|
||||||
|
|||||||
+20
-2
@@ -17,8 +17,6 @@ export interface School {
|
|||||||
school_type_code: string | null;
|
school_type_code: string | null;
|
||||||
religious_denomination: string | null;
|
religious_denomination: string | null;
|
||||||
age_range: string | null;
|
age_range: string | null;
|
||||||
has_sixth_form?: boolean | null;
|
|
||||||
status?: string | null; // GIAS establishment status ("Open" / "Open, but proposed to close")
|
|
||||||
|
|
||||||
// Address
|
// Address
|
||||||
address1: string | null;
|
address1: string | null;
|
||||||
@@ -101,6 +99,25 @@ export interface OfstedInspection {
|
|||||||
rc_sixth_form: number | null;
|
rc_sixth_form: number | null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export interface OfstedParentView {
|
||||||
|
survey_date: string | null;
|
||||||
|
total_responses: number | null;
|
||||||
|
q_happy_pct: number | null;
|
||||||
|
q_safe_pct: number | null;
|
||||||
|
q_behaviour_pct: number | null;
|
||||||
|
q_bullying_pct: number | null;
|
||||||
|
q_communication_pct: number | null;
|
||||||
|
q_progress_pct: number | null;
|
||||||
|
q_teaching_pct: number | null;
|
||||||
|
q_information_pct: number | null;
|
||||||
|
q_curriculum_pct: number | null;
|
||||||
|
q_future_pct: number | null;
|
||||||
|
q_leadership_pct: number | null;
|
||||||
|
q_wellbeing_pct: number | null;
|
||||||
|
q_recommend_pct: number | null;
|
||||||
|
q_sen_pct: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
export interface SchoolCensus {
|
export interface SchoolCensus {
|
||||||
year: number;
|
year: number;
|
||||||
total_pupils: number | null;
|
total_pupils: number | null;
|
||||||
@@ -295,6 +312,7 @@ export interface SchoolDetailsResponse {
|
|||||||
absence_data: AbsenceData | null;
|
absence_data: AbsenceData | null;
|
||||||
// Supplementary data (null until Kestra populates)
|
// Supplementary data (null until Kestra populates)
|
||||||
ofsted: OfstedInspection | null;
|
ofsted: OfstedInspection | null;
|
||||||
|
parent_view: OfstedParentView | null;
|
||||||
census: SchoolCensus | null;
|
census: SchoolCensus | null;
|
||||||
admissions: SchoolAdmissions | null;
|
admissions: SchoolAdmissions | null;
|
||||||
/** All available admissions years, oldest first. Drives the multi-year trend view. */
|
/** All available admissions years, oldest first. Drives the multi-year trend view. */
|
||||||
|
|||||||
@@ -317,57 +317,6 @@ export function getTrendColor(trend: 'up' | 'down' | 'stable'): string {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Broad shape of a KS2/KS4 metric, used to scale chart axes and format values.
|
|
||||||
*/
|
|
||||||
export type MetricKind = 'percentage' | 'progress' | 'score';
|
|
||||||
|
|
||||||
export function metricKind(metric: string): MetricKind {
|
|
||||||
if (metric.includes('progress')) return 'progress';
|
|
||||||
if (metric.includes('pct') || metric.includes('rate')) return 'percentage';
|
|
||||||
return 'score';
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Fit a chart y-axis to the data instead of a fixed frame, so clustered
|
|
||||||
* series remain distinguishable. Padding keeps a minimum span so noise is
|
|
||||||
* not magnified into drama.
|
|
||||||
*
|
|
||||||
* - percentage: pad and snap to 5s; cap at 100; floor at 0 only when the
|
|
||||||
* data is non-negative (some trend metrics have `pct` in the key but hold
|
|
||||||
* negative year-over-year deltas).
|
|
||||||
* - progress: symmetric around 0 so the zero line always shows.
|
|
||||||
* - score (Attainment 8, scaled scores): pad and snap to integers; floor at
|
|
||||||
* 0 only when the data is non-negative.
|
|
||||||
*/
|
|
||||||
export function computeYBounds(
|
|
||||||
values: Array<number | null | undefined>,
|
|
||||||
kind: MetricKind,
|
|
||||||
): { min?: number; max?: number } {
|
|
||||||
const nums = values.filter((v): v is number => typeof v === 'number' && Number.isFinite(v));
|
|
||||||
if (nums.length === 0) return {};
|
|
||||||
|
|
||||||
const lo = Math.min(...nums);
|
|
||||||
const hi = Math.max(...nums);
|
|
||||||
|
|
||||||
if (kind === 'progress') {
|
|
||||||
const reach = Math.max(2, Math.ceil(Math.max(Math.abs(lo), Math.abs(hi)) + 0.5));
|
|
||||||
return { min: -reach, max: reach };
|
|
||||||
}
|
|
||||||
|
|
||||||
if (kind === 'percentage') {
|
|
||||||
const pad = Math.max(5, Math.round((hi - lo) * 0.2));
|
|
||||||
const min = Math.floor((lo - pad) / 5) * 5;
|
|
||||||
const max = Math.min(100, Math.ceil((hi + pad) / 5) * 5);
|
|
||||||
return { min: lo >= 0 ? Math.max(0, min) : min, max };
|
|
||||||
}
|
|
||||||
|
|
||||||
// score
|
|
||||||
const pad = Math.max(2, (hi - lo) * 0.2);
|
|
||||||
const min = Math.floor(lo - pad);
|
|
||||||
return { min: lo >= 0 ? Math.max(0, min) : min, max: Math.ceil(hi + pad) };
|
|
||||||
}
|
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// Local Storage Utilities
|
// Local Storage Utilities
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -718,18 +667,3 @@ export function buildOfstedListBadge(school: {
|
|||||||
|
|
||||||
return { label: 'Not yet inspected', cssClass: 'ofstedPending' };
|
return { label: 'Not yet inspected', cssClass: 'ofstedPending' };
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
|
||||||
// Establishment status
|
|
||||||
// ============================================================================
|
|
||||||
|
|
||||||
export const PROPOSED_TO_CLOSE_STATUS = 'Open, but proposed to close';
|
|
||||||
|
|
||||||
/**
|
|
||||||
* GIAS lists some operating schools as "Open, but proposed to close".
|
|
||||||
* They remain open (and may stay open if the proposal is withdrawn), but the
|
|
||||||
* UI marks them so families check with the local authority before applying.
|
|
||||||
*/
|
|
||||||
export function isProposedToClose(school: { status?: string | null }): boolean {
|
|
||||||
return school.status === PROPOSED_TO_CLOSE_STATUS;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -3,10 +3,21 @@ const nextConfig = {
|
|||||||
// Enable standalone output for Docker
|
// Enable standalone output for Docker
|
||||||
output: 'standalone',
|
output: 'standalone',
|
||||||
|
|
||||||
// The /api/* and /sitemap.xml proxies to the FastAPI backend are route
|
// API Proxy to FastAPI backend
|
||||||
// handlers (app/api/[...path]/route.ts, app/sitemap.xml/route.ts) rather
|
async rewrites() {
|
||||||
// than rewrites, so the backend host is read from FASTAPI_URL at runtime
|
const apiUrl = process.env.FASTAPI_URL || 'http://localhost:8000/api';
|
||||||
// instead of being baked into the build.
|
const backendUrl = apiUrl.replace(/\/api$/, '');
|
||||||
|
return [
|
||||||
|
{
|
||||||
|
source: '/api/:path*',
|
||||||
|
destination: `${apiUrl}/:path*`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
source: '/sitemap.xml',
|
||||||
|
destination: `${backendUrl}/sitemap.xml`,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
},
|
||||||
|
|
||||||
// Image optimization
|
// Image optimization
|
||||||
images: {
|
images: {
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ RUN pip install --no-cache-dir \
|
|||||||
./plugins/extractors/tap-uk-gias \
|
./plugins/extractors/tap-uk-gias \
|
||||||
./plugins/extractors/tap-uk-ees \
|
./plugins/extractors/tap-uk-ees \
|
||||||
./plugins/extractors/tap-uk-ofsted \
|
./plugins/extractors/tap-uk-ofsted \
|
||||||
|
./plugins/extractors/tap-uk-parent-view \
|
||||||
./plugins/extractors/tap-uk-fbit \
|
./plugins/extractors/tap-uk-fbit \
|
||||||
./plugins/extractors/tap-uk-idaci
|
./plugins/extractors/tap-uk-idaci
|
||||||
|
|
||||||
|
|||||||
@@ -38,31 +38,6 @@ default_args = {
|
|||||||
"retry_delay": timedelta(minutes=5),
|
"retry_delay": timedelta(minutes=5),
|
||||||
}
|
}
|
||||||
|
|
||||||
# The backend caches the marts DataFrame at startup; after any rebuild the
|
|
||||||
# cache must be invalidated or the API serves stale (or empty) data until the
|
|
||||||
# container restarts.
|
|
||||||
INVALIDATE_CACHE_CMD = """
|
|
||||||
set -e
|
|
||||||
BACKEND_URL="${BACKEND_URL:-http://backend:80}"
|
|
||||||
ADMIN_KEY="${ADMIN_API_KEY:-changeme}"
|
|
||||||
|
|
||||||
echo "Calling $BACKEND_URL/api/admin/reload ..."
|
|
||||||
|
|
||||||
response=$(curl -s -o /tmp/reload_response.json -w "%{http_code}" \\
|
|
||||||
--connect-timeout 10 --max-time 120 \\
|
|
||||||
-X POST "$BACKEND_URL/api/admin/reload" \\
|
|
||||||
-H "X-API-Key: $ADMIN_KEY" \\
|
|
||||||
-H "Content-Type: application/json")
|
|
||||||
|
|
||||||
echo "HTTP status: $response"
|
|
||||||
cat /tmp/reload_response.json
|
|
||||||
|
|
||||||
if [ "$response" != "200" ]; then
|
|
||||||
echo "ERROR: backend cache reload failed (HTTP $response)"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
"""
|
|
||||||
|
|
||||||
|
|
||||||
# ── Daily DAG (GIAS + downstream) ──────────────────────────────────────
|
# ── Daily DAG (GIAS + downstream) ──────────────────────────────────────
|
||||||
|
|
||||||
@@ -108,7 +83,7 @@ print(f'Validation passed: {{count}} GIAS rows')
|
|||||||
|
|
||||||
dbt_build = BashOperator(
|
dbt_build = BashOperator(
|
||||||
task_id="dbt_build",
|
task_id="dbt_build",
|
||||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_gias_establishments+ stg_gias_links+ gias_code_names+ --exclude int_ks2_with_lineage+ int_ks4_with_lineage+",
|
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_gias_establishments+ stg_gias_links+ --exclude int_ks2_with_lineage+ int_ks4_with_lineage+",
|
||||||
)
|
)
|
||||||
|
|
||||||
sync_typesense = BashOperator(
|
sync_typesense = BashOperator(
|
||||||
@@ -116,12 +91,7 @@ print(f'Validation passed: {{count}} GIAS rows')
|
|||||||
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
||||||
)
|
)
|
||||||
|
|
||||||
invalidate_cache = BashOperator(
|
extract_group >> validate_raw >> dbt_build >> sync_typesense
|
||||||
task_id="invalidate_cache",
|
|
||||||
bash_command=INVALIDATE_CACHE_CMD,
|
|
||||||
)
|
|
||||||
|
|
||||||
extract_group >> validate_raw >> dbt_build >> sync_typesense >> invalidate_cache
|
|
||||||
|
|
||||||
|
|
||||||
# ── Monthly DAG (Ofsted) ───────────────────────────────────────────────
|
# ── Monthly DAG (Ofsted) ───────────────────────────────────────────────
|
||||||
@@ -151,12 +121,7 @@ with DAG(
|
|||||||
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
||||||
)
|
)
|
||||||
|
|
||||||
invalidate_cache_ofsted = BashOperator(
|
extract_ofsted >> dbt_build_ofsted >> sync_typesense_ofsted
|
||||||
task_id="invalidate_cache",
|
|
||||||
bash_command=INVALIDATE_CACHE_CMD,
|
|
||||||
)
|
|
||||||
|
|
||||||
extract_ofsted >> dbt_build_ofsted >> sync_typesense_ofsted >> invalidate_cache_ofsted
|
|
||||||
|
|
||||||
|
|
||||||
# ── Annual DAG (EES: KS2, KS4, Census, Admissions) ───────────────────
|
# ── Annual DAG (EES: KS2, KS4, Census, Admissions) ───────────────────
|
||||||
@@ -188,12 +153,32 @@ with DAG(
|
|||||||
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
bash_command=f"cd {PIPELINE_DIR} && python scripts/sync_typesense.py",
|
||||||
)
|
)
|
||||||
|
|
||||||
invalidate_cache_ees = BashOperator(
|
extract_ees_group >> dbt_build_ees >> sync_typesense_ees
|
||||||
task_id="invalidate_cache",
|
|
||||||
bash_command=INVALIDATE_CACHE_CMD,
|
|
||||||
|
# ── Monthly DAG (Parent View) ──────────────────────────────────────────
|
||||||
|
|
||||||
|
with DAG(
|
||||||
|
dag_id="school_data_monthly_parent_view",
|
||||||
|
default_args=default_args,
|
||||||
|
description="Monthly Ofsted Parent View extraction and transform",
|
||||||
|
schedule="0 3 1 * *",
|
||||||
|
start_date=datetime(2025, 1, 1),
|
||||||
|
catchup=False,
|
||||||
|
tags=["school-compare", "monthly"],
|
||||||
|
) as monthly_parent_view_dag:
|
||||||
|
|
||||||
|
extract_parent_view = BashOperator(
|
||||||
|
task_id="extract_parent_view",
|
||||||
|
bash_command=f"cd {PIPELINE_DIR} && {MELTANO_BIN} run tap-uk-parent-view target-postgres",
|
||||||
)
|
)
|
||||||
|
|
||||||
extract_ees_group >> dbt_build_ees >> sync_typesense_ees >> invalidate_cache_ees
|
dbt_build_parent_view = BashOperator(
|
||||||
|
task_id="dbt_build",
|
||||||
|
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_parent_view+ fact_parent_view+",
|
||||||
|
)
|
||||||
|
|
||||||
|
extract_parent_view >> dbt_build_parent_view
|
||||||
|
|
||||||
|
|
||||||
# ── Annual DAG (IDACI Deprivation) ────────────────────────────────────
|
# ── Annual DAG (IDACI Deprivation) ────────────────────────────────────
|
||||||
@@ -218,9 +203,4 @@ with DAG(
|
|||||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_idaci+ fact_deprivation+",
|
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_idaci+ fact_deprivation+",
|
||||||
)
|
)
|
||||||
|
|
||||||
invalidate_cache_idaci = BashOperator(
|
extract_idaci >> dbt_build_idaci
|
||||||
task_id="invalidate_cache",
|
|
||||||
bash_command=INVALIDATE_CACHE_CMD,
|
|
||||||
)
|
|
||||||
|
|
||||||
extract_idaci >> dbt_build_idaci >> invalidate_cache_idaci
|
|
||||||
|
|||||||
@@ -50,6 +50,11 @@ plugins:
|
|||||||
kind: string
|
kind: string
|
||||||
description: Ofsted Management Information download URL
|
description: Ofsted Management Information download URL
|
||||||
|
|
||||||
|
- name: tap-uk-parent-view
|
||||||
|
namespace: uk_parent_view
|
||||||
|
pip_url: ./plugins/extractors/tap-uk-parent-view
|
||||||
|
executable: tap-uk-parent-view
|
||||||
|
|
||||||
- name: tap-uk-fbit
|
- name: tap-uk-fbit
|
||||||
namespace: uk_fbit
|
namespace: uk_fbit
|
||||||
pip_url: ./plugins/extractors/tap-uk-fbit
|
pip_url: ./plugins/extractors/tap-uk-fbit
|
||||||
|
|||||||
@@ -31,22 +31,15 @@ class GIASEstablishmentsStream(Stream):
|
|||||||
schema = th.PropertiesList(
|
schema = th.PropertiesList(
|
||||||
th.Property("URN", th.IntegerType, required=True),
|
th.Property("URN", th.IntegerType, required=True),
|
||||||
th.Property("EstablishmentName", th.StringType),
|
th.Property("EstablishmentName", th.StringType),
|
||||||
th.Property("TypeOfEstablishment (code)", th.StringType),
|
|
||||||
th.Property("TypeOfEstablishment (name)", th.StringType),
|
th.Property("TypeOfEstablishment (name)", th.StringType),
|
||||||
th.Property("PhaseOfEducation (code)", th.StringType),
|
|
||||||
th.Property("PhaseOfEducation (name)", th.StringType),
|
th.Property("PhaseOfEducation (name)", th.StringType),
|
||||||
th.Property("OfficialSixthForm (code)", th.StringType),
|
|
||||||
th.Property("OfficialSixthForm (name)", th.StringType),
|
|
||||||
th.Property("LA (code)", th.StringType),
|
th.Property("LA (code)", th.StringType),
|
||||||
th.Property("LA (name)", th.StringType),
|
th.Property("LA (name)", th.StringType),
|
||||||
th.Property("EstablishmentNumber", th.StringType),
|
th.Property("EstablishmentNumber", th.StringType),
|
||||||
th.Property("EstablishmentStatus (code)", th.StringType),
|
|
||||||
th.Property("EstablishmentStatus (name)", th.StringType),
|
th.Property("EstablishmentStatus (name)", th.StringType),
|
||||||
th.Property("Postcode", th.StringType),
|
th.Property("Postcode", th.StringType),
|
||||||
th.Property("Gender (name)", th.StringType),
|
th.Property("Gender (name)", th.StringType),
|
||||||
th.Property("ReligiousCharacter (code)", th.StringType),
|
|
||||||
th.Property("ReligiousCharacter (name)", th.StringType),
|
th.Property("ReligiousCharacter (name)", th.StringType),
|
||||||
th.Property("AdmissionsPolicy (code)", th.StringType),
|
|
||||||
th.Property("AdmissionsPolicy (name)", th.StringType),
|
th.Property("AdmissionsPolicy (name)", th.StringType),
|
||||||
th.Property("SchoolCapacity", th.StringType),
|
th.Property("SchoolCapacity", th.StringType),
|
||||||
th.Property("NumberOfPupils", th.StringType),
|
th.Property("NumberOfPupils", th.StringType),
|
||||||
|
|||||||
@@ -68,19 +68,6 @@ COLUMN_PRIORITY = {
|
|||||||
"ungraded_inspection_date": [
|
"ungraded_inspection_date": [
|
||||||
"Date of latest ungraded inspection",
|
"Date of latest ungraded inspection",
|
||||||
],
|
],
|
||||||
# Report Card fields (post-Nov 2025 framework). Confirmed verbatim MI
|
|
||||||
# headers per diagnose_compare_gaps.py's Task 1(c) findings. No MI column
|
|
||||||
# currently exists for early-years or sixth-form report-card grades, so
|
|
||||||
# those two fields are deliberately omitted here (see schema below) --
|
|
||||||
# they stay absent from every record, same as the existing `report_url`
|
|
||||||
# pattern for fields with no COLUMN_PRIORITY entry.
|
|
||||||
"rc_safeguarding_met": ["Safeguarding standards"],
|
|
||||||
"rc_inclusion": ["Inclusion"],
|
|
||||||
"rc_curriculum_teaching": ["Curriculum and teaching"],
|
|
||||||
"rc_achievement": ["Achievement"],
|
|
||||||
"rc_attendance_behaviour": ["Attendance and behaviour"],
|
|
||||||
"rc_personal_development": ["Personal development and wellbeing"],
|
|
||||||
"rc_leadership_governance": ["Leadership and governance"],
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -124,17 +111,6 @@ class OfstedInspectionsStream(Stream):
|
|||||||
th.Property("sixth_form_provision", th.StringType),
|
th.Property("sixth_form_provision", th.StringType),
|
||||||
th.Property("ungraded_outcome", th.StringType),
|
th.Property("ungraded_outcome", th.StringType),
|
||||||
th.Property("ungraded_inspection_date", th.StringType),
|
th.Property("ungraded_inspection_date", th.StringType),
|
||||||
th.Property("rc_safeguarding_met", th.StringType),
|
|
||||||
th.Property("rc_inclusion", th.StringType),
|
|
||||||
th.Property("rc_curriculum_teaching", th.StringType),
|
|
||||||
th.Property("rc_achievement", th.StringType),
|
|
||||||
th.Property("rc_attendance_behaviour", th.StringType),
|
|
||||||
th.Property("rc_personal_development", th.StringType),
|
|
||||||
th.Property("rc_leadership_governance", th.StringType),
|
|
||||||
# No MI column exists for these yet; declared for forward
|
|
||||||
# compatibility with the mart schema, always emitted as absent/NULL.
|
|
||||||
th.Property("rc_early_years", th.StringType),
|
|
||||||
th.Property("rc_sixth_form", th.StringType),
|
|
||||||
th.Property("report_url", th.StringType),
|
th.Property("report_url", th.StringType),
|
||||||
).to_dict()
|
).to_dict()
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,18 @@
|
|||||||
|
[build-system]
|
||||||
|
requires = ["setuptools>=68", "wheel"]
|
||||||
|
build-backend = "setuptools.build_meta"
|
||||||
|
|
||||||
|
[project]
|
||||||
|
name = "tap-uk-parent-view"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Singer tap for UK Ofsted Parent View survey data"
|
||||||
|
requires-python = ">=3.10"
|
||||||
|
dependencies = [
|
||||||
|
"singer-sdk~=0.53",
|
||||||
|
"requests>=2.31",
|
||||||
|
"pandas>=2.0",
|
||||||
|
"openpyxl>=3.1",
|
||||||
|
]
|
||||||
|
|
||||||
|
[project.scripts]
|
||||||
|
tap-uk-parent-view = "tap_uk_parent_view.tap:TapUKParentView.cli"
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""tap-uk-parent-view: Singer tap for Ofsted Parent View survey data."""
|
||||||
@@ -0,0 +1,151 @@
|
|||||||
|
"""Parent View Singer tap — extracts survey data from Ofsted Parent View open data portal."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
import re
|
||||||
|
from datetime import date
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
import requests
|
||||||
|
from singer_sdk import Stream, Tap
|
||||||
|
from singer_sdk import typing as th
|
||||||
|
|
||||||
|
OPEN_DATA_PAGE = "https://parentview.ofsted.gov.uk/open-data"
|
||||||
|
|
||||||
|
|
||||||
|
def _positive_pct(row: pd.Series, q_col_base: str) -> float | None:
|
||||||
|
"""Sum 'Strongly agree' + 'Agree' percentages for a question."""
|
||||||
|
strongly = row.get(f"{q_col_base} - Strongly agree %") or row.get(f"{q_col_base} - Strongly Agree %")
|
||||||
|
agree = row.get(f"{q_col_base} - Agree %")
|
||||||
|
try:
|
||||||
|
total = 0.0
|
||||||
|
if pd.notna(strongly):
|
||||||
|
total += float(strongly)
|
||||||
|
if pd.notna(agree):
|
||||||
|
total += float(agree)
|
||||||
|
return round(total, 1) if total > 0 else None
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
class ParentViewStream(Stream):
|
||||||
|
"""Stream: Parent View survey responses per school."""
|
||||||
|
|
||||||
|
name = "parent_view"
|
||||||
|
primary_keys = ["urn"]
|
||||||
|
replication_key = None
|
||||||
|
|
||||||
|
schema = th.PropertiesList(
|
||||||
|
th.Property("urn", th.IntegerType, required=True),
|
||||||
|
th.Property("survey_date", th.StringType),
|
||||||
|
th.Property("total_responses", th.IntegerType),
|
||||||
|
th.Property("q_happy_pct", th.NumberType),
|
||||||
|
th.Property("q_safe_pct", th.NumberType),
|
||||||
|
th.Property("q_behaviour_pct", th.NumberType),
|
||||||
|
th.Property("q_bullying_pct", th.NumberType),
|
||||||
|
th.Property("q_communication_pct", th.NumberType),
|
||||||
|
th.Property("q_progress_pct", th.NumberType),
|
||||||
|
th.Property("q_teaching_pct", th.NumberType),
|
||||||
|
th.Property("q_information_pct", th.NumberType),
|
||||||
|
th.Property("q_curriculum_pct", th.NumberType),
|
||||||
|
th.Property("q_future_pct", th.NumberType),
|
||||||
|
th.Property("q_leadership_pct", th.NumberType),
|
||||||
|
th.Property("q_wellbeing_pct", th.NumberType),
|
||||||
|
th.Property("q_recommend_pct", th.NumberType),
|
||||||
|
).to_dict()
|
||||||
|
|
||||||
|
def _discover_download_url(self) -> str:
|
||||||
|
"""Scrape the open data page for the download link."""
|
||||||
|
resp = requests.get(OPEN_DATA_PAGE, timeout=30)
|
||||||
|
resp.raise_for_status()
|
||||||
|
urls = re.findall(r'href="([^"]+\.(?:xlsx|csv|zip))"', resp.text, re.IGNORECASE)
|
||||||
|
if not urls:
|
||||||
|
msg = "No download link found on Parent View open data page"
|
||||||
|
raise RuntimeError(msg)
|
||||||
|
url = urls[0]
|
||||||
|
if not url.startswith("http"):
|
||||||
|
url = "https://parentview.ofsted.gov.uk" + url
|
||||||
|
return url
|
||||||
|
|
||||||
|
def get_records(self, context):
|
||||||
|
url = self._discover_download_url()
|
||||||
|
self.logger.info("Downloading Parent View data: %s", url)
|
||||||
|
|
||||||
|
resp = requests.get(url, timeout=120)
|
||||||
|
resp.raise_for_status()
|
||||||
|
|
||||||
|
if url.endswith(".xlsx"):
|
||||||
|
df = pd.read_excel(io.BytesIO(resp.content))
|
||||||
|
else:
|
||||||
|
df = pd.read_csv(
|
||||||
|
io.BytesIO(resp.content),
|
||||||
|
encoding="latin-1",
|
||||||
|
low_memory=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Normalise URN column
|
||||||
|
urn_col = next((c for c in df.columns if c.strip().upper() == "URN"), None)
|
||||||
|
if not urn_col:
|
||||||
|
self.logger.error("URN column not found. Columns: %s", list(df.columns)[:20])
|
||||||
|
return
|
||||||
|
|
||||||
|
df.rename(columns={urn_col: "urn"}, inplace=True)
|
||||||
|
df["urn"] = pd.to_numeric(df["urn"], errors="coerce")
|
||||||
|
df = df.dropna(subset=["urn"])
|
||||||
|
|
||||||
|
# Find total responses column
|
||||||
|
resp_col = next(
|
||||||
|
(c for c in df.columns if "total" in c.lower() and "respon" in c.lower()),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
|
||||||
|
today = date.today().isoformat()
|
||||||
|
|
||||||
|
for _, row in df.iterrows():
|
||||||
|
try:
|
||||||
|
urn = int(row["urn"])
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
continue
|
||||||
|
|
||||||
|
total = None
|
||||||
|
if resp_col and pd.notna(row.get(resp_col)):
|
||||||
|
try:
|
||||||
|
total = int(row[resp_col])
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
yield {
|
||||||
|
"urn": urn,
|
||||||
|
"survey_date": today,
|
||||||
|
"total_responses": total,
|
||||||
|
"q_happy_pct": _positive_pct(row, "Q1"),
|
||||||
|
"q_safe_pct": _positive_pct(row, "Q2"),
|
||||||
|
"q_behaviour_pct": _positive_pct(row, "Q3"),
|
||||||
|
"q_bullying_pct": _positive_pct(row, "Q4"),
|
||||||
|
"q_communication_pct": _positive_pct(row, "Q5"),
|
||||||
|
"q_progress_pct": _positive_pct(row, "Q7"),
|
||||||
|
"q_teaching_pct": _positive_pct(row, "Q8"),
|
||||||
|
"q_information_pct": _positive_pct(row, "Q9"),
|
||||||
|
"q_curriculum_pct": _positive_pct(row, "Q10"),
|
||||||
|
"q_future_pct": _positive_pct(row, "Q11"),
|
||||||
|
"q_leadership_pct": _positive_pct(row, "Q12"),
|
||||||
|
"q_wellbeing_pct": _positive_pct(row, "Q13"),
|
||||||
|
"q_recommend_pct": _positive_pct(row, "Q14"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class TapUKParentView(Tap):
|
||||||
|
"""Singer tap for UK Ofsted Parent View."""
|
||||||
|
|
||||||
|
name = "tap-uk-parent-view"
|
||||||
|
config_jsonschema = th.PropertiesList(
|
||||||
|
th.Property("download_url", th.StringType, description="Direct URL to Parent View data file"),
|
||||||
|
).to_dict()
|
||||||
|
|
||||||
|
def discover_streams(self):
|
||||||
|
return [ParentViewStream(self)]
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
TapUKParentView.cli()
|
||||||
@@ -1,305 +0,0 @@
|
|||||||
"""Diagnose the three data gaps blocking the compare-screen redesign.
|
|
||||||
|
|
||||||
Run from repo root (network access required, no DB needed):
|
|
||||||
uv run --with singer-sdk --with pandas --with requests \
|
|
||||||
python pipeline/scripts/diagnose_compare_gaps.py
|
|
||||||
|
|
||||||
(singer_sdk is a transitive import of tap_uk_ees.tap / tap_uk_ofsted.tap and
|
|
||||||
is not part of the repo's default environment, hence the `uv run --with`.)
|
|
||||||
"""
|
|
||||||
import io
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
|
|
||||||
sys.path.insert(0, "pipeline/plugins/extractors/tap-uk-ees")
|
|
||||||
sys.path.insert(0, "pipeline/plugins/extractors/tap-uk-ofsted")
|
|
||||||
from tap_uk_ees.tap import ( # noqa: E402
|
|
||||||
_KS2_NATIONAL_COL_MAP,
|
|
||||||
_KS2_NATIONAL_CSV_URL,
|
|
||||||
download_release_zip,
|
|
||||||
get_all_releases,
|
|
||||||
)
|
|
||||||
from tap_uk_ofsted.tap import discover_csv_url # noqa: E402
|
|
||||||
|
|
||||||
TIMEOUT = 120
|
|
||||||
|
|
||||||
|
|
||||||
def check_national_gps_science():
|
|
||||||
print("\n=== (a) National catalogue CSV: GPS/science columns ===")
|
|
||||||
resp = requests.get(_KS2_NATIONAL_CSV_URL, timeout=TIMEOUT)
|
|
||||||
resp.raise_for_status()
|
|
||||||
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False)
|
|
||||||
df.columns = [c.strip().lower() for c in df.columns]
|
|
||||||
for csv_col in ("pt_gps_exp", "pt_scita_exp", "avg_readscore", "avg_matscore", "avg_gpsscore"):
|
|
||||||
status = "PRESENT" if csv_col in df.columns else "MISSING"
|
|
||||||
print(f" {csv_col}: {status}")
|
|
||||||
gps_like = [c for c in df.columns if "gps" in c or "scita" in c or "sci" in c]
|
|
||||||
print(f" all gps/science-ish columns: {gps_like}")
|
|
||||||
if "geographic_level" in df.columns:
|
|
||||||
nat = df[df["geographic_level"].str.strip().str.lower() == "national"]
|
|
||||||
else:
|
|
||||||
print(" geographic_level column missing — cannot isolate national rows")
|
|
||||||
return
|
|
||||||
print(f" national rows time_periods: {sorted(nat['time_period'].unique())}")
|
|
||||||
# Sample the values our map would read for the latest year
|
|
||||||
latest = nat[nat["time_period"] == nat["time_period"].max()]
|
|
||||||
for csv_col, field in _KS2_NATIONAL_COL_MAP.items():
|
|
||||||
val = latest.iloc[0].get(csv_col, "<col missing>") if len(latest) else "<no row>"
|
|
||||||
print(f" {field} <- {csv_col} = {val!r}")
|
|
||||||
|
|
||||||
|
|
||||||
def check_ks2_attainment_years_subjects():
|
|
||||||
print("\n=== (b) EES KS2 attainment: years & subject labels ===")
|
|
||||||
releases = get_all_releases("key-stage-2-attainment")
|
|
||||||
print(f" releases found: {[r['time_period'] for r in releases]}")
|
|
||||||
for release in releases:
|
|
||||||
try:
|
|
||||||
zf = download_release_zip(release["id"])
|
|
||||||
except Exception as e:
|
|
||||||
print(f" {release['time_period']}: DOWNLOAD FAILED: {e}")
|
|
||||||
continue
|
|
||||||
name = next((n for n in zf.namelist()
|
|
||||||
if "ks2_school_attainment_data" in n and n.endswith(".csv")), None)
|
|
||||||
if not name:
|
|
||||||
print(f" {release['time_period']}: NO school attainment CSV in ZIP")
|
|
||||||
print(f" all CSVs in zip: {[n for n in zf.namelist() if n.endswith('.csv')]}")
|
|
||||||
continue
|
|
||||||
with zf.open(name) as f:
|
|
||||||
df = pd.read_csv(f, dtype=str, keep_default_na=False, nrows=200000)
|
|
||||||
years = sorted(df["time_period"].unique())
|
|
||||||
subjects = sorted(df["subject"].unique())
|
|
||||||
print(f" release {release['time_period']}: time_periods={years}")
|
|
||||||
print(f" subjects={subjects}")
|
|
||||||
|
|
||||||
|
|
||||||
def check_ofsted_report_card_columns():
|
|
||||||
print("\n=== (c) Ofsted MI CSV: report-card columns ===")
|
|
||||||
url = discover_csv_url()
|
|
||||||
print(f" MI file: {url}")
|
|
||||||
if url is None or not url.lower().endswith(".csv"):
|
|
||||||
print(f" URL is not a CSV (likely ODS) — stopping this section. url={url!r}")
|
|
||||||
return
|
|
||||||
resp = requests.get(url, timeout=TIMEOUT)
|
|
||||||
resp.raise_for_status()
|
|
||||||
df = pd.read_csv(io.BytesIO(resp.content), dtype=str, keep_default_na=False, nrows=5)
|
|
||||||
rc_like = [c for c in df.columns
|
|
||||||
if re.search(r"report card|inclusion|curriculum|achievement|safeguard|well.?being|governance", c, re.I)]
|
|
||||||
print(f" candidate report-card columns ({len(rc_like)}):")
|
|
||||||
for c in rc_like:
|
|
||||||
print(f" - {c!r}")
|
|
||||||
print(f" all columns ({len(df.columns)}):")
|
|
||||||
for c in df.columns:
|
|
||||||
print(f" - {c!r}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
check_national_gps_science()
|
|
||||||
check_ks2_attainment_years_subjects()
|
|
||||||
check_ofsted_report_card_columns()
|
|
||||||
|
|
||||||
|
|
||||||
# FINDINGS 2026-07-12: run via
|
|
||||||
# uv run --with singer-sdk --with pandas --with requests \
|
|
||||||
# python pipeline/scripts/diagnose_compare_gaps.py
|
|
||||||
#
|
|
||||||
# (a) National catalogue CSV (GPS/science) — NOT a source-data problem.
|
|
||||||
# pt_gps_exp, pt_scita_exp, avg_readscore, avg_matscore, avg_gpsscore are
|
|
||||||
# all PRESENT in the catalogue CSV and hold real numeric values for the
|
|
||||||
# latest national row (time_period 202425: pt_gps_exp='72.6' ->
|
|
||||||
# gps_expected_pct; pt_scita_exp='81.6' -> science_expected_pct).
|
|
||||||
# national time_periods present: 201516, 201617, 201718, 201819, 201920,
|
|
||||||
# 202021, 202122, 202223, 202324, 202425 (COVID years 201920/202021 are
|
|
||||||
# present as rows but suppressed with 'x' per the module docstring, not
|
|
||||||
# absent). So _KS2_NATIONAL_COL_MAP is correct and the extractor's own
|
|
||||||
# read of the source is fine end-to-end -- the NULLs in
|
|
||||||
# marts.fact_ks2_national_averages are NOT caused by a missing/renamed
|
|
||||||
# source column. The gap must be introduced downstream of the tap
|
|
||||||
# (staging/mart SQL, a stale/incomplete load, or a dbt model not
|
|
||||||
# selecting these two columns) -- Task 5/6 should look at the dbt
|
|
||||||
# staging model for ees_ks2_national and the mart definition, not the
|
|
||||||
# tap/column-map.
|
|
||||||
#
|
|
||||||
# (b) EES KS2 attainment (school-level, "key-stage-2-attainment" publication)
|
|
||||||
# releases found (via get_all_releases): [None, '202425', '202324',
|
|
||||||
# '202223', '202122']. The `None` entry is the *current/latest* release
|
|
||||||
# (its slug doesn't parse to a 6-digit time_period by _slug_to_time_period,
|
|
||||||
# but the CSV inside carries time_period='202425' -- same data as the
|
|
||||||
# 202425-labelled release).
|
|
||||||
#
|
|
||||||
# Only two of the four releases contain a school-level attainment CSV
|
|
||||||
# matching "ks2_school_attainment_data*.csv":
|
|
||||||
# - release None (latest): HAS IT -> time_periods=['202425']
|
|
||||||
# subjects=['Grammar, punctuation and spelling', 'Maths', 'Reading',
|
|
||||||
# 'Reading, writing and maths', 'Science', 'Writing']
|
|
||||||
# - release 202324: HAS IT -> time_periods=['202324']
|
|
||||||
# subjects= same 6 labels as above
|
|
||||||
# - release 202223: NO school attainment CSV in ZIP. This
|
|
||||||
# release's ZIP instead contains only LA/regional/national/MAT-level
|
|
||||||
# files (e.g. ks2_regional_and_local_authority_*, ks2_multi_academy
|
|
||||||
# _trusts_*, ks2_national_*); no data/*school*attainment*.csv file
|
|
||||||
# exists at all in this release's package. This CONFIRMS the
|
|
||||||
# "subject-level 2022/23 is NULL in prod" symptom: the source
|
|
||||||
# release literally does not publish a school-level attainment file
|
|
||||||
# for 202223 under this filename pattern -- it's not a tap bug.
|
|
||||||
# - release 202122: NO school attainment CSV in ZIP. Same
|
|
||||||
# situation: ZIP has only LA/regional/national-level files (e.g.
|
|
||||||
# ks2_regional_and_local_authority_2016_to_2022_revised.csv,
|
|
||||||
# ks2_national_school_characteristics_2016_to_2022_revised.csv);
|
|
||||||
# no school-level attainment CSV present. This CONFIRMS "school-level
|
|
||||||
# 2021/22 is absent" -- again a genuine source-data absence, not an
|
|
||||||
# extractor bug.
|
|
||||||
# Implication for Tasks 5/6/7: 202122 and 202223 school-level attainment
|
|
||||||
# cannot be backfilled from the "key-stage-2-attainment" EES publication
|
|
||||||
# via this filename pattern -- those two years must either be sourced
|
|
||||||
# from a different EES dataset/file (e.g. one of the *_school_location_
|
|
||||||
# and_pupil_characteristics or *_school_type_and_pupil_characteristics
|
|
||||||
# files present in those ZIPs, which may carry school-level rows under a
|
|
||||||
# different filename), left NULL with an explicit "source unavailable"
|
|
||||||
# note, or backfilled from the legacy DfE "Compare School Performance"
|
|
||||||
# wide-format CSVs referenced elsewhere in tap.py. Subject labels to use
|
|
||||||
# when a source *is* found for 202324/202425:
|
|
||||||
# 'Grammar, punctuation and spelling', 'Maths', 'Reading',
|
|
||||||
# 'Reading, writing and maths', 'Science', 'Writing'
|
|
||||||
# (Reading, writing and maths spans reading+writing+maths combined --
|
|
||||||
# this is the RWM row.)
|
|
||||||
#
|
|
||||||
# (c) Ofsted MI CSV (report-card columns) — confirmed PRESENT.
|
|
||||||
# discover_csv_url() resolved to (as at run time, latest inspections
|
|
||||||
# 31 May 2026):
|
|
||||||
# https://assets.publishing.service.gov.uk/media/6a27c45be13080622db38815/
|
|
||||||
# Management_information_-_state-funded_schools_-_latest_inspections_as_at_31_May_2026.csv
|
|
||||||
# This is a real .csv (not .ods) so section (c) ran to completion.
|
|
||||||
# Exact report-card column headers (7 grade columns + their paired date
|
|
||||||
# columns, all present verbatim, case/spacing exactly as below):
|
|
||||||
# 'Safeguarding standards' / 'Safeguarding standards - date of grade'
|
|
||||||
# 'Inclusion' / 'Inclusion - date of grade'
|
|
||||||
# 'Curriculum and teaching' / 'Curriculum and teaching - date of grade'
|
|
||||||
# 'Achievement' / 'Achievement - date of grade'
|
|
||||||
# 'Attendance and behaviour' / 'Attendance and behaviour - date of grade'
|
|
||||||
# 'Personal development and wellbeing' / 'Personal development and wellbeing - date of grade'
|
|
||||||
# 'Leadership and governance' / 'Leadership and governance - date of grade'
|
|
||||||
# Plus a related pass/fail-style field:
|
|
||||||
# 'Latest OEIF safeguarding is effective?' (note: double space in the
|
|
||||||
# header, verbatim from source -- preserve exactly when mapping)
|
|
||||||
# These are the new-style "report card" single-word-area grades
|
|
||||||
# (introduced alongside the "Attendance and behaviour" split from
|
|
||||||
# "Personal development"); they coexist in the same CSV with the legacy
|
|
||||||
# 5-judgement OEIF columns ('Latest OEIF overall effectiveness',
|
|
||||||
# 'Latest OEIF quality of education', 'Latest OEIF behaviour and
|
|
||||||
# attitudes', 'Latest OEIF personal development', 'Latest OEIF
|
|
||||||
# effectiveness of leadership and management'). Task 7 should map the 7
|
|
||||||
# report-card columns above (grade + date pairs, 6 of them, plus the
|
|
||||||
# safeguarding-effective flag) rather than inventing new column names.
|
|
||||||
|
|
||||||
# TASK 6 VERIFICATION 2026-07-12: 2021/22 legacy KS2 school-level archive
|
|
||||||
#
|
|
||||||
# RESULT: BLOCKED at the source-data level. School-level KS2 attainment for
|
|
||||||
# academic year 2021/22 was never published anywhere publicly by DfE -- not
|
|
||||||
# in EES (confirmed by Task 1's finding (b) above), not in the legacy
|
|
||||||
# "Compare School Performance" download wizard, and not as a standalone
|
|
||||||
# performance-tables archive/ODS on assets.publishing.service.gov.uk. This
|
|
||||||
# is a deliberate DfE decision, not a gap in our extraction logic.
|
|
||||||
#
|
|
||||||
# Confirming quote (Key stage 2 attainment 2021/22 release notes, via
|
|
||||||
# https://explore-education-statistics.service.gov.uk/find-statistics/
|
|
||||||
# key-stage-2-attainment/2021-22):
|
|
||||||
# "We will not publish key stage 2 data for academic year 2021/22 in
|
|
||||||
# performance tables (also known as Compare School and College
|
|
||||||
# Performance)." ... "The Department will, however, still produce the
|
|
||||||
# normal suite of key stage 2 accountability measures at school and
|
|
||||||
# multi-academy trust level and share these securely with primary
|
|
||||||
# schools, academy trusts and local authorities to inform school
|
|
||||||
# improvement discussions."
|
|
||||||
# (i.e. school-level 202122 KS2 results exist internally at DfE but were
|
|
||||||
# withheld from every public channel: performance tables/CSCP, EES, and by
|
|
||||||
# extension the legacy DfE archives the current legacy_ks2_urls entries in
|
|
||||||
# meltano.yml were sourced from.)
|
|
||||||
#
|
|
||||||
# What was tried:
|
|
||||||
# 1. Direct download URL pattern from the task brief:
|
|
||||||
# https://www.compare-school-performance.service.gov.uk/download-data?download=true®ions=0&filters=KS2&fileformat=csv&year=2021-2022&meta=false
|
|
||||||
# -> HTTP 404, HTML error page (not a CSV/ZIP). Saved response inspected;
|
|
||||||
# confirmed 404 via response headers (`content-type: text/html`).
|
|
||||||
# 2. Walked the actual multi-step download wizard at
|
|
||||||
# https://www.compare-school-performance.service.gov.uk/download-data
|
|
||||||
# with a browser User-Agent and a cookie jar, replicating the GET-based
|
|
||||||
# form steps: currentstep=year (downloadYear=2021-2022) -> currentstep=
|
|
||||||
# region (regiontype=all&la=0) -> currentstep=datatypes. On the final
|
|
||||||
# "datatypes" step, the checkbox list for 2021-2022 has NO "ks2" (or
|
|
||||||
# "ks2mats") option at all -- only ks4/ks4prov/ks4underlying/ks5* /
|
|
||||||
# pupil-destination/absence/census/mats checkboxes are present.
|
|
||||||
# Control check: repeating the same wizard walk for downloadYear=
|
|
||||||
# 2018-2019, 2022-2023 and 2023-2024 shows a "ks2" (and "ks2mats")
|
|
||||||
# checkbox present in all three; downloadYear=2020-2021 (COVID-cancelled
|
|
||||||
# KS2 SATs year) also has NO ks2 checkbox, matching the pattern for a
|
|
||||||
# year where school-level KS2 genuinely isn't published. 2021-2022
|
|
||||||
# behaves identically to the cancelled 2020-2021 year, not like the
|
|
||||||
# normal 2018-2019/2022-2023/2023-2024 years.
|
|
||||||
# 3. Web search for a standalone KS2 2022 performance-tables archive
|
|
||||||
# (e.g. "england_ks2final" for 2022) on assets.publishing.service.gov.uk
|
|
||||||
# found no such file; only unrelated 2022/2023-dated documents.
|
|
||||||
#
|
|
||||||
# No ZIP was ever obtained -- /tmp/dfe-2021-2022-ks2.zip contains the 404
|
|
||||||
# HTML error page from attempt (1) above, not a real archive. It contains
|
|
||||||
# no england_ks2final.csv (there is no ZIP to look inside).
|
|
||||||
#
|
|
||||||
# Column-map check (brief's Step 1): NOT RUN -- there is no 2021/22
|
|
||||||
# england_ks2final.csv to check headers against. This is moot until/unless
|
|
||||||
# a non-public source (e.g. a manual/internal DfE extract) becomes
|
|
||||||
# available; _LEGACY_KS2_COLUMN_MAP itself is unchanged and untested here.
|
|
||||||
#
|
|
||||||
# Recommendation: mark 202122 school-level KS2 as a genuine, permanent
|
|
||||||
# source-data gap (not a backfill candidate) unless the project can obtain
|
|
||||||
# the internal DfE extract DfE says it shared "securely with primary
|
|
||||||
# schools, academy trusts and local authorities" -- that is not a route
|
|
||||||
# available to this pipeline. Task 6's meltano.yml change (Step 2) and the
|
|
||||||
# filebrowser upload should NOT proceed for 202122; there is nothing to
|
|
||||||
# upload.
|
|
||||||
|
|
||||||
# TASK 7 VALUE SAMPLE 2026-07-12: live value_counts() over the 7 report-card
|
|
||||||
# columns (plus the related safeguarding-effective flag) in the same MI CSV
|
|
||||||
# resolved by discover_csv_url() as at run time (31 May 2026 inspections
|
|
||||||
# file). Blank cells read as the literal string 'NULL' (matches
|
|
||||||
# keep_default_na=False in tap.py). Observed non-blank values, verbatim:
|
|
||||||
#
|
|
||||||
# 'Safeguarding standards': 'Met' (1319), 'Not met' (10)
|
|
||||||
# 'Inclusion': 'Expected standard' (710),
|
|
||||||
# 'Strong standard' (447), 'Needs attention' (130), 'Exceptional' (23),
|
|
||||||
# 'Urgent improvement' (19)
|
|
||||||
# 'Curriculum and teaching': 'Expected standard' (797),
|
|
||||||
# 'Needs attention' (287), 'Strong standard' (206),
|
|
||||||
# 'Urgent improvement' (28), 'Exceptional' (11)
|
|
||||||
# 'Achievement': 'Expected standard' (701),
|
|
||||||
# 'Needs attention' (364), 'Strong standard' (207),
|
|
||||||
# 'Urgent improvement' (39), 'Exceptional' (18)
|
|
||||||
# 'Attendance and behaviour': 'Expected standard' (699),
|
|
||||||
# 'Strong standard' (405), 'Needs attention' (188),
|
|
||||||
# 'Urgent improvement' (21), 'Exceptional' (16)
|
|
||||||
# 'Personal development and wellbeing': 'Expected standard' (728),
|
|
||||||
# 'Strong standard' (504), 'Needs attention' (66), 'Exceptional' (23),
|
|
||||||
# 'Urgent improvement' (8)
|
|
||||||
# 'Leadership and governance': 'Expected standard' (813),
|
|
||||||
# 'Strong standard' (292), 'Needs attention' (172),
|
|
||||||
# 'Urgent improvement' (34), 'Exceptional' (18)
|
|
||||||
# 'Latest OEIF safeguarding is effective?' (note double space, not used by
|
|
||||||
# Task 7 -- kept for completeness): 'Yes' (12970), 'No' (96)
|
|
||||||
#
|
|
||||||
# So the 6 graded report-card columns share exactly one 5-value vocabulary:
|
|
||||||
# {'Exceptional', 'Strong standard', 'Expected standard', 'Needs attention',
|
|
||||||
# 'Urgent improvement'} -- no 'Attention needed' variant was observed
|
|
||||||
# anywhere, so parse_report_card_grade.sql does NOT need that speculative
|
|
||||||
# branch from the task brief. 'Safeguarding standards' is a separate
|
|
||||||
# two-value vocabulary {'Met', 'Not met'}.
|
|
||||||
#
|
|
||||||
# Collision check: 'Achievement' matches by EXACT list-membership
|
|
||||||
# (`candidate in df_columns`, a Python list containment check against the
|
|
||||||
# full column-name list, not a substring/regex match) against only
|
|
||||||
# ['Achievement', 'Achievement - date of grade'] -- the date-paired column
|
|
||||||
# has a different exact string and is never selected. Same check for
|
|
||||||
# 'Safeguarding standards' found only itself, its own date-of-grade column,
|
|
||||||
# and the unrelated 'Latest OEIF safeguarding is effective?' column (not
|
|
||||||
# mapped to any rc_* field). No legacy OEIF column is accidentally consumed
|
|
||||||
# by an rc_ mapping.
|
|
||||||
@@ -1,144 +0,0 @@
|
|||||||
"""Generate GIAS code->name dictionaries from the live bulk CSV.
|
|
||||||
|
|
||||||
Writes:
|
|
||||||
- backend/gias_codes.py (canonical Python module)
|
|
||||||
- pipeline/scripts/gias_codes.py (byte-identical copy)
|
|
||||||
- pipeline/transform/seeds/gias_code_names.csv (dbt seed for drift test)
|
|
||||||
|
|
||||||
Run from the repo root whenever the dbt drift test warns that DfE
|
|
||||||
added/renamed a value: python pipeline/scripts/generate_gias_codes.py
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import io
|
|
||||||
import sys
|
|
||||||
from datetime import date, timedelta
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
|
|
||||||
GIAS_URL = (
|
|
||||||
"https://ea-edubase-api-prod.azurewebsites.net"
|
|
||||||
"/edubase/downloads/public/edubasealldata{date}.csv"
|
|
||||||
)
|
|
||||||
|
|
||||||
# (CSV code column, CSV name column, python dict name, seed field key)
|
|
||||||
FIELDS = [
|
|
||||||
("TypeOfEstablishment (code)", "TypeOfEstablishment (name)", "SCHOOL_TYPE", "school_type"),
|
|
||||||
("EstablishmentStatus (code)", "EstablishmentStatus (name)", "ESTABLISHMENT_STATUS", "establishment_status"),
|
|
||||||
("PhaseOfEducation (code)", "PhaseOfEducation (name)", "PHASE_OF_EDUCATION", "phase_of_education"),
|
|
||||||
("OfficialSixthForm (code)", "OfficialSixthForm (name)", "OFFICIAL_SIXTH_FORM", "official_sixth_form"),
|
|
||||||
("ReligiousCharacter (code)", "ReligiousCharacter (name)", "RELIGIOUS_CHARACTER", "religious_character"),
|
|
||||||
("AdmissionsPolicy (code)", "AdmissionsPolicy (name)", "ADMISSIONS_POLICY", "admissions_policy"),
|
|
||||||
]
|
|
||||||
|
|
||||||
MODULE_HEADER = '''"""GIAS code -> name dictionaries.
|
|
||||||
|
|
||||||
GENERATED by pipeline/scripts/generate_gias_codes.py from the GIAS bulk CSV
|
|
||||||
— do not edit by hand; rerun the script when the dbt drift test warns.
|
|
||||||
The canonical file is backend/gias_codes.py; pipeline/scripts/gias_codes.py
|
|
||||||
must be byte-identical (enforced by backend/tests/test_gias_codes.py).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import math
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
'''
|
|
||||||
|
|
||||||
MODULE_FOOTER = '''
|
|
||||||
|
|
||||||
def translate(code, mapping: dict[int, str]) -> str | None:
|
|
||||||
"""Translate a GIAS code to its display name.
|
|
||||||
|
|
||||||
None/NaN -> None (column absent or suppressed). Unknown codes degrade to
|
|
||||||
"Unknown (<code>)" with a warning so a new DfE value never blanks the UI.
|
|
||||||
"""
|
|
||||||
if code is None or (isinstance(code, float) and math.isnan(code)):
|
|
||||||
return None
|
|
||||||
code = int(code)
|
|
||||||
if code not in mapping:
|
|
||||||
logger.warning("Unknown GIAS code %s (not in dictionary)", code)
|
|
||||||
return f"Unknown ({code})"
|
|
||||||
return mapping[code]
|
|
||||||
'''
|
|
||||||
|
|
||||||
|
|
||||||
def download_csv() -> pd.DataFrame:
|
|
||||||
for day in (date.today(), date.today() - timedelta(days=1)):
|
|
||||||
url = GIAS_URL.format(date=day.strftime("%Y%m%d"))
|
|
||||||
print(f"Downloading {url}")
|
|
||||||
resp = requests.get(url, timeout=300)
|
|
||||||
if resp.status_code == 404:
|
|
||||||
continue
|
|
||||||
resp.raise_for_status()
|
|
||||||
return pd.read_csv(
|
|
||||||
io.StringIO(resp.content.decode("latin-1")),
|
|
||||||
dtype=str, keep_default_na=False,
|
|
||||||
)
|
|
||||||
sys.exit("GIAS CSV not available for today or yesterday")
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
|
||||||
repo = Path(__file__).resolve().parents[2]
|
|
||||||
df = download_csv()
|
|
||||||
|
|
||||||
module_parts = [MODULE_HEADER]
|
|
||||||
seed_rows: list[tuple[str, int, str]] = []
|
|
||||||
|
|
||||||
for code_col, name_col, dict_name, field_key in FIELDS:
|
|
||||||
pairs = (
|
|
||||||
df[[code_col, name_col]]
|
|
||||||
.loc[lambda d: d[code_col] != ""]
|
|
||||||
.drop_duplicates()
|
|
||||||
)
|
|
||||||
by_code: dict[int, set] = {}
|
|
||||||
for c, n in pairs.itertuples(index=False):
|
|
||||||
by_code.setdefault(int(c), set()).add(n)
|
|
||||||
mapping = []
|
|
||||||
for code, names in sorted(by_code.items()):
|
|
||||||
named = sorted(n for n in names if n != "")
|
|
||||||
if len(named) > 1:
|
|
||||||
sys.exit(f"{code_col}: code {code} maps to multiple names {named} — investigate before generating")
|
|
||||||
# Codes that only ever appear with a blank (name) are GIAS
|
|
||||||
# "not recorded" sentinels (e.g. ReligiousCharacter 99,
|
|
||||||
# AdmissionsPolicy 9). Map them to "" so the API serves the same
|
|
||||||
# empty string the old name pipeline did — the "Unknown (<code>)"
|
|
||||||
# path is reserved for genuinely new codes.
|
|
||||||
mapping.append((code, named[0] if named else ""))
|
|
||||||
lines = [f"{dict_name}: dict[int, str] = {{"]
|
|
||||||
for code, name in mapping:
|
|
||||||
escaped = name.replace('"', '\\"')
|
|
||||||
lines.append(f' {code}: "{escaped}",')
|
|
||||||
lines.append("}\n")
|
|
||||||
module_parts.append("\n".join(lines))
|
|
||||||
seed_rows += [(field_key, code, name) for code, name in mapping]
|
|
||||||
|
|
||||||
module = "\n".join(module_parts) + MODULE_FOOTER
|
|
||||||
|
|
||||||
(repo / "backend" / "gias_codes.py").write_text(module)
|
|
||||||
(repo / "pipeline" / "scripts" / "gias_codes.py").write_text(module)
|
|
||||||
|
|
||||||
seed_path = repo / "pipeline" / "transform" / "seeds" / "gias_code_names.csv"
|
|
||||||
with open(seed_path, "w", newline="") as fh:
|
|
||||||
import csv
|
|
||||||
w = csv.writer(fh)
|
|
||||||
w.writerow(["field", "code", "name"])
|
|
||||||
w.writerows(seed_rows)
|
|
||||||
|
|
||||||
print(f"Wrote backend/gias_codes.py, pipeline/scripts/gias_codes.py, {seed_path.name}")
|
|
||||||
print("\nKey codes for the dbt work (Task 3):")
|
|
||||||
for field in ("establishment_status", "phase_of_education", "official_sixth_form"):
|
|
||||||
print(f" {field}:")
|
|
||||||
for f, code, name in seed_rows:
|
|
||||||
if f == field:
|
|
||||||
print(f" {code} = {name}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
@@ -1,155 +0,0 @@
|
|||||||
"""GIAS code -> name dictionaries.
|
|
||||||
|
|
||||||
GENERATED by pipeline/scripts/generate_gias_codes.py from the GIAS bulk CSV
|
|
||||||
— do not edit by hand; rerun the script when the dbt drift test warns.
|
|
||||||
The canonical file is backend/gias_codes.py; pipeline/scripts/gias_codes.py
|
|
||||||
must be byte-identical (enforced by backend/tests/test_gias_codes.py).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import math
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
SCHOOL_TYPE: dict[int, str] = {
|
|
||||||
1: "Community school",
|
|
||||||
2: "Voluntary aided school",
|
|
||||||
3: "Voluntary controlled school",
|
|
||||||
5: "Foundation school",
|
|
||||||
6: "City technology college",
|
|
||||||
7: "Community special school",
|
|
||||||
8: "Non-maintained special school",
|
|
||||||
10: "Other independent special school",
|
|
||||||
11: "Other independent school",
|
|
||||||
12: "Foundation special school",
|
|
||||||
14: "Pupil referral unit",
|
|
||||||
15: "Local authority nursery school",
|
|
||||||
18: "Further education",
|
|
||||||
24: "Secure units",
|
|
||||||
25: "Offshore schools",
|
|
||||||
26: "Service children's education",
|
|
||||||
27: "Miscellaneous",
|
|
||||||
28: "Academy sponsor led",
|
|
||||||
29: "Higher education institutions",
|
|
||||||
30: "Welsh establishment",
|
|
||||||
31: "Sixth form centres",
|
|
||||||
32: "Special post 16 institution",
|
|
||||||
33: "Academy special sponsor led",
|
|
||||||
34: "Academy converter",
|
|
||||||
35: "Free schools",
|
|
||||||
36: "Free schools special",
|
|
||||||
37: "British schools overseas",
|
|
||||||
38: "Free schools alternative provision",
|
|
||||||
39: "Free schools 16 to 19",
|
|
||||||
40: "University technical college",
|
|
||||||
41: "Studio schools",
|
|
||||||
42: "Academy alternative provision converter",
|
|
||||||
43: "Academy alternative provision sponsor led",
|
|
||||||
44: "Academy special converter",
|
|
||||||
45: "Academy 16-19 converter",
|
|
||||||
46: "Academy 16 to 19 sponsor led",
|
|
||||||
49: "Online provider",
|
|
||||||
56: "Institution funded by other government department",
|
|
||||||
57: "Academy secure 16 to 19",
|
|
||||||
}
|
|
||||||
|
|
||||||
ESTABLISHMENT_STATUS: dict[int, str] = {
|
|
||||||
1: "Open",
|
|
||||||
2: "Closed",
|
|
||||||
3: "Open, but proposed to close",
|
|
||||||
4: "Proposed to open",
|
|
||||||
}
|
|
||||||
|
|
||||||
PHASE_OF_EDUCATION: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
1: "Nursery",
|
|
||||||
2: "Primary",
|
|
||||||
3: "Middle deemed primary",
|
|
||||||
4: "Secondary",
|
|
||||||
5: "Middle deemed secondary",
|
|
||||||
6: "16 plus",
|
|
||||||
7: "All-through",
|
|
||||||
}
|
|
||||||
|
|
||||||
OFFICIAL_SIXTH_FORM: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
1: "Has a sixth form",
|
|
||||||
2: "Does not have a sixth form",
|
|
||||||
9: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
RELIGIOUS_CHARACTER: dict[int, str] = {
|
|
||||||
0: "Does not apply",
|
|
||||||
2: "Church of England",
|
|
||||||
3: "Roman Catholic",
|
|
||||||
4: "Methodist",
|
|
||||||
5: "Jewish",
|
|
||||||
6: "None",
|
|
||||||
7: "Muslim",
|
|
||||||
8: "Seventh Day Adventist",
|
|
||||||
9: "Church of England/Methodist",
|
|
||||||
10: "Methodist/Church of England",
|
|
||||||
11: "Church of England/Roman Catholic",
|
|
||||||
12: "Church of England/United Reformed Church",
|
|
||||||
13: "Roman Catholic/Church of England",
|
|
||||||
14: "Quaker",
|
|
||||||
15: "Christian",
|
|
||||||
16: "United Reformed Church",
|
|
||||||
17: "Congregational Church",
|
|
||||||
18: "Free Church",
|
|
||||||
19: "Church of England/Free Church",
|
|
||||||
20: "Church of England/Christian",
|
|
||||||
21: "Sikh",
|
|
||||||
22: "Greek Orthodox",
|
|
||||||
24: "Buddhist",
|
|
||||||
25: "Hindu",
|
|
||||||
26: "Moravian",
|
|
||||||
28: "Inter- / non- denominational",
|
|
||||||
29: "Multi-faith",
|
|
||||||
30: "Church of England/Methodist/United Reform Church/Baptist",
|
|
||||||
31: "Anglican",
|
|
||||||
32: "Anglican/Christian",
|
|
||||||
33: "Anglican/Evangelical",
|
|
||||||
34: "Anglican/Church of England",
|
|
||||||
35: "Catholic",
|
|
||||||
36: "Charadi Jewish",
|
|
||||||
37: "Christian/Evangelical",
|
|
||||||
38: "Christian Science",
|
|
||||||
39: "Christian/Methodist",
|
|
||||||
40: "Christian/non-denominational",
|
|
||||||
41: "Church of England/Evangelical",
|
|
||||||
42: "Islam",
|
|
||||||
43: "Orthodox Jewish",
|
|
||||||
44: "Plymouth Brethren Christian Church",
|
|
||||||
45: "Protestant",
|
|
||||||
46: "Protestant/Evangelical",
|
|
||||||
47: "Reformed Baptist",
|
|
||||||
48: "Roman Catholic/Anglican",
|
|
||||||
49: "Sunni Deobandi",
|
|
||||||
99: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
ADMISSIONS_POLICY: dict[int, str] = {
|
|
||||||
0: "Not applicable",
|
|
||||||
2: "Selective",
|
|
||||||
4: "Non-selective",
|
|
||||||
9: "",
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def translate(code, mapping: dict[int, str]) -> str | None:
|
|
||||||
"""Translate a GIAS code to its display name.
|
|
||||||
|
|
||||||
None/NaN -> None (column absent or suppressed). Unknown codes degrade to
|
|
||||||
"Unknown (<code>)" with a warning so a new DfE value never blanks the UI.
|
|
||||||
"""
|
|
||||||
if code is None or (isinstance(code, float) and math.isnan(code)):
|
|
||||||
return None
|
|
||||||
code = int(code)
|
|
||||||
if code not in mapping:
|
|
||||||
logger.warning("Unknown GIAS code %s (not in dictionary)", code)
|
|
||||||
return f"Unknown ({code})"
|
|
||||||
return mapping[code]
|
|
||||||
@@ -19,8 +19,6 @@ import psycopg2
|
|||||||
import psycopg2.extras
|
import psycopg2.extras
|
||||||
import typesense
|
import typesense
|
||||||
|
|
||||||
from gias_codes import PHASE_OF_EDUCATION, RELIGIOUS_CHARACTER, SCHOOL_TYPE, translate
|
|
||||||
|
|
||||||
COLLECTION_SCHEMA = {
|
COLLECTION_SCHEMA = {
|
||||||
"fields": [
|
"fields": [
|
||||||
{"name": "urn", "type": "int32"},
|
{"name": "urn", "type": "int32"},
|
||||||
@@ -46,10 +44,10 @@ QUERY_BASE = """
|
|||||||
SELECT
|
SELECT
|
||||||
s.urn,
|
s.urn,
|
||||||
s.school_name,
|
s.school_name,
|
||||||
s.phase_code,
|
s.phase,
|
||||||
s.school_type_code,
|
s.school_type,
|
||||||
l.local_authority_name as local_authority,
|
l.local_authority_name as local_authority,
|
||||||
s.religious_character_code,
|
s.religious_character,
|
||||||
s.ofsted_grade,
|
s.ofsted_grade,
|
||||||
l.postcode,
|
l.postcode,
|
||||||
s.headteacher_name,
|
s.headteacher_name,
|
||||||
@@ -87,15 +85,14 @@ def build_document(row: dict) -> dict:
|
|||||||
"id": str(row["urn"]),
|
"id": str(row["urn"]),
|
||||||
"urn": row["urn"],
|
"urn": row["urn"],
|
||||||
"school_name": row["school_name"] or "",
|
"school_name": row["school_name"] or "",
|
||||||
"phase": translate(row["phase_code"], PHASE_OF_EDUCATION) or "",
|
"phase": row["phase"] or "",
|
||||||
"school_type": translate(row["school_type_code"], SCHOOL_TYPE) or "",
|
"school_type": row["school_type"] or "",
|
||||||
"local_authority": row["local_authority"] or "",
|
"local_authority": row["local_authority"] or "",
|
||||||
"postcode": row["postcode"] or "",
|
"postcode": row["postcode"] or "",
|
||||||
}
|
}
|
||||||
|
|
||||||
religious_character = translate(row.get("religious_character_code"), RELIGIOUS_CHARACTER)
|
if row.get("religious_character"):
|
||||||
if religious_character:
|
doc["religious_character"] = row["religious_character"]
|
||||||
doc["religious_character"] = religious_character
|
|
||||||
if row.get("ofsted_grade"):
|
if row.get("ofsted_grade"):
|
||||||
doc["ofsted_rating"] = OFSTED_LABELS.get(row["ofsted_grade"], "")
|
doc["ofsted_rating"] = OFSTED_LABELS.get(row["ofsted_grade"], "")
|
||||||
if row.get("headteacher_name"):
|
if row.get("headteacher_name"):
|
||||||
|
|||||||
@@ -1,17 +0,0 @@
|
|||||||
-- Macro: Parse Ofsted Report Card grade (post-Nov 2025 framework) from text
|
|
||||||
-- into the 5-point scale. Real values confirmed via a live sample of the MI
|
|
||||||
-- CSV (see pipeline/scripts/diagnose_compare_gaps.py's
|
|
||||||
-- "TASK 7 VALUE SAMPLE 2026-07-12" note) -- unrecognised text (including the
|
|
||||||
-- 'NULL' sentinel used by the source CSV for blanks) parses to NULL, never
|
|
||||||
-- errors.
|
|
||||||
|
|
||||||
{% macro parse_report_card_grade(column_name) %}
|
|
||||||
case lower(trim(nullif({{ column_name }}, 'NULL')))
|
|
||||||
when 'exceptional' then 1
|
|
||||||
when 'strong standard' then 2
|
|
||||||
when 'expected standard' then 3
|
|
||||||
when 'needs attention' then 4
|
|
||||||
when 'urgent improvement' then 5
|
|
||||||
else null
|
|
||||||
end
|
|
||||||
{% endmacro %}
|
|
||||||
@@ -15,11 +15,8 @@ current_ks2 as (
|
|||||||
year, total_pupils, eligible_pupils,
|
year, total_pupils, eligible_pupils,
|
||||||
rwm_expected_pct, rwm_high_pct,
|
rwm_expected_pct, rwm_high_pct,
|
||||||
reading_expected_pct, reading_high_pct, reading_avg_score, reading_progress,
|
reading_expected_pct, reading_high_pct, reading_avg_score, reading_progress,
|
||||||
reading_progress_lower_ci, reading_progress_upper_ci,
|
|
||||||
writing_expected_pct, writing_high_pct, writing_progress,
|
writing_expected_pct, writing_high_pct, writing_progress,
|
||||||
writing_progress_lower_ci, writing_progress_upper_ci, writing_working_towards_pct,
|
|
||||||
maths_expected_pct, maths_high_pct, maths_avg_score, maths_progress,
|
maths_expected_pct, maths_high_pct, maths_avg_score, maths_progress,
|
||||||
maths_progress_lower_ci, maths_progress_upper_ci,
|
|
||||||
gps_expected_pct, gps_high_pct, gps_avg_score, science_expected_pct,
|
gps_expected_pct, gps_high_pct, gps_avg_score, science_expected_pct,
|
||||||
reading_absence_pct, writing_absence_pct, maths_absence_pct, gps_absence_pct, science_absence_pct,
|
reading_absence_pct, writing_absence_pct, maths_absence_pct, gps_absence_pct, science_absence_pct,
|
||||||
rwm_expected_boys_pct, rwm_high_boys_pct, rwm_expected_girls_pct, rwm_high_girls_pct,
|
rwm_expected_boys_pct, rwm_high_boys_pct, rwm_expected_girls_pct, rwm_high_girls_pct,
|
||||||
@@ -36,11 +33,8 @@ predecessor_ks2 as (
|
|||||||
ks2.year, ks2.total_pupils, ks2.eligible_pupils,
|
ks2.year, ks2.total_pupils, ks2.eligible_pupils,
|
||||||
ks2.rwm_expected_pct, ks2.rwm_high_pct,
|
ks2.rwm_expected_pct, ks2.rwm_high_pct,
|
||||||
ks2.reading_expected_pct, ks2.reading_high_pct, ks2.reading_avg_score, ks2.reading_progress,
|
ks2.reading_expected_pct, ks2.reading_high_pct, ks2.reading_avg_score, ks2.reading_progress,
|
||||||
ks2.reading_progress_lower_ci, ks2.reading_progress_upper_ci,
|
|
||||||
ks2.writing_expected_pct, ks2.writing_high_pct, ks2.writing_progress,
|
ks2.writing_expected_pct, ks2.writing_high_pct, ks2.writing_progress,
|
||||||
ks2.writing_progress_lower_ci, ks2.writing_progress_upper_ci, ks2.writing_working_towards_pct,
|
|
||||||
ks2.maths_expected_pct, ks2.maths_high_pct, ks2.maths_avg_score, ks2.maths_progress,
|
ks2.maths_expected_pct, ks2.maths_high_pct, ks2.maths_avg_score, ks2.maths_progress,
|
||||||
ks2.maths_progress_lower_ci, ks2.maths_progress_upper_ci,
|
|
||||||
ks2.gps_expected_pct, ks2.gps_high_pct, ks2.gps_avg_score, ks2.science_expected_pct,
|
ks2.gps_expected_pct, ks2.gps_high_pct, ks2.gps_avg_score, ks2.science_expected_pct,
|
||||||
ks2.reading_absence_pct, ks2.writing_absence_pct, ks2.maths_absence_pct, ks2.gps_absence_pct, ks2.science_absence_pct,
|
ks2.reading_absence_pct, ks2.writing_absence_pct, ks2.maths_absence_pct, ks2.gps_absence_pct, ks2.science_absence_pct,
|
||||||
ks2.rwm_expected_boys_pct, ks2.rwm_high_boys_pct, ks2.rwm_expected_girls_pct, ks2.rwm_high_girls_pct,
|
ks2.rwm_expected_boys_pct, ks2.rwm_high_boys_pct, ks2.rwm_expected_girls_pct, ks2.rwm_high_girls_pct,
|
||||||
|
|||||||
@@ -18,8 +18,7 @@ current_ks4 as (
|
|||||||
english_maths_strong_pass_pct, english_maths_standard_pass_pct,
|
english_maths_strong_pass_pct, english_maths_standard_pass_pct,
|
||||||
ebacc_entry_pct, ebacc_strong_pass_pct, ebacc_standard_pass_pct, ebacc_avg_score,
|
ebacc_entry_pct, ebacc_strong_pass_pct, ebacc_standard_pass_pct, ebacc_avg_score,
|
||||||
gcse_grade_91_pct,
|
gcse_grade_91_pct,
|
||||||
sen_pct, sen_support_pct, sen_ehcp_pct,
|
sen_pct, sen_support_pct, sen_ehcp_pct
|
||||||
progress_8_banding, attainment_8_disadvantage_gap, progress_8_disadvantage_gap
|
|
||||||
from all_ks4
|
from all_ks4
|
||||||
),
|
),
|
||||||
|
|
||||||
@@ -35,8 +34,7 @@ predecessor_ks4 as (
|
|||||||
ks4.english_maths_strong_pass_pct, ks4.english_maths_standard_pass_pct,
|
ks4.english_maths_strong_pass_pct, ks4.english_maths_standard_pass_pct,
|
||||||
ks4.ebacc_entry_pct, ks4.ebacc_strong_pass_pct, ks4.ebacc_standard_pass_pct, ks4.ebacc_avg_score,
|
ks4.ebacc_entry_pct, ks4.ebacc_strong_pass_pct, ks4.ebacc_standard_pass_pct, ks4.ebacc_avg_score,
|
||||||
ks4.gcse_grade_91_pct,
|
ks4.gcse_grade_91_pct,
|
||||||
ks4.sen_pct, ks4.sen_support_pct, ks4.sen_ehcp_pct,
|
ks4.sen_pct, ks4.sen_support_pct, ks4.sen_ehcp_pct
|
||||||
ks4.progress_8_banding, ks4.attainment_8_disadvantage_gap, ks4.progress_8_disadvantage_gap
|
|
||||||
from all_ks4 ks4
|
from all_ks4 ks4
|
||||||
inner join {{ ref('int_school_lineage') }} lin
|
inner join {{ ref('int_school_lineage') }} lin
|
||||||
on ks4.urn = lin.predecessor_urn
|
on ks4.urn = lin.predecessor_urn
|
||||||
|
|||||||
@@ -8,46 +8,18 @@ models:
|
|||||||
tests: [not_null, unique]
|
tests: [not_null, unique]
|
||||||
- name: school_name
|
- name: school_name
|
||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: phase_code
|
- name: phase
|
||||||
description: >
|
description: >
|
||||||
GIAS PhaseOfEducation code (2 = Primary, 4 = Secondary, 7 = All-through,
|
Primary / Secondary / All-through etc. May be null for a small number
|
||||||
etc. — see seeds/gias_code_names.csv). May be null for a small number
|
|
||||||
of independent schools where GIAS publishes "Not Applicable", no
|
of independent schools where GIAS publishes "Not Applicable", no
|
||||||
statutory age range, and the school name gives no hint.
|
statutory age range, and the school name gives no hint.
|
||||||
tests:
|
tests:
|
||||||
- not_null:
|
- not_null:
|
||||||
severity: warn
|
severity: warn
|
||||||
- name: has_sixth_form
|
- name: status
|
||||||
description: >
|
|
||||||
Authoritative sixth-form flag from GIAS OfficialSixthForm.
|
|
||||||
"Has a sixth form" => true; "Does not have a sixth form" and
|
|
||||||
"Not applicable" => false; blank GIAS value falls back to
|
|
||||||
statutory_high_age >= 18. Replaces the age_range-contains-"18"
|
|
||||||
heuristic (spec 2026-07-07 §3).
|
|
||||||
tests:
|
|
||||||
- not_null
|
|
||||||
- accepted_values:
|
|
||||||
values: [true, false]
|
|
||||||
- name: status_code
|
|
||||||
description: GIAS EstablishmentStatus code (1 = Open, 3 = Open but proposed to close)
|
|
||||||
tests:
|
tests:
|
||||||
- accepted_values:
|
- accepted_values:
|
||||||
values: [1, 3]
|
values: ["Open"]
|
||||||
- name: school_type_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [1, 2, 3, 5, 6, 7, 8, 10, 11, 12, 14, 15, 18, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 49, 56, 57]
|
|
||||||
- name: religious_character_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 99]
|
|
||||||
- name: admissions_policy_code
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
severity: warn
|
|
||||||
values: [0, 2, 4, 9]
|
|
||||||
|
|
||||||
- name: dim_location
|
- name: dim_location
|
||||||
description: School location dimension with PostGIS geometry
|
description: School location dimension with PostGIS geometry
|
||||||
@@ -86,13 +58,6 @@ models:
|
|||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: year
|
- name: year
|
||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: reading_progress_lower_ci
|
|
||||||
- name: reading_progress_upper_ci
|
|
||||||
- name: writing_progress_lower_ci
|
|
||||||
- name: writing_progress_upper_ci
|
|
||||||
- name: writing_working_towards_pct
|
|
||||||
- name: maths_progress_lower_ci
|
|
||||||
- name: maths_progress_upper_ci
|
|
||||||
tests:
|
tests:
|
||||||
- unique:
|
- unique:
|
||||||
column_name: "urn || '-' || year"
|
column_name: "urn || '-' || year"
|
||||||
@@ -104,15 +69,6 @@ models:
|
|||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: year
|
- name: year
|
||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: progress_8_banding
|
|
||||||
tests:
|
|
||||||
- accepted_values:
|
|
||||||
values: ['Well above average', 'Above average', 'Average', 'Below average', 'Well below average']
|
|
||||||
config:
|
|
||||||
where: "progress_8_banding is not null"
|
|
||||||
severity: warn
|
|
||||||
- name: attainment_8_disadvantage_gap
|
|
||||||
- name: progress_8_disadvantage_gap
|
|
||||||
tests:
|
tests:
|
||||||
- unique:
|
- unique:
|
||||||
column_name: "urn || '-' || year"
|
column_name: "urn || '-' || year"
|
||||||
@@ -140,11 +96,6 @@ models:
|
|||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: year
|
- name: year
|
||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
- name: second_preference_offers
|
|
||||||
- name: third_preference_offers
|
|
||||||
- name: cross_la_applications
|
|
||||||
- name: cross_la_offers
|
|
||||||
- name: total_offers
|
|
||||||
|
|
||||||
- name: fact_finance
|
- name: fact_finance
|
||||||
description: School financial data — one row per URN per year
|
description: School financial data — one row per URN per year
|
||||||
@@ -154,6 +105,12 @@ models:
|
|||||||
- name: year
|
- name: year
|
||||||
tests: [not_null]
|
tests: [not_null]
|
||||||
|
|
||||||
|
- name: fact_parent_view
|
||||||
|
description: Parent View survey responses
|
||||||
|
columns:
|
||||||
|
- name: urn
|
||||||
|
tests: [not_null]
|
||||||
|
|
||||||
- name: fact_ks2_national_averages
|
- name: fact_ks2_national_averages
|
||||||
description: Official DfE KS2 national headline averages — one row per academic year
|
description: Official DfE KS2 national headline averages — one row per academic year
|
||||||
columns:
|
columns:
|
||||||
|
|||||||
@@ -31,5 +31,4 @@ select
|
|||||||
else null
|
else null
|
||||||
end as longitude
|
end as longitude
|
||||||
from {{ ref('stg_gias_establishments') }} s
|
from {{ ref('stg_gias_establishments') }} s
|
||||||
-- Must match dim_school's status filter exactly (the API inner-joins the two).
|
where s.status = 'Open'
|
||||||
where s.status_code in (1, 3)
|
|
||||||
|
|||||||
@@ -19,17 +19,16 @@ select
|
|||||||
s.urn,
|
s.urn,
|
||||||
s.local_authority_code * 1000 + s.establishment_number as laestab,
|
s.local_authority_code * 1000 + s.establishment_number as laestab,
|
||||||
s.school_name,
|
s.school_name,
|
||||||
-- Phase in GIAS code space (see seeds/gias_code_names.csv):
|
|
||||||
-- 2 = Primary, 4 = Secondary, 7 = All-through, 0 = Not applicable.
|
|
||||||
case
|
case
|
||||||
-- 1. Trust GIAS phase when it's a real value (0 = the catch-all "Not Applicable")
|
-- 1. Trust GIAS phase when it's a real value (not the catch-all "Not Applicable")
|
||||||
when s.phase_code is not null and s.phase_code != 0
|
when s.phase is not null
|
||||||
then s.phase_code
|
and lower(trim(s.phase)) not in ('not applicable', '', 'unknown')
|
||||||
|
then s.phase
|
||||||
-- 2. Infer from statutory age range (independent schools still publish these)
|
-- 2. Infer from statutory age range (independent schools still publish these)
|
||||||
when s.statutory_high_age is not null and s.statutory_high_age <= 11 then 2
|
when s.statutory_high_age is not null and s.statutory_high_age <= 11 then 'Primary'
|
||||||
when s.statutory_low_age is not null and s.statutory_low_age >= 11 then 4
|
when s.statutory_low_age is not null and s.statutory_low_age >= 11 then 'Secondary'
|
||||||
when s.statutory_low_age is not null and s.statutory_high_age is not null
|
when s.statutory_low_age is not null and s.statutory_high_age is not null
|
||||||
and s.statutory_low_age < 11 and s.statutory_high_age > 11 then 7
|
and s.statutory_low_age < 11 and s.statutory_high_age > 11 then 'All-through'
|
||||||
-- 3. Fallback: infer from school name (covers independents with missing ages)
|
-- 3. Fallback: infer from school name (covers independents with missing ages)
|
||||||
when s.school_name ilike '%primary%'
|
when s.school_name ilike '%primary%'
|
||||||
or s.school_name ilike '%infant%'
|
or s.school_name ilike '%infant%'
|
||||||
@@ -37,29 +36,22 @@ select
|
|||||||
or s.school_name ilike '%preparatory%'
|
or s.school_name ilike '%preparatory%'
|
||||||
or s.school_name ilike '% prep school%'
|
or s.school_name ilike '% prep school%'
|
||||||
or s.school_name ilike '% prep %'
|
or s.school_name ilike '% prep %'
|
||||||
then 2
|
then 'Primary'
|
||||||
when s.school_name ilike '%secondary%'
|
when s.school_name ilike '%secondary%'
|
||||||
or s.school_name ilike '%high school%'
|
or s.school_name ilike '%high school%'
|
||||||
or s.school_name ilike '%grammar%'
|
or s.school_name ilike '%grammar%'
|
||||||
or s.school_name ilike '%senior school%'
|
or s.school_name ilike '%senior school%'
|
||||||
or s.school_name ilike '%upper school%'
|
or s.school_name ilike '%upper school%'
|
||||||
then 4
|
then 'Secondary'
|
||||||
-- 4. Give up — null renders no phase pill
|
-- 4. Give up — leave phase null so the UI renders no pill
|
||||||
else null
|
else null
|
||||||
end as phase_code,
|
end as phase,
|
||||||
s.school_type_code,
|
s.school_type,
|
||||||
s.academy_trust_name,
|
s.academy_trust_name,
|
||||||
s.academy_trust_uid,
|
s.academy_trust_uid,
|
||||||
s.religious_character_code,
|
s.religious_character,
|
||||||
s.gender,
|
s.gender,
|
||||||
s.statutory_low_age || '-' || s.statutory_high_age as age_range,
|
s.statutory_low_age || '-' || s.statutory_high_age as age_range,
|
||||||
-- GIAS OfficialSixthForm in code space: 1 = has, 2 = does not, 0 = N/A.
|
|
||||||
-- Null (rare, new establishments) falls back to the statutory age range.
|
|
||||||
case
|
|
||||||
when s.official_sixth_form_code = 1 then true
|
|
||||||
when s.official_sixth_form_code in (0, 2) then false
|
|
||||||
else coalesce(s.statutory_high_age >= 18, false)
|
|
||||||
end as has_sixth_form,
|
|
||||||
s.capacity,
|
s.capacity,
|
||||||
s.total_pupils,
|
s.total_pupils,
|
||||||
concat_ws(' ', s.head_title, s.head_first_name, s.head_last_name) as headteacher_name,
|
concat_ws(' ', s.head_title, s.head_first_name, s.head_last_name) as headteacher_name,
|
||||||
@@ -67,9 +59,9 @@ select
|
|||||||
s.telephone,
|
s.telephone,
|
||||||
s.open_date,
|
s.open_date,
|
||||||
s.close_date,
|
s.close_date,
|
||||||
s.status_code,
|
s.status,
|
||||||
s.nursery_provision,
|
s.nursery_provision,
|
||||||
s.admissions_policy_code,
|
s.admissions_policy,
|
||||||
|
|
||||||
-- Latest Ofsted (populated after monthly Ofsted pipeline runs)
|
-- Latest Ofsted (populated after monthly Ofsted pipeline runs)
|
||||||
{% if ofsted_relation is not none %}
|
{% if ofsted_relation is not none %}
|
||||||
@@ -88,6 +80,4 @@ from schools s
|
|||||||
{% if ofsted_relation is not none %}
|
{% if ofsted_relation is not none %}
|
||||||
left join {{ ref('int_ofsted_latest') }} o on s.urn = o.urn
|
left join {{ ref('int_ofsted_latest') }} o on s.urn = o.urn
|
||||||
{% endif %}
|
{% endif %}
|
||||||
-- 1 = Open; 3 = Open, but proposed to close (still operating; drops out when
|
where s.status = 'Open'
|
||||||
-- GIAS flips to Closed — marts fully rebuild each run).
|
|
||||||
where s.status_code in (1, 3)
|
|
||||||
|
|||||||
@@ -5,14 +5,9 @@ select
|
|||||||
year,
|
year,
|
||||||
school_phase,
|
school_phase,
|
||||||
places_offered,
|
places_offered,
|
||||||
total_offers,
|
|
||||||
total_applications,
|
total_applications,
|
||||||
first_preference_applications,
|
first_preference_applications,
|
||||||
first_preference_offers,
|
first_preference_offers,
|
||||||
second_preference_offers,
|
|
||||||
third_preference_offers,
|
|
||||||
cross_la_applications,
|
|
||||||
cross_la_offers,
|
|
||||||
first_preference_offer_pct,
|
first_preference_offer_pct,
|
||||||
oversubscription_ratio,
|
oversubscription_ratio,
|
||||||
oversubscribed,
|
oversubscribed,
|
||||||
|
|||||||
@@ -15,20 +15,13 @@ select
|
|||||||
reading_high_pct,
|
reading_high_pct,
|
||||||
reading_avg_score,
|
reading_avg_score,
|
||||||
reading_progress,
|
reading_progress,
|
||||||
reading_progress_lower_ci,
|
|
||||||
reading_progress_upper_ci,
|
|
||||||
writing_expected_pct,
|
writing_expected_pct,
|
||||||
writing_high_pct,
|
writing_high_pct,
|
||||||
writing_progress,
|
writing_progress,
|
||||||
writing_progress_lower_ci,
|
|
||||||
writing_progress_upper_ci,
|
|
||||||
writing_working_towards_pct,
|
|
||||||
maths_expected_pct,
|
maths_expected_pct,
|
||||||
maths_high_pct,
|
maths_high_pct,
|
||||||
maths_avg_score,
|
maths_avg_score,
|
||||||
maths_progress,
|
maths_progress,
|
||||||
maths_progress_lower_ci,
|
|
||||||
maths_progress_upper_ci,
|
|
||||||
gps_expected_pct,
|
gps_expected_pct,
|
||||||
gps_high_pct,
|
gps_high_pct,
|
||||||
gps_avg_score,
|
gps_avg_score,
|
||||||
|
|||||||
@@ -16,9 +16,6 @@ select
|
|||||||
progress_8_score,
|
progress_8_score,
|
||||||
progress_8_lower_ci,
|
progress_8_lower_ci,
|
||||||
progress_8_upper_ci,
|
progress_8_upper_ci,
|
||||||
progress_8_banding,
|
|
||||||
attainment_8_disadvantage_gap,
|
|
||||||
progress_8_disadvantage_gap,
|
|
||||||
progress_8_english,
|
progress_8_english,
|
||||||
progress_8_maths,
|
progress_8_maths,
|
||||||
progress_8_ebacc,
|
progress_8_ebacc,
|
||||||
|
|||||||
@@ -0,0 +1,20 @@
|
|||||||
|
-- Mart: Parent View survey responses — one row per URN (latest survey)
|
||||||
|
|
||||||
|
select
|
||||||
|
urn,
|
||||||
|
survey_date,
|
||||||
|
total_responses,
|
||||||
|
q_happy_pct,
|
||||||
|
q_safe_pct,
|
||||||
|
q_behaviour_pct,
|
||||||
|
q_bullying_pct,
|
||||||
|
q_communication_pct,
|
||||||
|
q_progress_pct,
|
||||||
|
q_teaching_pct,
|
||||||
|
q_information_pct,
|
||||||
|
q_curriculum_pct,
|
||||||
|
q_future_pct,
|
||||||
|
q_leadership_pct,
|
||||||
|
q_wellbeing_pct,
|
||||||
|
q_recommend_pct
|
||||||
|
from {{ ref('stg_parent_view') }}
|
||||||
@@ -25,20 +25,13 @@ select
|
|||||||
ks2.reading_high_pct,
|
ks2.reading_high_pct,
|
||||||
ks2.reading_avg_score,
|
ks2.reading_avg_score,
|
||||||
ks2.reading_progress,
|
ks2.reading_progress,
|
||||||
ks2.reading_progress_lower_ci,
|
|
||||||
ks2.reading_progress_upper_ci,
|
|
||||||
ks2.writing_expected_pct,
|
ks2.writing_expected_pct,
|
||||||
ks2.writing_high_pct,
|
ks2.writing_high_pct,
|
||||||
ks2.writing_progress,
|
ks2.writing_progress,
|
||||||
ks2.writing_progress_lower_ci,
|
|
||||||
ks2.writing_progress_upper_ci,
|
|
||||||
ks2.writing_working_towards_pct,
|
|
||||||
ks2.maths_expected_pct,
|
ks2.maths_expected_pct,
|
||||||
ks2.maths_high_pct,
|
ks2.maths_high_pct,
|
||||||
ks2.maths_avg_score,
|
ks2.maths_avg_score,
|
||||||
ks2.maths_progress,
|
ks2.maths_progress,
|
||||||
ks2.maths_progress_lower_ci,
|
|
||||||
ks2.maths_progress_upper_ci,
|
|
||||||
ks2.gps_expected_pct,
|
ks2.gps_expected_pct,
|
||||||
ks2.gps_high_pct,
|
ks2.gps_high_pct,
|
||||||
ks2.gps_avg_score,
|
ks2.gps_avg_score,
|
||||||
@@ -68,9 +61,6 @@ select
|
|||||||
ks4.progress_8_maths,
|
ks4.progress_8_maths,
|
||||||
ks4.progress_8_ebacc,
|
ks4.progress_8_ebacc,
|
||||||
ks4.progress_8_open,
|
ks4.progress_8_open,
|
||||||
ks4.progress_8_banding,
|
|
||||||
ks4.attainment_8_disadvantage_gap,
|
|
||||||
ks4.progress_8_disadvantage_gap,
|
|
||||||
ks4.english_maths_strong_pass_pct,
|
ks4.english_maths_strong_pass_pct,
|
||||||
ks4.english_maths_standard_pass_pct,
|
ks4.english_maths_standard_pass_pct,
|
||||||
ks4.ebacc_entry_pct,
|
ks4.ebacc_entry_pct,
|
||||||
|
|||||||
@@ -53,6 +53,9 @@ sources:
|
|||||||
|
|
||||||
# Phonics: no school-level data on EES (only national/LA level)
|
# Phonics: no school-level data on EES (only national/LA level)
|
||||||
|
|
||||||
|
- name: parent_view
|
||||||
|
description: Ofsted Parent View survey responses
|
||||||
|
|
||||||
- name: fbit_finance
|
- name: fbit_finance
|
||||||
description: Financial benchmarking data from FBIT API
|
description: Financial benchmarking data from FBIT API
|
||||||
|
|
||||||
|
|||||||
@@ -32,11 +32,6 @@ renamed as (
|
|||||||
{{ safe_numeric('times_put_as_any_preferred_school') }}::integer as total_applications,
|
{{ safe_numeric('times_put_as_any_preferred_school') }}::integer as total_applications,
|
||||||
{{ safe_numeric('times_put_as_1st_preference') }}::integer as first_preference_applications,
|
{{ safe_numeric('times_put_as_1st_preference') }}::integer as first_preference_applications,
|
||||||
|
|
||||||
-- Cross-borough demand: applications naming this school from families
|
|
||||||
-- living in another local authority, and offers made to them.
|
|
||||||
{{ safe_numeric('"all_applications_from_another_LA"') }}::integer as cross_la_applications,
|
|
||||||
{{ safe_numeric('"offers_to_applicants_from_another_LA"') }}::integer as cross_la_offers,
|
|
||||||
|
|
||||||
-- Proportions
|
-- Proportions
|
||||||
-- first_preference_offer_pct: of families who listed this school FIRST,
|
-- first_preference_offer_pct: of families who listed this school FIRST,
|
||||||
-- the percentage that received an offer. 0–100 scale.
|
-- the percentage that received an offer. 0–100 scale.
|
||||||
|
|||||||
@@ -39,12 +39,6 @@ pivoted as (
|
|||||||
max(case when subject = 'Reading'
|
max(case when subject = 'Reading'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('progress_measure_score') }} end) as reading_progress,
|
then {{ safe_numeric('progress_measure_score') }} end) as reading_progress,
|
||||||
max(case when subject = 'Reading'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as reading_progress_lower_ci,
|
|
||||||
max(case when subject = 'Reading'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as reading_progress_upper_ci,
|
|
||||||
max(case when subject = 'Reading'
|
max(case when subject = 'Reading'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as reading_absence_pct,
|
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as reading_absence_pct,
|
||||||
@@ -59,15 +53,6 @@ pivoted as (
|
|||||||
max(case when subject = 'Writing'
|
max(case when subject = 'Writing'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('progress_measure_score') }} end) as writing_progress,
|
then {{ safe_numeric('progress_measure_score') }} end) as writing_progress,
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as writing_progress_lower_ci,
|
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as writing_progress_upper_ci,
|
|
||||||
max(case when subject = 'Writing'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('working_towards_expected_standard_pupil_percent') }} end) as writing_working_towards_pct,
|
|
||||||
max(case when subject = 'Writing'
|
max(case when subject = 'Writing'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as writing_absence_pct,
|
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as writing_absence_pct,
|
||||||
@@ -85,12 +70,6 @@ pivoted as (
|
|||||||
max(case when subject = 'Maths'
|
max(case when subject = 'Maths'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('progress_measure_score') }} end) as maths_progress,
|
then {{ safe_numeric('progress_measure_score') }} end) as maths_progress,
|
||||||
max(case when subject = 'Maths'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_lower_conf_interval') }} end) as maths_progress_lower_ci,
|
|
||||||
max(case when subject = 'Maths'
|
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
|
||||||
then {{ safe_numeric('progress_measure_upper_conf_interval') }} end) as maths_progress_upper_ci,
|
|
||||||
max(case when subject = 'Maths'
|
max(case when subject = 'Maths'
|
||||||
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
and breakdown_topic = 'All pupils' and breakdown = 'Total'
|
||||||
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as maths_absence_pct,
|
then {{ safe_numeric('absent_or_not_able_to_access_percent') }} end) as maths_absence_pct,
|
||||||
@@ -164,20 +143,13 @@ select
|
|||||||
p.reading_high_pct,
|
p.reading_high_pct,
|
||||||
p.reading_avg_score,
|
p.reading_avg_score,
|
||||||
p.reading_progress,
|
p.reading_progress,
|
||||||
p.reading_progress_lower_ci,
|
|
||||||
p.reading_progress_upper_ci,
|
|
||||||
p.writing_expected_pct,
|
p.writing_expected_pct,
|
||||||
p.writing_high_pct,
|
p.writing_high_pct,
|
||||||
p.writing_progress,
|
p.writing_progress,
|
||||||
p.writing_progress_lower_ci,
|
|
||||||
p.writing_progress_upper_ci,
|
|
||||||
p.writing_working_towards_pct,
|
|
||||||
p.maths_expected_pct,
|
p.maths_expected_pct,
|
||||||
p.maths_high_pct,
|
p.maths_high_pct,
|
||||||
p.maths_avg_score,
|
p.maths_avg_score,
|
||||||
p.maths_progress,
|
p.maths_progress,
|
||||||
p.maths_progress_lower_ci,
|
|
||||||
p.maths_progress_upper_ci,
|
|
||||||
p.gps_expected_pct,
|
p.gps_expected_pct,
|
||||||
p.gps_high_pct,
|
p.gps_high_pct,
|
||||||
p.gps_avg_score,
|
p.gps_avg_score,
|
||||||
|
|||||||
@@ -31,10 +31,4 @@ select
|
|||||||
|
|
||||||
from {{ source('raw', 'ees_ks2_national') }}
|
from {{ source('raw', 'ees_ks2_national') }}
|
||||||
where time_period ~ '^[0-9]+$'
|
where time_period ~ '^[0-9]+$'
|
||||||
-- 2015/16 was the first year of the current expected-standard tests, so it's
|
and cast(trim(time_period) as integer) >= 201617
|
||||||
-- the correct floor (not 2016/17 -- that excluded a real, comparable national
|
|
||||||
-- row). GPS/science/scaled-score columns are already mapped correctly end to
|
|
||||||
-- end (tap.py's _KS2_NATIONAL_COL_MAP + this model select them fine); the
|
|
||||||
-- prod NULLs for those fields are stale raw.ees_ks2_national data from before
|
|
||||||
-- the map covered them, not a mapping bug -- no map change accompanies this fix.
|
|
||||||
and cast(trim(time_period) as integer) >= 201516
|
|
||||||
|
|||||||
@@ -62,16 +62,7 @@ info as (
|
|||||||
{{ safe_numeric('ks2_scaledscore_average') }} as prior_attainment_avg,
|
{{ safe_numeric('ks2_scaledscore_average') }} as prior_attainment_avg,
|
||||||
{{ safe_numeric('sen_pupil_percent') }} as sen_pct,
|
{{ safe_numeric('sen_pupil_percent') }} as sen_pct,
|
||||||
{{ safe_numeric('sen_with_ehcp_pupil_percent') }} as sen_ehcp_pct,
|
{{ safe_numeric('sen_with_ehcp_pupil_percent') }} as sen_ehcp_pct,
|
||||||
{{ safe_numeric('sen_no_ehcp_pupil_percent') }} as sen_support_pct,
|
{{ safe_numeric('sen_no_ehcp_pupil_percent') }} as sen_support_pct
|
||||||
-- EES suppression sentinels (z/c/x/q/u) and blanks must not reach the
|
|
||||||
-- mart as banding labels
|
|
||||||
case
|
|
||||||
when lower(trim(progress8_banding)) in ('', 'z', 'c', 'x', 'q', 'u', 'null')
|
|
||||||
then null
|
|
||||||
else trim(progress8_banding)
|
|
||||||
end as progress_8_banding,
|
|
||||||
{{ safe_numeric('attainment8_diffn') }} as attainment_8_disadvantage_gap,
|
|
||||||
{{ safe_numeric('progress8_diffn') }} as progress_8_disadvantage_gap
|
|
||||||
from {{ source('raw', 'ees_ks4_info') }}
|
from {{ source('raw', 'ees_ks4_info') }}
|
||||||
where school_urn is not null
|
where school_urn is not null
|
||||||
)
|
)
|
||||||
@@ -111,10 +102,7 @@ select
|
|||||||
-- Context
|
-- Context
|
||||||
i.sen_pct,
|
i.sen_pct,
|
||||||
i.sen_ehcp_pct,
|
i.sen_ehcp_pct,
|
||||||
i.sen_support_pct,
|
i.sen_support_pct
|
||||||
i.progress_8_banding,
|
|
||||||
i.attainment_8_disadvantage_gap,
|
|
||||||
i.progress_8_disadvantage_gap
|
|
||||||
|
|
||||||
from all_pupils p
|
from all_pupils p
|
||||||
left join info i on p.urn = i.urn and p.year = i.year
|
left join info i on p.urn = i.urn and p.year = i.year
|
||||||
|
|||||||
@@ -12,12 +12,11 @@ renamed as (
|
|||||||
"LA (name)" as local_authority_name,
|
"LA (name)" as local_authority_name,
|
||||||
cast(nullif("EstablishmentNumber", '') as integer) as establishment_number,
|
cast(nullif("EstablishmentNumber", '') as integer) as establishment_number,
|
||||||
"EstablishmentName" as school_name,
|
"EstablishmentName" as school_name,
|
||||||
cast(nullif(trim("TypeOfEstablishment (code)"), '') as integer) as school_type_code,
|
"TypeOfEstablishment (name)" as school_type,
|
||||||
cast(nullif(trim("PhaseOfEducation (code)"), '') as integer) as phase_code,
|
"PhaseOfEducation (name)" as phase,
|
||||||
cast(nullif(trim("OfficialSixthForm (code)"), '') as integer) as official_sixth_form_code,
|
|
||||||
"Gender (name)" as gender,
|
"Gender (name)" as gender,
|
||||||
cast(nullif(trim("ReligiousCharacter (code)"), '') as integer) as religious_character_code,
|
"ReligiousCharacter (name)" as religious_character,
|
||||||
cast(nullif(trim("AdmissionsPolicy (code)"), '') as integer) as admissions_policy_code,
|
"AdmissionsPolicy (name)" as admissions_policy,
|
||||||
"SchoolCapacity" as capacity,
|
"SchoolCapacity" as capacity,
|
||||||
cast(nullif("NumberOfPupils", '') as integer) as total_pupils,
|
cast(nullif("NumberOfPupils", '') as integer) as total_pupils,
|
||||||
"HeadTitle (name)" as head_title,
|
"HeadTitle (name)" as head_title,
|
||||||
@@ -30,7 +29,7 @@ renamed as (
|
|||||||
"Town" as town,
|
"Town" as town,
|
||||||
"County (name)" as county,
|
"County (name)" as county,
|
||||||
"Postcode" as postcode,
|
"Postcode" as postcode,
|
||||||
cast(nullif(trim("EstablishmentStatus (code)"), '') as integer) as status_code,
|
"EstablishmentStatus (name)" as status,
|
||||||
case when "OpenDate" = '' then null else to_date("OpenDate", 'DD-MM-YYYY') end as open_date,
|
case when "OpenDate" = '' then null else to_date("OpenDate", 'DD-MM-YYYY') end as open_date,
|
||||||
case when "CloseDate" = '' then null else to_date("CloseDate", 'DD-MM-YYYY') end as close_date,
|
case when "CloseDate" = '' then null else to_date("CloseDate", 'DD-MM-YYYY') end as close_date,
|
||||||
"Trusts (name)" as academy_trust_name,
|
"Trusts (name)" as academy_trust_name,
|
||||||
|
|||||||
@@ -17,23 +17,13 @@ select
|
|||||||
{{ safe_numeric('reading_high_pct') }} as reading_high_pct,
|
{{ safe_numeric('reading_high_pct') }} as reading_high_pct,
|
||||||
{{ safe_numeric('reading_avg_score') }} as reading_avg_score,
|
{{ safe_numeric('reading_avg_score') }} as reading_avg_score,
|
||||||
{{ safe_numeric('reading_progress') }} as reading_progress,
|
{{ safe_numeric('reading_progress') }} as reading_progress,
|
||||||
-- Progress CIs / working-towards: not published in the legacy CSVs.
|
|
||||||
-- Typed placeholders keep positional alignment with stg_ees_ks2 in
|
|
||||||
-- int_ks2_with_lineage's UNION ALL.
|
|
||||||
null::numeric as reading_progress_lower_ci,
|
|
||||||
null::numeric as reading_progress_upper_ci,
|
|
||||||
{{ safe_numeric('writing_expected_pct') }} as writing_expected_pct,
|
{{ safe_numeric('writing_expected_pct') }} as writing_expected_pct,
|
||||||
{{ safe_numeric('writing_high_pct') }} as writing_high_pct,
|
{{ safe_numeric('writing_high_pct') }} as writing_high_pct,
|
||||||
{{ safe_numeric('writing_progress') }} as writing_progress,
|
{{ safe_numeric('writing_progress') }} as writing_progress,
|
||||||
null::numeric as writing_progress_lower_ci,
|
|
||||||
null::numeric as writing_progress_upper_ci,
|
|
||||||
null::numeric as writing_working_towards_pct,
|
|
||||||
{{ safe_numeric('maths_expected_pct') }} as maths_expected_pct,
|
{{ safe_numeric('maths_expected_pct') }} as maths_expected_pct,
|
||||||
{{ safe_numeric('maths_high_pct') }} as maths_high_pct,
|
{{ safe_numeric('maths_high_pct') }} as maths_high_pct,
|
||||||
{{ safe_numeric('maths_avg_score') }} as maths_avg_score,
|
{{ safe_numeric('maths_avg_score') }} as maths_avg_score,
|
||||||
{{ safe_numeric('maths_progress') }} as maths_progress,
|
{{ safe_numeric('maths_progress') }} as maths_progress,
|
||||||
null::numeric as maths_progress_lower_ci,
|
|
||||||
null::numeric as maths_progress_upper_ci,
|
|
||||||
{{ safe_numeric('gps_expected_pct') }} as gps_expected_pct,
|
{{ safe_numeric('gps_expected_pct') }} as gps_expected_pct,
|
||||||
{{ safe_numeric('gps_high_pct') }} as gps_high_pct,
|
{{ safe_numeric('gps_high_pct') }} as gps_high_pct,
|
||||||
{{ safe_numeric('gps_avg_score') }} as gps_avg_score,
|
{{ safe_numeric('gps_avg_score') }} as gps_avg_score,
|
||||||
|
|||||||
@@ -41,13 +41,8 @@ select
|
|||||||
|
|
||||||
-- SEN
|
-- SEN
|
||||||
null::numeric as sen_pct,
|
null::numeric as sen_pct,
|
||||||
{{ safe_numeric('sen_ehcp_pct') }} as sen_ehcp_pct,
|
|
||||||
{{ safe_numeric('sen_support_pct') }} as sen_support_pct,
|
{{ safe_numeric('sen_support_pct') }} as sen_support_pct,
|
||||||
|
{{ safe_numeric('sen_ehcp_pct') }} as sen_ehcp_pct
|
||||||
-- Progress 8 banding & disadvantage gaps (not published in legacy format)
|
|
||||||
null::text as progress_8_banding,
|
|
||||||
null::numeric as attainment_8_disadvantage_gap,
|
|
||||||
null::numeric as progress_8_disadvantage_gap
|
|
||||||
|
|
||||||
from {{ source('raw', 'legacy_ks4') }}
|
from {{ source('raw', 'legacy_ks4') }}
|
||||||
where urn is not null
|
where urn is not null
|
||||||
|
|||||||
@@ -33,23 +33,17 @@ renamed as (
|
|||||||
nullif(trim(ungraded_outcome), 'NULL') as ungraded_outcome,
|
nullif(trim(ungraded_outcome), 'NULL') as ungraded_outcome,
|
||||||
{{ parse_ungraded_outcome('ungraded_outcome') }}::integer as ungraded_grade,
|
{{ parse_ungraded_outcome('ungraded_outcome') }}::integer as ungraded_grade,
|
||||||
|
|
||||||
-- Report Card fields (post-Nov 2025 framework), 5-point scale:
|
-- Report Card fields (post-Nov 2025 framework)
|
||||||
-- 1 Exceptional · 2 Strong standard · 3 Expected standard
|
-- TODO: add rc_* columns to tap-uk-ofsted schema once CSV column names are confirmed
|
||||||
-- · 4 Needs attention · 5 Urgent improvement
|
null::text as rc_safeguarding_met,
|
||||||
case lower(trim(nullif(rc_safeguarding_met, 'NULL')))
|
null::text as rc_inclusion,
|
||||||
when 'met' then true
|
null::text as rc_curriculum_teaching,
|
||||||
when 'not met' then false
|
null::text as rc_achievement,
|
||||||
end as rc_safeguarding_met,
|
null::text as rc_attendance_behaviour,
|
||||||
{{ parse_report_card_grade('rc_inclusion') }}::integer as rc_inclusion,
|
null::text as rc_personal_development,
|
||||||
{{ parse_report_card_grade('rc_curriculum_teaching') }}::integer as rc_curriculum_teaching,
|
null::text as rc_leadership_governance,
|
||||||
{{ parse_report_card_grade('rc_achievement') }}::integer as rc_achievement,
|
null::text as rc_early_years,
|
||||||
{{ parse_report_card_grade('rc_attendance_behaviour') }}::integer as rc_attendance_behaviour,
|
null::text as rc_sixth_form,
|
||||||
{{ parse_report_card_grade('rc_personal_development') }}::integer as rc_personal_development,
|
|
||||||
{{ parse_report_card_grade('rc_leadership_governance') }}::integer as rc_leadership_governance,
|
|
||||||
-- No MI column exists for these yet (see tap.py); the tap never
|
|
||||||
-- emits rc_early_years/rc_sixth_form, so these stay NULL.
|
|
||||||
null::integer as rc_early_years,
|
|
||||||
null::integer as rc_sixth_form,
|
|
||||||
|
|
||||||
report_url
|
report_url
|
||||||
from source
|
from source
|
||||||
|
|||||||
@@ -0,0 +1,30 @@
|
|||||||
|
-- Staging model: Ofsted Parent View survey responses
|
||||||
|
-- The tap computes positive percentages (Strongly agree + Agree) per question.
|
||||||
|
|
||||||
|
with source as (
|
||||||
|
select * from {{ source('raw', 'parent_view') }}
|
||||||
|
),
|
||||||
|
|
||||||
|
renamed as (
|
||||||
|
select
|
||||||
|
cast(urn as integer) as urn,
|
||||||
|
cast(survey_date as date) as survey_date,
|
||||||
|
cast(total_responses as integer) as total_responses,
|
||||||
|
cast(q_happy_pct as numeric) as q_happy_pct,
|
||||||
|
cast(q_safe_pct as numeric) as q_safe_pct,
|
||||||
|
cast(q_behaviour_pct as numeric) as q_behaviour_pct,
|
||||||
|
cast(q_bullying_pct as numeric) as q_bullying_pct,
|
||||||
|
cast(q_communication_pct as numeric) as q_communication_pct,
|
||||||
|
cast(q_progress_pct as numeric) as q_progress_pct,
|
||||||
|
cast(q_teaching_pct as numeric) as q_teaching_pct,
|
||||||
|
cast(q_information_pct as numeric) as q_information_pct,
|
||||||
|
cast(q_curriculum_pct as numeric) as q_curriculum_pct,
|
||||||
|
cast(q_future_pct as numeric) as q_future_pct,
|
||||||
|
cast(q_leadership_pct as numeric) as q_leadership_pct,
|
||||||
|
cast(q_wellbeing_pct as numeric) as q_wellbeing_pct,
|
||||||
|
cast(q_recommend_pct as numeric) as q_recommend_pct
|
||||||
|
from source
|
||||||
|
where urn is not null
|
||||||
|
)
|
||||||
|
|
||||||
|
select * from renamed
|
||||||
@@ -1,108 +0,0 @@
|
|||||||
field,code,name
|
|
||||||
school_type,1,Community school
|
|
||||||
school_type,2,Voluntary aided school
|
|
||||||
school_type,3,Voluntary controlled school
|
|
||||||
school_type,5,Foundation school
|
|
||||||
school_type,6,City technology college
|
|
||||||
school_type,7,Community special school
|
|
||||||
school_type,8,Non-maintained special school
|
|
||||||
school_type,10,Other independent special school
|
|
||||||
school_type,11,Other independent school
|
|
||||||
school_type,12,Foundation special school
|
|
||||||
school_type,14,Pupil referral unit
|
|
||||||
school_type,15,Local authority nursery school
|
|
||||||
school_type,18,Further education
|
|
||||||
school_type,24,Secure units
|
|
||||||
school_type,25,Offshore schools
|
|
||||||
school_type,26,Service children's education
|
|
||||||
school_type,27,Miscellaneous
|
|
||||||
school_type,28,Academy sponsor led
|
|
||||||
school_type,29,Higher education institutions
|
|
||||||
school_type,30,Welsh establishment
|
|
||||||
school_type,31,Sixth form centres
|
|
||||||
school_type,32,Special post 16 institution
|
|
||||||
school_type,33,Academy special sponsor led
|
|
||||||
school_type,34,Academy converter
|
|
||||||
school_type,35,Free schools
|
|
||||||
school_type,36,Free schools special
|
|
||||||
school_type,37,British schools overseas
|
|
||||||
school_type,38,Free schools alternative provision
|
|
||||||
school_type,39,Free schools 16 to 19
|
|
||||||
school_type,40,University technical college
|
|
||||||
school_type,41,Studio schools
|
|
||||||
school_type,42,Academy alternative provision converter
|
|
||||||
school_type,43,Academy alternative provision sponsor led
|
|
||||||
school_type,44,Academy special converter
|
|
||||||
school_type,45,Academy 16-19 converter
|
|
||||||
school_type,46,Academy 16 to 19 sponsor led
|
|
||||||
school_type,49,Online provider
|
|
||||||
school_type,56,Institution funded by other government department
|
|
||||||
school_type,57,Academy secure 16 to 19
|
|
||||||
establishment_status,1,Open
|
|
||||||
establishment_status,2,Closed
|
|
||||||
establishment_status,3,"Open, but proposed to close"
|
|
||||||
establishment_status,4,Proposed to open
|
|
||||||
phase_of_education,0,Not applicable
|
|
||||||
phase_of_education,1,Nursery
|
|
||||||
phase_of_education,2,Primary
|
|
||||||
phase_of_education,3,Middle deemed primary
|
|
||||||
phase_of_education,4,Secondary
|
|
||||||
phase_of_education,5,Middle deemed secondary
|
|
||||||
phase_of_education,6,16 plus
|
|
||||||
phase_of_education,7,All-through
|
|
||||||
official_sixth_form,0,Not applicable
|
|
||||||
official_sixth_form,1,Has a sixth form
|
|
||||||
official_sixth_form,2,Does not have a sixth form
|
|
||||||
official_sixth_form,9,
|
|
||||||
religious_character,0,Does not apply
|
|
||||||
religious_character,2,Church of England
|
|
||||||
religious_character,3,Roman Catholic
|
|
||||||
religious_character,4,Methodist
|
|
||||||
religious_character,5,Jewish
|
|
||||||
religious_character,6,None
|
|
||||||
religious_character,7,Muslim
|
|
||||||
religious_character,8,Seventh Day Adventist
|
|
||||||
religious_character,9,Church of England/Methodist
|
|
||||||
religious_character,10,Methodist/Church of England
|
|
||||||
religious_character,11,Church of England/Roman Catholic
|
|
||||||
religious_character,12,Church of England/United Reformed Church
|
|
||||||
religious_character,13,Roman Catholic/Church of England
|
|
||||||
religious_character,14,Quaker
|
|
||||||
religious_character,15,Christian
|
|
||||||
religious_character,16,United Reformed Church
|
|
||||||
religious_character,17,Congregational Church
|
|
||||||
religious_character,18,Free Church
|
|
||||||
religious_character,19,Church of England/Free Church
|
|
||||||
religious_character,20,Church of England/Christian
|
|
||||||
religious_character,21,Sikh
|
|
||||||
religious_character,22,Greek Orthodox
|
|
||||||
religious_character,24,Buddhist
|
|
||||||
religious_character,25,Hindu
|
|
||||||
religious_character,26,Moravian
|
|
||||||
religious_character,28,Inter- / non- denominational
|
|
||||||
religious_character,29,Multi-faith
|
|
||||||
religious_character,30,Church of England/Methodist/United Reform Church/Baptist
|
|
||||||
religious_character,31,Anglican
|
|
||||||
religious_character,32,Anglican/Christian
|
|
||||||
religious_character,33,Anglican/Evangelical
|
|
||||||
religious_character,34,Anglican/Church of England
|
|
||||||
religious_character,35,Catholic
|
|
||||||
religious_character,36,Charadi Jewish
|
|
||||||
religious_character,37,Christian/Evangelical
|
|
||||||
religious_character,38,Christian Science
|
|
||||||
religious_character,39,Christian/Methodist
|
|
||||||
religious_character,40,Christian/non-denominational
|
|
||||||
religious_character,41,Church of England/Evangelical
|
|
||||||
religious_character,42,Islam
|
|
||||||
religious_character,43,Orthodox Jewish
|
|
||||||
religious_character,44,Plymouth Brethren Christian Church
|
|
||||||
religious_character,45,Protestant
|
|
||||||
religious_character,46,Protestant/Evangelical
|
|
||||||
religious_character,47,Reformed Baptist
|
|
||||||
religious_character,48,Roman Catholic/Anglican
|
|
||||||
religious_character,49,Sunni Deobandi
|
|
||||||
religious_character,99,
|
|
||||||
admissions_policy,0,Not applicable
|
|
||||||
admissions_policy,2,Selective
|
|
||||||
admissions_policy,4,Non-selective
|
|
||||||
admissions_policy,9,
|
|
||||||
|
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user