Compare commits
225
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
41d3f3b971 | ||
|
|
c013265cb4 | ||
|
|
ad9d3b67f7 | ||
|
|
ca47d08186 | ||
|
|
62a6bfaf0c | ||
|
|
e344298440 | ||
|
|
8020191832 | ||
|
|
5f9caad7f4 | ||
|
|
e65c93b68e | ||
|
|
e3f21a5bc7 | ||
|
|
dea435a906 | ||
|
|
dd5b48e612 | ||
|
|
a88139a539 | ||
|
|
9b765125ad | ||
|
|
ccf0892a0e | ||
|
|
0450f8ecd6 | ||
|
|
27d83f9bd0 | ||
|
|
19c574edb0 | ||
|
|
e78ec14e2e | ||
|
|
fb3ef7d2b9 | ||
|
|
49ac96b487 | ||
|
|
214c80663e | ||
|
|
84caee9f72 | ||
|
|
c992d7f3b9 | ||
|
|
17bfb4a3f7 | ||
|
|
15b8923e60 | ||
|
|
354244f755 | ||
|
|
3e2fa4a419 | ||
|
|
50546ecf22 | ||
|
|
99d62ef748 | ||
|
|
68452681f8 | ||
|
|
0cc4f52816 | ||
|
|
eb13ab0b5e | ||
|
|
fa49164143 | ||
|
|
cf3c773f86 | ||
|
|
4b54c25943 | ||
|
|
e7645d1ba5 | ||
|
|
29b5f85952 | ||
|
|
c077c27720 | ||
|
|
e2fc7a8f15 | ||
|
|
355a5a841c | ||
|
|
96deab7d58 | ||
|
|
e9886361d2 | ||
|
|
8ebe461435 | ||
|
|
2002529137 | ||
|
|
bd7c8593d9 | ||
|
|
0c414680fd | ||
|
|
e211e1376d | ||
|
|
74418ca6b9 | ||
|
|
5df8c93420 | ||
|
|
ebf9c12446 | ||
|
|
ca4ddd2b12 | ||
|
|
dff3e210ab | ||
|
|
37bbda1da1 | ||
|
|
a37da15008 | ||
|
|
cff3854e63 | ||
|
|
1bb3e0360f | ||
|
|
4e4b30e812 | ||
|
|
0e177ca2ec | ||
|
|
367a07c15d | ||
|
|
983a581555 | ||
|
|
b34feb8e98 | ||
|
|
cc99865bd4 | ||
|
|
587cfe3f0b | ||
|
|
0a4c051ee5 | ||
|
|
343b40c645 | ||
|
|
d1688ac150 | ||
|
|
271ffe92d4 | ||
|
|
0571d1c0ff | ||
|
|
180d6e9b3e | ||
|
|
80f405123e | ||
|
|
077aca6008 | ||
|
|
9dba5ff1ff | ||
|
|
f530a912bc | ||
|
|
029fe8d8a6 | ||
|
|
cd1c5d1e1a | ||
|
|
cd6a45bf7d | ||
|
|
5c0ccc693d | ||
|
|
151cf4bc80 | ||
|
|
83dc5ae5dc | ||
|
|
2175dccb7c | ||
|
|
bd2a6c385b | ||
|
|
4e0d8bcf87 | ||
|
|
91314a80b5 | ||
|
|
52b00ac752 | ||
|
|
b571d9c549 | ||
|
|
8a23e3657d | ||
|
|
e4e8f02599 | ||
|
|
b62dc17532 | ||
|
|
4d7762d796 | ||
|
|
3650f7d8b7 | ||
|
|
dfce308f1f | ||
|
|
64ae71d7ab | ||
|
|
7ab084dd3a | ||
|
|
5ad1cbfb53 | ||
|
|
0c901cd0d1 | ||
|
|
7b41218e6e | ||
|
|
9b75f54206 | ||
|
|
38bc17cab3 | ||
|
|
dc156058fe | ||
|
|
1d8858fbda | ||
|
|
eaf5e5d180 | ||
|
|
be780ebe13 | ||
|
|
b6c2cd5116 | ||
|
|
dc79d653e5 | ||
|
|
b0d5334e06 | ||
|
|
d65eb58883 | ||
|
|
7f5f0fb676 | ||
|
|
47f3591ed8 | ||
|
|
6fc7fce948 | ||
|
|
eb6d918650 | ||
|
|
d47ac71c47 | ||
|
|
17e5371e9c | ||
|
|
e2c63a9905 | ||
|
|
3f3c5953f6 | ||
|
|
124c6702a9 | ||
|
|
e25722d9ab | ||
|
|
07d586d0ad | ||
|
|
b793640507 | ||
|
|
21a5d18f59 | ||
|
|
f614414070 | ||
|
|
310b63b0cb | ||
|
|
c2c76c5817 | ||
|
|
c5a4d106da | ||
|
|
2437ffce42 | ||
|
|
eb648f3f76 | ||
|
|
74e5fffc10 | ||
|
|
748ef32180 | ||
|
|
b0c4ea8282 | ||
|
|
e236669fde | ||
|
|
fb5a0928bd | ||
|
|
264edd2e3a | ||
|
|
cd2cbe7be6 | ||
|
|
73182d0c0c | ||
|
|
cbe3a9a772 | ||
|
|
2e9b5c83c5 | ||
|
|
102397fe69 | ||
|
|
68a192e430 | ||
|
|
ccd5074c90 | ||
|
|
2b4cf20d75 | ||
|
|
cef2f77149 | ||
|
|
68b6417149 | ||
|
|
c5719ef362 | ||
|
|
5e5b61987a | ||
|
|
c564566432 | ||
|
|
9188626051 | ||
|
|
7ae9ecdc36 | ||
|
|
1980d79eee | ||
|
|
9423f11567 | ||
|
|
576013d627 | ||
|
|
7c08138fe4 | ||
|
|
a7829d591a | ||
|
|
1ed4470fc2 | ||
|
|
7a16b1b52f | ||
|
|
cf9d41b476 | ||
|
|
e820e7fecd | ||
|
|
4fdeb70a93 | ||
|
|
9a1f56c431 | ||
|
|
ade9dbb3ba | ||
|
|
d1a8596208 | ||
|
|
a3c09d9b67 | ||
|
|
a7f4c86464 | ||
|
|
0804566736 | ||
|
|
55363cbd18 | ||
|
|
868eb344f5 | ||
|
|
0b15497c09 | ||
|
|
d55f6cce23 | ||
|
|
3236efa846 | ||
|
|
d5a6db289d | ||
|
|
d8ccb5b733 | ||
|
|
0fa1a292c7 | ||
|
|
c3f044bd65 | ||
|
|
d2115364ae | ||
|
|
28cf0a342c | ||
|
|
d88e77f459 | ||
|
|
06eb433db5 | ||
|
|
1a6d349dad | ||
|
|
75d3534d82 | ||
|
|
ff041544f2 | ||
|
|
59265f78b6 | ||
|
|
6e0a278340 | ||
|
|
e651dd0d65 | ||
|
|
22c113fc29 | ||
|
|
e953ee7c5f | ||
|
|
413d86cc3c | ||
|
|
4f01fbdedb | ||
|
|
c3ba7aae0d | ||
|
|
54a30de0d8 | ||
|
|
c30ad1db07 | ||
|
|
7424cef7c6 | ||
|
|
01ccbb8e82 | ||
|
|
c339c2f1a1 | ||
|
|
e2ca3d79f9 | ||
|
|
c2364bf09e | ||
|
|
43e0621728 | ||
|
|
d1358cc00f | ||
|
|
865a69b54d | ||
|
|
9cc87c41bb | ||
|
|
8967966eef | ||
|
|
4a9a5c734b | ||
|
|
4e82e6c916 | ||
|
|
d4340a8fdd | ||
|
|
bb2f7a5841 | ||
|
|
1cb5314c53 | ||
|
|
4cea26b813 | ||
|
|
dbb74d9b60 | ||
|
|
9545aec7f4 | ||
|
|
3365ebcb3a | ||
|
|
6d79bd3331 | ||
|
|
24e114dee7 | ||
|
|
6c5db0c266 | ||
|
|
6f749ed21f | ||
|
|
d423826840 | ||
|
|
d3c63ccc6d | ||
|
|
b93eb3a691 | ||
|
|
6b871ce1e9 | ||
|
|
c981d89137 | ||
|
|
de5e790112 | ||
|
|
42138fc402 | ||
|
|
c5af476213 | ||
|
|
de853b90b3 | ||
|
|
759d9f5cea | ||
|
|
555d3f0a7d | ||
|
|
ecc847091c | ||
|
|
b187a478c9 |
No files matched your search
+13
-5
@@ -20,7 +20,7 @@ PORT=80
|
||||
# =============================================================================
|
||||
# CORS
|
||||
# =============================================================================
|
||||
# Comma-separated list of allowed origins
|
||||
# JSON array of allowed origins (pydantic-settings format)
|
||||
# In production, only include your actual domain
|
||||
ALLOWED_ORIGINS=["https://schoolcompare.co.uk"]
|
||||
|
||||
@@ -33,13 +33,21 @@ ADMIN_API_KEY=CHANGE_THIS_TO_A_SECURE_RANDOM_KEY
|
||||
|
||||
# Rate limiting (requests per minute per IP)
|
||||
RATE_LIMIT_PER_MINUTE=60
|
||||
RATE_LIMIT_BURST=10
|
||||
GLOBAL_RATE_LIMIT_PER_MINUTE=3000
|
||||
|
||||
# Maximum request body size in bytes (default 1MB)
|
||||
MAX_REQUEST_SIZE=1048576
|
||||
|
||||
# =============================================================================
|
||||
# API
|
||||
# SEARCH AND OPTIONAL FEATURE FLAGS
|
||||
# =============================================================================
|
||||
DEFAULT_PAGE_SIZE=50
|
||||
MAX_PAGE_SIZE=100
|
||||
TYPESENSE_URL=http://localhost:8108
|
||||
TYPESENSE_API_KEY=CHANGE_THIS_TO_YOUR_TYPESENSE_KEY
|
||||
|
||||
# Empty URL disables Unleash-backed flags. Match the managed environment when used.
|
||||
UNLEASH_URL=
|
||||
UNLEASH_API_TOKEN=
|
||||
|
||||
# Page-size limits are currently declared by route Query parameters.
|
||||
# DEFAULT_PAGE_SIZE, MAX_PAGE_SIZE and RATE_LIMIT_BURST are not reliable tuning
|
||||
# controls in the current routes; see docs/LEGACY_CODE.md.
|
||||
+73
-17
@@ -5,6 +5,11 @@ on:
|
||||
branches:
|
||||
- main
|
||||
|
||||
# Serialise the entire build/deploy/test cycle: no other run can move staging tags.
|
||||
concurrency:
|
||||
group: staging-release
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
REGISTRY: privaterepo.sitaru.org
|
||||
BACKEND_IMAGE_NAME: ${{ gitea.repository }}-backend
|
||||
@@ -12,7 +17,18 @@ env:
|
||||
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
|
||||
|
||||
jobs:
|
||||
prepare:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
build_id: ${{ steps.identity.outputs.build_id }}
|
||||
steps:
|
||||
- id: identity
|
||||
run: python3 -c 'import uuid; print("build_id=" + uuid.uuid4().hex)' >> "$GITHUB_OUTPUT"
|
||||
|
||||
build-backend:
|
||||
needs: [prepare]
|
||||
outputs:
|
||||
digest: ${{ steps.build.outputs.digest }}
|
||||
name: Build Backend (FastAPI)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -46,17 +62,24 @@ jobs:
|
||||
type=raw,value=staging
|
||||
|
||||
- name: Build and push Backend Docker image
|
||||
id: build
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: .
|
||||
file: ./Dockerfile
|
||||
push: true
|
||||
build-args: |
|
||||
BUILD_SHA=${{ gitea.sha }}
|
||||
BUILD_ID=${{ needs.prepare.outputs.build_id }}
|
||||
tags: ${{ steps.meta-backend.outputs.tags }}
|
||||
labels: ${{ steps.meta-backend.outputs.labels }}
|
||||
cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache
|
||||
cache-to: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache,mode=max
|
||||
|
||||
build-frontend:
|
||||
needs: [prepare]
|
||||
outputs:
|
||||
digest: ${{ steps.build.outputs.digest }}
|
||||
name: Build Frontend (Next.js)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -90,18 +113,23 @@ jobs:
|
||||
type=raw,value=staging
|
||||
|
||||
- name: Build and push Frontend Docker image
|
||||
id: build
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: ./nextjs-app
|
||||
file: ./nextjs-app/Dockerfile
|
||||
push: true
|
||||
build-args: |
|
||||
BUILD_SHA=${{ gitea.sha }}
|
||||
BUILD_ID=${{ needs.prepare.outputs.build_id }}
|
||||
tags: ${{ steps.meta-frontend.outputs.tags }}
|
||||
labels: ${{ steps.meta-frontend.outputs.labels }}
|
||||
build-args: |
|
||||
FASTAPI_URL=http://backend:80/api
|
||||
# Cache disabled due to registry size limits
|
||||
|
||||
build-pipeline:
|
||||
needs: [prepare]
|
||||
outputs:
|
||||
digest: ${{ steps.build.outputs.digest }}
|
||||
name: Build Pipeline (Meltano + dbt + Airflow)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -135,11 +163,15 @@ jobs:
|
||||
type=raw,value=staging
|
||||
|
||||
- name: Build and push Pipeline Docker image
|
||||
id: build
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: ./pipeline
|
||||
file: ./pipeline/Dockerfile
|
||||
push: true
|
||||
build-args: |
|
||||
BUILD_SHA=${{ gitea.sha }}
|
||||
BUILD_ID=${{ needs.prepare.outputs.build_id }}
|
||||
tags: ${{ steps.meta-pipeline.outputs.tags }}
|
||||
labels: ${{ steps.meta-pipeline.outputs.labels }}
|
||||
cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.PIPELINE_IMAGE_NAME }}:buildcache
|
||||
@@ -148,30 +180,23 @@ jobs:
|
||||
deploy-staging:
|
||||
name: Deploy to Staging
|
||||
runs-on: ubuntu-latest
|
||||
needs: [build-backend, build-frontend, build-pipeline]
|
||||
needs: [prepare, build-backend, build-frontend, build-pipeline]
|
||||
steps:
|
||||
- name: Trigger staging stack update
|
||||
run: curl -fsSk -X POST "${{ secrets.PORTAINER_STAGING_WEBHOOK }}"
|
||||
|
||||
- name: Wait for staging to become healthy
|
||||
run: |
|
||||
echo "Polling ${STAGING_BASE_URL} for up to 5 minutes..."
|
||||
for i in $(seq 1 60); do
|
||||
if curl -fsS -o /dev/null --max-time 10 "${STAGING_BASE_URL}/"; then
|
||||
echo "Staging is up (attempt $i)"
|
||||
exit 0
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
echo "Staging did not become healthy in time" >&2
|
||||
exit 1
|
||||
- uses: actions/checkout@v4
|
||||
- name: Verify deployed release identity
|
||||
run: python3 scripts/ci/release.py wait
|
||||
env:
|
||||
STAGING_BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||
EXPECTED_SHA: ${{ gitea.sha }}
|
||||
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
|
||||
|
||||
e2e-staging:
|
||||
name: E2E Journeys against Staging
|
||||
runs-on: ubuntu-latest
|
||||
needs: [deploy-staging]
|
||||
needs: [prepare, deploy-staging, build-backend, build-frontend, build-pipeline]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
@@ -187,11 +212,42 @@ jobs:
|
||||
npm ci
|
||||
npx playwright install --with-deps chromium
|
||||
|
||||
- name: Verify release before journeys
|
||||
run: python3 scripts/ci/release.py wait --timeout 10
|
||||
env:
|
||||
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||
EXPECTED_SHA: ${{ gitea.sha }}
|
||||
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
|
||||
|
||||
- name: Run E2E journeys
|
||||
working-directory: e2e
|
||||
run: npx playwright test
|
||||
env:
|
||||
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||
EXPECTED_SHA: ${{ gitea.sha }}
|
||||
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
|
||||
|
||||
- name: Verify release after journeys
|
||||
run: python3 scripts/ci/release.py wait --timeout 10
|
||||
env:
|
||||
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
|
||||
EXPECTED_SHA: ${{ gitea.sha }}
|
||||
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
|
||||
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
- uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ gitea.actor }}
|
||||
password: ${{ secrets.REGISTRY_TOKEN }}
|
||||
- name: Mark tested image digests as verified
|
||||
run: python3 scripts/ci/release.py verify
|
||||
env:
|
||||
EXPECTED_SHA: ${{ gitea.sha }}
|
||||
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
|
||||
BACKEND_DIGEST: ${{ needs.build-backend.outputs.digest }}
|
||||
FRONTEND_DIGEST: ${{ needs.build-frontend.outputs.digest }}
|
||||
PIPELINE_DIGEST: ${{ needs.build-pipeline.outputs.digest }}
|
||||
|
||||
# Production deployment is a second, manual approval: see promote.yml
|
||||
# ("Promote to Production (manual)") and docs/DEPLOY.md.
|
||||
@@ -68,13 +68,13 @@ jobs:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install -r requirements.txt pytest "httpx<0.28"
|
||||
run: pip install -r requirements.txt pytest "httpx<0.28" pyyaml
|
||||
|
||||
- name: Import smoke test
|
||||
run: python -c "from backend.app import app; print('backend imports OK')"
|
||||
|
||||
- name: Backend unit tests
|
||||
run: python -m pytest backend/tests -q
|
||||
run: python -m pytest backend/tests pipeline/tests scripts/ci/tests -q
|
||||
|
||||
build-backend:
|
||||
name: Build Backend (no push)
|
||||
|
||||
@@ -97,33 +97,15 @@ jobs:
|
||||
username: ${{ gitea.actor }}
|
||||
password: ${{ secrets.REGISTRY_TOKEN }}
|
||||
|
||||
- name: Retag approved images as prod (keeping rollback pointer)
|
||||
run: |
|
||||
SHORT_SHA="${{ steps.resolve.outputs.short }}"
|
||||
for IMAGE in \
|
||||
"${REGISTRY}/${BACKEND_IMAGE_NAME}" \
|
||||
"${REGISTRY}/${FRONTEND_IMAGE_NAME}" \
|
||||
"${REGISTRY}/${PIPELINE_IMAGE_NAME}"; do
|
||||
# Keep a rollback pointer before moving :prod
|
||||
docker buildx imagetools create -t "${IMAGE}:prod-previous" "${IMAGE}:prod" || true
|
||||
docker buildx imagetools create -t "${IMAGE}:prod" "${IMAGE}:${SHORT_SHA}"
|
||||
echo "Promoted ${IMAGE}:${SHORT_SHA} -> :prod"
|
||||
done
|
||||
- name: Resolve verified digests and promote the complete image set
|
||||
run: python3 scripts/ci/release.py promote --output release.json
|
||||
env:
|
||||
EXPECTED_SHA: ${{ steps.resolve.outputs.full }}
|
||||
|
||||
- name: Trigger production stack update
|
||||
run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}"
|
||||
|
||||
- name: Wait for production to become healthy
|
||||
run: |
|
||||
echo "Polling ${PROD_BASE_URL} for up to 5 minutes..."
|
||||
for i in $(seq 1 60); do
|
||||
if curl -fsS -o /dev/null --max-time 10 "${PROD_BASE_URL}/"; then
|
||||
echo "Production is up (attempt $i)"
|
||||
exit 0
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
echo "Production did not become healthy in time" >&2
|
||||
exit 1
|
||||
- name: Verify production release identity
|
||||
run: python3 scripts/ci/release.py wait --release release.json
|
||||
env:
|
||||
PROD_BASE_URL: ${{ secrets.PROD_BASE_URL }}
|
||||
BASE_URL: ${{ secrets.PROD_BASE_URL }}
|
||||
+13
-187
@@ -1,191 +1,17 @@
|
||||
# Docker Deployment Guide
|
||||
# Docker deployment
|
||||
|
||||
## Quick Start
|
||||
The maintained deployment runbook is [docs/DEPLOY.md](docs/DEPLOY.md).
|
||||
|
||||
Deploy the complete SchoolCompare stack (PostgreSQL + FastAPI + Next.js) with one command:
|
||||
- Production: `docker-compose.portainer.yml`, using `:prod` images.
|
||||
- Staging: `docker-compose.portainer.staging.yml`, using `:staging` images.
|
||||
- Builds and deployment: `.gitea/workflows/deploy.yml`.
|
||||
- Human-approved production promotion: `.gitea/workflows/promote.yml`.
|
||||
|
||||
```bash
|
||||
docker-compose up -d
|
||||
```
|
||||
The generic `docker-compose.yml` is not a supported one-command onboarding path:
|
||||
it still uses `:latest` tags that the release workflow no longer publishes and
|
||||
lacks the full current CMS setup. Review the [legacy inventory](docs/LEGACY_CODE.md)
|
||||
before using old compose examples. Starting an empty database does not populate
|
||||
school marts.
|
||||
|
||||
This will start:
|
||||
- **PostgreSQL** on port 5432 (database)
|
||||
- **FastAPI** on port 8000 (backend API)
|
||||
- **Next.js** on port 3000 (frontend)
|
||||
|
||||
## Service Details
|
||||
|
||||
### PostgreSQL Database
|
||||
- **Port**: 5432
|
||||
- **Container**: `schoolcompare_db`
|
||||
- **Credentials**:
|
||||
- User: `schoolcompare`
|
||||
- Password: `schoolcompare`
|
||||
- Database: `schoolcompare`
|
||||
- **Volume**: `postgres_data` (persistent storage)
|
||||
|
||||
### FastAPI Backend
|
||||
- **Port**: 8000 → 80 (container)
|
||||
- **Container**: `schoolcompare_backend`
|
||||
- **Built from**: Root `Dockerfile`
|
||||
- **API Endpoint**: http://localhost:8000/api
|
||||
- **Health Check**: http://localhost:8000/api/data-info
|
||||
|
||||
### Next.js Frontend
|
||||
- **Port**: 3000
|
||||
- **Container**: `schoolcompare_nextjs`
|
||||
- **Built from**: `nextjs-app/Dockerfile`
|
||||
- **URL**: http://localhost:3000
|
||||
- **Connects to**: Backend via internal network
|
||||
|
||||
## Commands
|
||||
|
||||
### Start all services
|
||||
```bash
|
||||
docker-compose up -d
|
||||
```
|
||||
|
||||
### View logs
|
||||
```bash
|
||||
# All services
|
||||
docker-compose logs -f
|
||||
|
||||
# Specific service
|
||||
docker-compose logs -f nextjs
|
||||
docker-compose logs -f backend
|
||||
docker-compose logs -f db
|
||||
```
|
||||
|
||||
### Check status
|
||||
```bash
|
||||
docker-compose ps
|
||||
```
|
||||
|
||||
### Stop all services
|
||||
```bash
|
||||
docker-compose down
|
||||
```
|
||||
|
||||
### Rebuild after code changes
|
||||
```bash
|
||||
# Rebuild and restart specific service
|
||||
docker-compose up -d --build nextjs
|
||||
|
||||
# Rebuild all services
|
||||
docker-compose up -d --build
|
||||
```
|
||||
|
||||
### Clean restart (remove volumes)
|
||||
```bash
|
||||
docker-compose down -v
|
||||
docker-compose up -d
|
||||
```
|
||||
|
||||
## Initial Database Setup
|
||||
|
||||
After first start, you may need to initialize the database:
|
||||
|
||||
```bash
|
||||
# Enter the backend container
|
||||
docker exec -it schoolcompare_backend bash
|
||||
|
||||
# Run migrations or data loading
|
||||
python -m backend.data_loader
|
||||
```
|
||||
|
||||
## Accessing Services
|
||||
|
||||
Once running:
|
||||
- **Frontend**: http://localhost:3000
|
||||
- **Backend API**: http://localhost:8000/api
|
||||
- **API Docs**: http://localhost:8000/docs (Swagger UI)
|
||||
- **Database**: localhost:5432 (use any PostgreSQL client)
|
||||
|
||||
## Environment Variables
|
||||
|
||||
Create a `.env` file in the root directory to customize:
|
||||
|
||||
```env
|
||||
# Database
|
||||
POSTGRES_USER=schoolcompare
|
||||
POSTGRES_PASSWORD=your_secure_password
|
||||
POSTGRES_DB=schoolcompare
|
||||
|
||||
# Backend
|
||||
DATABASE_URL=postgresql://schoolcompare:your_secure_password@db:5432/schoolcompare
|
||||
|
||||
# Frontend (for client-side access)
|
||||
NEXT_PUBLIC_API_URL=http://localhost:8000/api
|
||||
```
|
||||
|
||||
Then run:
|
||||
```bash
|
||||
docker-compose up -d
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Backend not connecting to database
|
||||
```bash
|
||||
# Check database health
|
||||
docker-compose ps
|
||||
|
||||
# View backend logs
|
||||
docker-compose logs backend
|
||||
|
||||
# Restart backend
|
||||
docker-compose restart backend
|
||||
```
|
||||
|
||||
### Frontend not connecting to backend
|
||||
```bash
|
||||
# Check backend health
|
||||
curl http://localhost:8000/api/data-info
|
||||
|
||||
# Check Next.js environment variables
|
||||
docker exec schoolcompare_nextjs env | grep API
|
||||
```
|
||||
|
||||
### Port already in use
|
||||
```bash
|
||||
# Change ports in docker-compose.yml
|
||||
# For example, change "3000:3000" to "3001:3000"
|
||||
```
|
||||
|
||||
### Rebuild from scratch
|
||||
```bash
|
||||
docker-compose down -v
|
||||
docker system prune -a
|
||||
docker-compose up -d --build
|
||||
```
|
||||
|
||||
## Production Deployment
|
||||
|
||||
For production, update the following:
|
||||
|
||||
1. **Use secure passwords** in `.env` file
|
||||
2. **Configure reverse proxy** (Nginx) in front of Next.js
|
||||
3. **Enable HTTPS** with SSL certificates
|
||||
4. **Set production environment variables**:
|
||||
```env
|
||||
NODE_ENV=production
|
||||
POSTGRES_PASSWORD=<strong-password>
|
||||
```
|
||||
5. **Backup database** regularly:
|
||||
```bash
|
||||
docker exec schoolcompare_db pg_dump -U schoolcompare schoolcompare > backup.sql
|
||||
```
|
||||
|
||||
## Network Architecture
|
||||
|
||||
```
|
||||
Internet
|
||||
↓
|
||||
Next.js (port 3000) ← User browsers
|
||||
↓ (internal network)
|
||||
FastAPI (port 8000) ← API calls
|
||||
↓ (internal network)
|
||||
PostgreSQL (port 5432) ← Data queries
|
||||
```
|
||||
|
||||
All services communicate via the `schoolcompare-network` Docker network.
|
||||
For architecture, configuration and test commands, see
|
||||
[ARCHITECTURE.md](docs/ARCHITECTURE.md) and [DEVELOPMENT.md](docs/DEVELOPMENT.md).
|
||||
@@ -24,6 +24,12 @@ RUN pip install --no-cache-dir -r requirements.txt
|
||||
COPY backend/ ./backend/
|
||||
COPY scripts/ ./scripts/
|
||||
|
||||
ARG BUILD_SHA=development
|
||||
ARG BUILD_ID=development
|
||||
LABEL io.schoolcompare.build-id=$BUILD_ID
|
||||
LABEL io.schoolcompare.commit=$BUILD_SHA
|
||||
RUN python -c 'import json,sys; open("backend/build-info.json", "w").write(json.dumps({"sha":sys.argv[1],"build_id":sys.argv[2]}))' "$BUILD_SHA" "$BUILD_ID"
|
||||
|
||||
# Expose the application port
|
||||
EXPOSE 80
|
||||
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
> Historical migration record, retained for context. Setup and architecture claims below may be obsolete. Use [README.md](README.md), [architecture](docs/ARCHITECTURE.md) and [deployment](docs/DEPLOY.md) for current guidance.
|
||||
|
||||
# SchoolCompare: Vanilla JS → Next.js Migration Summary
|
||||
|
||||
## Overview
|
||||
|
||||
@@ -1,214 +1,67 @@
|
||||
# Primary School Compass 🧒📚
|
||||
# SchoolCompare
|
||||
|
||||
A modern web application for comparing **primary school (KS2)** performance data in **Wandsworth and Merton** over the last 5 years. Built with FastAPI and vanilla JavaScript with Chart.js visualizations.
|
||||
SchoolCompare compares schools across England: primary (KS2), secondary (KS4),
|
||||
all-through and post-16 provision, with coverage depending on the source dataset.
|
||||
It provides school search, postcode maps, comparisons, rankings, place pages,
|
||||
Ofsted information, admissions and destination measures. Editorial content lives
|
||||
in a Payload CMS blog.
|
||||
|
||||

|
||||

|
||||

|
||||
## Start here
|
||||
|
||||
## Features
|
||||
- [Architecture and data flow](docs/ARCHITECTURE.md)
|
||||
- [Development and validation](docs/DEVELOPMENT.md)
|
||||
- [Deployment and promotion](docs/DEPLOY.md)
|
||||
- [Legacy and unused-code inventory](docs/LEGACY_CODE.md)
|
||||
- [Frontend conventions](nextjs-app/README.md)
|
||||
- [CMS publishing](nextjs-app/docs/PUBLISHING.md)
|
||||
|
||||
- 📊 **Interactive Charts** - Visualize KS2 performance trends over time
|
||||
- 🔍 **Smart Search** - Find primary schools by name in Wandsworth & Merton
|
||||
- ⚖️ **Side-by-Side Comparison** - Compare up to 5 schools simultaneously
|
||||
- 🏆 **Rankings** - View top-performing primary schools by various KS2 metrics
|
||||
- 📱 **Responsive Design** - Works beautifully on desktop and mobile
|
||||
## Repository map
|
||||
|
||||
## Key Metrics (KS2)
|
||||
| Path | Responsibility |
|
||||
|---|---|
|
||||
| `backend/` | FastAPI routes, cached school data, read-only SQLAlchemy mappings, feature flags |
|
||||
| `nextjs-app/` | Next.js App Router, React UI, Payload CMS, frontend tests |
|
||||
| `pipeline/plugins/extractors/` | Custom Singer taps for GIAS, EES, Ofsted and other datasets |
|
||||
| `pipeline/transform/` | dbt staging/intermediate models, marts, seeds and data tests |
|
||||
| `pipeline/dags/` | Airflow extraction, transformation and publication workflows |
|
||||
| `pipeline/scripts/` | Search indexing, code generation and operational diagnostics |
|
||||
| `e2e/` | Playwright journeys against a running environment |
|
||||
| `.gitea/workflows/` | PR checks, staging deployment and manual production promotion |
|
||||
| `scripts/` | CI review tooling and historical data utilities; see the legacy inventory |
|
||||
| `docs/superpowers/`, `mockups/` | Design history and prototypes, not application entry points |
|
||||
|
||||
The application tracks these Key Stage 2 performance indicators:
|
||||
## Runtime
|
||||
|
||||
| Metric | Description |
|
||||
|--------|-------------|
|
||||
| **Reading Progress** | Progress in reading from KS1 to KS2 |
|
||||
| **Writing Progress** | Progress in writing from KS1 to KS2 |
|
||||
| **Maths Progress** | Progress in maths from KS1 to KS2 |
|
||||
| **Reading Expected %** | Percentage meeting expected standard in reading |
|
||||
| **Writing Expected %** | Percentage meeting expected standard in writing |
|
||||
| **Maths Expected %** | Percentage meeting expected standard in maths |
|
||||
| **Reading, Writing & Maths Combined %** | Percentage meeting expected standard in all three subjects |
|
||||
The public site is **Next.js**, not the FastAPI root page. Browser `/api/*`
|
||||
requests pass through a Next.js route handler to FastAPI. Server-rendered pages
|
||||
call FastAPI directly using `FASTAPI_URL`, including its `/api` suffix.
|
||||
|
||||
## Quick Start
|
||||
PostgreSQL/PostGIS stores school data. Meltano/Singer extracts source data;
|
||||
dbt builds `marts.*`; FastAPI reads those tables. Typesense serves text search
|
||||
and autocomplete. Payload runs inside Next.js and owns a separate `payload`
|
||||
database schema and uploaded media.
|
||||
|
||||
### 1. Clone and Setup
|
||||
There is **no automatic CSV import or sample dataset on startup**. A working
|
||||
school-data environment needs populated marts from the pipeline or an approved
|
||||
database snapshot. See [development](docs/DEVELOPMENT.md) before choosing a setup.
|
||||
|
||||
```bash
|
||||
cd school_results
|
||||
## Validation
|
||||
|
||||
# Create virtual environment
|
||||
python -m venv venv
|
||||
source venv/bin/activate # On Windows: venv\Scripts\activate
|
||||
|
||||
# Install dependencies
|
||||
pip install -r requirements.txt
|
||||
```sh
|
||||
cd nextjs-app
|
||||
npm ci
|
||||
npm run typecheck
|
||||
npm test -- --runInBand
|
||||
```
|
||||
|
||||
### 2. Run the Application
|
||||
Backend checks, pipeline validation, runtime versions and E2E requirements are
|
||||
listed in [DEVELOPMENT.md](docs/DEVELOPMENT.md). No `npm run lint` script is
|
||||
currently defined.
|
||||
|
||||
```bash
|
||||
# Start the server
|
||||
python -m uvicorn backend.app:app --reload --port 8000
|
||||
```
|
||||
|
||||
Then open http://localhost:8000 in your browser.
|
||||
|
||||
The app will run with **sample data** by default, showing **110 primary schools** (66 in Wandsworth, 44 in Merton) with 5 years of KS2 performance data.
|
||||
|
||||
### 3. (Optional) Use Real Data
|
||||
|
||||
To use real UK school performance data:
|
||||
|
||||
1. Visit [Compare School Performance - Download Data](https://www.compare-school-performance.service.gov.uk/download-data)
|
||||
|
||||
2. Download **Key Stage 2** data for the years you want (2019-2024)
|
||||
- Select "Key Stage 2" as the data type
|
||||
|
||||
3. Place the CSV files in the `data/` folder
|
||||
|
||||
4. Restart the server - it will automatically load and filter to Wandsworth & Merton schools
|
||||
|
||||
**Note:** The app only displays schools in Wandsworth and Merton. Data from other areas will be filtered out.
|
||||
|
||||
See the helper script for more details:
|
||||
```bash
|
||||
python scripts/download_data.py
|
||||
```
|
||||
|
||||
## Project Structure
|
||||
|
||||
```
|
||||
school_results/
|
||||
├── backend/
|
||||
│ └── app.py # FastAPI application with all API endpoints
|
||||
├── frontend/
|
||||
│ ├── index.html # Main HTML page
|
||||
│ ├── styles.css # Styling (warm, editorial design)
|
||||
│ └── app.js # Frontend JavaScript
|
||||
├── data/
|
||||
│ └── .gitkeep # Place CSV data files here
|
||||
├── scripts/
|
||||
│ └── download_data.py # Helper for downloading/processing data
|
||||
├── requirements.txt # Python dependencies
|
||||
└── README.md
|
||||
```
|
||||
|
||||
## API Endpoints
|
||||
|
||||
| Endpoint | Description |
|
||||
|----------|-------------|
|
||||
| `GET /api/schools` | List schools with optional search/filter |
|
||||
| `GET /api/schools/{urn}` | Get detailed data for a specific school |
|
||||
| `GET /api/compare?urns=...` | Compare multiple schools |
|
||||
| `GET /api/rankings` | Get school rankings by metric |
|
||||
| `GET /api/filters` | Get available filter options |
|
||||
| `GET /api/metrics` | Get available performance metrics |
|
||||
|
||||
### Example API Usage
|
||||
|
||||
```bash
|
||||
# Search for schools
|
||||
curl "http://localhost:8000/api/schools?search=academy"
|
||||
|
||||
# Get school details
|
||||
curl "http://localhost:8000/api/schools/100001"
|
||||
|
||||
# Compare schools
|
||||
curl "http://localhost:8000/api/compare?urns=100001,100002,100003"
|
||||
|
||||
# Get rankings
|
||||
curl "http://localhost:8000/api/rankings?metric=rwm_expected_pct&year=2024"
|
||||
```
|
||||
|
||||
## Data Format
|
||||
|
||||
If using your own CSV data, ensure it includes these columns (or similar):
|
||||
|
||||
| Column | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| URN | Integer | Unique Reference Number |
|
||||
| SCHNAME | String | School name |
|
||||
| LA | String | Local Authority (must be Wandsworth or Merton) |
|
||||
| READPROG | Float | Reading progress score |
|
||||
| WRITPROG | Float | Writing progress score |
|
||||
| MATPROG | Float | Maths progress score |
|
||||
| PTRWM_EXP | Float | % meeting expected standard in reading, writing & maths |
|
||||
| PTREAD_EXP | Float | % meeting expected standard in reading |
|
||||
| PTWRIT_EXP | Float | % meeting expected standard in writing |
|
||||
| PTMAT_EXP | Float | % meeting expected standard in maths |
|
||||
|
||||
The application normalizes column names automatically and filters to only show Wandsworth and Merton schools.
|
||||
|
||||
## Technology Stack
|
||||
|
||||
- **Backend**: FastAPI (Python) - High-performance async API framework
|
||||
- **Frontend**: Vanilla JavaScript with Chart.js
|
||||
- **Styling**: Custom CSS with CSS variables for theming
|
||||
- **Data**: Pandas for CSV processing
|
||||
|
||||
## Design Philosophy
|
||||
|
||||
The UI features a warm, editorial design inspired by quality publications:
|
||||
- **Typography**: DM Sans for body text, Playfair Display for headings
|
||||
- **Color Palette**: Warm cream background with coral and teal accents
|
||||
- **Interactions**: Smooth animations and hover effects
|
||||
- **Charts**: Clean, readable data visualizations
|
||||
|
||||
## Development
|
||||
|
||||
```bash
|
||||
# Run with auto-reload
|
||||
python -m uvicorn backend.app:app --reload --port 8000
|
||||
|
||||
# Or run directly
|
||||
python backend/app.py
|
||||
```
|
||||
|
||||
## Coverage
|
||||
|
||||
This application is specifically designed for:
|
||||
|
||||
- **School Phase**: Primary schools only (Key Stage 2)
|
||||
- **Geographic Area**: Wandsworth and Merton (London boroughs)
|
||||
- **Time Period**: Last 5 years of data (2020-2024)
|
||||
|
||||
Note: 2021 data shows as unavailable because SATs were cancelled due to COVID-19.
|
||||
|
||||
## Data Source
|
||||
|
||||
Data is sourced from the UK Government's [Compare School Performance](https://www.compare-school-performance.service.gov.uk/) service, which provides official school performance data for England.
|
||||
|
||||
**Important**: When using real data, please comply with the [terms of use](https://www.compare-school-performance.service.gov.uk/download-data) and data protection regulations.
|
||||
|
||||
## Scheduled Jobs
|
||||
|
||||
### Geocoding Schools (Cron Job)
|
||||
|
||||
School postcodes are geocoded by a scheduled job, not on-demand. This improves performance and reduces API calls.
|
||||
|
||||
**Setup the cron job** (runs weekly on Sunday at 2am):
|
||||
|
||||
```bash
|
||||
# Edit crontab
|
||||
crontab -e
|
||||
|
||||
# Add this line (adjust paths as needed):
|
||||
0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1
|
||||
```
|
||||
|
||||
**Manual run:**
|
||||
```bash
|
||||
# Geocode only schools missing coordinates
|
||||
python scripts/geocode_schools.py
|
||||
|
||||
# Force re-geocode all schools
|
||||
python scripts/geocode_schools.py --force
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT License - feel free to use this project for educational purposes.
|
||||
|
||||
---
|
||||
|
||||
Built with ❤️ for Wandsworth & Merton families
|
||||
## Deployment
|
||||
|
||||
Work on a feature branch and open a PR. Merging to `main` builds images and
|
||||
deploys staging. Production promotion is a separate, human-triggered Gitea
|
||||
workflow. Use [DEPLOY.md](docs/DEPLOY.md) and the Portainer compose files as the
|
||||
operational references. The generic compose examples still reference `:latest`,
|
||||
which the current release workflow does not publish.
|
||||
+591
-52
@@ -6,6 +6,7 @@ Uses real data from UK Government Compare School Performance downloads.
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
import time
|
||||
from contextlib import asynccontextmanager
|
||||
from datetime import datetime, timezone
|
||||
from typing import Optional
|
||||
@@ -15,7 +16,7 @@ import pandas as pd
|
||||
from fastapi import FastAPI, HTTPException, Query, Request, Depends, Header
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.middleware.gzip import GZipMiddleware
|
||||
from fastapi.responses import FileResponse, Response
|
||||
from fastapi.responses import FileResponse, JSONResponse, Response
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from slowapi import Limiter, _rate_limit_exceeded_handler
|
||||
from slowapi.util import get_remote_address
|
||||
@@ -25,7 +26,8 @@ from starlette.middleware.base import BaseHTTPMiddleware
|
||||
import asyncio
|
||||
from .config import settings
|
||||
from .data_loader import (
|
||||
clear_cache,
|
||||
build_latest_school_data,
|
||||
load_school_data_as_dataframe,
|
||||
compute_benchmarks,
|
||||
load_school_data,
|
||||
load_latest_school_data,
|
||||
@@ -33,22 +35,26 @@ from .data_loader import (
|
||||
get_supplementary_data,
|
||||
get_supplementary_data_batch,
|
||||
search_schools_typesense,
|
||||
suggest_schools_typesense,
|
||||
)
|
||||
from .data_loader import get_data_info as get_db_info
|
||||
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||
from . import flags
|
||||
from .places import build_place_index, build_place_registry, places_for_urn
|
||||
from .schemas import METRIC_DEFINITIONS, PHASE_GROUPS, PHASE_ORDER, RANKING_COLUMNS, SCHOOL_COLUMNS
|
||||
from .school_groups import (
|
||||
FAITH_GROUPS,
|
||||
FAITH_KEYS,
|
||||
TYPE_GROUPS,
|
||||
faith_groups_for,
|
||||
type_group_for,
|
||||
type_group_key,
|
||||
)
|
||||
from .nearby_schools import select_nearby
|
||||
from .utils import clean_for_json, convert_to_native
|
||||
|
||||
# Values to exclude from filter dropdowns (empty strings, non-applicable labels)
|
||||
EXCLUDED_FILTER_VALUES = {"", "Not applicable", "Does not apply"}
|
||||
|
||||
# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they include.
|
||||
# All-through schools appear in both primary and secondary results.
|
||||
PHASE_GROUPS: dict[str, set[str]] = {
|
||||
"primary": {"primary", "middle deemed primary", "all-through"},
|
||||
"secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"},
|
||||
"all-through": {"all-through"},
|
||||
}
|
||||
|
||||
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
|
||||
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
|
||||
BASE_URL = "https://www.schoolcompare.co.uk"
|
||||
@@ -58,6 +64,16 @@ MAX_SLUG_LENGTH = 60
|
||||
# regenerate endpoint after a pipeline run.
|
||||
_sitemaps: dict[str, str] | None = None
|
||||
|
||||
# Built from the same DataFrame the sitemap uses, so places and sitemap can
|
||||
# never describe different corpora. Reset by the same admin endpoint.
|
||||
_place_registry: dict | None = None
|
||||
# Cached beside the registry, and invalidated by identity against it — see
|
||||
# get_place_index. Never cleared independently.
|
||||
_place_index: dict | None = None
|
||||
_place_index_source: dict | None = None
|
||||
|
||||
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
text = text.lower()
|
||||
@@ -101,7 +117,7 @@ def _has_publishable_data(row) -> bool:
|
||||
|
||||
|
||||
def _url_element(loc: str, lastmod: str | None = None) -> str:
|
||||
"""One <url> entry. No priority or changefreq — Google ignores both."""
|
||||
"""One <url> entry. No priority or changefreq, Google ignores both."""
|
||||
body = f"<loc>{loc}</loc>"
|
||||
if lastmod:
|
||||
body += f"<lastmod>{lastmod}</lastmod>"
|
||||
@@ -170,6 +186,32 @@ SITEMAP_CHUNK_SIZE = 10_000
|
||||
SITEMAP_CHILD_PREFIX = "/sitemaps"
|
||||
|
||||
|
||||
def get_place_registry() -> dict:
|
||||
"""The place registry, built once and cached for the process."""
|
||||
global _place_registry
|
||||
if _place_registry is None:
|
||||
_place_registry = build_place_registry(load_school_data())
|
||||
return _place_registry
|
||||
|
||||
|
||||
def get_place_index() -> dict:
|
||||
"""URN → its published places, cached against the registry it came from.
|
||||
|
||||
Invalidation is an identity check rather than a second flag to remember to
|
||||
clear. Anything that drops `_place_registry` — the tests all do — gets a
|
||||
fresh registry object here, which no longer matches the one the index was
|
||||
built from, so the index rebuilds with it. A separate `_place_index = None`
|
||||
would be one more thing to forget, and a stale reverse index is exactly the
|
||||
bug that would put links to another dataset's places on a school page.
|
||||
"""
|
||||
global _place_index, _place_index_source
|
||||
registry = get_place_registry()
|
||||
if _place_index is None or _place_index_source is not registry:
|
||||
_place_index = build_place_index(registry)
|
||||
_place_index_source = registry
|
||||
return _place_index
|
||||
|
||||
|
||||
def _urlset(rows: list[str]) -> str:
|
||||
return "\n".join([
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
@@ -179,9 +221,110 @@ def _urlset(rows: list[str]) -> str:
|
||||
])
|
||||
|
||||
|
||||
def build_sitemaps() -> dict[str, str]:
|
||||
def _place_url(place) -> str:
|
||||
"""The canonical path for a place. Two namespaces, per the spec.
|
||||
|
||||
Towns and localities share /schools/[place]; authorities take their own
|
||||
prefix because 67 town names collide with an authority name and neither
|
||||
set contains the other.
|
||||
"""
|
||||
if place.kind == "authority":
|
||||
return f"/schools/authority/{place.slug}"
|
||||
if place.kind == "outcode":
|
||||
return f"/schools/near/{place.slug}"
|
||||
return f"/schools/{place.slug}"
|
||||
|
||||
|
||||
def _places_payload(urn: int) -> list[dict]:
|
||||
"""The published places containing this school, as the school page needs
|
||||
them: a name to write in the link, a count so the anchor can say what it
|
||||
leads to, and the canonical path.
|
||||
|
||||
`phases` carries the phase variants this school actually appears on, which
|
||||
is usually one and is two for an all-through school — it is listed on both
|
||||
pages, so there is no tie to break.
|
||||
|
||||
Membership is read straight from the registry's own `phase_urns` rather
|
||||
than re-derived from the school's phase string. The registry is the one
|
||||
place that decides which phases a place publishes and who is on them;
|
||||
computing it a second time here is how a page comes to link a school to a
|
||||
phase page that does not list it, or to a route that does not exist. That
|
||||
is also why outcodes need no special case: they carry empty `phase_urns`,
|
||||
so they report no phase links on their own.
|
||||
"""
|
||||
payload = []
|
||||
for place in places_for_urn(get_place_index(), int(urn)):
|
||||
phases = [
|
||||
{
|
||||
"phase": phase,
|
||||
"count": len(phase_urns),
|
||||
"url": f"{_place_url(place)}/{phase}",
|
||||
}
|
||||
for phase, phase_urns in sorted(place.phase_urns.items())
|
||||
if int(urn) in phase_urns
|
||||
]
|
||||
payload.append({
|
||||
"kind": place.kind,
|
||||
"slug": place.slug,
|
||||
"name": place.name,
|
||||
"count": len(place.urns),
|
||||
"url": _place_url(place),
|
||||
"phases": phases,
|
||||
})
|
||||
return payload
|
||||
|
||||
|
||||
def _nearby_schools_payload(urn: int) -> list[dict]:
|
||||
"""The nearest eligible schools this page may offer, closest first.
|
||||
|
||||
Phase and reach are read from the school's own row inside select_nearby,
|
||||
so nothing here can hand it a phase that disagrees with the data.
|
||||
|
||||
Wrapped: a failure in selection must never 500 a page that is otherwise
|
||||
complete, which is the posture get_supplementary_data already takes. The
|
||||
section simply does not render.
|
||||
"""
|
||||
try:
|
||||
return select_nearby(load_latest_school_data(), int(urn))
|
||||
except Exception:
|
||||
import logging
|
||||
|
||||
logging.getLogger(__name__).exception(
|
||||
"Nearby schools selection failed for urn=%s", urn
|
||||
)
|
||||
return []
|
||||
|
||||
|
||||
def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
|
||||
"""A <url> per place, plus a phase variant wherever that phase clears the
|
||||
threshold on its own.
|
||||
|
||||
Phase is part of the query — "primary schools in beccles" — so each
|
||||
variant is its own indexable page. Submitting only the bare place URL left
|
||||
~950 of them reachable by nothing: absent from every sitemap, and not
|
||||
linked from the place page either.
|
||||
"""
|
||||
rows: list[str] = []
|
||||
if registry is None:
|
||||
registry = get_place_registry()
|
||||
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)):
|
||||
if p.kind not in kinds:
|
||||
continue
|
||||
rows.append(_url_element(BASE_URL + _place_url(p)))
|
||||
# Which phases a place publishes is the registry's decision alone —
|
||||
# outcodes report none, because the spec gives them no phase route.
|
||||
# Repeating that rule here was how the page and the sitemap came to
|
||||
# disagree about which URLs exist.
|
||||
for phase in ("primary", "secondary"):
|
||||
if p.publishes_phase(phase):
|
||||
rows.append(_url_element(f"{BASE_URL}{_place_url(p)}/{phase}"))
|
||||
return rows
|
||||
|
||||
|
||||
def build_sitemaps(df=None, registry=None) -> dict[str, str]:
|
||||
"""Build the sitemap index and every child, keyed by name."""
|
||||
df = load_school_data()
|
||||
if df is None:
|
||||
df = load_school_data()
|
||||
|
||||
children: dict[str, str] = {
|
||||
"static.xml": _urlset(
|
||||
@@ -196,6 +339,17 @@ def build_sitemaps() -> dict[str, str]:
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"schools-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# Separate children per family: Search Console reports coverage per
|
||||
# submitted sitemap, which is how the location layer's indexation is
|
||||
# measured apart from the school pages'.
|
||||
for label, kinds in (("places", ("town", "locality", "authority")),
|
||||
("outcodes", ("outcode",))):
|
||||
rows = _place_sitemap_rows(kinds, registry)
|
||||
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
|
||||
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
|
||||
for n, chunk in enumerate(chunks, start=1):
|
||||
children[f"{label}-{n}.xml"] = _urlset(chunk)
|
||||
|
||||
# On a sitemap index, lastmod means "when this sitemap file last changed",
|
||||
# so generation time is the correct value here — unlike on a <url>, where
|
||||
# it would be a claim about content we cannot support.
|
||||
@@ -227,12 +381,111 @@ def clean_filter_values(series: pd.Series) -> list[str]:
|
||||
)
|
||||
|
||||
|
||||
def order_phases(phases: list[str]) -> list[str]:
|
||||
"""Phases in the order a child meets them; any GIAS adds later follow, A-Z."""
|
||||
rank = {p: i for i, p in enumerate(PHASE_ORDER)}
|
||||
return sorted(phases, key=lambda p: (rank.get(p.lower(), len(rank)), p))
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# SECURITY MIDDLEWARE & HELPERS
|
||||
# =============================================================================
|
||||
|
||||
# Rate limiter
|
||||
limiter = Limiter(key_func=get_remote_address)
|
||||
def client_key(request: Request) -> str:
|
||||
"""The rate-limit bucket: the real caller, not the proxy in front of them.
|
||||
|
||||
`get_remote_address` reads request.client.host. In staging and production
|
||||
the backend has no published ports and sits on the internal network, so its
|
||||
only caller is the Next proxy — meaning every browser user on the site
|
||||
shared one bucket. Measured before this fix: 70 concurrent requests to
|
||||
/api/schools returned 60 OK and 10 refused.
|
||||
|
||||
CF-Connecting-IP first, because Cloudflare (in front of both environments)
|
||||
sets it on every origin request and *overwrites* any client-supplied value,
|
||||
which a parsed X-Forwarded-For chain does not guarantee. The XFF fallback is
|
||||
forgeable, but only by a caller already inside the Docker network, which is
|
||||
the one place nothing untrusted can reach.
|
||||
"""
|
||||
cf = request.headers.get("cf-connecting-ip")
|
||||
if cf:
|
||||
return cf.strip()
|
||||
xff = request.headers.get("x-forwarded-for")
|
||||
if xff:
|
||||
return xff.split(",")[0].strip()
|
||||
return get_remote_address(request)
|
||||
|
||||
|
||||
# Per-client limiter. Paired with the global ceiling below — the two do
|
||||
# different jobs and neither substitutes for the other.
|
||||
limiter = Limiter(key_func=client_key)
|
||||
|
||||
|
||||
# --- The ceiling no header can raise ----------------------------------------
|
||||
#
|
||||
# client_key trusts CF-Connecting-IP, and nothing in this process can tell an
|
||||
# edge-set header from an attacker-set one. That distinction can only be made
|
||||
# at Cloudflare, with Authenticated Origin Pulls or an origin firewall. A
|
||||
# caller reaching the origin directly could otherwise mint a fresh rate-limit
|
||||
# bucket per request and evade per-client limits entirely — which would make
|
||||
# correct keying a net regression against abuse, since the single shared bucket
|
||||
# it replaced at least capped everyone at 60/minute together.
|
||||
#
|
||||
# So per-client limits give fairness, and this gives the origin a hard total.
|
||||
# It does not make the header trustworthy; it bounds what trusting it can cost.
|
||||
# The header problem itself is closed at Cloudflare, not here.
|
||||
#
|
||||
# [window_start_monotonic, count], or None before the first request. A fixed
|
||||
# window is crude, which is right for a backstop: it has to be obviously
|
||||
# correct rather than fair.
|
||||
_global_window: Optional[list] = None
|
||||
|
||||
# The container healthcheck runs `curl http://localhost:80/api/data-info` from
|
||||
# inside the container. Starving it would fail the check, restart the
|
||||
# container, and turn a load spike into an outage loop — the ceiling exists to
|
||||
# protect the origin, not to kill it.
|
||||
_LOCAL_HOSTS = frozenset({"127.0.0.1", "::1", "localhost"})
|
||||
|
||||
|
||||
def exempt_from_ceiling(request: Request) -> bool:
|
||||
"""Whether the ceiling should ignore this request.
|
||||
|
||||
Its own function so the rule is testable without standing up a server —
|
||||
and so the healthcheck exemption is somewhere a reader can find it.
|
||||
"""
|
||||
if not request.url.path.startswith("/api/"):
|
||||
return True
|
||||
# The peer address, never the Host header: Host is set by the caller and
|
||||
# would hand every attacker an exemption.
|
||||
return (request.client.host if request.client else "") in _LOCAL_HOSTS
|
||||
|
||||
|
||||
class GlobalRateLimitMiddleware(BaseHTTPMiddleware):
|
||||
"""A cap on total /api/ traffic, independent of any client identity."""
|
||||
|
||||
async def dispatch(self, request: Request, call_next):
|
||||
global _global_window
|
||||
|
||||
if exempt_from_ceiling(request):
|
||||
return await call_next(request)
|
||||
|
||||
now = time.monotonic()
|
||||
# One event loop, and no await between the read and the write, so this
|
||||
# sequence is atomic without a lock.
|
||||
if _global_window is None or now - _global_window[0] >= 60:
|
||||
_global_window = [now, 0]
|
||||
_global_window[1] += 1
|
||||
|
||||
if _global_window[1] > settings.global_rate_limit_per_minute:
|
||||
return JSONResponse(
|
||||
# Distinguishable from slowapi's per-client 429: an operator
|
||||
# reading logs has to be able to tell "one noisy client" from
|
||||
# "the origin is saturated".
|
||||
{"detail": "The service is at capacity. Please retry shortly."},
|
||||
status_code=429,
|
||||
headers={"Retry-After":
|
||||
str(max(1, int(60 - (now - _global_window[0]))))},
|
||||
)
|
||||
return await call_next(request)
|
||||
|
||||
|
||||
class SecurityHeadersMiddleware(BaseHTTPMiddleware):
|
||||
@@ -290,6 +543,7 @@ CACHE_RULES: list[tuple[str, tuple[int, int, int]]] = [
|
||||
("/api/schools/", (300, 3600, 86400)), # /api/schools/{urn}
|
||||
("/api/rankings", (60, 600, 3600)),
|
||||
("/api/compare", (60, 600, 3600)),
|
||||
("/api/suggest", (60, 3600, 86400)), # autosuggest
|
||||
("/api/schools", (30, 300, 1800)), # search list
|
||||
]
|
||||
|
||||
@@ -367,7 +621,41 @@ def verify_admin_api_key(x_api_key: str = Header(None)) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def _with_whole_school_pupils(rows: pd.DataFrame, source: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Set total_pupils to the size of the school.
|
||||
|
||||
fact_performance's total_pupils is the cohort a year's results were
|
||||
measured on. For a secondary that is the GCSE year group alone (Burntwood:
|
||||
245 against 1,462 on roll). Cards, map popups and place rows label it
|
||||
"pupils", so they take the register's whole-school count instead, and
|
||||
nothing when the register has none.
|
||||
"""
|
||||
whole = source["gias_total_pupils"] if "gias_total_pupils" in source.columns else None
|
||||
return rows.assign(total_pupils=whole)
|
||||
|
||||
|
||||
def _with_type_group(rows: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Name each row's search-filter type group, or None for a type in no group.
|
||||
|
||||
The school page and the search rows print it ("State school") in place of
|
||||
the GIAS establishment type, and an independent school gets a Fee-paying
|
||||
flag from it.
|
||||
"""
|
||||
if "school_type" not in rows.columns:
|
||||
return rows
|
||||
return rows.assign(type_group=rows["school_type"].map(type_group_for))
|
||||
|
||||
|
||||
# Input validation helpers
|
||||
def _names_in_group(names: pd.Series, in_group) -> set:
|
||||
"""The distinct names in a column that a group predicate accepts.
|
||||
|
||||
Evaluated once per distinct name rather than per row, so a filter over
|
||||
every school costs a few dozen lookups.
|
||||
"""
|
||||
return {n for n in names.dropna().unique() if in_group(n)}
|
||||
|
||||
|
||||
def sanitize_search_input(value: Optional[str], max_length: int = 100) -> Optional[str]:
|
||||
"""Sanitize search input to prevent injection attacks."""
|
||||
if value is None:
|
||||
@@ -395,6 +683,7 @@ def validate_postcode(postcode: Optional[str]) -> Optional[str]:
|
||||
async def lifespan(app: FastAPI):
|
||||
"""Application lifespan - startup and shutdown events."""
|
||||
global _sitemaps
|
||||
flags.init()
|
||||
print("Loading school data from marts...")
|
||||
df = load_school_data()
|
||||
if df.empty:
|
||||
@@ -436,6 +725,10 @@ app.add_middleware(CacheAndETagMiddleware)
|
||||
app.add_middleware(SecurityHeadersMiddleware)
|
||||
app.add_middleware(RequestSizeLimitMiddleware)
|
||||
app.add_middleware(GZipMiddleware, minimum_size=512)
|
||||
# Added last, so it is outermost and refuses before anything downstream does
|
||||
# work. A ceiling that only applies after the expensive part has run is not a
|
||||
# ceiling.
|
||||
app.add_middleware(GlobalRateLimitMiddleware)
|
||||
|
||||
# CORS middleware - restricted for production
|
||||
app.add_middleware(
|
||||
@@ -473,6 +766,15 @@ async def get_config():
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/release")
|
||||
async def release_identity():
|
||||
import json
|
||||
from pathlib import Path
|
||||
path = Path(__file__).with_name("build-info.json")
|
||||
identity = json.loads(path.read_text()) if path.exists() else {"sha": "development", "build_id": "development"}
|
||||
return JSONResponse(identity, headers={"Cache-Control": "no-store"})
|
||||
|
||||
|
||||
@app.get("/api/schools")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_schools(
|
||||
@@ -482,7 +784,7 @@ async def get_schools(
|
||||
None, description="Filter by local authority", max_length=100
|
||||
),
|
||||
school_type: Optional[str] = Query(None, description="Filter by school type", max_length=100),
|
||||
phase: Optional[str] = Query(None, description="Filter by phase: primary, secondary, all-through", max_length=50),
|
||||
phase: Optional[str] = Query(None, description="Filter by phase: primary or secondary (grouped), or any GIAS phase name (exact)", max_length=50),
|
||||
postcode: Optional[str] = Query(None, description="Search near postcode", max_length=10),
|
||||
radius: float = Query(5.0, ge=0.1, le=5, description="Search radius in miles"),
|
||||
page: int = Query(1, ge=1, le=1000, description="Page number"),
|
||||
@@ -490,6 +792,7 @@ async def get_schools(
|
||||
gender: Optional[str] = Query(None, description="Filter by gender (Mixed/Boys/Girls)", max_length=50),
|
||||
admissions_policy: Optional[str] = Query(None, description="Filter by admissions policy", max_length=100),
|
||||
has_sixth_form: Optional[str] = Query(None, description="Filter by sixth form presence: yes/no", max_length=3),
|
||||
faith: Optional[str] = Query(None, description="Filter by faith group key", max_length=40),
|
||||
):
|
||||
"""
|
||||
Get list of schools with pagination.
|
||||
@@ -502,6 +805,7 @@ async def get_schools(
|
||||
local_authority = sanitize_search_input(local_authority)
|
||||
school_type = sanitize_search_input(school_type)
|
||||
phase = sanitize_search_input(phase)
|
||||
faith = sanitize_search_input(faith)
|
||||
postcode = validate_postcode(postcode)
|
||||
|
||||
# Load the pre-computed latest-year snapshot (cached after first request / startup).
|
||||
@@ -509,7 +813,7 @@ async def get_schools(
|
||||
df_latest = load_latest_school_data()
|
||||
|
||||
if df_latest.empty:
|
||||
return {"schools": [], "total": 0, "page": page, "page_size": 0}
|
||||
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
|
||||
|
||||
# Use configured default if not specified
|
||||
if page_size is None:
|
||||
@@ -517,11 +821,13 @@ async def get_schools(
|
||||
|
||||
# Phase filter — uses PHASE_GROUPS so all-through/middle schools appear
|
||||
# in the correct phase(s) rather than being invisible to both filters.
|
||||
# Any other GIAS phase (nursery, 16 plus, middle deemed ...) is an exact
|
||||
# match. It must never fall through to no filter: the search page offers
|
||||
# every phase, and "Nursery" used to return the whole result set.
|
||||
if phase:
|
||||
phase_lower = phase.lower().replace("_", "-")
|
||||
allowed = PHASE_GROUPS.get(phase_lower)
|
||||
if allowed:
|
||||
df_latest = df_latest[df_latest["phase"].str.lower().isin(allowed)]
|
||||
allowed = PHASE_GROUPS.get(phase_lower, {phase_lower})
|
||||
df_latest = df_latest[df_latest["phase"].fillna("").str.lower().isin(allowed)]
|
||||
|
||||
# Secondary-specific filters (after phase filter)
|
||||
if gender:
|
||||
@@ -540,6 +846,22 @@ async def get_schools(
|
||||
flag = df_latest["age_range"].str.contains("18", na=False)
|
||||
df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag]
|
||||
|
||||
# Faith group (backend/school_groups.py). A joint school is in every faith
|
||||
# its label names; a missing religious character is "none". An unknown key
|
||||
# matches nothing rather than being ignored, so a typo cannot show all.
|
||||
if faith:
|
||||
faith_key = faith.lower()
|
||||
if faith_key in FAITH_KEYS and "religious_denomination" in df_latest.columns:
|
||||
column = df_latest["religious_denomination"]
|
||||
matches = column.isin(_names_in_group(column, lambda n: faith_key in faith_groups_for(n)))
|
||||
# _names_in_group skips missing names; a missing religious
|
||||
# character is "No religious character".
|
||||
if faith_key == "none":
|
||||
matches = matches | column.isna()
|
||||
df_latest = df_latest[matches]
|
||||
else:
|
||||
df_latest = df_latest.iloc[0:0]
|
||||
|
||||
# Include key result metrics for display on cards
|
||||
location_cols = ["latitude", "longitude"]
|
||||
result_cols = [
|
||||
@@ -562,7 +884,7 @@ async def get_schools(
|
||||
if c in df_latest.columns
|
||||
]
|
||||
# fact_performance guarantees one row per (urn, year); df_latest has one row per urn.
|
||||
schools_df = df_latest[available_cols]
|
||||
schools_df = _with_whole_school_pupils(df_latest[available_cols], df_latest)
|
||||
|
||||
# Location-based search (uses pre-geocoded data from database)
|
||||
search_coords = None
|
||||
@@ -608,8 +930,8 @@ async def get_schools(
|
||||
|
||||
# Apply filters
|
||||
if search:
|
||||
ts_urns = search_schools_typesense(search)
|
||||
if ts_urns:
|
||||
ts_urns = await asyncio.to_thread(search_schools_typesense, search)
|
||||
if ts_urns is not None:
|
||||
urn_order = {urn: i for i, urn in enumerate(ts_urns)}
|
||||
schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy()
|
||||
schools_df["_ts_rank"] = schools_df["urn"].map(urn_order)
|
||||
@@ -617,9 +939,9 @@ async def get_schools(
|
||||
else:
|
||||
# Fallback: Typesense unavailable, use substring match
|
||||
search_lower = search.lower()
|
||||
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False)
|
||||
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False, regex=False)
|
||||
if "address" in schools_df.columns:
|
||||
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False)
|
||||
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False, regex=False)
|
||||
schools_df = schools_df[mask]
|
||||
|
||||
if local_authority:
|
||||
@@ -627,10 +949,17 @@ async def get_schools(
|
||||
schools_df["local_authority"].str.lower() == local_authority.lower()
|
||||
]
|
||||
|
||||
# A type group key (backend/school_groups.py), old keys included, or for
|
||||
# an old link a raw GIAS type label, matched exactly as before.
|
||||
if school_type:
|
||||
schools_df = schools_df[
|
||||
schools_df["school_type"].str.lower() == school_type.lower()
|
||||
]
|
||||
type_key = type_group_key(school_type)
|
||||
if type_key:
|
||||
column = schools_df["school_type"]
|
||||
schools_df = schools_df[
|
||||
column.isin(_names_in_group(column, lambda n: type_group_for(n) == type_key))
|
||||
]
|
||||
else:
|
||||
schools_df = schools_df[schools_df["school_type"].str.lower() == school_type.lower()]
|
||||
|
||||
# Compute result-scoped filter values (before pagination).
|
||||
# Gender and admissions are secondary-only filters — scope them to schools
|
||||
@@ -639,7 +968,7 @@ async def get_schools(
|
||||
result_filters = {
|
||||
"local_authorities": clean_filter_values(schools_df["local_authority"]) if "local_authority" in schools_df.columns else [],
|
||||
"school_types": clean_filter_values(schools_df["school_type"]) if "school_type" in schools_df.columns else [],
|
||||
"phases": clean_filter_values(schools_df["phase"]) if "phase" in schools_df.columns else [],
|
||||
"phases": order_phases(clean_filter_values(schools_df["phase"])) if "phase" in schools_df.columns else [],
|
||||
"genders": clean_filter_values(schools_df.loc[_sec_mask, "gender"]) if "gender" in schools_df.columns and _sec_mask.any() else [],
|
||||
"admissions_policies": clean_filter_values(schools_df.loc[_sec_mask, "admissions_policy"]) if "admissions_policy" in schools_df.columns and _sec_mask.any() else [],
|
||||
}
|
||||
@@ -648,7 +977,7 @@ async def get_schools(
|
||||
total = len(schools_df)
|
||||
start_idx = (page - 1) * page_size
|
||||
end_idx = start_idx + page_size
|
||||
schools_df = schools_df.iloc[start_idx:end_idx]
|
||||
schools_df = _with_type_group(schools_df.iloc[start_idx:end_idx])
|
||||
|
||||
return {
|
||||
"schools": clean_for_json(schools_df),
|
||||
@@ -678,7 +1007,7 @@ async def get_school_details(request: Request, urn: int):
|
||||
df = load_school_data()
|
||||
|
||||
if df.empty:
|
||||
raise HTTPException(status_code=404, detail="No data available")
|
||||
raise HTTPException(status_code=503, detail="School data temporarily unavailable")
|
||||
|
||||
school_data = df[df["urn"] == urn]
|
||||
|
||||
@@ -712,6 +1041,7 @@ async def get_school_details(request: Request, urn: int):
|
||||
"school_name": latest.get("school_name", ""),
|
||||
"local_authority": latest.get("local_authority", ""),
|
||||
"school_type": latest.get("school_type", ""),
|
||||
"type_group": type_group_for(latest.get("school_type")),
|
||||
"address": latest.get("address", ""),
|
||||
"religious_denomination": latest.get("religious_denomination", ""),
|
||||
"age_range": latest.get("age_range", ""),
|
||||
@@ -736,17 +1066,35 @@ async def get_school_details(request: Request, urn: int):
|
||||
|
||||
return {
|
||||
"school_info": school_info,
|
||||
# Where this school sits in the location layer, for the page's link
|
||||
# module and breadcrumb. Derived from the same registry the place
|
||||
# pages and the sitemap use, so a link is never offered for a page
|
||||
# that does not exist. Empty is a valid answer: a school whose town
|
||||
# and authority both fall below the publish threshold has nowhere to
|
||||
# point, and the page renders without the module.
|
||||
"places": _places_payload(urn),
|
||||
# The nearest eligible schools, closest first. Always present on a
|
||||
# build with this code; the frontend treats absent and empty
|
||||
# identically, which is what lets the two images deploy independently.
|
||||
"nearby_schools": _nearby_schools_payload(urn),
|
||||
"yearly_data": clean_for_json(school_data),
|
||||
# Supplementary data (null if not yet populated by Kestra)
|
||||
"ofsted": supplementary.get("ofsted"),
|
||||
"census": supplementary.get("census"),
|
||||
"admissions": supplementary.get("admissions"),
|
||||
"admissions_history": supplementary.get("admissions_history") or [],
|
||||
"admission_distance": supplementary.get("admission_distance"),
|
||||
# Behind a flag, and withheld at the source rather than rendered-but-
|
||||
# hidden: this endpoint is public and unauthenticated, so a field left
|
||||
# in the payload is a published field. The key is absent, not null —
|
||||
# null would state that this school has no cut-off, which is a
|
||||
# different claim from "we are not publishing cut-offs".
|
||||
**({"admission_distance": supplementary.get("admission_distance")}
|
||||
if flags.is_enabled("admission_distance") else {}),
|
||||
"sen_detail": supplementary.get("sen_detail"),
|
||||
"phonics": supplementary.get("phonics"),
|
||||
"deprivation": supplementary.get("deprivation"),
|
||||
"finance": supplementary.get("finance"),
|
||||
"destinations": supplementary.get("destinations"),
|
||||
}
|
||||
|
||||
|
||||
@@ -887,15 +1235,29 @@ async def get_filter_options(request: Request):
|
||||
"local_authorities": [],
|
||||
"school_types": [],
|
||||
"years": [],
|
||||
"school_type_groups": [],
|
||||
"faiths": [],
|
||||
}
|
||||
|
||||
# Phases: return values from data, ordered sensibly
|
||||
phases = clean_filter_values(df["phase"]) if "phase" in df.columns else []
|
||||
# Phases: the values in the data, in the order a child meets them
|
||||
phases = order_phases(clean_filter_values(df["phase"])) if "phase" in df.columns else []
|
||||
|
||||
secondary_df = df[df["attainment_8_score"].notna()] if "attainment_8_score" in df.columns else df.iloc[0:0]
|
||||
genders = clean_filter_values(secondary_df["gender"]) if "gender" in secondary_df.columns else []
|
||||
admissions_policies = clean_filter_values(secondary_df["admissions_policy"]) if "admissions_policy" in secondary_df.columns else []
|
||||
|
||||
def offered(groups, present):
|
||||
return [{"value": key, "label": label} for key, label, _ in groups if key in present]
|
||||
|
||||
type_groups_present = (
|
||||
{type_group_for(n) for n in df["school_type"].dropna().unique()} - {None}
|
||||
if "school_type" in df.columns else set()
|
||||
)
|
||||
faiths_present = (
|
||||
{f for n in df["religious_denomination"].unique() for f in faith_groups_for(n)}
|
||||
if "religious_denomination" in df.columns else set()
|
||||
)
|
||||
|
||||
return {
|
||||
"local_authorities": clean_filter_values(df["local_authority"]) if "local_authority" in df.columns else [],
|
||||
"school_types": clean_filter_values(df["school_type"]) if "school_type" in df.columns else [],
|
||||
@@ -903,6 +1265,8 @@ async def get_filter_options(request: Request):
|
||||
"phases": phases,
|
||||
"genders": genders,
|
||||
"admissions_policies": admissions_policies,
|
||||
"school_type_groups": offered(TYPE_GROUPS, type_groups_present),
|
||||
"faiths": offered(FAITH_GROUPS, faiths_present),
|
||||
}
|
||||
|
||||
|
||||
@@ -1117,6 +1481,145 @@ async def get_rankings(
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/places")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def list_places(request: Request):
|
||||
"""Every published place. The sitemap and the link modules read this."""
|
||||
registry = get_place_registry()
|
||||
return {"places": [
|
||||
{"kind": p.kind, "slug": p.slug, "name": p.name, "count": len(p.urns)}
|
||||
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug))
|
||||
]}
|
||||
|
||||
|
||||
@app.get("/api/places/{kind}/{slug}")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_place(request: Request, kind: str, slug: str,
|
||||
phase: Optional[str] = None):
|
||||
"""One place: its schools ranked, and its local averages."""
|
||||
if kind not in VALID_PLACE_KINDS:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
registry = get_place_registry()
|
||||
place = registry.get(f"{kind}:{slug}")
|
||||
if place is None:
|
||||
raise HTTPException(status_code=404, detail="No such place")
|
||||
|
||||
df = load_latest_school_data()
|
||||
rows = df[df["urn"].isin(place.urns)]
|
||||
|
||||
if phase:
|
||||
wanted = PHASE_GROUPS.get(phase.lower())
|
||||
if wanted and "phase" in rows.columns:
|
||||
rows = rows[rows["phase"].fillna("").str.lower().isin(wanted)]
|
||||
|
||||
# The metric the page shows, and averages.
|
||||
metric = "attainment_8_score" if phase == "secondary" else "rwm_expected_pct"
|
||||
|
||||
# Alphabetical, not by score. A place page is read by someone looking for
|
||||
# a school they can name, and scanning for it is what the order should
|
||||
# serve. /rankings is where the league-table ordering lives, and it keeps
|
||||
# sorting by metric.
|
||||
if "school_name" in rows.columns:
|
||||
rows = rows.sort_values("school_name", key=lambda c: c.str.lower())
|
||||
|
||||
averages = {
|
||||
m: (None if m not in rows.columns or rows[m].dropna().empty
|
||||
else float(rows[m].dropna().mean()))
|
||||
for m in ("rwm_expected_pct", "attainment_8_score")
|
||||
}
|
||||
|
||||
# dict.fromkeys, not a list: SCHOOL_COLUMNS already ends with latitude and
|
||||
# longitude, so concatenating them again selected each twice and pandas
|
||||
# dropped one of every duplicated pair with a "columns are not unique"
|
||||
# warning. Ordered de-duplication keeps the column order and the warning
|
||||
# cannot come back.
|
||||
#
|
||||
# parliamentary_constituency is not in SCHOOL_COLUMNS and the place table
|
||||
# shows it. The `in rows.columns` guard is what keeps a mart the pipeline
|
||||
# has not rebuilt working: it and nursery_provision are the optional GIAS
|
||||
# columns data_loader degrades to NULL.
|
||||
cols = [c for c in dict.fromkeys(
|
||||
SCHOOL_COLUMNS + ["latitude", "longitude", "phase",
|
||||
"parliamentary_constituency",
|
||||
"rwm_expected_pct", "attainment_8_score",
|
||||
"total_pupils"])
|
||||
if c in rows.columns]
|
||||
|
||||
return {
|
||||
"place": {"kind": place.kind, "slug": place.slug, "name": place.name,
|
||||
"count": len(place.urns),
|
||||
"parent_authority": place.parent_authority,
|
||||
# Every authority the place meaningfully sits in. SW19 is
|
||||
# mostly Merton but partly Wandsworth; naming one asserts
|
||||
# something false.
|
||||
#
|
||||
# The slug is null where that authority has no page of its
|
||||
# own: City of London and the Isles of Scilly hold fewer
|
||||
# schools than the threshold. Naming them is still right;
|
||||
# linking them would be a 404.
|
||||
"authorities": [
|
||||
{"name": name,
|
||||
"slug": (_slugify(name)
|
||||
if f"authority:{_slugify(name)}" in registry
|
||||
else None),
|
||||
"count": n}
|
||||
for name, n in place.authorities
|
||||
],
|
||||
# Only phases that clear the threshold, so the page links
|
||||
# variants that exist rather than 404s.
|
||||
"phases": [ph for ph in ("primary", "secondary")
|
||||
if place.publishes_phase(ph)]},
|
||||
"schools": clean_for_json(
|
||||
_with_type_group(_with_whole_school_pupils(rows[cols], rows))),
|
||||
"averages": averages,
|
||||
}
|
||||
|
||||
|
||||
# Two characters. One is not a query — it matches thousands of schools and the
|
||||
# response is useless, so it is not worth a round trip.
|
||||
SUGGEST_MIN_QUERY = 2
|
||||
|
||||
|
||||
@app.get("/api/suggest")
|
||||
@limiter.limit("120/minute")
|
||||
async def suggest_schools(
|
||||
request: Request,
|
||||
q: str = Query("", max_length=100),
|
||||
limit: int = Query(8, ge=1, le=20),
|
||||
):
|
||||
"""School name suggestions, from Typesense alone.
|
||||
|
||||
Deliberately not a mode of /api/schools: that path filters and sorts the
|
||||
full in-memory DataFrame, which is far too expensive to run per keystroke.
|
||||
|
||||
Nothing here returns an error for ordinary input. A short query, no
|
||||
matches, or Typesense being unreachable are all 200 with an empty list —
|
||||
a dropdown that quietly does not appear is the right failure for a
|
||||
keystroke path, and there is no DataFrame fallback because the 25,000-row
|
||||
substring scan is precisely what this endpoint exists to avoid.
|
||||
|
||||
120/minute rather than the default 60: a 200 ms debounce makes typing
|
||||
legitimately bursty.
|
||||
"""
|
||||
query = q.strip()
|
||||
if len(query) < SUGGEST_MIN_QUERY:
|
||||
return {"suggestions": []}
|
||||
return {"suggestions": suggest_schools_typesense(query, limit)}
|
||||
|
||||
|
||||
@app.get("/api/flags")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_feature_flags(request: Request):
|
||||
"""Every declared flag and its current value.
|
||||
|
||||
Internal only. The Next proxy denies this path, because the response names
|
||||
every unreleased feature the codebase knows about — which is exactly what
|
||||
shipping dark is meant to keep quiet.
|
||||
"""
|
||||
return flags.all_flags()
|
||||
|
||||
|
||||
@app.get("/api/data-info")
|
||||
@limiter.limit(f"{settings.rate_limit_per_minute}/minute")
|
||||
async def get_data_info(request: Request):
|
||||
@@ -1162,20 +1665,51 @@ async def get_data_info(request: Request):
|
||||
}
|
||||
|
||||
|
||||
_publication_lock = asyncio.Lock()
|
||||
|
||||
|
||||
def _prepare_publication(df):
|
||||
if df.empty:
|
||||
raise ValueError("Refusing to publish an empty school dataset")
|
||||
if not {"urn", "year", "school_name"}.issubset(df.columns):
|
||||
raise ValueError("School dataset is missing required columns")
|
||||
if df["urn"].isna().any() or df.duplicated(["urn", "year"]).any():
|
||||
raise ValueError("School dataset has missing URNs or duplicate school years")
|
||||
latest = build_latest_school_data(df)
|
||||
registry = build_place_registry(df)
|
||||
index = build_place_index(registry)
|
||||
sitemaps = build_sitemaps(df, registry)
|
||||
return df, latest, registry, index, sitemaps
|
||||
|
||||
|
||||
def _publish(prepared):
|
||||
# Called on the event loop with no await: routes cannot observe half a swap.
|
||||
# The application currently runs one worker; replicas require coordination.
|
||||
from . import data_loader
|
||||
global _place_registry, _place_index, _place_index_source, _sitemaps
|
||||
df, latest, registry, index, sitemaps = prepared
|
||||
data_loader._df_cache = df
|
||||
data_loader._df_latest_cache = latest
|
||||
_place_registry = registry
|
||||
_place_index = index
|
||||
_place_index_source = registry
|
||||
_sitemaps = sitemaps
|
||||
|
||||
|
||||
@app.post("/api/admin/reload")
|
||||
@limiter.limit("5/minute")
|
||||
async def reload_data(
|
||||
request: Request,
|
||||
_: bool = Depends(verify_admin_api_key)
|
||||
):
|
||||
"""
|
||||
Admin endpoint to force data reload (useful after data updates).
|
||||
Requires X-API-Key header with valid admin API key.
|
||||
"""
|
||||
clear_cache()
|
||||
await asyncio.to_thread(load_school_data)
|
||||
await asyncio.to_thread(load_latest_school_data)
|
||||
return {"status": "reloaded"}
|
||||
async def reload_data(request: Request, _: bool = Depends(verify_admin_api_key)):
|
||||
"""Validate a complete replacement before publishing it; retain data on failure."""
|
||||
async with _publication_lock:
|
||||
try:
|
||||
df = await asyncio.to_thread(load_school_data_as_dataframe)
|
||||
prepared = await asyncio.to_thread(_prepare_publication, df)
|
||||
except Exception as exc:
|
||||
import logging
|
||||
logging.getLogger(__name__).exception("Dataset reload failed")
|
||||
raise HTTPException(status_code=503, detail="Dataset reload failed; previous data retained") from exc
|
||||
_publish(prepared)
|
||||
return {"status": "reloaded", "schools": len(prepared[1])}
|
||||
|
||||
|
||||
|
||||
@@ -1227,11 +1761,16 @@ async def regenerate_sitemap(
|
||||
request: Request,
|
||||
_: bool = Depends(verify_admin_api_key),
|
||||
):
|
||||
"""Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
|
||||
global _sitemaps
|
||||
_sitemaps = build_sitemaps()
|
||||
n = sum(x.count("<url>") for x in _sitemaps.values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
|
||||
"""Rebuild derived publication data without clearing the live registry."""
|
||||
async with _publication_lock:
|
||||
try:
|
||||
prepared = await asyncio.to_thread(_prepare_publication, load_school_data())
|
||||
except Exception as exc:
|
||||
raise HTTPException(status_code=503, detail="Sitemap rebuild failed; previous data retained") from exc
|
||||
_publish(prepared)
|
||||
n = sum(x.count("<url>") for x in prepared[4].values())
|
||||
return {"status": "ok", "urls": n, "sitemaps": len(prepared[4])}
|
||||
|
||||
|
||||
|
||||
# Mount static files directly (must be after all routes to avoid catching API calls)
|
||||
|
||||
@@ -35,6 +35,11 @@ class Settings(BaseSettings):
|
||||
# Security
|
||||
admin_api_key: str = Field(default_factory=lambda: secrets.token_urlsafe(32))
|
||||
rate_limit_per_minute: int = 60 # Requests per minute per IP
|
||||
# A ceiling on total /api/ traffic, independent of any client identity.
|
||||
# client_key trusts headers only Cloudflare can vouch for, so a caller
|
||||
# reaching the origin directly could otherwise mint a fresh bucket per
|
||||
# request. See GlobalRateLimitMiddleware in backend/app.py.
|
||||
global_rate_limit_per_minute: int = 3000
|
||||
rate_limit_burst: int = 10 # Allow burst of requests
|
||||
max_request_size: int = 1024 * 1024 # 1MB max request size
|
||||
|
||||
@@ -42,6 +47,16 @@ class Settings(BaseSettings):
|
||||
typesense_url: str = "http://localhost:8108"
|
||||
typesense_api_key: str = ""
|
||||
|
||||
# Feature flags (Unleash). An empty unleash_url disables flags entirely and
|
||||
# every flag evaluates False — the correct behaviour for local development
|
||||
# and CI, and the reason no test needs a running Unleash.
|
||||
unleash_url: str = ""
|
||||
unleash_api_token: str = ""
|
||||
unleash_app_name: str = "schoolcompare-backend"
|
||||
# On a named volume, so a restart during an Unleash outage keeps
|
||||
# last-known state instead of reverting a released feature to dark.
|
||||
unleash_cache_directory: str = "/app/.unleash"
|
||||
|
||||
# Analytics
|
||||
ga_measurement_id: Optional[str] = "G-J0PCVT14NY" # Google Analytics 4 Measurement ID
|
||||
|
||||
|
||||
+347
-18
@@ -20,6 +20,7 @@ from .models import (
|
||||
DimSchool, DimLocation, KS2Performance,
|
||||
FactOfstedInspection, FactAdmissions, FactAdmissionDistance,
|
||||
FactDeprivation, FactFinance, FactPupilCharacteristics,
|
||||
FactKs4Destinations, FactKs5Destinations,
|
||||
)
|
||||
from .ofsted_codes import ofsted_page_url, report_card_labels
|
||||
from .schemas import SCHOOL_TYPE_MAP
|
||||
@@ -83,22 +84,111 @@ def _get_typesense_client():
|
||||
return None
|
||||
|
||||
|
||||
def search_schools_typesense(query: str, limit: int = 250) -> List[int]:
|
||||
"""Search Typesense. Returns URNs in relevance order, or [] if unavailable."""
|
||||
SEARCH_PAGE_SIZE = 250
|
||||
# Search results are filtered again by the API (authority, phase, postcode,
|
||||
# etc.), so one page is too small for scoped searches. Keep the candidate set
|
||||
# bounded, though: a broad query must not turn into an unbounded sequence of
|
||||
# Typesense requests. Four pages is enough to preserve useful scoped matches
|
||||
# while putting a hard ceiling on latency and upstream load.
|
||||
SEARCH_MAX_CANDIDATES = 1_000
|
||||
|
||||
|
||||
def search_schools_typesense(query: str) -> Optional[List[int]]:
|
||||
"""Return a bounded set of matching URNs in relevance order.
|
||||
|
||||
``None`` means Typesense is unavailable; ``[]`` is a valid zero-match
|
||||
result. The API applies its remaining filters after this search, so the
|
||||
first few pages are fetched rather than only the first page. Once the
|
||||
candidate ceiling is reached, the relevance-ordered prefix is returned on
|
||||
purpose; fetching every match would make common or adversarial queries
|
||||
unbounded.
|
||||
"""
|
||||
client = _get_typesense_client()
|
||||
if client is None:
|
||||
return None
|
||||
urns: list[int] = []
|
||||
fetched = 0
|
||||
try:
|
||||
page = 1
|
||||
while fetched < SEARCH_MAX_CANDIDATES:
|
||||
page_size = min(SEARCH_PAGE_SIZE, SEARCH_MAX_CANDIDATES - fetched)
|
||||
result = client.collections["schools"].documents.search({
|
||||
"q": query,
|
||||
"query_by": "school_name,local_authority,postcode",
|
||||
"per_page": page_size,
|
||||
"page": page,
|
||||
"typo_tokens_threshold": 1,
|
||||
})
|
||||
hits = result.get("hits", [])
|
||||
urns.extend(int(h["document"]["urn"]) for h in hits)
|
||||
fetched += len(hits)
|
||||
if fetched >= result.get("found", fetched):
|
||||
return list(dict.fromkeys(urns))
|
||||
if not hits:
|
||||
raise ValueError("Search pagination ended before all matches arrived")
|
||||
page += 1
|
||||
logging.getLogger(__name__).info(
|
||||
"Typesense search capped at %d candidates for query %r",
|
||||
SEARCH_MAX_CANDIDATES,
|
||||
query,
|
||||
)
|
||||
return list(dict.fromkeys(urns))
|
||||
except Exception:
|
||||
logging.getLogger(__name__).exception("School search unavailable")
|
||||
return None
|
||||
|
||||
|
||||
# The most a public endpoint will return in one response.
|
||||
SUGGEST_MAX_LIMIT = 20
|
||||
|
||||
# Fields a suggestion row carries, and the default when the document omits an
|
||||
# optional one. phase and school_type are optional in the Typesense schema.
|
||||
_SUGGEST_FIELDS = ("school_name", "local_authority", "postcode",
|
||||
"phase", "school_type")
|
||||
|
||||
|
||||
def suggest_schools_typesense(query: str, limit: int = 8) -> List[dict]:
|
||||
"""Autosuggest rows straight from Typesense. Never raises.
|
||||
|
||||
Returns documents rather than URNs, unlike search_schools_typesense, so the
|
||||
caller needs no DataFrame. Every field below is already in the index — see
|
||||
pipeline/scripts/sync_typesense.py — which is what makes this cheap enough
|
||||
to run per keystroke.
|
||||
"""
|
||||
client = _get_typesense_client()
|
||||
if client is None:
|
||||
return []
|
||||
try:
|
||||
result = client.collections["schools"].documents.search({
|
||||
"q": query,
|
||||
"query_by": "school_name,local_authority,postcode",
|
||||
"per_page": min(limit, 250),
|
||||
"query_by": "school_name,local_authority",
|
||||
"per_page": max(1, min(limit, SUGGEST_MAX_LIMIT)),
|
||||
"typo_tokens_threshold": 1,
|
||||
})
|
||||
return [int(h["document"]["urn"]) for h in result.get("hits", [])]
|
||||
except Exception:
|
||||
# A dropdown that quietly stops appearing is the right failure here.
|
||||
return []
|
||||
|
||||
rows = []
|
||||
for hit in result.get("hits", []) or []:
|
||||
doc = (hit or {}).get("document") or {}
|
||||
try:
|
||||
urn = int(doc["urn"])
|
||||
except (KeyError, TypeError, ValueError):
|
||||
# Skip the row, keep the rest. Typesense declares urn as int32 so
|
||||
# this should be unreachable, but the index is a separate system
|
||||
# that something other than this code can reindex — and "never
|
||||
# raises" is a promise the keystroke path actually depends on.
|
||||
# Dropping one malformed document is right; blanking the whole
|
||||
# dropdown, or serving a suggestion pointing at /school/0, is not.
|
||||
logging.getLogger(__name__).warning(
|
||||
"skipping malformed suggestion document: %r", doc)
|
||||
continue
|
||||
row = {"urn": urn}
|
||||
row.update({f: str(doc.get(f, "") or "") for f in _SUGGEST_FIELDS})
|
||||
rows.append(row)
|
||||
return rows
|
||||
|
||||
|
||||
def normalize_school_type(school_type: Optional[str]) -> Optional[str]:
|
||||
"""Convert cryptic school type codes to user-friendly names."""
|
||||
@@ -135,16 +225,6 @@ def geocode_single_postcode(postcode: str) -> Optional[Tuple[float, float]]:
|
||||
return None
|
||||
|
||||
|
||||
def haversine_distance(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
|
||||
"""Calculate great-circle distance between two points (miles)."""
|
||||
from math import radians, cos, sin, asin, sqrt
|
||||
lat1, lon1, lat2, lon2 = map(radians, [lat1, lon1, lat2, lon2])
|
||||
dlat = lat2 - lat1
|
||||
dlon = lon2 - lon1
|
||||
a = sin(dlat / 2) ** 2 + cos(lat1) * cos(lat2) * sin(dlon / 2) ** 2
|
||||
return 2 * asin(sqrt(a)) * 3956
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# MAIN DATA LOAD — joins dim_school + dim_location + fact_performance
|
||||
# fact_performance is a merged KS2+KS4 table (one row per URN per year).
|
||||
@@ -459,7 +539,12 @@ def load_latest_school_data() -> pd.DataFrame:
|
||||
if _df_latest_cache is not None:
|
||||
return _df_latest_cache
|
||||
|
||||
df = load_school_data()
|
||||
_df_latest_cache = build_latest_school_data(load_school_data())
|
||||
return _df_latest_cache
|
||||
|
||||
|
||||
def build_latest_school_data(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Build a replacement snapshot without mutating the published caches."""
|
||||
if df.empty:
|
||||
return df
|
||||
|
||||
@@ -492,8 +577,7 @@ def load_latest_school_data() -> pd.DataFrame:
|
||||
df_latest = pd.concat([df_latest, df_no_perf], ignore_index=True)
|
||||
|
||||
print(f"Latest-snapshot cache built: {len(df_latest)} schools")
|
||||
_df_latest_cache = df_latest
|
||||
return _df_latest_cache
|
||||
return df_latest
|
||||
|
||||
|
||||
def clear_cache():
|
||||
@@ -764,6 +848,218 @@ def _finance_dict(f) -> dict:
|
||||
}
|
||||
|
||||
|
||||
# Destination measures that are totals DfE published itself, rather than one of
|
||||
# the categories that partition the cohort.
|
||||
_AGGREGATE_MEASURES = {"agg_sustained_education", "agg_sustained_all"}
|
||||
|
||||
|
||||
def _format_cohort_year(year) -> str | None:
|
||||
"""202223 -> '2022/23'.
|
||||
|
||||
The section has to date its own cohort. Destination measures run about two
|
||||
GCSE years behind the results shown above them on the same page, so an
|
||||
undated figure reads as stale data rather than as a different question.
|
||||
"""
|
||||
if not year:
|
||||
return None
|
||||
text = str(year)
|
||||
if len(text) == 6:
|
||||
return f"{text[:4]}/{text[4:6]}"
|
||||
if len(text) == 8:
|
||||
return f"{text[:4]}/{text[6:8]}"
|
||||
return text
|
||||
|
||||
|
||||
_PUPIL_GROUPS = ("disadvantaged", "other", "all")
|
||||
|
||||
|
||||
def _lone_hidden_groups(groups: dict) -> list:
|
||||
"""Pupil groups hiding exactly one category — solvable by subtraction."""
|
||||
return [
|
||||
key for key, group in groups.items()
|
||||
if sum(1 for c in group["categories"] if c["status"] == "suppressed") == 1
|
||||
]
|
||||
|
||||
|
||||
def _lone_hidden_categories(groups: dict) -> list:
|
||||
"""Categories hidden in exactly one of several pupil groups."""
|
||||
lone = []
|
||||
categories = {c["category"] for g in groups.values() for c in g["categories"]}
|
||||
for category in categories:
|
||||
found = [
|
||||
c for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == category
|
||||
]
|
||||
hidden = [c for c in found if c["status"] == "suppressed"]
|
||||
if len(hidden) == 1 and len(found) > 1:
|
||||
lone.append(category)
|
||||
return lone
|
||||
|
||||
|
||||
def disclosure_invariant_holds(groups: dict) -> bool:
|
||||
"""Every row and every column hides none, or at least two.
|
||||
|
||||
Public so the tests can assert it directly rather than re-deriving it.
|
||||
"""
|
||||
return not _lone_hidden_groups(groups) and not _lone_hidden_categories(groups)
|
||||
|
||||
|
||||
def _mask_for_disclosure(groups: dict) -> None:
|
||||
"""Withhold further cells until nothing suppressed can be solved for.
|
||||
|
||||
Not rendering a figure is not the same as not publishing it. This endpoint
|
||||
is public and unauthenticated, so anything left in the payload is
|
||||
published, whatever the UI chooses to draw — the same reasoning the
|
||||
admission_distance field carries in app.py.
|
||||
|
||||
Two identities let a caller solve for a withheld cell:
|
||||
|
||||
* within a pupil group, the categories sum to the cohort, so a group with
|
||||
exactly ONE suppressed category gives it away as cohort - sum(rest);
|
||||
* across groups, disadvantaged + other = all for every category, so a
|
||||
category suppressed in exactly ONE of the three gives itself away.
|
||||
|
||||
DfE's own answer is secondary suppression: withhold a second cell so the
|
||||
residual spans two unknowns and identifies neither.
|
||||
|
||||
Where no companion can do that — a sparse cohort whose every other category
|
||||
is `not_applicable`, which is common in special schools and alternative
|
||||
provision — there is nothing left to withhold, so the pupil group is
|
||||
DROPPED entirely. An earlier version simply gave up here and returned with
|
||||
the violation intact and no signal, which is the one outcome this function
|
||||
must never produce: a disclosure-control pass that fails silently is worse
|
||||
than none, because everything downstream trusts it.
|
||||
|
||||
Mutates `groups` in place. Guaranteed to return with
|
||||
disclosure_invariant_holds(groups) true.
|
||||
"""
|
||||
|
||||
def suppress(cell):
|
||||
if cell["status"] == "published":
|
||||
cell["status"] = "suppressed"
|
||||
cell["pupils"] = None
|
||||
cell["percentage"] = None
|
||||
return True
|
||||
return False
|
||||
|
||||
def add_companion(candidates) -> bool:
|
||||
"""Withhold a second cell so the residual spans two unknowns.
|
||||
|
||||
The companion must carry pupils. Suppressing a zero looks like
|
||||
secondary suppression and protects nothing: the residual still equals
|
||||
the original withheld figure exactly. Returns False when no cell can
|
||||
do the job, which escalates to dropping the group.
|
||||
"""
|
||||
published = [c for c in candidates if c["status"] == "published"]
|
||||
useful = sorted(
|
||||
(c for c in published if (c["pupils"] or 0) > 0),
|
||||
key=lambda c: c["pupils"],
|
||||
)
|
||||
if useful:
|
||||
return suppress(useful[0])
|
||||
# Every remaining cell is zero or not applicable: withholding any of
|
||||
# them leaves the residual equal to the original figure.
|
||||
return False
|
||||
|
||||
# Fixpoint: each new suppression can break the other identity. Terminates
|
||||
# because every pass either adds a suppression, drops a group, or stops.
|
||||
while not disclosure_invariant_holds(groups):
|
||||
changed = False
|
||||
|
||||
for category in _lone_hidden_categories(groups):
|
||||
siblings = [
|
||||
c for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == category
|
||||
]
|
||||
if add_companion(siblings):
|
||||
changed = True
|
||||
|
||||
for key in _lone_hidden_groups(groups):
|
||||
if add_companion(groups[key]["categories"]):
|
||||
changed = True
|
||||
|
||||
if changed:
|
||||
continue
|
||||
|
||||
# Nothing left to withhold. Drop the groups that are still solvable,
|
||||
# and any category still solvable across the groups that remain.
|
||||
for key in _lone_hidden_groups(groups):
|
||||
del groups[key]
|
||||
changed = True
|
||||
|
||||
for category in _lone_hidden_categories(groups):
|
||||
for group in groups.values():
|
||||
for cell in group["categories"]:
|
||||
if cell["category"] == category and suppress(cell):
|
||||
changed = True
|
||||
|
||||
if not changed:
|
||||
# Unreachable given the two escalations above, but a masking pass
|
||||
# must never spin or exit unsafely. Withhold everything.
|
||||
groups.clear()
|
||||
return
|
||||
|
||||
|
||||
def _destinations_block(rows: list) -> dict | None:
|
||||
"""Shape destination rows for one phase into the API's block.
|
||||
|
||||
Applies secondary suppression before returning, so no caller of this public
|
||||
endpoint can solve for a figure DfE withheld. See _mask_for_disclosure.
|
||||
|
||||
Aggregate measures are dropped entirely. DfE publishes them, and they would
|
||||
be useful for a "what is published for this group" fallback, but nothing
|
||||
renders them today and an aggregate spanning exactly one suppressed
|
||||
component names that component. An unused field that leaks is not a
|
||||
trade-off worth carrying — re-add them with their own guard if the fallback
|
||||
is ever built.
|
||||
|
||||
Deliberately computes no residual, no "remaining pupils" figure, and no
|
||||
total that would close a gap left by a suppressed category.
|
||||
"""
|
||||
if not rows:
|
||||
return None
|
||||
|
||||
years = [r["year"] for r in rows if r.get("year") is not None]
|
||||
if not years:
|
||||
return None
|
||||
latest_year = max(years)
|
||||
rows = [r for r in rows if r.get("year") == latest_year]
|
||||
|
||||
groups: dict = {}
|
||||
for row in rows:
|
||||
group = groups.setdefault(
|
||||
row["pupil_group"],
|
||||
{"cohort": row.get("cohort_pupils"), "categories": []},
|
||||
)
|
||||
measure = row["destination_measure"]
|
||||
published = row.get("status") == "published"
|
||||
# Belt and braces: percentage is derived from the same source cell as
|
||||
# pupils, but publishing one without the other would hand back the
|
||||
# cohort (pupils / percentage) and with it the residual.
|
||||
cell = {
|
||||
"category": measure,
|
||||
"pupils": row.get("pupils") if published else None,
|
||||
"percentage": row.get("percentage") if published else None,
|
||||
"status": row.get("status"),
|
||||
}
|
||||
if measure in _AGGREGATE_MEASURES:
|
||||
continue
|
||||
group["categories"].append(cell)
|
||||
|
||||
if not groups:
|
||||
return None
|
||||
|
||||
_mask_for_disclosure(groups)
|
||||
|
||||
# Masking can empty the block entirely — a sparse cohort where no group
|
||||
# could be made safe. Return None so the section is absent rather than
|
||||
# rendering an empty shell.
|
||||
if not groups:
|
||||
return None
|
||||
|
||||
return {"cohort_year": _format_cohort_year(latest_year), "groups": groups}
|
||||
|
||||
|
||||
def _empty_supplementary() -> dict:
|
||||
return {
|
||||
"ofsted": None,
|
||||
@@ -775,6 +1071,7 @@ def _empty_supplementary() -> dict:
|
||||
"phonics": None,
|
||||
"deprivation": None,
|
||||
"finance": None,
|
||||
"destinations": None,
|
||||
}
|
||||
|
||||
|
||||
@@ -902,6 +1199,38 @@ def get_supplementary_data_batch(db: Session, urns: list[int]) -> dict:
|
||||
result[f.urn]["finance"] = _finance_dict(f)
|
||||
_safe(_finance)
|
||||
|
||||
# Destinations — KS4 and 16-18. Both marts are long-format, so every row
|
||||
# for a URN is collected and _destinations_block picks the latest year and
|
||||
# shapes the pupil groups. A phase with no rows serialises as null rather
|
||||
# than an empty shell, so the frontend renders nothing rather than an empty
|
||||
# section.
|
||||
def _destinations():
|
||||
from collections import defaultdict
|
||||
|
||||
def _collect(model):
|
||||
per_urn = defaultdict(list)
|
||||
for r in db.query(model).filter(model.urn.in_(urns)).all():
|
||||
per_urn[r.urn].append({
|
||||
"year": r.year,
|
||||
"pupil_group": r.pupil_group,
|
||||
"destination_measure": r.destination_measure,
|
||||
"cohort_pupils": r.cohort_pupils,
|
||||
"pupils": r.pupils,
|
||||
"percentage": r.percentage,
|
||||
"status": r.status,
|
||||
})
|
||||
return per_urn
|
||||
|
||||
ks4_rows = _collect(FactKs4Destinations)
|
||||
ks5_rows = _collect(FactKs5Destinations)
|
||||
for urn in urns:
|
||||
ks4 = _destinations_block(ks4_rows.get(urn, []))
|
||||
ks5 = _destinations_block(ks5_rows.get(urn, []))
|
||||
result[urn]["destinations"] = (
|
||||
{"ks4": ks4, "ks5": ks5} if (ks4 or ks5) else None
|
||||
)
|
||||
_safe(_destinations)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
"""Feature flags: what can be switched, and what is switched right now.
|
||||
|
||||
Ship-dark, not a kill switch. Flags let work merge and deploy without becoming
|
||||
visible; they are expected to flip about monthly, by a person, deliberately.
|
||||
Nothing here does percentage rollouts or user targeting — the site has no user
|
||||
identity to target.
|
||||
|
||||
Unleash holds the state. It does not hold the list. REGISTRY below is that
|
||||
list, and it exists for three reasons: the SDK evaluates an unknown flag to
|
||||
False, so without a registry that is an *undeclared* False, indistinguishable
|
||||
from a typo; /api/flags needs a key set to return when Unleash is unreachable;
|
||||
and a flag in the UI but not in the registry is orphaned and should be visibly
|
||||
so rather than quietly authoritative.
|
||||
|
||||
Every flag defaults to False. There is no per-flag default, because a flag that
|
||||
defaults on is a kill switch, and this is not one.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import date
|
||||
|
||||
from .config import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A flag is temporary scaffolding. See test_a_flag_older_than_the_limit.
|
||||
MAX_FLAG_AGE_DAYS = 90
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Flag:
|
||||
# One string: the registry key, the Unleash flag name, and the JSON key in
|
||||
# /api/flags. snake_case, matching the API's existing convention. No case
|
||||
# transformation anywhere, so there is no mapping layer to get wrong.
|
||||
name: str
|
||||
description: str # one line: what turning this on reveals
|
||||
added: date # for the staleness tripwire
|
||||
|
||||
|
||||
REGISTRY: dict[str, Flag] = {
|
||||
f.name: f for f in (
|
||||
Flag(
|
||||
name="admission_distance",
|
||||
description=(
|
||||
"The last-distance-offered figure on the Admissions tile and "
|
||||
"the 'How far away are you?' section on school pages."
|
||||
),
|
||||
added=date(2026, 8, 23),
|
||||
),
|
||||
Flag(
|
||||
name="school_autosuggest",
|
||||
description=(
|
||||
"School name suggestions as you type in the main search box."
|
||||
),
|
||||
added=date(2026, 8, 26),
|
||||
),
|
||||
Flag(
|
||||
name="about_page",
|
||||
description=(
|
||||
"The /about page, its footer link, its sitemap entry, and the "
|
||||
"named-author byline on every blog post."
|
||||
),
|
||||
added=date(2026, 9, 8),
|
||||
),
|
||||
Flag(
|
||||
name="blog",
|
||||
description=(
|
||||
"The /blog index, post pages, the RSS feed, their footer link "
|
||||
"and their sitemap entries. Not /admin: posts must be "
|
||||
"writable before the blog is readable."
|
||||
),
|
||||
added=date(2026, 9, 8),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
_client = None
|
||||
|
||||
|
||||
def init() -> None:
|
||||
"""Start the Unleash client, or log why flags are all off.
|
||||
|
||||
Called once from the app lifespan. Never raises: a flag system that can
|
||||
stop the API from booting is worse than one that is switched off.
|
||||
"""
|
||||
global _client
|
||||
if not settings.unleash_url or not settings.unleash_api_token:
|
||||
logger.warning(
|
||||
"Unleash is not configured (UNLEASH_URL / UNLEASH_API_TOKEN); "
|
||||
"every feature flag evaluates to False.")
|
||||
return
|
||||
|
||||
try:
|
||||
from UnleashClient import UnleashClient
|
||||
|
||||
_client = UnleashClient(
|
||||
url=settings.unleash_url,
|
||||
app_name=settings.unleash_app_name,
|
||||
custom_headers={"Authorization": settings.unleash_api_token},
|
||||
cache_directory=settings.unleash_cache_directory,
|
||||
refresh_interval=15,
|
||||
)
|
||||
_client.initialize_client()
|
||||
logger.info("Unleash client initialised against %s", settings.unleash_url)
|
||||
except Exception:
|
||||
# Fail closed and keep serving. The SDK also evaluates everything False
|
||||
# until its first successful sync, so this is the same direction.
|
||||
_client = None
|
||||
logger.exception("Unleash client failed to start; flags are all False.")
|
||||
|
||||
|
||||
def is_enabled(name: str) -> bool:
|
||||
"""Whether `name` is on. False for anything unknown, unreachable or broken."""
|
||||
if name not in REGISTRY:
|
||||
logger.error(
|
||||
"undeclared feature flag %r was evaluated; returning False. "
|
||||
"Add it to backend/flags.py REGISTRY or fix the name.", name)
|
||||
return False
|
||||
if _client is None:
|
||||
return False
|
||||
try:
|
||||
return bool(_client.is_enabled(
|
||||
name, fallback_function=lambda feature_name, context: False))
|
||||
except Exception:
|
||||
logger.exception("flag %r failed to evaluate; returning False", name)
|
||||
return False
|
||||
|
||||
|
||||
def all_flags() -> dict[str, bool]:
|
||||
"""Every declared flag and its current value. Serves /api/flags."""
|
||||
return {name: is_enabled(name) for name in REGISTRY}
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Curated London localities, defined by the postcode districts they cover.
|
||||
|
||||
The GIAS `town` field puts 1,819 London schools under the single value
|
||||
"London", so it cannot answer "schools in Battersea" — a query that appears in
|
||||
the Search Console baseline. No single field can: parliamentary constituency
|
||||
gives Battersea but not Canary Wharf; postcodes.io's admin_ward gives Canary
|
||||
Wharf but not Battersea; neither gives Clapham or Shoreditch, which are postal
|
||||
and colloquial rather than administrative.
|
||||
|
||||
So this is curated. Where a locality ends is a judgement, not a fact, and a
|
||||
reviewable file is the honest place for a judgement. No new ingestion is
|
||||
needed — the corpus already carries postcodes.
|
||||
|
||||
This is the canonical copy. `pipeline/transform/seeds/locality_outcodes.csv`
|
||||
mirrors it for anyone querying the warehouse directly; the backend image does
|
||||
not contain `pipeline/`, which is why the module rather than the seed is
|
||||
canonical. Same arrangement as `backend/gias_codes.py`.
|
||||
|
||||
A locality whose outcodes hold fewer than MIN_SCHOOLS schools is not
|
||||
published, so a typo produces no page rather than an empty one. Places that
|
||||
fail that check are logged at startup, because a locality you meant to publish
|
||||
quietly not appearing is the failure worth hearing about.
|
||||
|
||||
Two rules for anything added here.
|
||||
|
||||
**Sub-borough districts only.** A London borough is a local authority and
|
||||
already has a page at /schools/authority/[la] covering all of its schools; a
|
||||
locality defined by two or three outcodes would be a partial, near-duplicate
|
||||
subset of it. Hackney, Islington, Greenwich and Ealing were all in the first
|
||||
draft for that reason and have been removed.
|
||||
|
||||
**The slug must not match a GIAS town.** "Richmond" did — GIAS has a Richmond
|
||||
in North Yorkshire with 37 schools — so the London one could never publish.
|
||||
The registry skips any locality that collides and logs it.
|
||||
"""
|
||||
|
||||
# slug -> (display name, outcodes)
|
||||
LOCALITY_OUTCODES: dict[str, tuple[str, tuple[str, ...]]] = {
|
||||
"battersea": ("Battersea", ("SW11",)),
|
||||
"canary-wharf": ("Canary Wharf", ("E14",)),
|
||||
"clapham": ("Clapham", ("SW4",)),
|
||||
"shoreditch": ("Shoreditch", ("EC2A", "E1")),
|
||||
"peckham": ("Peckham", ("SE15",)),
|
||||
"brixton": ("Brixton", ("SW2", "SW9")),
|
||||
"camden-town": ("Camden Town", ("NW1",)),
|
||||
"wimbledon": ("Wimbledon", ("SW19",)),
|
||||
"putney": ("Putney", ("SW15",)),
|
||||
"fulham": ("Fulham", ("SW6",)),
|
||||
"chiswick": ("Chiswick", ("W4",)),
|
||||
"stratford": ("Stratford", ("E15",)),
|
||||
"walthamstow": ("Walthamstow", ("E17",)),
|
||||
"tooting": ("Tooting", ("SW17",)),
|
||||
"dulwich": ("Dulwich", ("SE21", "SE22")),
|
||||
}
|
||||
@@ -1,512 +0,0 @@
|
||||
"""
|
||||
Database migration logic for importing CSV data.
|
||||
Used by both CLI script and automatic startup migration.
|
||||
"""
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import requests
|
||||
|
||||
from .config import settings
|
||||
from .database import Base, engine, get_db_session
|
||||
from .models import School, SchoolResult
|
||||
from .schemas import (
|
||||
COLUMN_MAPPINGS,
|
||||
LA_CODE_TO_NAME,
|
||||
NULL_VALUES,
|
||||
SCHOOL_TYPE_MAP,
|
||||
)
|
||||
|
||||
|
||||
def parse_numeric(value) -> Optional[float]:
|
||||
"""Parse a numeric value, handling special cases."""
|
||||
if pd.isna(value):
|
||||
return None
|
||||
if isinstance(value, (int, float)):
|
||||
return float(value) if not np.isnan(value) else None
|
||||
str_val = str(value).strip().upper()
|
||||
if str_val in NULL_VALUES or str_val == "":
|
||||
return None
|
||||
# Remove percentage signs if present
|
||||
str_val = str_val.replace("%", "")
|
||||
try:
|
||||
return float(str_val)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def extract_year_from_folder(folder_name: str) -> Optional[int]:
|
||||
"""Extract year from folder name like '2023-2024'."""
|
||||
match = re.search(r"(\d{4})-(\d{4})", folder_name)
|
||||
if match:
|
||||
return int(match.group(2))
|
||||
match = re.search(r"(\d{4})", folder_name)
|
||||
if match:
|
||||
return int(match.group(1))
|
||||
return None
|
||||
|
||||
|
||||
def geocode_postcodes_bulk(postcodes: list) -> Dict[str, tuple]:
|
||||
"""
|
||||
Geocode postcodes in bulk using postcodes.io API.
|
||||
Returns dict of postcode -> (latitude, longitude).
|
||||
"""
|
||||
results = {}
|
||||
valid_postcodes = [
|
||||
p.strip().upper()
|
||||
for p in postcodes
|
||||
if p and isinstance(p, str) and len(p.strip()) >= 5
|
||||
]
|
||||
valid_postcodes = list(set(valid_postcodes))
|
||||
|
||||
if not valid_postcodes:
|
||||
return results
|
||||
|
||||
batch_size = 100
|
||||
total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size
|
||||
|
||||
for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)):
|
||||
batch = valid_postcodes[batch_start : batch_start + batch_size]
|
||||
print(
|
||||
f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)..."
|
||||
)
|
||||
|
||||
try:
|
||||
response = requests.post(
|
||||
"https://api.postcodes.io/postcodes",
|
||||
json={"postcodes": batch},
|
||||
timeout=30,
|
||||
)
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
for item in data.get("result", []):
|
||||
if item and item.get("result"):
|
||||
pc = item["query"].upper()
|
||||
lat = item["result"].get("latitude")
|
||||
lon = item["result"].get("longitude")
|
||||
if lat and lon:
|
||||
results[pc] = (lat, lon)
|
||||
except Exception as e:
|
||||
print(f" Warning: Geocoding batch failed: {e}")
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def load_csv_data(data_dir: Path) -> pd.DataFrame:
|
||||
"""Load all CSV data from data directory."""
|
||||
all_data = []
|
||||
|
||||
for folder in sorted(data_dir.iterdir()):
|
||||
if not folder.is_dir():
|
||||
continue
|
||||
|
||||
year = extract_year_from_folder(folder.name)
|
||||
if not year:
|
||||
continue
|
||||
|
||||
# Specifically look for the KS2 results file
|
||||
ks2_file = folder / "england_ks2final.csv"
|
||||
if not ks2_file.exists():
|
||||
continue
|
||||
|
||||
csv_file = ks2_file
|
||||
print(f" Loading {csv_file.name} (year {year})...")
|
||||
|
||||
try:
|
||||
df = pd.read_csv(csv_file, encoding="latin-1", low_memory=False)
|
||||
except Exception as e:
|
||||
print(f" Error loading {csv_file}: {e}")
|
||||
continue
|
||||
|
||||
# Rename columns
|
||||
df.rename(columns=COLUMN_MAPPINGS, inplace=True)
|
||||
df["year"] = year
|
||||
|
||||
# Handle local authority name
|
||||
la_name_cols = ["LANAME", "LA (name)", "LA_NAME", "LA NAME"]
|
||||
la_name_col = next((c for c in la_name_cols if c in df.columns), None)
|
||||
|
||||
if la_name_col and la_name_col != "local_authority":
|
||||
df["local_authority"] = df[la_name_col]
|
||||
elif "LEA" in df.columns:
|
||||
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
|
||||
df["local_authority"] = (
|
||||
df["local_authority_code"]
|
||||
.map(LA_CODE_TO_NAME)
|
||||
.fillna(df["LEA"].astype(str))
|
||||
)
|
||||
|
||||
# Store LEA code
|
||||
if "LEA" in df.columns:
|
||||
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
|
||||
|
||||
# Map school type
|
||||
if "school_type_code" in df.columns:
|
||||
df["school_type"] = (
|
||||
df["school_type_code"]
|
||||
.map(SCHOOL_TYPE_MAP)
|
||||
.fillna(df["school_type_code"])
|
||||
)
|
||||
|
||||
# Create combined address
|
||||
addr_parts = ["address1", "address2", "town", "postcode"]
|
||||
for col in addr_parts:
|
||||
if col not in df.columns:
|
||||
df[col] = None
|
||||
|
||||
df["address"] = df.apply(
|
||||
lambda r: ", ".join(
|
||||
str(v)
|
||||
for v in [
|
||||
r.get("address1"),
|
||||
r.get("address2"),
|
||||
r.get("town"),
|
||||
r.get("postcode"),
|
||||
]
|
||||
if pd.notna(v) and str(v).strip()
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
all_data.append(df)
|
||||
print(f" Loaded {len(df)} records")
|
||||
|
||||
if all_data:
|
||||
result = pd.concat(all_data, ignore_index=True)
|
||||
print(f"\nTotal records loaded: {len(result)}")
|
||||
print(f"Unique schools: {result['urn'].nunique()}")
|
||||
print(f"Years: {sorted(result['year'].unique())}")
|
||||
return result
|
||||
|
||||
return pd.DataFrame()
|
||||
|
||||
|
||||
def migrate_data(df: pd.DataFrame, geocode: bool = False, geocode_cache: dict = None):
|
||||
"""Migrate DataFrame data to database."""
|
||||
|
||||
if geocode_cache is None:
|
||||
geocode_cache = {}
|
||||
|
||||
# Clean URN column - convert to integer, drop invalid values
|
||||
df = df.copy()
|
||||
df["urn"] = pd.to_numeric(df["urn"], errors="coerce")
|
||||
df = df.dropna(subset=["urn"])
|
||||
df["urn"] = df["urn"].astype(int)
|
||||
|
||||
# Group by URN to get unique schools (use latest year's data)
|
||||
school_data = (
|
||||
df.sort_values("year", ascending=False).groupby("urn").first().reset_index()
|
||||
)
|
||||
print(f"\nMigrating {len(school_data)} unique schools...")
|
||||
|
||||
# Geocode postcodes that aren't already in the cache
|
||||
geocoded = dict(geocode_cache) # start with preserved coordinates
|
||||
if geocode and "postcode" in df.columns:
|
||||
cached_postcodes = {
|
||||
str(row.get("postcode", "")).strip().upper()
|
||||
for _, row in school_data.iterrows()
|
||||
if int(float(str(row.get("urn", 0) or 0))) in geocode_cache
|
||||
}
|
||||
postcodes_needed = [
|
||||
p for p in df["postcode"].dropna().unique()
|
||||
if str(p).strip().upper() not in cached_postcodes
|
||||
]
|
||||
if postcodes_needed:
|
||||
print(f"\nGeocoding {len(postcodes_needed)} postcodes ({len(geocode_cache)} restored from cache)...")
|
||||
fresh = geocode_postcodes_bulk(postcodes_needed)
|
||||
geocoded.update(fresh)
|
||||
print(f" Successfully geocoded {len(fresh)} new postcodes")
|
||||
else:
|
||||
print(f"\nAll {len(geocode_cache)} postcodes restored from cache, skipping geocoding.")
|
||||
|
||||
with get_db_session() as db:
|
||||
# Create schools
|
||||
urn_to_school_id = {}
|
||||
schools_created = 0
|
||||
|
||||
for _, row in school_data.iterrows():
|
||||
# Safely parse URN - handle None, NaN, whitespace, and invalid values
|
||||
urn_val = row.get("urn")
|
||||
urn = None
|
||||
if pd.notna(urn_val):
|
||||
try:
|
||||
urn_str = str(urn_val).strip()
|
||||
if urn_str:
|
||||
urn = int(float(urn_str)) # Handle "12345.0" format
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if not urn:
|
||||
continue
|
||||
|
||||
# Skip if we've already added this URN (handles duplicates in source data)
|
||||
if urn in urn_to_school_id:
|
||||
continue
|
||||
|
||||
# Get geocoding data
|
||||
postcode = row.get("postcode")
|
||||
lat, lon = None, None
|
||||
if postcode and pd.notna(postcode):
|
||||
coords = geocoded.get(str(postcode).strip().upper())
|
||||
if coords:
|
||||
lat, lon = coords
|
||||
|
||||
# Safely parse local_authority_code
|
||||
la_code = None
|
||||
la_code_val = row.get("local_authority_code")
|
||||
if pd.notna(la_code_val):
|
||||
try:
|
||||
la_code_str = str(la_code_val).strip()
|
||||
if la_code_str:
|
||||
la_code = int(float(la_code_str))
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
school = School(
|
||||
urn=urn,
|
||||
school_name=row.get("school_name")
|
||||
if pd.notna(row.get("school_name"))
|
||||
else "Unknown",
|
||||
local_authority=row.get("local_authority")
|
||||
if pd.notna(row.get("local_authority"))
|
||||
else None,
|
||||
local_authority_code=la_code,
|
||||
school_type=row.get("school_type")
|
||||
if pd.notna(row.get("school_type"))
|
||||
else None,
|
||||
school_type_code=row.get("school_type_code")
|
||||
if pd.notna(row.get("school_type_code"))
|
||||
else None,
|
||||
religious_denomination=row.get("religious_denomination")
|
||||
if pd.notna(row.get("religious_denomination"))
|
||||
else None,
|
||||
age_range=row.get("age_range")
|
||||
if pd.notna(row.get("age_range"))
|
||||
else None,
|
||||
address1=row.get("address1") if pd.notna(row.get("address1")) else None,
|
||||
address2=row.get("address2") if pd.notna(row.get("address2")) else None,
|
||||
town=row.get("town") if pd.notna(row.get("town")) else None,
|
||||
postcode=row.get("postcode") if pd.notna(row.get("postcode")) else None,
|
||||
latitude=lat,
|
||||
longitude=lon,
|
||||
)
|
||||
db.add(school)
|
||||
db.flush() # Get the ID
|
||||
urn_to_school_id[urn] = school.id
|
||||
schools_created += 1
|
||||
|
||||
if schools_created % 1000 == 0:
|
||||
print(f" Created {schools_created} schools...")
|
||||
|
||||
print(f" Created {schools_created} schools")
|
||||
|
||||
# Create results
|
||||
print(f"\nMigrating {len(df)} yearly results...")
|
||||
results_created = 0
|
||||
|
||||
for _, row in df.iterrows():
|
||||
# Safely parse URN
|
||||
urn_val = row.get("urn")
|
||||
urn = None
|
||||
if pd.notna(urn_val):
|
||||
try:
|
||||
urn_str = str(urn_val).strip()
|
||||
if urn_str:
|
||||
urn = int(float(urn_str))
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if not urn or urn not in urn_to_school_id:
|
||||
continue
|
||||
|
||||
school_id = urn_to_school_id[urn]
|
||||
|
||||
# Safely parse year
|
||||
year_val = row.get("year")
|
||||
year = None
|
||||
if pd.notna(year_val):
|
||||
try:
|
||||
year = int(float(str(year_val).strip()))
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if not year:
|
||||
continue
|
||||
|
||||
result = SchoolResult(
|
||||
school_id=school_id,
|
||||
year=year,
|
||||
total_pupils=parse_numeric(row.get("total_pupils")),
|
||||
eligible_pupils=parse_numeric(row.get("eligible_pupils")),
|
||||
# Expected Standard
|
||||
rwm_expected_pct=parse_numeric(row.get("rwm_expected_pct")),
|
||||
reading_expected_pct=parse_numeric(row.get("reading_expected_pct")),
|
||||
writing_expected_pct=parse_numeric(row.get("writing_expected_pct")),
|
||||
maths_expected_pct=parse_numeric(row.get("maths_expected_pct")),
|
||||
gps_expected_pct=parse_numeric(row.get("gps_expected_pct")),
|
||||
science_expected_pct=parse_numeric(row.get("science_expected_pct")),
|
||||
# Higher Standard
|
||||
rwm_high_pct=parse_numeric(row.get("rwm_high_pct")),
|
||||
reading_high_pct=parse_numeric(row.get("reading_high_pct")),
|
||||
writing_high_pct=parse_numeric(row.get("writing_high_pct")),
|
||||
maths_high_pct=parse_numeric(row.get("maths_high_pct")),
|
||||
gps_high_pct=parse_numeric(row.get("gps_high_pct")),
|
||||
# Progress
|
||||
reading_progress=parse_numeric(row.get("reading_progress")),
|
||||
writing_progress=parse_numeric(row.get("writing_progress")),
|
||||
maths_progress=parse_numeric(row.get("maths_progress")),
|
||||
# Averages
|
||||
reading_avg_score=parse_numeric(row.get("reading_avg_score")),
|
||||
maths_avg_score=parse_numeric(row.get("maths_avg_score")),
|
||||
gps_avg_score=parse_numeric(row.get("gps_avg_score")),
|
||||
# Context
|
||||
disadvantaged_pct=parse_numeric(row.get("disadvantaged_pct")),
|
||||
eal_pct=parse_numeric(row.get("eal_pct")),
|
||||
sen_support_pct=parse_numeric(row.get("sen_support_pct")),
|
||||
sen_ehcp_pct=parse_numeric(row.get("sen_ehcp_pct")),
|
||||
stability_pct=parse_numeric(row.get("stability_pct")),
|
||||
# Absence
|
||||
reading_absence_pct=parse_numeric(row.get("reading_absence_pct")),
|
||||
gps_absence_pct=parse_numeric(row.get("gps_absence_pct")),
|
||||
maths_absence_pct=parse_numeric(row.get("maths_absence_pct")),
|
||||
writing_absence_pct=parse_numeric(row.get("writing_absence_pct")),
|
||||
science_absence_pct=parse_numeric(row.get("science_absence_pct")),
|
||||
# Gender
|
||||
rwm_expected_boys_pct=parse_numeric(row.get("rwm_expected_boys_pct")),
|
||||
rwm_expected_girls_pct=parse_numeric(row.get("rwm_expected_girls_pct")),
|
||||
rwm_high_boys_pct=parse_numeric(row.get("rwm_high_boys_pct")),
|
||||
rwm_high_girls_pct=parse_numeric(row.get("rwm_high_girls_pct")),
|
||||
# Disadvantaged
|
||||
rwm_expected_disadvantaged_pct=parse_numeric(
|
||||
row.get("rwm_expected_disadvantaged_pct")
|
||||
),
|
||||
rwm_expected_non_disadvantaged_pct=parse_numeric(
|
||||
row.get("rwm_expected_non_disadvantaged_pct")
|
||||
),
|
||||
disadvantaged_gap=parse_numeric(row.get("disadvantaged_gap")),
|
||||
# 3-Year
|
||||
rwm_expected_3yr_pct=parse_numeric(row.get("rwm_expected_3yr_pct")),
|
||||
reading_avg_3yr=parse_numeric(row.get("reading_avg_3yr")),
|
||||
maths_avg_3yr=parse_numeric(row.get("maths_avg_3yr")),
|
||||
)
|
||||
db.add(result)
|
||||
results_created += 1
|
||||
|
||||
if results_created % 10000 == 0:
|
||||
print(f" Created {results_created} results...")
|
||||
db.flush()
|
||||
|
||||
print(f" Created {results_created} results")
|
||||
|
||||
# Commit all changes
|
||||
db.commit()
|
||||
print("\nMigration complete!")
|
||||
|
||||
|
||||
def _apply_schema_alterations():
|
||||
"""
|
||||
Add new columns to existing tables using ALTER TABLE … ADD COLUMN IF NOT EXISTS.
|
||||
Safe to run on every migration — no-ops if the column already exists.
|
||||
Add entries here whenever models.py gains new columns on an existing table.
|
||||
"""
|
||||
alterations = [
|
||||
# v4: Ofsted Report Card columns
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS framework VARCHAR(20)",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_safeguarding_met BOOLEAN",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_inclusion INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_curriculum_teaching INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_achievement INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_attendance_behaviour INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_personal_development INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_leadership_governance INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_early_years INTEGER",
|
||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_sixth_form INTEGER",
|
||||
]
|
||||
from sqlalchemy import text as sa_text
|
||||
with engine.connect() as conn:
|
||||
for stmt in alterations:
|
||||
try:
|
||||
conn.execute(sa_text(stmt))
|
||||
except Exception as e:
|
||||
print(f" Warning: alteration skipped ({e})")
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _apply_schema_drops():
|
||||
"""
|
||||
Drop tables retired from the schema. Idempotent (DROP … IF EXISTS), so it's
|
||||
safe to run on every migration. Add entries here when a model is removed.
|
||||
"""
|
||||
drops = [
|
||||
# v6: Ofsted Parent View feature removed
|
||||
"DROP TABLE IF EXISTS marts.fact_parent_view CASCADE",
|
||||
]
|
||||
from sqlalchemy import text as sa_text
|
||||
with engine.connect() as conn:
|
||||
for stmt in drops:
|
||||
try:
|
||||
conn.execute(sa_text(stmt))
|
||||
except Exception as e:
|
||||
print(f" Warning: drop skipped ({e})")
|
||||
conn.commit()
|
||||
|
||||
|
||||
def run_full_migration(geocode: bool = False) -> bool:
|
||||
"""
|
||||
Run a complete migration: drop all tables and reimport from CSV.
|
||||
|
||||
Returns True if successful, False if no data found.
|
||||
Raises exception on error.
|
||||
"""
|
||||
# Preserve existing geocoding so a reimport doesn't throw away coordinates
|
||||
# that took a long time to compute.
|
||||
geocode_cache: dict[int, tuple[float, float]] = {}
|
||||
inspector = __import__("sqlalchemy").inspect(engine)
|
||||
if "schools" in inspector.get_table_names():
|
||||
try:
|
||||
with get_db_session() as db:
|
||||
rows = db.execute(
|
||||
__import__("sqlalchemy").text(
|
||||
"SELECT urn, latitude, longitude FROM schools "
|
||||
"WHERE latitude IS NOT NULL AND longitude IS NOT NULL"
|
||||
)
|
||||
).fetchall()
|
||||
geocode_cache = {r.urn: (r.latitude, r.longitude) for r in rows}
|
||||
print(f" Saved {len(geocode_cache)} existing geocoded coordinates.")
|
||||
except Exception as e:
|
||||
print(f" Warning: could not save geocode cache: {e}")
|
||||
|
||||
# Only drop the core KS2 tables — leave supplementary tables (ofsted, census,
|
||||
# finance, etc.) intact so a reimport doesn't wipe integrator-populated data.
|
||||
# schema_version is NOT dropped: it persists so restarts don't re-trigger migration.
|
||||
ks2_tables = ["school_results", "schools"]
|
||||
print(f"Dropping core tables: {ks2_tables} ...")
|
||||
inspector = __import__("sqlalchemy").inspect(engine)
|
||||
existing = set(inspector.get_table_names())
|
||||
for tname in ks2_tables:
|
||||
if tname in existing:
|
||||
Base.metadata.tables[tname].drop(bind=engine)
|
||||
|
||||
print("Creating all tables...")
|
||||
Base.metadata.create_all(bind=engine)
|
||||
|
||||
# ALTER existing supplementary tables to add any new columns.
|
||||
# create_all() only creates missing tables; it won't add columns to tables
|
||||
# that already exist from an older schema version. These statements are
|
||||
# idempotent (IF NOT EXISTS) so they're safe to run on every migration.
|
||||
print("Applying column additions to supplementary tables...")
|
||||
_apply_schema_alterations()
|
||||
|
||||
print("Dropping retired tables...")
|
||||
_apply_schema_drops()
|
||||
|
||||
print("\nLoading CSV data...")
|
||||
df = load_csv_data(settings.data_dir)
|
||||
|
||||
if df.empty:
|
||||
print("Warning: No CSV data found to migrate!")
|
||||
return False
|
||||
|
||||
migrate_data(df, geocode=geocode, geocode_cache=geocode_cache)
|
||||
return True
|
||||
@@ -321,3 +321,48 @@ class Ks2NationalAverage(Base):
|
||||
gps_high_pct = Column(Float)
|
||||
gps_avg_score = Column(Float)
|
||||
science_expected_pct = Column(Float)
|
||||
|
||||
|
||||
class FactKs4Destinations(Base):
|
||||
"""KS4 leavers destinations — one row per URN, year, pupil group, measure.
|
||||
|
||||
Long format rather than wide because pupil_group is a real third dimension.
|
||||
`status` is load-bearing: 'suppressed' means DfE withheld a figure it
|
||||
considered disclosive and the page must print "withheld"; 'not_applicable'
|
||||
means the measure does not apply and the page must print nothing. `pupils`
|
||||
is null for both, so collapsing status to a null check loses the
|
||||
difference — and the categories sum to the cohort, so a consumer that
|
||||
treats a withheld cell as zero republishes what DfE hid.
|
||||
"""
|
||||
__tablename__ = "fact_ks4_destinations"
|
||||
__table_args__ = (
|
||||
Index("ix_ks4_dest_urn_year", "urn", "year"),
|
||||
MARTS,
|
||||
)
|
||||
|
||||
urn = Column(Integer, primary_key=True)
|
||||
year = Column(Integer, primary_key=True)
|
||||
pupil_group = Column(String(20), primary_key=True)
|
||||
destination_measure = Column(String(40), primary_key=True)
|
||||
cohort_pupils = Column(Integer)
|
||||
pupils = Column(Integer)
|
||||
percentage = Column(Float)
|
||||
status = Column(String(20))
|
||||
|
||||
|
||||
class FactKs5Destinations(Base):
|
||||
"""16-18 study leavers destinations — same grain as FactKs4Destinations."""
|
||||
__tablename__ = "fact_ks5_destinations"
|
||||
__table_args__ = (
|
||||
Index("ix_ks5_dest_urn_year", "urn", "year"),
|
||||
MARTS,
|
||||
)
|
||||
|
||||
urn = Column(Integer, primary_key=True)
|
||||
year = Column(Integer, primary_key=True)
|
||||
pupil_group = Column(String(20), primary_key=True)
|
||||
destination_measure = Column(String(40), primary_key=True)
|
||||
cohort_pupils = Column(Integer)
|
||||
pupils = Column(Integer)
|
||||
percentage = Column(Float)
|
||||
status = Column(String(20))
|
||||
@@ -0,0 +1,263 @@
|
||||
"""Which nearby schools a detail page may offer as alternatives.
|
||||
|
||||
HARD FILTERS decide eligibility, and encode claims the section is not allowed
|
||||
to make. A selective school is not an alternative to a non-selective one, a
|
||||
special school is not comparable to a mainstream one, and a Girls school is not
|
||||
an option for a Boys school's reader. They never relax, at any distance, even
|
||||
where that means the section does not render at all.
|
||||
|
||||
DISTANCE decides the order, and nothing else does.
|
||||
|
||||
An earlier version ranked by intake similarity first and used distance only as
|
||||
a tiebreak. That put a Catholic school 2.9 miles away above the community
|
||||
school 0.3 miles down the road, and — because the row filled from the best tier
|
||||
before widening — filled all six slots with faith matches while omitting every
|
||||
school a parent could actually walk to. For a primary, a school that far is not
|
||||
a weaker option; it is not an option. Distance is a constraint and intake is a
|
||||
preference, and the ranking now says so.
|
||||
|
||||
Similarity survives as `shared`: what a candidate genuinely has in common with
|
||||
this school, reported on its card, so a reader applies their own weighting
|
||||
instead of having ours applied for them.
|
||||
|
||||
Pure functions over a DataFrame: no I/O, no FastAPI, no database.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .schemas import PHASE_GROUPS
|
||||
|
||||
# Three fit the row; the rest are behind the carousel arrows.
|
||||
MAX_SCHOOLS = 6
|
||||
MINIMUM = 2
|
||||
|
||||
# How far the section will reach, in miles, when nothing closer exists.
|
||||
#
|
||||
# A sanity bound rather than a target: ordering by distance already handles
|
||||
# density, so a school in a dense area fills all six slots inside a mile and
|
||||
# never sees this. It decides one thing — what happens where the area is
|
||||
# sparse — and the answer differs by phase because catchments do. Primary
|
||||
# catchments are routinely under a mile; beyond two, a primary is not a weaker
|
||||
# option but not an option, and no section is the honest answer.
|
||||
PRIMARY_RADIUS_MILES = 2.0
|
||||
SECONDARY_RADIUS_MILES = 6.0
|
||||
POST16_RADIUS_MILES = 10.0
|
||||
|
||||
EARTH_RADIUS_MILES = 3958.8
|
||||
|
||||
_SPECIAL = re.compile(r"\bspecial\b|pupil referral|alternative provision", re.I)
|
||||
|
||||
# Values that mean "this school has no religious character".
|
||||
_NO_FAITH = {"", "none", "does not apply", "not applicable"}
|
||||
|
||||
|
||||
def is_special_provision(school_type: str | None) -> bool:
|
||||
"""Mirror of isSpecialSchool() in nextjs-app/lib/utils.ts.
|
||||
|
||||
Special schools carry a mainstream phase, so phase alone cannot identify
|
||||
them. The two implementations must agree: a school the frontend treats as
|
||||
special for benchmarking but this treats as mainstream would be dropped
|
||||
from its own England comparison and then offered as a peer to a mainstream
|
||||
school on the next page along.
|
||||
"""
|
||||
return bool(_SPECIAL.search(school_type or ""))
|
||||
|
||||
|
||||
def is_selective(admissions_policy: str | None) -> bool:
|
||||
"""Strictly selective. Unknown counts as non-selective, which is the safe
|
||||
direction: it can only ever exclude a pairing, never invent one."""
|
||||
return (admissions_policy or "").strip().lower() == "selective"
|
||||
|
||||
|
||||
def faith_key(denomination: str | None) -> str:
|
||||
value = (denomination or "").strip().lower()
|
||||
return "" if value in _NO_FAITH else value
|
||||
|
||||
|
||||
def faith_label(denomination: str | None) -> str:
|
||||
return denomination.strip() if faith_key(denomination) else "No religious character"
|
||||
|
||||
|
||||
def genders_compatible(a: str | None, b: str | None) -> bool:
|
||||
single = {"boys", "girls"}
|
||||
left, right = (a or "").strip().lower(), (b or "").strip().lower()
|
||||
return not (left in single and right in single and left != right)
|
||||
|
||||
|
||||
def is_secondary_phase(phase: str | None) -> bool:
|
||||
"""Whether this phase takes the secondary side: secondary group membership,
|
||||
minus all-through.
|
||||
|
||||
Membership is read from PHASE_GROUPS rather than tested with `"secondary" in
|
||||
phase`, because that substring misses "16 plus" — GIAS phase 6, which
|
||||
PHASE_GROUPS deliberately files as secondary. The substring version fails
|
||||
silently rather than loudly: a sixth-form college is simply handed the
|
||||
primary bucket and offered infant schools as peers.
|
||||
|
||||
All-through is the exception. PHASE_GROUPS lists it on both sides because it
|
||||
belongs on both phases' place pages, but the detail page renders it with the
|
||||
primary template, and the metric follows the template.
|
||||
"""
|
||||
text = (phase or "").strip().lower()
|
||||
return text != "all-through" and text in PHASE_GROUPS["secondary"]
|
||||
|
||||
|
||||
def radius_miles(phase: str | None) -> float:
|
||||
"""How far this phase's section will reach when nothing closer exists."""
|
||||
if (phase or "").strip().lower() == "16 plus":
|
||||
return POST16_RADIUS_MILES
|
||||
return SECONDARY_RADIUS_MILES if is_secondary_phase(phase) else PRIMARY_RADIUS_MILES
|
||||
|
||||
|
||||
def _phase_group(is_secondary: bool) -> set[str]:
|
||||
return PHASE_GROUPS["secondary" if is_secondary else "primary"]
|
||||
|
||||
|
||||
def _haversine_miles(lat1: float, lon1: float, lat2, lon2):
|
||||
"""Vectorised, matching the postcode search in app.py."""
|
||||
lat1_r, lon1_r = np.radians(lat1), np.radians(lon1)
|
||||
lat2_r, lon2_r = np.radians(lat2.astype(float)), np.radians(lon2.astype(float))
|
||||
dlat, dlon = lat2_r - lat1_r, lon2_r - lon1_r
|
||||
a = np.sin(dlat / 2) ** 2 + np.cos(lat1_r) * np.cos(lat2_r) * np.sin(dlon / 2) ** 2
|
||||
return 2 * EARTH_RADIUS_MILES * np.arcsin(np.sqrt(a))
|
||||
|
||||
|
||||
def _native(value):
|
||||
"""NaN and numpy scalars both reach JSONResponse badly; normalise here so
|
||||
the caller never has to remember to."""
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, np.generic):
|
||||
value = value.item()
|
||||
if isinstance(value, float) and np.isnan(value):
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def _mask(series: pd.Series, predicate) -> pd.Series:
|
||||
"""A boolean mask that survives an empty frame.
|
||||
|
||||
`Series.apply` on an empty Series returns an empty *DataFrame*, and using
|
||||
that as a mask silently drops every column — so the next column lookup
|
||||
raises KeyError rather than yielding no rows. This is not hypothetical: a
|
||||
special school with no special school near it empties the frame at the
|
||||
provision filter, which is the ordinary case for most special schools.
|
||||
"""
|
||||
return pd.Series([predicate(value) for value in series], index=series.index, dtype=bool)
|
||||
|
||||
|
||||
def _shared(subject: pd.Series, candidate: pd.Series, is_secondary: bool) -> list[str]:
|
||||
"""What this candidate genuinely has in common with the subject.
|
||||
|
||||
Empty is a real answer, and renders no chips at all. A card claiming a
|
||||
shared characteristic it does not have would be worse than a bare one —
|
||||
and since these no longer affect the order, an empty list costs the school
|
||||
nothing but its place in the row, which distance already decided.
|
||||
"""
|
||||
shared: list[str] = []
|
||||
|
||||
gender = str(subject.get("gender") or "").strip()
|
||||
if gender and str(candidate.get("gender") or "").strip().lower() == gender.lower():
|
||||
shared.append(gender)
|
||||
|
||||
if is_secondary:
|
||||
policy = str(candidate.get("admissions_policy") or "").strip()
|
||||
subject_policy = str(subject.get("admissions_policy") or "").strip()
|
||||
if (
|
||||
policy
|
||||
and policy.lower() == subject_policy.lower()
|
||||
and policy.lower() not in {"not applicable", "unknown"}
|
||||
):
|
||||
shared.append(policy)
|
||||
|
||||
if faith_key(candidate.get("religious_denomination")) == faith_key(
|
||||
subject.get("religious_denomination")
|
||||
):
|
||||
shared.append(faith_label(candidate.get("religious_denomination")))
|
||||
|
||||
return shared
|
||||
|
||||
|
||||
def select_nearby(frame: pd.DataFrame, urn: int) -> list[dict]:
|
||||
"""The nearest eligible schools, closest first — at most MAX_SCHOOLS, and
|
||||
none at all below MINIMUM.
|
||||
|
||||
The phase is read from the subject's own row rather than passed in, so a
|
||||
caller cannot hand this a phase that disagrees with the data it selects
|
||||
from.
|
||||
"""
|
||||
subject_rows = frame[frame["urn"] == urn]
|
||||
if subject_rows.empty:
|
||||
return []
|
||||
subject = subject_rows.iloc[0]
|
||||
|
||||
lat, lon = _native(subject.get("latitude")), _native(subject.get("longitude"))
|
||||
if lat is None or lon is None:
|
||||
return []
|
||||
|
||||
phase = subject.get("phase")
|
||||
is_secondary = is_secondary_phase(phase)
|
||||
reach = radius_miles(phase)
|
||||
metric_key = "attainment_8_score" if is_secondary else "rwm_expected_pct"
|
||||
|
||||
candidates = frame[frame["urn"] != urn].copy()
|
||||
for column in ("latitude", "longitude"):
|
||||
candidates = candidates[candidates[column].notna()]
|
||||
if candidates.empty:
|
||||
return []
|
||||
|
||||
# ── Hard filters ────────────────────────────────────────────────────
|
||||
allowed_phases = _phase_group(is_secondary)
|
||||
candidates = candidates[
|
||||
candidates["phase"].fillna("").str.lower().isin(allowed_phases)
|
||||
]
|
||||
candidates = candidates[candidates["status"].fillna("").str.lower().str.startswith("open")]
|
||||
|
||||
subject_special = is_special_provision(subject.get("school_type"))
|
||||
special = _mask(candidates["school_type"], is_special_provision)
|
||||
candidates = candidates[special if subject_special else ~special]
|
||||
|
||||
subject_selective = is_selective(subject.get("admissions_policy"))
|
||||
selective = _mask(candidates["admissions_policy"], is_selective)
|
||||
candidates = candidates[selective if subject_selective else ~selective]
|
||||
|
||||
subject_gender = subject.get("gender")
|
||||
candidates = candidates[
|
||||
_mask(candidates["gender"], lambda g: genders_compatible(subject_gender, g))
|
||||
]
|
||||
if candidates.empty:
|
||||
return []
|
||||
|
||||
candidates["distance_miles"] = _haversine_miles(
|
||||
lat, lon, candidates["latitude"].values, candidates["longitude"].values
|
||||
).round(1)
|
||||
|
||||
# ── Nearest first, and nothing else has a say ───────────────────────
|
||||
within = candidates[candidates["distance_miles"] <= reach]
|
||||
if len(within) < MINIMUM:
|
||||
return []
|
||||
|
||||
selected = within.sort_values(["distance_miles", "urn"]).head(MAX_SCHOOLS)
|
||||
return [
|
||||
{
|
||||
"urn": int(row["urn"]),
|
||||
"school_name": str(row.get("school_name") or ""),
|
||||
"distance_miles": float(row["distance_miles"]),
|
||||
"school_type": _native(row.get("school_type")),
|
||||
"age_range": _native(row.get("age_range")),
|
||||
# Each peer's own phase, not the subject's: the pool is a phase
|
||||
# group, so an all-through school can sit beside a primary. The
|
||||
# compare basket counts it against both of its tabs.
|
||||
"phase": _native(row.get("phase")),
|
||||
"shared": _shared(subject, row, is_secondary),
|
||||
"metric_value": _native(row.get(metric_key)),
|
||||
"metric_key": metric_key,
|
||||
"metric_year": _native(row.get("year")),
|
||||
}
|
||||
for _, row in selected.iterrows()
|
||||
]
|
||||
@@ -0,0 +1,356 @@
|
||||
"""The place registry: what places the site publishes, and what is in each.
|
||||
|
||||
One module owns this question. The pages, the sitemap and the internal-link
|
||||
modules all read from here, so the threshold and the collision rules exist in
|
||||
exactly one place and are testable without a browser or a database.
|
||||
|
||||
Two namespaces, never one. 67 viable town names collide with a local
|
||||
authority name, and the authority is the larger set in only 43 of them —
|
||||
postal towns cross authority boundaries, so neither can absorb the other.
|
||||
Keys are "<kind>:<slug>" so the collision cannot reappear in the dict.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Five schools with publishable data. Below this a place has nothing to say
|
||||
# that a list of schools does not, and publishing it is index bloat.
|
||||
MIN_SCHOOLS = 5
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Place:
|
||||
kind: str # "town" | "locality" | "authority" | "outcode"
|
||||
slug: str
|
||||
name: str
|
||||
urns: tuple[int, ...]
|
||||
parent_authority: str | None # authority NAME, for the 301 target
|
||||
# Every authority the place meaningfully sits in, largest first. A quarter
|
||||
# of outcodes and a third of towns straddle a boundary — SW19 is mostly
|
||||
# Merton but partly Wandsworth — so naming only one asserts something
|
||||
# false. parent_authority stays single because a redirect needs one
|
||||
# target; this is what the page shows.
|
||||
authorities: tuple[tuple[str, int], ...] = ()
|
||||
# URNs per phase, so the per-phase threshold can be applied without
|
||||
# re-querying. A place with 30 primaries and 2 secondaries publishes a
|
||||
# primary variant and no secondary one.
|
||||
phase_urns: dict[str, tuple[int, ...]] = field(default_factory=dict)
|
||||
|
||||
def publishes_phase(self, phase: str) -> bool:
|
||||
return len(self.phase_urns.get(phase, ())) >= MIN_SCHOOLS
|
||||
|
||||
@property
|
||||
def key(self) -> str:
|
||||
return f"{self.kind}:{self.slug}"
|
||||
|
||||
|
||||
def _publishable_urns(df) -> set[int]:
|
||||
"""URNs with something a page could state, deduplicated across years."""
|
||||
from backend.app import _PUBLISHABLE_FIELDS
|
||||
|
||||
cols = [c for c in _PUBLISHABLE_FIELDS if c in df.columns]
|
||||
if not cols:
|
||||
return set()
|
||||
return set(df.loc[df[cols].notna().any(axis=1), "urn"].astype(int))
|
||||
|
||||
|
||||
# The measure a phase page is built around. A page with no results in this
|
||||
# column has nothing a list of school names does not already give.
|
||||
_PHASE_METRIC = {
|
||||
"primary": "rwm_expected_pct",
|
||||
"secondary": "attainment_8_score",
|
||||
}
|
||||
|
||||
|
||||
def _phase_urns(group, publishable: set[int]) -> dict[str, tuple[int, ...]]:
|
||||
"""URNs per phase, counting only schools with a result for that phase.
|
||||
|
||||
Not merely "publishable". A school with an Ofsted grade and no results is
|
||||
worth a page of its own and belongs in the place list, but it cannot
|
||||
populate a phase page's results column — and the threshold is there to ask
|
||||
whether that column will have anything in it.
|
||||
|
||||
Counting publishable schools instead let /schools/kent/primary publish
|
||||
with none of its five rows carrying a result, and left 44 phase pages
|
||||
majority-blank. It is the same rule as "no page without a local average",
|
||||
which was never extended per phase.
|
||||
|
||||
All-through schools count toward both phases, matching the PHASE_GROUPS
|
||||
mapping the search filters already use.
|
||||
"""
|
||||
from backend.app import PHASE_GROUPS
|
||||
|
||||
if "phase" not in group.columns:
|
||||
return {}
|
||||
lowered = group["phase"].fillna("").str.lower()
|
||||
|
||||
out: dict[str, tuple[int, ...]] = {}
|
||||
for phase in ("primary", "secondary"):
|
||||
wanted = PHASE_GROUPS.get(phase, set())
|
||||
subset = group[lowered.isin(wanted)]
|
||||
|
||||
# The page lists every school of the phase; the threshold counts only
|
||||
# those carrying a result, so a mostly-empty table never publishes.
|
||||
metric = _PHASE_METRIC[phase]
|
||||
with_result = (
|
||||
{int(u) for u in subset.loc[subset[metric].notna(), "urn"]}
|
||||
if metric in subset.columns else set()
|
||||
)
|
||||
if len(with_result & publishable) < MIN_SCHOOLS:
|
||||
continue
|
||||
|
||||
urns = tuple(sorted({int(u) for u in subset["urn"]} & publishable))
|
||||
if urns:
|
||||
out[phase] = urns
|
||||
return out
|
||||
|
||||
|
||||
# A place is described by an authority when it holds at least a tenth of the
|
||||
# schools, and at least two. GIAS carries occasional postcode errors — EN6
|
||||
# lists two Shropshire schools among fourteen in Hertfordshire — and a bare
|
||||
# "any authority present" rule would print those as though they were real.
|
||||
# There is deliberately no cap on how many are named. An earlier cut stopped
|
||||
# at three, which silently dropped the fourth in exactly the case where the
|
||||
# information matters most — a genuinely fragmented place. The share rule is
|
||||
# the only limit, and it already bounds the list at ten.
|
||||
_AUTHORITY_MIN_SHARE = 0.10
|
||||
_AUTHORITY_MIN_SCHOOLS = 2
|
||||
|
||||
|
||||
def _authorities(group) -> tuple[tuple[str, int], ...]:
|
||||
"""Authorities this place meaningfully sits in, largest first."""
|
||||
from backend.app import EXCLUDED_FILTER_VALUES
|
||||
|
||||
if "local_authority" not in group.columns:
|
||||
return ()
|
||||
counts = group["local_authority"].dropna().value_counts()
|
||||
total = int(counts.sum())
|
||||
if not total:
|
||||
return ()
|
||||
|
||||
kept = [
|
||||
(str(name), int(n)) for name, n in counts.items()
|
||||
if str(name) not in EXCLUDED_FILTER_VALUES
|
||||
and n >= _AUTHORITY_MIN_SCHOOLS
|
||||
and n / total >= _AUTHORITY_MIN_SHARE
|
||||
]
|
||||
# A place too small or too fragmented for the share rule still names its
|
||||
# largest authority, or the page would say nothing about where it is.
|
||||
if not kept:
|
||||
for name, n in counts.items():
|
||||
if str(name) not in EXCLUDED_FILTER_VALUES:
|
||||
return ((str(name), int(n)),)
|
||||
return ()
|
||||
return tuple(kept)
|
||||
|
||||
|
||||
def _parent_authority(authorities: tuple[tuple[str, int], ...]) -> str | None:
|
||||
"""The 301 target: the largest authority a place sits in.
|
||||
|
||||
Derived from `authorities` rather than computed separately. The first cut
|
||||
used `mode()` here while `authorities` used `value_counts()`, and on an
|
||||
exact tie pandas does not guarantee the two pick the same name — so the
|
||||
redirect could have pointed somewhere other than the authority the page
|
||||
named first. One computation, one answer.
|
||||
|
||||
Deriving it also inherits the sentinel filter, so a place can no longer
|
||||
redirect to /schools/authority/does-not-apply.
|
||||
"""
|
||||
return authorities[0][0] if authorities else None
|
||||
|
||||
|
||||
def _group(df, column: str, kind: str, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per distinct SLUG in `column` that clears the threshold.
|
||||
|
||||
Grouped by slug, not by raw value, because GIAS spells the same place
|
||||
several ways and they all resolve to one URL. Five town slugs come from
|
||||
more than one spelling: "London" (1,819 schools) and "LONDON" (12) both
|
||||
slugify to `london`; Weston-super-Mare is split 14/19 across two
|
||||
spellings; Newcastle-under-Lyme across three.
|
||||
|
||||
Grouping by raw value meant the later group simply overwrote the earlier
|
||||
one in this dict — so /schools/london could have shown twelve schools
|
||||
instead of 1,819, silently and depending on row order.
|
||||
|
||||
The display name is the most common spelling, which is the one a reader
|
||||
expects to see.
|
||||
"""
|
||||
from backend.app import _slugify
|
||||
|
||||
if column not in df.columns:
|
||||
return {}
|
||||
|
||||
working = df.assign(_slug=df[column].map(
|
||||
lambda v: _slugify(str(v).strip()) if isinstance(v, str) and v.strip() else None))
|
||||
working = working[working["_slug"].notna() & (working["_slug"] != "")]
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for slug, group in working.groupby("_slug"):
|
||||
slug = str(slug)
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
spellings = group[column].dropna().value_counts()
|
||||
if spellings.empty:
|
||||
continue
|
||||
name = str(spellings.index[0]).strip()
|
||||
authorities = () if kind == "authority" else _authorities(group)
|
||||
place = Place(
|
||||
kind=kind, slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(authorities),
|
||||
authorities=authorities,
|
||||
phase_urns=_phase_urns(group, publishable),
|
||||
)
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
# "SW11 2AA" -> "SW11". Two letters max, one or two digits, optional letter.
|
||||
_OUTCODE_RE = re.compile(r"^([A-Z]{1,2}\d{1,2}[A-Z]?)\s")
|
||||
|
||||
|
||||
def _outcode(postcode) -> str | None:
|
||||
if not isinstance(postcode, str):
|
||||
return None
|
||||
m = _OUTCODE_RE.match(postcode.upper().strip())
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def _outcode_places(df, publishable: set[int]) -> dict[str, Place]:
|
||||
"""One Place per postcode district clearing the threshold.
|
||||
|
||||
These carry no phase variants: nobody searches "primary schools in SW11",
|
||||
so the spec gives them no /primary or /secondary route. `phase_urns` is
|
||||
left empty rather than computed and then filtered downstream — the
|
||||
registry is the one place that decides which phases a place publishes,
|
||||
and the page links whatever it reports.
|
||||
|
||||
Computing them here put a link to a route that does not exist on every one
|
||||
of the 1,720 outcode pages.
|
||||
"""
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
working = working[working["_oc"].notna()]
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for oc, group in working.groupby("_oc"):
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
continue
|
||||
authorities = _authorities(group)
|
||||
place = Place(kind="outcode", slug=str(oc).lower(), name=str(oc),
|
||||
urns=urns, parent_authority=_parent_authority(authorities),
|
||||
authorities=authorities)
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
def _locality_places(df, publishable: set[int],
|
||||
town_slugs: set[str]) -> dict[str, Place]:
|
||||
"""One Place per curated locality clearing the threshold."""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
if "postcode" not in df.columns:
|
||||
return {}
|
||||
working = df.assign(_oc=df["postcode"].map(_outcode))
|
||||
|
||||
out: dict[str, Place] = {}
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
if slug in town_slugs:
|
||||
# Skip, do not raise. The guard exists so a locality never
|
||||
# silently shadows a town — skipping achieves that, and the error
|
||||
# log makes it loud.
|
||||
#
|
||||
# Raising here took down sitemap generation for all 25,000 school
|
||||
# pages when "richmond" met the GIAS town Richmond in North
|
||||
# Yorkshire. Worse, GIAS town names change without any code change,
|
||||
# so a raise means curated data can break the site spontaneously.
|
||||
# A curation mistake must cost one page, not the sitemap.
|
||||
logger.error(
|
||||
"locality %r collides with the published town of the same "
|
||||
"slug and has been skipped; rename it or remove it", slug)
|
||||
continue
|
||||
group = working[working["_oc"].isin(outcodes)]
|
||||
urns = tuple(sorted({int(u) for u in group["urn"]} & publishable))
|
||||
if len(urns) < MIN_SCHOOLS:
|
||||
# Not an error — a locality can legitimately be too small. Logged
|
||||
# because one you meant to publish quietly vanishing is the
|
||||
# failure worth hearing about.
|
||||
logger.warning(
|
||||
"locality %s (%s) has %d publishable schools, below the "
|
||||
"threshold of %d - not published",
|
||||
slug, ", ".join(outcodes), len(urns), MIN_SCHOOLS)
|
||||
continue
|
||||
authorities = _authorities(group)
|
||||
place = Place(kind="locality", slug=slug, name=name, urns=urns,
|
||||
parent_authority=_parent_authority(authorities),
|
||||
authorities=authorities,
|
||||
phase_urns=_phase_urns(group, publishable))
|
||||
out[place.key] = place
|
||||
return out
|
||||
|
||||
|
||||
# Ordered authority → town/locality → outcode, widest first, because that is
|
||||
# the order a breadcrumb reads. The link module re-sorts for its own purposes.
|
||||
_PLACE_ORDER = {"authority": 0, "town": 1, "locality": 2, "outcode": 3}
|
||||
|
||||
|
||||
def build_place_index(registry: dict[str, Place]) -> dict[int, tuple[Place, ...]]:
|
||||
"""URN → the published places containing it, built once per registry.
|
||||
|
||||
The reverse of the registry, and the thing school pages link out through.
|
||||
Derived from the registry rather than maintained beside it, so the two
|
||||
cannot disagree about which places exist: a place below the publish
|
||||
threshold is absent from the registry, so it is absent from here too, and
|
||||
a link is never offered for a page that does not exist.
|
||||
|
||||
Built as an index rather than scanned per call because /api/schools/{urn}
|
||||
is the site's highest-traffic endpoint. Scanning meant walking every place
|
||||
and doing a tuple membership test against each — on the order of 10^5
|
||||
comparisons per request, repeated for every school page view. One pass at
|
||||
registry-build time replaces all of it with a dict lookup.
|
||||
"""
|
||||
grouped: dict[int, list[Place]] = {}
|
||||
for place in registry.values():
|
||||
for urn in place.urns:
|
||||
grouped.setdefault(int(urn), []).append(place)
|
||||
|
||||
return {
|
||||
urn: tuple(sorted(places,
|
||||
key=lambda p: (_PLACE_ORDER.get(p.kind, 9), p.slug)))
|
||||
for urn, places in grouped.items()
|
||||
}
|
||||
|
||||
|
||||
def places_for_urn(index: dict[int, tuple[Place, ...]], urn: int) -> tuple[Place, ...]:
|
||||
"""The published places containing this school, widest first.
|
||||
|
||||
Empty is a real answer, not a failure: a school whose town and authority
|
||||
both fall below the publish threshold has nowhere to link, and the page
|
||||
renders without the module.
|
||||
"""
|
||||
return index.get(int(urn), ())
|
||||
|
||||
|
||||
def build_place_registry(df) -> dict[str, Place]:
|
||||
"""Every place the site publishes, keyed by "<kind>:<slug>"."""
|
||||
if df.empty or "urn" not in df.columns:
|
||||
return {}
|
||||
|
||||
publishable = _publishable_urns(df)
|
||||
registry: dict[str, Place] = {}
|
||||
registry.update(_group(df, "local_authority", "authority", publishable))
|
||||
|
||||
towns = _group(df, "town", "town", publishable)
|
||||
registry.update(towns)
|
||||
|
||||
town_slugs = {p.slug for p in towns.values()}
|
||||
registry.update(_locality_places(df, publishable, town_slugs))
|
||||
registry.update(_outcode_places(df, publishable))
|
||||
return registry
|
||||
@@ -532,6 +532,31 @@ RANKING_COLUMNS = [
|
||||
"gcse_grade_91_pct",
|
||||
]
|
||||
|
||||
# Maps user-facing phase filter values to the GIAS PhaseOfEducation values they
|
||||
# include. All-through schools appear in both primary and secondary results,
|
||||
# which is why this is a set per phase rather than a single string comparison.
|
||||
#
|
||||
# Lives here rather than in app.py because nearby_schools.py needs it too, and
|
||||
# importing app from there would be a cycle.
|
||||
PHASE_GROUPS: dict[str, set[str]] = {
|
||||
"primary": {"primary", "middle deemed primary", "all-through"},
|
||||
"secondary": {"secondary", "middle deemed secondary", "all-through", "16 plus"},
|
||||
"all-through": {"all-through"},
|
||||
}
|
||||
|
||||
# GIAS phases in the order a child meets them, for the phase filter's options.
|
||||
# All-through spans the whole path, so it follows the stages. Lowercased, as
|
||||
# PHASE_GROUPS is, so a change of case in the GIAS label keeps its place.
|
||||
PHASE_ORDER: list[str] = [
|
||||
"nursery",
|
||||
"primary",
|
||||
"middle deemed primary",
|
||||
"middle deemed secondary",
|
||||
"secondary",
|
||||
"16 plus",
|
||||
"all-through",
|
||||
]
|
||||
|
||||
# School listing columns
|
||||
SCHOOL_COLUMNS = [
|
||||
"urn",
|
||||
@@ -544,6 +569,7 @@ SCHOOL_COLUMNS = [
|
||||
"religious_denomination",
|
||||
"age_range",
|
||||
"has_sixth_form",
|
||||
"nursery_provision",
|
||||
"status",
|
||||
"gender",
|
||||
"admissions_policy",
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
The 34 GIAS establishment types describe governance and funding, which for a
|
||||
mainstream state school barely changes what a parent experiences. The search
|
||||
filter offers five groups a parent recognises instead, and a faith filter in
|
||||
place of the faith signal that "Voluntary aided" and "Voluntary controlled"
|
||||
only half carry. See
|
||||
docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md.
|
||||
|
||||
Groups are defined over GIAS codes and looked up by the translated name,
|
||||
because the DataFrame the API filters carries names only: the codes are
|
||||
replaced at load (data_loader.translate_gias_code_columns), the legacy-name
|
||||
mart fallback never had them, and test fixtures are written in names.
|
||||
"""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
|
||||
# (key, label, GIAS TypeOfEstablishment codes), in the order shown.
|
||||
TYPE_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Every mainstream state school, academy or council-run: Academy sponsor
|
||||
# led, Academy converter, Free schools, University technical college,
|
||||
# Studio schools, City technology college, Community, Voluntary aided,
|
||||
# Voluntary controlled, Foundation, LA nursery. Academy against council-run
|
||||
# was two near-halves of one pool, and did not follow the difference a
|
||||
# parent feels most, admissions: voluntary aided and foundation schools
|
||||
# set their own, as academies do. Faith, which voluntary aided mostly
|
||||
# meant, has its own filter.
|
||||
("state", "State school (free)",
|
||||
frozenset({28, 34, 35, 40, 41, 6, 1, 2, 3, 5, 15})),
|
||||
("independent", "Independent (fee-paying)", frozenset({11})),
|
||||
# Every special type, independent ones included (usually funded by the
|
||||
# council through an EHCP, so SEND provision to a parent, not private
|
||||
# school), and Special post 16 institutions.
|
||||
("special", "Special school (SEND)", frozenset({7, 8, 10, 12, 32, 33, 36, 44})),
|
||||
# Further education, Sixth form centres, the 16-19 academies and free schools.
|
||||
("post16", "Sixth form or college", frozenset({18, 31, 39, 45, 46})),
|
||||
# Pupil referral units and AP academies and free schools. Last: parents do
|
||||
# not apply to these; the local authority places children there.
|
||||
("alternative", "Alternative provision", frozenset({14, 38, 42, 43})),
|
||||
)
|
||||
|
||||
# In no group, so reachable only under "Any school type": Secure units,
|
||||
# Miscellaneous, Higher education institutions, Online provider, Institution
|
||||
# funded by other government department, Academy secure 16 to 19.
|
||||
UNOFFERED_TYPE_CODES: frozenset[int] = frozenset({24, 27, 29, 49, 56, 57})
|
||||
|
||||
# (key, label, GIAS ReligiousCharacter codes), in the order shown. A joint
|
||||
# school is in every faith its label names; a generic "Christian" beside a
|
||||
# named church adds nothing.
|
||||
FAITH_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Does not apply, None, and 99 (a blank label).
|
||||
("none", "No religious character", frozenset({0, 6, 99})),
|
||||
("church_of_england", "Church of England",
|
||||
frozenset({2, 9, 10, 11, 12, 13, 19, 20, 30, 31, 32, 33, 34, 41, 48})),
|
||||
("roman_catholic", "Roman Catholic", frozenset({3, 11, 13, 35, 48})),
|
||||
# 28 "Inter- / non- denominational" is how GIAS files Christian schools
|
||||
# tied to no one church.
|
||||
("other_christian", "Other Christian",
|
||||
frozenset({4, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 22, 26, 28, 30, 33,
|
||||
37, 38, 39, 40, 41, 44, 45, 46, 47})),
|
||||
("jewish", "Jewish", frozenset({5, 36, 43})),
|
||||
("muslim", "Muslim", frozenset({7, 42, 49})),
|
||||
("other_faith", "Other faith", frozenset({21, 24, 25, 29})),
|
||||
)
|
||||
|
||||
TYPE_GROUP_KEYS: frozenset[str] = frozenset(k for k, _, _ in TYPE_GROUPS)
|
||||
|
||||
# Keys a group was offered under before, so their links keep working: "state"
|
||||
# was "academy" and "council" until 2026-10-02.
|
||||
TYPE_GROUP_ALIASES: dict[str, str] = {"academy": "state", "council": "state"}
|
||||
FAITH_KEYS: frozenset[str] = frozenset(k for k, _, _ in FAITH_GROUPS)
|
||||
|
||||
|
||||
def _key(name: str) -> str:
|
||||
return name.strip().lower()
|
||||
|
||||
|
||||
_TYPE_GROUP_BY_NAME: dict[str, str] = {
|
||||
_key(SCHOOL_TYPE[code]): key
|
||||
for key, _, codes in TYPE_GROUPS
|
||||
for code in codes
|
||||
if code in SCHOOL_TYPE
|
||||
}
|
||||
|
||||
_FAITHS_BY_NAME: dict[str, tuple[str, ...]] = {}
|
||||
for _faith, _, _codes in FAITH_GROUPS:
|
||||
for _code in sorted(_codes):
|
||||
if _code in RELIGIOUS_CHARACTER:
|
||||
_name = _key(RELIGIOUS_CHARACTER[_code])
|
||||
_FAITHS_BY_NAME[_name] = _FAITHS_BY_NAME.get(_name, ()) + (_faith,)
|
||||
|
||||
|
||||
def type_group_key(value: str) -> Optional[str]:
|
||||
"""The type group a school_type URL value names, old keys included, or
|
||||
None when it names no group (an old link's raw GIAS type)."""
|
||||
v = value.strip().lower()
|
||||
v = TYPE_GROUP_ALIASES.get(v, v)
|
||||
return v if v in TYPE_GROUP_KEYS else None
|
||||
|
||||
|
||||
def type_group_for(name: object) -> Optional[str]:
|
||||
"""The type group of a GIAS establishment type name, or None."""
|
||||
if not isinstance(name, str):
|
||||
return None
|
||||
return _TYPE_GROUP_BY_NAME.get(_key(name))
|
||||
|
||||
|
||||
def faith_groups_for(name: object) -> tuple[str, ...]:
|
||||
"""The faith groups of a GIAS religious character name.
|
||||
|
||||
A missing or blank name is "No religious character". A name the
|
||||
dictionary does not know has no faith, so it matches no faith option.
|
||||
"""
|
||||
if not isinstance(name, str):
|
||||
return ("none",) if name is None or pd.isna(name) else ()
|
||||
return _FAITHS_BY_NAME.get(_key(name), ())
|
||||
@@ -0,0 +1,269 @@
|
||||
"""The destinations serialiser's contract.
|
||||
|
||||
Not rendering a figure is not the same as not publishing it. This endpoint is
|
||||
public and unauthenticated, so whatever the payload carries is published,
|
||||
whatever the UI draws. The categories sum to the cohort and the pupil groups
|
||||
sum to each other, so a lone suppressed cell is solvable by subtraction — the
|
||||
serialiser adds secondary suppression to prevent it.
|
||||
|
||||
See docs/superpowers/specs/2026-08-28-destination-measures-design.md.
|
||||
"""
|
||||
|
||||
from backend.data_loader import (
|
||||
_destinations_block, _format_cohort_year, disclosure_invariant_holds,
|
||||
)
|
||||
|
||||
|
||||
def _row(group, measure, pupils, status, cohort=180, percentage=None, year=202223):
|
||||
return {
|
||||
"pupil_group": group,
|
||||
"destination_measure": measure,
|
||||
"pupils": pupils,
|
||||
"percentage": percentage,
|
||||
"status": status,
|
||||
"cohort_pupils": cohort,
|
||||
"year": year,
|
||||
}
|
||||
|
||||
|
||||
def test_suppressed_category_serialises_as_suppressed_with_null_pupils():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
_row("all", "sixth_form_college", None, "suppressed"),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
cats = {c["category"]: c for c in block["groups"]["all"]["categories"]}
|
||||
assert cats["sixth_form_college"]["status"] == "suppressed"
|
||||
assert cats["sixth_form_college"]["pupils"] is None
|
||||
assert cats["sixth_form_college"]["percentage"] is None
|
||||
|
||||
|
||||
def test_published_category_keeps_its_figures():
|
||||
block = _destinations_block([
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
])
|
||||
cat = block["groups"]["all"]["categories"][0]
|
||||
assert cat["pupils"] == 75
|
||||
assert cat["percentage"] == 41.7
|
||||
assert cat["status"] == "published"
|
||||
|
||||
|
||||
def test_only_the_latest_year_is_served():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 60, "published", year=202122),
|
||||
_row("all", "school_sixth_form", 75, "published", year=202223),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert block["cohort_year"] == "2022/23"
|
||||
assert len(block["groups"]["all"]["categories"]) == 1
|
||||
assert block["groups"]["all"]["categories"][0]["pupils"] == 75
|
||||
|
||||
|
||||
def test_all_three_pupil_groups_are_carried():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published"),
|
||||
_row("disadvantaged", "school_sixth_form", 17, "published", cohort=62),
|
||||
_row("other", "school_sixth_form", 58, "published", cohort=118),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert set(block["groups"]) == {"all", "disadvantaged", "other"}
|
||||
assert block["groups"]["disadvantaged"]["cohort"] == 62
|
||||
|
||||
|
||||
def test_cohort_year_is_reported_so_the_page_can_date_itself():
|
||||
block = _destinations_block([_row("all", "school_sixth_form", 75, "published")])
|
||||
assert block["cohort_year"] == "2022/23"
|
||||
|
||||
|
||||
def test_format_cohort_year_handles_the_six_digit_form():
|
||||
assert _format_cohort_year(202223) == "2022/23"
|
||||
assert _format_cohort_year(None) is None
|
||||
|
||||
|
||||
def test_empty_rows_yield_none_not_an_empty_shell():
|
||||
assert _destinations_block([]) is None
|
||||
|
||||
|
||||
# ── Disclosure control ──────────────────────────────────────────────────────
|
||||
#
|
||||
# The rendering guards in lib/destinations.ts stop a withheld figure being
|
||||
# DRAWN. They do nothing about it being COMPUTED: this endpoint is public and
|
||||
# unauthenticated, so whatever the payload carries is published. These tests
|
||||
# are the ones that matter.
|
||||
|
||||
def _solve_residual(group):
|
||||
"""What any caller can work out: cohort minus everything published."""
|
||||
published = [c["pupils"] for c in group["categories"] if c["pupils"] is not None]
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
return group["cohort"] - sum(published), len(hidden)
|
||||
|
||||
|
||||
def test_a_lone_suppressed_category_cannot_be_solved_for():
|
||||
"""Whitley Bay High School's real 2022/23 disadvantaged group: further
|
||||
education withheld, everything else published, cohort 41. Before secondary
|
||||
suppression the payload gave the answer away as 41 - 23 = 18."""
|
||||
rows = [
|
||||
_row("disadvantaged", "school_sixth_form", 15, "published", cohort=41),
|
||||
_row("disadvantaged", "sixth_form_college", 0, "published", cohort=41),
|
||||
_row("disadvantaged", "further_education", None, "suppressed", cohort=41),
|
||||
_row("disadvantaged", "apprenticeship", 1, "published", cohort=41),
|
||||
_row("disadvantaged", "employment", 2, "published", cohort=41),
|
||||
_row("disadvantaged", "not_sustained", 3, "published", cohort=41),
|
||||
_row("disadvantaged", "not_captured", 2, "published", cohort=41),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["disadvantaged"]
|
||||
residual, hidden = _solve_residual(group)
|
||||
assert hidden >= 2, "a lone suppressed cell must gain a companion"
|
||||
assert residual != 18, "the withheld figure is recoverable from the payload"
|
||||
|
||||
|
||||
def test_every_group_hides_none_or_at_least_two_categories():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published"),
|
||||
_row("all", "sixth_form_college", None, "suppressed"),
|
||||
_row("all", "further_education", 61, "published"),
|
||||
_row("all", "apprenticeship", 8, "published"),
|
||||
_row("all", "employment", 6, "published"),
|
||||
_row("all", "not_sustained", 5, "published"),
|
||||
_row("all", "not_captured", 4, "published"),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
assert len(hidden) >= 2
|
||||
|
||||
|
||||
def test_a_category_hidden_in_one_group_is_hidden_in_a_second():
|
||||
"""disadvantaged + other = all for every category, so a category withheld
|
||||
in exactly one of the three is recoverable from the other two."""
|
||||
rows = []
|
||||
for measure, a, d, o in [
|
||||
("school_sixth_form", 75, None, 58),
|
||||
("further_education", 61, 27, 34),
|
||||
("apprenticeship", 8, 4, 4),
|
||||
("employment", 6, 1, 5),
|
||||
("not_sustained", 5, 3, 2),
|
||||
("not_captured", 4, 2, 2),
|
||||
]:
|
||||
rows.append(_row("all", measure, a, "published", cohort=159))
|
||||
rows.append(_row("disadvantaged", measure, d,
|
||||
"published" if d is not None else "suppressed", cohort=37))
|
||||
rows.append(_row("other", measure, o, "published", cohort=122))
|
||||
|
||||
groups = _destinations_block(rows)["groups"]
|
||||
measures = {c["category"] for g in groups.values() for c in g["categories"]}
|
||||
assert len(measures) == 6, "the fixture's six measures must all be checked"
|
||||
|
||||
for measure in sorted(measures):
|
||||
hidden = sum(
|
||||
1 for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == measure and c["status"] == "suppressed"
|
||||
)
|
||||
# The invariant is "none, or at least two" — not "at least two".
|
||||
assert hidden != 1, f"{measure} is solvable across the pupil groups"
|
||||
|
||||
|
||||
def test_a_suppressed_cell_never_keeps_its_percentage():
|
||||
"""percentage / pupils would hand back the cohort, and with it the residual."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
_row("all", "sixth_form_college", None, "suppressed", percentage=11.7),
|
||||
_row("all", "further_education", 61, "published", percentage=33.9),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
for cell in group["categories"]:
|
||||
if cell["status"] != "published":
|
||||
assert cell["pupils"] is None
|
||||
assert cell["percentage"] is None
|
||||
|
||||
|
||||
def test_aggregates_are_not_served():
|
||||
"""An aggregate spanning exactly one suppressed component names it, and
|
||||
nothing renders them today."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published"),
|
||||
_row("all", "agg_sustained_all", 171, "published"),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
assert [c["category"] for c in group["categories"]] == ["school_sixth_form"]
|
||||
assert "aggregates" not in group
|
||||
|
||||
|
||||
def test_a_fully_published_group_is_left_alone():
|
||||
"""Secondary suppression must not cost anything where nothing is withheld —
|
||||
this is the all-pupils view on every mainstream secondary."""
|
||||
rows = [
|
||||
_row("all", m, p, "published")
|
||||
for m, p in [("school_sixth_form", 75), ("sixth_form_college", 21),
|
||||
("further_education", 61), ("apprenticeship", 8),
|
||||
("employment", 6), ("not_sustained", 5), ("not_captured", 4)]
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
assert all(c["status"] == "published" for c in group["categories"])
|
||||
assert len(group["categories"]) == 7
|
||||
|
||||
|
||||
def test_the_invariant_is_asserted_directly_not_re_derived():
|
||||
"""A group with one suppressed category and nothing else to withhold."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
|
||||
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
|
||||
_row("all", "further_education", None, "not_applicable", cohort=9),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert block is None or disclosure_invariant_holds(block["groups"])
|
||||
|
||||
|
||||
def test_a_sparse_cohort_with_no_companion_drops_the_group():
|
||||
"""Special schools and AP routinely have one suppressed category and every
|
||||
other one not applicable. There is nothing left to withhold, so the group
|
||||
goes — an earlier version returned here with the violation intact."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
|
||||
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
|
||||
_row("all", "further_education", None, "not_applicable", cohort=9),
|
||||
_row("all", "apprenticeship", None, "not_applicable", cohort=9),
|
||||
_row("all", "employment", None, "not_applicable", cohort=9),
|
||||
_row("all", "not_sustained", None, "not_applicable", cohort=9),
|
||||
_row("all", "not_captured", None, "not_applicable", cohort=9),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert block is None or "all" not in block["groups"], (
|
||||
"a group that cannot be made safe must not be served"
|
||||
)
|
||||
|
||||
|
||||
def test_zeros_are_not_treated_as_a_usable_companion():
|
||||
"""Suppressing a zero protects nothing — the residual is unchanged. With
|
||||
only zeros available the group must be dropped, not falsely 'fixed'."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=5),
|
||||
_row("all", "sixth_form_college", 0, "published", cohort=5),
|
||||
_row("all", "further_education", 0, "published", cohort=5),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
if block and "all" in block["groups"]:
|
||||
group = block["groups"]["all"]
|
||||
published = sum(c["pupils"] for c in group["categories"]
|
||||
if c["pupils"] is not None)
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
assert len(hidden) != 1, "a zero companion leaves the figure solvable"
|
||||
assert group["cohort"] - published != 5
|
||||
|
||||
|
||||
def test_masking_always_terminates_in_a_safe_state():
|
||||
"""Exhaustive over every suppression pattern of a four-category group."""
|
||||
from itertools import product
|
||||
MEASURES = ["school_sixth_form", "sixth_form_college",
|
||||
"further_education", "apprenticeship"]
|
||||
for statuses in product(["published", "suppressed", "not_applicable"],
|
||||
repeat=len(MEASURES)):
|
||||
rows = [
|
||||
_row("all", m, 3 if st == "published" else None, st, cohort=12)
|
||||
for m, st in zip(MEASURES, statuses)
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
if block is None:
|
||||
continue
|
||||
assert disclosure_invariant_holds(block["groups"]), (
|
||||
f"invariant broken for {statuses}"
|
||||
)
|
||||
@@ -0,0 +1,163 @@
|
||||
"""Tests for the feature flag layer (spec 2026-08-23).
|
||||
|
||||
None of these need a running Unleash. That is the point: an unset UNLEASH_URL
|
||||
means every flag is False, which is what local development and CI get.
|
||||
"""
|
||||
|
||||
from datetime import date, timedelta
|
||||
|
||||
from backend import flags
|
||||
|
||||
|
||||
def test_every_declared_flag_is_keyed_by_its_own_name():
|
||||
# One string is the registry key, the Unleash flag name and the JSON key.
|
||||
# A mismatch here would mean the UI toggles a flag the code never reads.
|
||||
for key, flag in flags.REGISTRY.items():
|
||||
assert key == flag.name
|
||||
|
||||
|
||||
def test_flag_names_are_snake_case():
|
||||
# Matches the API's existing convention (admission_distance,
|
||||
# rwm_expected_pct) so no case transformation exists to get wrong.
|
||||
for name in flags.REGISTRY:
|
||||
assert name == name.lower()
|
||||
assert "-" not in name and " " not in name
|
||||
|
||||
|
||||
def test_an_unconfigured_client_evaluates_every_flag_false(monkeypatch):
|
||||
monkeypatch.setattr(flags, "_client", None)
|
||||
for name in flags.REGISTRY:
|
||||
assert flags.is_enabled(name) is False
|
||||
|
||||
|
||||
def test_an_undeclared_flag_is_false_rather_than_an_error(monkeypatch):
|
||||
# A typo'd flag name must not raise in a request path. It is logged as an
|
||||
# error, because an undeclared flag is always a bug.
|
||||
monkeypatch.setattr(flags, "_client", None)
|
||||
assert flags.is_enabled("no_such_flag") is False
|
||||
|
||||
|
||||
def test_an_exploding_client_is_false_rather_than_a_500(monkeypatch):
|
||||
class Boom:
|
||||
def is_enabled(self, *a, **kw):
|
||||
raise RuntimeError("unleash is on fire")
|
||||
|
||||
monkeypatch.setattr(flags, "_client", Boom())
|
||||
name = next(iter(flags.REGISTRY))
|
||||
assert flags.is_enabled(name) is False
|
||||
|
||||
|
||||
def test_all_flags_reports_every_declared_flag(monkeypatch):
|
||||
monkeypatch.setattr(flags, "_client", None)
|
||||
assert set(flags.all_flags()) == set(flags.REGISTRY)
|
||||
assert all(v is False for v in flags.all_flags().values())
|
||||
|
||||
|
||||
def test_a_flag_older_than_the_limit_fails_this_test():
|
||||
"""A tripwire, not an assertion about correctness.
|
||||
|
||||
Flags are temporary scaffolding and the failure mode of every flag system
|
||||
is accumulation. This fails on the day a flag turns 90, on whatever PR
|
||||
happens to be open — which is the point: someone has to decide.
|
||||
|
||||
To fix: delete the flag and the branches that read it, or, if it genuinely
|
||||
still needs to exist, move its `added` date and say why in the commit.
|
||||
"""
|
||||
stale = [
|
||||
f.name for f in flags.REGISTRY.values()
|
||||
if date.today() - f.added > timedelta(days=flags.MAX_FLAG_AGE_DAYS)
|
||||
]
|
||||
assert not stale, (
|
||||
f"Flags older than {flags.MAX_FLAG_AGE_DAYS} days: {stale}. "
|
||||
"Remove the flag and the code branches it guards, or move its `added` "
|
||||
"date deliberately."
|
||||
)
|
||||
|
||||
|
||||
def _client():
|
||||
from fastapi.testclient import TestClient
|
||||
from backend import app as app_module
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_the_flags_endpoint_lists_every_declared_flag(monkeypatch):
|
||||
monkeypatch.setattr(flags, "_client", None)
|
||||
body = _client().get("/api/flags").json()
|
||||
assert set(body) == set(flags.REGISTRY)
|
||||
|
||||
|
||||
def test_the_flags_endpoint_answers_false_when_unleash_is_unreachable(monkeypatch):
|
||||
# The endpoint must still answer. A frontend that cannot read flags renders
|
||||
# everything dark, which is right; one that gets a 500 renders nothing.
|
||||
monkeypatch.setattr(flags, "_client", None)
|
||||
res = _client().get("/api/flags")
|
||||
assert res.status_code == 200
|
||||
assert all(v is False for v in res.json().values())
|
||||
|
||||
|
||||
def _school_payload(monkeypatch, *, flag_on: bool):
|
||||
"""Fetch one school's payload with the distance flag forced on or off.
|
||||
|
||||
The DataFrame shape is copied from test_school_details.py rather than
|
||||
minimised: the endpoint reads a wide set of GIAS columns, and a trimmed
|
||||
frame fails for reasons that have nothing to do with flags.
|
||||
"""
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from fastapi.testclient import TestClient
|
||||
from backend import app as app_module
|
||||
|
||||
df = pd.DataFrame([{
|
||||
"urn": 150275,
|
||||
"school_name": "West London Performing Arts Academy",
|
||||
"phase": "Secondary",
|
||||
"school_type": "Special post 16 institution",
|
||||
"trust_name": None,
|
||||
"religious_denomination": "Does not apply",
|
||||
"gender": None,
|
||||
"age_range": "16-25",
|
||||
"admissions_policy": None,
|
||||
"capacity": np.nan,
|
||||
"gias_total_pupils": np.nan,
|
||||
"headteacher_name": None,
|
||||
"website": None,
|
||||
"ofsted_grade": np.nan,
|
||||
"local_authority": "Ealing",
|
||||
"address": "268 Northfield Avenue, London, W5 4UB",
|
||||
"postcode": "W5 4UB",
|
||||
"latitude": 51.4986,
|
||||
"longitude": -0.3148,
|
||||
"year": np.nan,
|
||||
"total_pupils": np.nan,
|
||||
"eligible_pupils": np.nan,
|
||||
"rwm_expected_pct": np.nan,
|
||||
}])
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
# Two arguments: get_supplementary_data(db, urn). See backend/app.py.
|
||||
monkeypatch.setattr(
|
||||
app_module, "get_supplementary_data",
|
||||
lambda db, urn: {"admission_distance": {"distance_m": 772.49,
|
||||
"year": 2024}})
|
||||
monkeypatch.setattr(flags, "is_enabled", lambda name: flag_on)
|
||||
|
||||
client = TestClient(app_module.app, raise_server_exceptions=False)
|
||||
res = client.get("/api/schools/150275")
|
||||
assert res.status_code == 200, res.text
|
||||
return res.json()
|
||||
|
||||
|
||||
def test_the_distance_field_is_absent_when_the_flag_is_off(monkeypatch):
|
||||
"""Absent, not null, and withheld at the source.
|
||||
|
||||
/api/schools/ is public and unauthenticated. Leaving a withheld field in
|
||||
the payload while declining to render it hands the record to anyone who
|
||||
opens the network tab — the reasoning already recorded in c9a1892.
|
||||
"""
|
||||
body = _school_payload(monkeypatch, flag_on=False)
|
||||
assert "admission_distance" not in body
|
||||
|
||||
|
||||
def test_the_distance_field_is_present_when_the_flag_is_on(monkeypatch):
|
||||
body = _school_payload(monkeypatch, flag_on=True)
|
||||
assert body["admission_distance"]["distance_m"] == 772.49
|
||||
@@ -0,0 +1,395 @@
|
||||
"""Selection rules for the nearby-schools section.
|
||||
|
||||
Hard filters encode claims the section is not allowed to make — that a
|
||||
selective school is an alternative to a non-selective one, that a special
|
||||
school is comparable to a mainstream one, or that a Girls school is an option
|
||||
for a Boys school's reader. They decide who is eligible.
|
||||
|
||||
Distance decides the order, and nothing else does. An earlier version ranked by
|
||||
intake similarity first, which put a Catholic school 2.9 miles away above the
|
||||
community school 0.3 miles down the road — for a primary, a school that far is
|
||||
not a weaker option, it is not an option. Similarity is now reported on the
|
||||
card and never reorders the row.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from backend.nearby_schools import (
|
||||
is_secondary_phase,
|
||||
radius_miles,
|
||||
select_nearby,
|
||||
)
|
||||
|
||||
BASE_LAT, BASE_LON = 51.5000, -0.1000
|
||||
|
||||
|
||||
def _row(urn, name, **overrides):
|
||||
base = {
|
||||
"urn": urn,
|
||||
"school_name": name,
|
||||
"local_authority": "Testshire",
|
||||
"school_type": "Community school",
|
||||
"phase": "Primary",
|
||||
"age_range": "4-11",
|
||||
"status": "Open",
|
||||
"gender": "Mixed",
|
||||
"religious_denomination": "None",
|
||||
"admissions_policy": "Not applicable",
|
||||
"latitude": BASE_LAT,
|
||||
"longitude": BASE_LON,
|
||||
"year": 202425,
|
||||
"rwm_expected_pct": 70.0,
|
||||
"attainment_8_score": np.nan,
|
||||
}
|
||||
base.update(overrides)
|
||||
return base
|
||||
|
||||
|
||||
def _frame(*rows):
|
||||
return pd.DataFrame(list(rows))
|
||||
|
||||
|
||||
def _at(miles):
|
||||
"""A latitude `miles` north of BASE_LAT."""
|
||||
return BASE_LAT + miles / 69.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Order: distance, and only distance
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_returns_nearest_first():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "Mid", latitude=_at(1.0)),
|
||||
_row(100003, "Near", latitude=_at(0.4)),
|
||||
_row(100004, "Far", latitude=_at(1.8)),
|
||||
)
|
||||
result = select_nearby(frame, 100001)
|
||||
assert [s["urn"] for s in result] == [100003, 100002, 100004]
|
||||
assert result[0]["distance_miles"] == 0.4
|
||||
|
||||
|
||||
def test_a_faith_match_never_outranks_a_closer_school():
|
||||
"""The reported defect. A Catholic primary surrounded by Catholic primaries
|
||||
showed six of them and omitted the community school down the road."""
|
||||
frame = _frame(
|
||||
_row(100001, "St Jude's RC Primary", religious_denomination="Roman Catholic"),
|
||||
_row(100002, "Elm Grove Primary", religious_denomination="None", latitude=_at(0.3)),
|
||||
_row(100003, "Holy Cross RC", religious_denomination="Roman Catholic", latitude=_at(0.8)),
|
||||
_row(100004, "Sacred Heart RC", religious_denomination="Roman Catholic", latitude=_at(1.2)),
|
||||
_row(100005, "St Peter's RC", religious_denomination="Roman Catholic", latitude=_at(1.6)),
|
||||
)
|
||||
result = select_nearby(frame, 100001)
|
||||
assert result[0]["urn"] == 100002, "the nearest school leads, whatever its intake"
|
||||
assert [s["distance_miles"] for s in result] == sorted(s["distance_miles"] for s in result)
|
||||
|
||||
|
||||
def test_the_nearest_eligible_school_is_always_shown():
|
||||
"""Whatever else changes, a section titled "nearby" cannot omit the nearest
|
||||
school while listing one four times further away."""
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"),
|
||||
_row(100002, "Nearest", gender="Mixed", religious_denomination="None", latitude=_at(0.2)),
|
||||
*[
|
||||
_row(100010 + n, f"Match {n}", gender="Boys",
|
||||
religious_denomination="Roman Catholic", latitude=_at(0.9 + n * 0.1))
|
||||
for n in range(6)
|
||||
],
|
||||
)
|
||||
assert select_nearby(frame, 100001)[0]["urn"] == 100002
|
||||
|
||||
|
||||
def test_caps_at_six_taking_the_nearest():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
*[_row(100010 + n, f"Peer {n}", latitude=_at(0.1 * (n + 1))) for n in range(7)],
|
||||
)
|
||||
result = select_nearby(frame, 100001)
|
||||
assert len(result) == 6
|
||||
assert 100016 not in {s["urn"] for s in result}, "the seventh-nearest is the one dropped"
|
||||
|
||||
|
||||
def test_fewer_than_two_matches_returns_empty():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "Only neighbour", latitude=_at(0.5)),
|
||||
)
|
||||
assert select_nearby(frame, 100001) == []
|
||||
|
||||
|
||||
def test_excludes_the_subject_school():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "A", latitude=_at(0.5)),
|
||||
_row(100003, "B", latitude=_at(0.6)),
|
||||
)
|
||||
assert 100001 not in {s["urn"] for s in select_nearby(frame, 100001)}
|
||||
|
||||
|
||||
def test_a_school_is_never_listed_twice():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "A", latitude=_at(0.5)),
|
||||
_row(100003, "B", latitude=_at(0.6)),
|
||||
)
|
||||
result = select_nearby(frame, 100001)
|
||||
assert len(result) == len({s["urn"] for s in result})
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reach: a sanity bound, not a target
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_primary_does_not_reach_past_two_miles():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "Just inside", latitude=_at(1.9)),
|
||||
_row(100003, "Just outside", latitude=_at(2.4)),
|
||||
_row(100004, "Miles away", latitude=_at(4.0)),
|
||||
)
|
||||
# One inside the cap is below the minimum, so nothing renders at all —
|
||||
# a primary with nothing within two miles has no nearby schools.
|
||||
assert select_nearby(frame, 100001) == []
|
||||
|
||||
|
||||
def test_secondary_reaches_further_than_primary():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", phase="Secondary"),
|
||||
_row(100002, "A", phase="Secondary", latitude=_at(3.0)),
|
||||
_row(100003, "B", phase="Secondary", latitude=_at(5.5)),
|
||||
)
|
||||
assert {s["urn"] for s in select_nearby(frame, 100001)} == {100002, 100003}
|
||||
|
||||
|
||||
def test_each_card_carries_its_own_phase():
|
||||
# The compare basket limits each phase separately, so an all-through peer
|
||||
# must not inherit the subject's "Primary".
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "A", latitude=_at(0.5)),
|
||||
_row(100003, "B", phase="All-through", age_range="4-18", latitude=_at(0.6)),
|
||||
)
|
||||
phases = {s["urn"]: s["phase"] for s in select_nearby(frame, 100001)}
|
||||
assert phases == {100002: "Primary", 100003: "All-through"}
|
||||
|
||||
|
||||
def test_the_cap_follows_the_phase():
|
||||
assert radius_miles("Primary") == 2.0
|
||||
assert radius_miles("Middle deemed primary") == 2.0
|
||||
assert radius_miles("All-through") == 2.0
|
||||
assert radius_miles("Secondary") == 6.0
|
||||
assert radius_miles("Middle deemed secondary") == 6.0
|
||||
# Post-16 is the phase people travel furthest for.
|
||||
assert radius_miles("16 plus") == 10.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Hard filters: eligibility, never order
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_selective_never_meets_non_selective():
|
||||
frame = _frame(
|
||||
_row(100001, "Grammar", phase="Secondary", admissions_policy="Selective"),
|
||||
_row(100002, "Comp A", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.5)),
|
||||
_row(100003, "Comp B", phase="Secondary", admissions_policy="Non-selective", latitude=_at(0.6)),
|
||||
)
|
||||
assert select_nearby(frame, 100001) == []
|
||||
assert 100001 not in {s["urn"] for s in select_nearby(frame, 100002)}
|
||||
|
||||
|
||||
def test_special_schools_match_only_each_other():
|
||||
frame = _frame(
|
||||
_row(100001, "Special", school_type="Community special school"),
|
||||
_row(100002, "Mainstream A", latitude=_at(0.5)),
|
||||
_row(100003, "Mainstream B", latitude=_at(0.6)),
|
||||
)
|
||||
assert select_nearby(frame, 100001) == []
|
||||
assert select_nearby(frame, 100002) == []
|
||||
|
||||
|
||||
def test_boys_never_meets_girls():
|
||||
frame = _frame(
|
||||
_row(100001, "Boys School", gender="Boys"),
|
||||
_row(100002, "Girls School", gender="Girls", latitude=_at(0.5)),
|
||||
_row(100003, "Mixed School", gender="Mixed", latitude=_at(0.6)),
|
||||
_row(100004, "Another Mixed", gender="Mixed", latitude=_at(0.7)),
|
||||
)
|
||||
urns = {s["urn"] for s in select_nearby(frame, 100001)}
|
||||
assert 100002 not in urns
|
||||
assert urns == {100003, 100004}
|
||||
|
||||
|
||||
def test_closed_schools_and_missing_coordinates_are_dropped():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "Closed", status="Closed", latitude=_at(0.5)),
|
||||
_row(100003, "No coords", latitude=np.nan, longitude=np.nan),
|
||||
_row(100004, "Good A", latitude=_at(0.6)),
|
||||
_row(100005, "Good B", latitude=_at(0.7)),
|
||||
)
|
||||
assert {s["urn"] for s in select_nearby(frame, 100001)} == {100004, 100005}
|
||||
|
||||
|
||||
def test_all_through_is_offered_on_both_phase_sides():
|
||||
frame = _frame(
|
||||
_row(100001, "Primary subject", phase="Primary"),
|
||||
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
|
||||
_row(100003, "Primary peer", phase="Primary", latitude=_at(0.6)),
|
||||
)
|
||||
assert 100002 in {s["urn"] for s in select_nearby(frame, 100001)}
|
||||
|
||||
secondary = _frame(
|
||||
_row(100010, "Secondary subject", phase="Secondary"),
|
||||
_row(100002, "All through", phase="All-through", latitude=_at(0.5)),
|
||||
_row(100011, "Secondary peer", phase="Secondary", latitude=_at(0.6)),
|
||||
)
|
||||
assert 100002 in {s["urn"] for s in select_nearby(secondary, 100010)}
|
||||
|
||||
|
||||
def test_sixteen_plus_is_matched_against_secondary_not_primary():
|
||||
"""GIAS phase 6 is "16 plus", and PHASE_GROUPS puts it in the secondary
|
||||
group — a sixth-form college's peers are secondaries and other colleges,
|
||||
never primary schools. A substring test for "secondary" misses it silently:
|
||||
no crash, just a page offering infant schools to a sixth form."""
|
||||
frame = _frame(
|
||||
_row(100001, "Sixth Form College", phase="16 plus", age_range="16-19"),
|
||||
_row(100002, "Nearby Secondary", phase="Secondary", latitude=_at(0.5),
|
||||
attainment_8_score=52.0),
|
||||
_row(100003, "Nearby College", phase="16 plus", latitude=_at(0.6)),
|
||||
_row(100004, "Nearby Primary", phase="Primary", latitude=_at(0.1)),
|
||||
)
|
||||
result = select_nearby(frame, 100001)
|
||||
urns = {s["urn"] for s in result}
|
||||
assert 100004 not in urns, "a primary school is not a peer for a sixth form"
|
||||
assert urns == {100002, 100003}
|
||||
assert all(s["metric_key"] == "attainment_8_score" for s in result)
|
||||
|
||||
|
||||
def test_is_secondary_phase_agrees_with_the_phase_groups_it_selects_from():
|
||||
for phase in ("Secondary", "Middle deemed secondary", "16 plus"):
|
||||
assert is_secondary_phase(phase) is True, phase
|
||||
for phase in ("Primary", "Middle deemed primary", "Nursery", "", None):
|
||||
assert is_secondary_phase(phase) is False, phase
|
||||
# In PHASE_GROUPS an all-through school is on both sides, but it renders
|
||||
# with the primary template, and the metric follows the phase side.
|
||||
assert is_secondary_phase("All-through") is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# What the card reports
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_shared_lists_only_what_is_actually_shared():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", phase="Secondary", gender="Mixed",
|
||||
religious_denomination="None", admissions_policy="Non-selective"),
|
||||
_row(100002, "Full match", phase="Secondary", gender="Mixed",
|
||||
religious_denomination="None", admissions_policy="Non-selective", latitude=_at(0.5)),
|
||||
_row(100003, "Faith differs", phase="Secondary", gender="Mixed",
|
||||
religious_denomination="Church of England", admissions_policy="Non-selective", latitude=_at(0.6)),
|
||||
)
|
||||
by_urn = {s["urn"]: s for s in select_nearby(frame, 100001)}
|
||||
assert by_urn[100002]["shared"] == ["Mixed", "Non-selective", "No religious character"]
|
||||
assert by_urn[100003]["shared"] == ["Mixed", "Non-selective"]
|
||||
|
||||
|
||||
def test_a_shared_faith_is_named():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", religious_denomination="Roman Catholic"),
|
||||
_row(100002, "Also RC", religious_denomination="Roman Catholic", latitude=_at(0.4)),
|
||||
_row(100003, "Secular", religious_denomination="None", latitude=_at(0.5)),
|
||||
)
|
||||
by_urn = {s["urn"]: s for s in select_nearby(frame, 100001)}
|
||||
assert "Roman Catholic" in by_urn[100002]["shared"]
|
||||
assert by_urn[100003]["shared"] == ["Mixed"]
|
||||
|
||||
|
||||
def test_shared_is_empty_when_nothing_is_shared():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", gender="Boys", religious_denomination="Roman Catholic"),
|
||||
_row(100002, "A", gender="Mixed", religious_denomination="None", latitude=_at(0.4)),
|
||||
_row(100003, "B", gender="Mixed", religious_denomination="Church of England", latitude=_at(0.5)),
|
||||
)
|
||||
assert all(s["shared"] == [] for s in select_nearby(frame, 100001))
|
||||
|
||||
|
||||
def test_no_tier_is_reported_because_there_are_no_tiers():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "A", latitude=_at(0.4)),
|
||||
_row(100003, "B", latitude=_at(0.5)),
|
||||
)
|
||||
assert all("tier" not in s for s in select_nearby(frame, 100001))
|
||||
|
||||
|
||||
def test_metric_follows_the_phase_side_not_the_neighbour():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject", phase="Secondary", attainment_8_score=50.0),
|
||||
_row(100002, "A", phase="Secondary", attainment_8_score=52.8, latitude=_at(0.5)),
|
||||
_row(100003, "B", phase="Secondary", attainment_8_score=np.nan, latitude=_at(0.6)),
|
||||
)
|
||||
by_urn = {s["urn"]: s for s in select_nearby(frame, 100001)}
|
||||
assert by_urn[100002]["metric_key"] == "attainment_8_score"
|
||||
assert by_urn[100002]["metric_value"] == 52.8
|
||||
assert by_urn[100002]["metric_year"] == 202425
|
||||
assert by_urn[100003]["metric_value"] is None
|
||||
|
||||
|
||||
def test_values_are_json_safe_native_types():
|
||||
frame = _frame(
|
||||
_row(100001, "Subject"),
|
||||
_row(100002, "A", latitude=_at(0.5)),
|
||||
_row(100003, "B", latitude=_at(0.6)),
|
||||
)
|
||||
for school in select_nearby(frame, 100001):
|
||||
assert isinstance(school["urn"], int)
|
||||
assert isinstance(school["distance_miles"], float)
|
||||
assert not isinstance(school["metric_value"], np.generic)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The endpoint
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _endpoint_frame():
|
||||
return _frame(
|
||||
_row(100001, "Subject Primary"),
|
||||
_row(100002, "Neighbour A", latitude=_at(0.5)),
|
||||
_row(100003, "Neighbour B", latitude=_at(0.6)),
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _endpoint_frame)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _endpoint_frame)
|
||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_detail_payload_carries_nearby_schools(client):
|
||||
resp = client.get("/api/schools/100001")
|
||||
assert resp.status_code == 200, resp.text
|
||||
similar = resp.json()["nearby_schools"]
|
||||
assert [s["school_name"] for s in similar] == ["Neighbour A", "Neighbour B"]
|
||||
assert similar[0]["metric_key"] == "rwm_expected_pct"
|
||||
|
||||
|
||||
def test_a_failure_in_selection_does_not_break_the_page(client, monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
def _explode(*args, **kwargs):
|
||||
raise ValueError("selection blew up")
|
||||
|
||||
monkeypatch.setattr(app_module, "select_nearby", _explode)
|
||||
resp = client.get("/api/schools/100001")
|
||||
assert resp.status_code == 200, resp.text
|
||||
assert resp.json()["nearby_schools"] == []
|
||||
@@ -0,0 +1,106 @@
|
||||
"""The /api/schools phase filter.
|
||||
|
||||
The search page offers every GIAS phase, but the filter only knew the three
|
||||
grouped ones (primary, secondary, all-through). Anything else — nursery,
|
||||
16 plus, the middle-deemed phases — fell through to no filter at all, so
|
||||
"Nursery" returned the whole result set, mostly primaries.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
PHASES = {
|
||||
100001: "Nursery",
|
||||
100002: "Primary",
|
||||
100003: "Middle deemed primary",
|
||||
100004: "Secondary",
|
||||
100005: "Middle deemed secondary",
|
||||
100006: "16 plus",
|
||||
100007: "All-through",
|
||||
}
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Testshire",
|
||||
"school_type": "Academy",
|
||||
"address": "1 Test Street",
|
||||
"town": "Testtown",
|
||||
"postcode": "TS1 1AA",
|
||||
"religious_denomination": None,
|
||||
"age_range": "4-11",
|
||||
"has_sixth_form": None,
|
||||
"gender": "Mixed",
|
||||
"admissions_policy": None,
|
||||
"ofsted_grade": np.nan,
|
||||
"ofsted_date": None,
|
||||
"ofsted_framework": None,
|
||||
"latitude": 51.5,
|
||||
"longitude": -0.1,
|
||||
"year": 202425,
|
||||
"total_pupils": 300,
|
||||
"rwm_expected_pct": np.nan,
|
||||
"attainment_8_score": np.nan,
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": urn, "school_name": f"{phase} School", "phase": phase}
|
||||
for urn, phase in PHASES.items()
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def _urns(client, phase):
|
||||
resp = client.get("/api/schools", params={"phase": phase})
|
||||
assert resp.status_code == 200, resp.text
|
||||
return sorted(s["urn"] for s in resp.json()["schools"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("phase, urn", [
|
||||
("nursery", 100001),
|
||||
("16 plus", 100006),
|
||||
("middle deemed primary", 100003),
|
||||
("middle deemed secondary", 100005),
|
||||
])
|
||||
def test_an_ungrouped_phase_matches_exactly(client, phase, urn):
|
||||
assert _urns(client, phase) == [urn]
|
||||
|
||||
|
||||
def test_grouped_phases_still_take_in_their_related_phases(client):
|
||||
assert _urns(client, "primary") == [100002, 100003, 100007]
|
||||
assert _urns(client, "secondary") == [100004, 100005, 100006, 100007]
|
||||
|
||||
|
||||
def test_an_unknown_phase_returns_nothing_rather_than_everything(client):
|
||||
assert _urns(client, "kindergarten") == []
|
||||
|
||||
|
||||
def test_filters_lists_phases_in_the_order_a_child_meets_them(client):
|
||||
# Alphabetical put "16 plus" and "All-through" first and Nursery fifth.
|
||||
# All-through spans the whole path, so it comes after the stages.
|
||||
assert client.get("/api/filters").json()["phases"] == [
|
||||
"Nursery",
|
||||
"Primary",
|
||||
"Middle deemed primary",
|
||||
"Middle deemed secondary",
|
||||
"Secondary",
|
||||
"16 plus",
|
||||
"All-through",
|
||||
]
|
||||
|
||||
|
||||
def test_an_unknown_phase_follows_the_known_ones():
|
||||
from backend.app import order_phases
|
||||
|
||||
assert order_phases(["Secondary", "Zeta", "Alpha", "Nursery"]) == [
|
||||
"Nursery", "Secondary", "Alpha", "Zeta",
|
||||
]
|
||||
@@ -0,0 +1,497 @@
|
||||
"""Tests for the place registry (spec 2026-08-21).
|
||||
|
||||
The registry is built from the in-memory school DataFrame, so these build a
|
||||
small frame directly rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from backend.places import (MIN_SCHOOLS, build_place_index,
|
||||
build_place_registry, places_for_urn)
|
||||
|
||||
|
||||
def _df(rows: list[dict]) -> pd.DataFrame:
|
||||
base = {
|
||||
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": np.nan,
|
||||
"phase": "Primary", "postcode": "AA1 1AA",
|
||||
}
|
||||
return pd.DataFrame([{**base, **r} for r in rows])
|
||||
|
||||
|
||||
def _town(n: int, town: str, la: str, start: int = 100000, **kw) -> list[dict]:
|
||||
"""`start` offsets the URNs so two calls can describe different schools —
|
||||
the Bedford case needs two authorities' worth of distinct URNs in one
|
||||
town."""
|
||||
return [
|
||||
{"urn": start + i, "school_name": f"{town} School {i}",
|
||||
"town": town, "local_authority": la, **kw}
|
||||
for i in range(n)
|
||||
]
|
||||
|
||||
|
||||
def test_town_clearing_the_threshold_is_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
assert "town:brentwood" in reg
|
||||
assert reg["town:brentwood"].name == "Brentwood"
|
||||
assert len(reg["town:brentwood"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_town_below_the_threshold_is_not_published():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS - 1, "Crosby", "Sefton")))
|
||||
assert "town:crosby" not in reg
|
||||
|
||||
|
||||
def test_a_town_below_threshold_still_names_its_authority():
|
||||
# The route layer needs somewhere to 301 to.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS - 1, "Crosby", "Sefton") + _town(MIN_SCHOOLS, "Bootle", "Sefton")))
|
||||
assert "authority:sefton" in reg
|
||||
|
||||
|
||||
def test_town_and_authority_of_the_same_name_are_separate_places():
|
||||
# 67 real collisions. Neither set contains the other: Bedford the town has
|
||||
# 104 schools, Bedford the authority 86, because postal towns cross
|
||||
# authority boundaries.
|
||||
rows = (_town(MIN_SCHOOLS, "Bedford", "Bedford")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Central Bedfordshire", start=200000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
town, authority = reg["town:bedford"], reg["authority:bedford"]
|
||||
assert set(town.urns) != set(authority.urns)
|
||||
assert len(town.urns) == MIN_SCHOOLS * 2 # both authorities' schools
|
||||
assert len(authority.urns) == MIN_SCHOOLS # only this authority's
|
||||
|
||||
|
||||
def test_schools_without_publishable_data_do_not_count_toward_the_threshold():
|
||||
rows = _town(MIN_SCHOOLS, "Ghosttown", "Nowhere")
|
||||
for r in rows:
|
||||
r["rwm_expected_pct"] = np.nan
|
||||
r["ofsted_grade"] = np.nan
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "town:ghosttown" not in reg
|
||||
|
||||
|
||||
def test_blank_town_is_ignored():
|
||||
rows = _town(MIN_SCHOOLS, "", "Essex")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("town:") for k in reg)
|
||||
|
||||
|
||||
def test_a_school_is_counted_once_even_with_several_years_of_rows():
|
||||
rows = []
|
||||
for year in (202324, 202425):
|
||||
rows += [{**r, "year": year} for r in _town(MIN_SCHOOLS, "Beccles", "Suffolk")]
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:beccles"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_groups_schools_by_outcode(monkeypatch):
|
||||
# The GIAS town field collapses 1,819 London schools into "London", so a
|
||||
# locality is defined by its postcode districts instead.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"battersea": ("Battersea", ("SW11",))})
|
||||
rows = _town(MIN_SCHOOLS, "London", "Wandsworth")
|
||||
for r in rows:
|
||||
r["postcode"] = "SW11 2AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["locality:battersea"].name == "Battersea"
|
||||
assert len(reg["locality:battersea"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_locality_below_the_threshold_is_not_published(monkeypatch):
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"nowhere": ("Nowhere", ("ZZ99",))})
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "London", "Wandsworth")))
|
||||
assert "locality:nowhere" not in reg
|
||||
|
||||
|
||||
def test_a_locality_may_not_shadow_a_viable_town(monkeypatch, caplog):
|
||||
"""A colliding locality is skipped loudly, and the town survives.
|
||||
|
||||
This used to raise, which took down sitemap generation for all 25,000
|
||||
school pages the first time a curated slug met a real GIAS town. Curated
|
||||
data must not be able to break the site — and GIAS town names change with
|
||||
no code change at all, so the raise could fire spontaneously.
|
||||
"""
|
||||
import logging
|
||||
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
|
||||
with caplog.at_level(logging.ERROR):
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
assert "locality:brentwood" not in reg # skipped
|
||||
assert "town:brentwood" in reg # the town is untouched
|
||||
assert "brentwood" in caplog.text # and it was loud about it
|
||||
|
||||
|
||||
def test_a_locality_collision_does_not_break_the_rest_of_the_registry(monkeypatch):
|
||||
# The whole point of skipping rather than raising.
|
||||
from backend import localities
|
||||
monkeypatch.setattr(localities, "LOCALITY_OUTCODES",
|
||||
{"brentwood": ("Brentwood", ("CM13",))})
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert "authority:essex" in reg
|
||||
assert "outcode:cm13" in reg
|
||||
|
||||
|
||||
def test_outcode_places_are_built_from_postcodes():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "CM13 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:cm13"].name == "CM13"
|
||||
assert len(reg["outcode:cm13"].urns) == MIN_SCHOOLS
|
||||
|
||||
|
||||
def test_malformed_postcodes_do_not_create_places():
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
for r in rows:
|
||||
r["postcode"] = "not a postcode"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not any(k.startswith("outcode:") for k in reg)
|
||||
|
||||
|
||||
def test_every_curated_locality_is_structurally_valid():
|
||||
# Guards the hand-maintained file: real slug, real name, real outcodes.
|
||||
import re
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
assert LOCALITY_OUTCODES, "the curated locality list must not be empty"
|
||||
for slug, (name, outcodes) in LOCALITY_OUTCODES.items():
|
||||
assert re.fullmatch(r"[a-z0-9-]+", slug), slug
|
||||
assert name.strip() == name and name, slug
|
||||
assert outcodes, f"{slug} has no outcodes"
|
||||
for oc in outcodes:
|
||||
assert re.fullmatch(r"[A-Z]{1,2}\d{1,2}[A-Z]?", oc), (slug, oc)
|
||||
|
||||
|
||||
def test_the_pipeline_seed_mirrors_the_canonical_module():
|
||||
"""Two copies with no drift guard is worse than one copy.
|
||||
|
||||
backend/localities.py is canonical because the backend image does not
|
||||
contain pipeline/. The seed exists so the warehouse can join on the same
|
||||
definitions, and this is what stops the two diverging — the same
|
||||
arrangement assert_gias_code_names_match_seed.sql gives gias_codes.
|
||||
"""
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
seed_path = (Path(__file__).resolve().parents[2]
|
||||
/ "pipeline/transform/seeds/locality_outcodes.csv")
|
||||
assert seed_path.exists(), f"missing seed mirror at {seed_path}"
|
||||
|
||||
seed = {
|
||||
row["locality_slug"]: (row["locality_name"],
|
||||
tuple(row["outcodes"].split("|")))
|
||||
for row in csv.DictReader(seed_path.open())
|
||||
}
|
||||
assert seed == LOCALITY_OUTCODES
|
||||
|
||||
|
||||
def test_no_curated_locality_names_a_london_borough():
|
||||
"""Boroughs are authorities and already have a page.
|
||||
|
||||
A locality defined by two or three outcodes inside a borough would be a
|
||||
partial, near-duplicate subset of that authority page — the exact
|
||||
thin-content failure the two-namespace design exists to avoid. Hackney,
|
||||
Islington, Greenwich and Ealing were all in the first draft.
|
||||
|
||||
Hardcoded rather than read from the corpus because this must fail in CI,
|
||||
where there is no database.
|
||||
"""
|
||||
from backend.localities import LOCALITY_OUTCODES
|
||||
|
||||
boroughs = {
|
||||
"barking-and-dagenham", "barnet", "bexley", "brent", "bromley",
|
||||
"camden", "croydon", "ealing", "enfield", "greenwich", "hackney",
|
||||
"hammersmith-and-fulham", "haringey", "harrow", "havering",
|
||||
"hillingdon", "hounslow", "islington", "kensington-and-chelsea",
|
||||
"kingston-upon-thames", "lambeth", "lewisham", "merton", "newham",
|
||||
"redbridge", "richmond-upon-thames", "southwark", "sutton",
|
||||
"tower-hamlets", "waltham-forest", "wandsworth", "westminster",
|
||||
}
|
||||
named = boroughs & set(LOCALITY_OUTCODES)
|
||||
assert not named, (
|
||||
f"these are boroughs, not districts: {sorted(named)} - they already "
|
||||
"have an authority page covering every school"
|
||||
)
|
||||
|
||||
|
||||
def test_a_place_names_every_authority_it_straddles():
|
||||
"""SW19 is mostly Merton but partly Wandsworth.
|
||||
|
||||
A quarter of viable outcodes and a third of viable towns cross an
|
||||
authority boundary, so naming only the largest asserts something false.
|
||||
"""
|
||||
rows = (_town(26, "London", "Merton", start=300000)
|
||||
+ _town(7, "London", "Wandsworth", start=400000))
|
||||
for r in rows:
|
||||
r["postcode"] = "SW19 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
names = [n for n, _ in reg["outcode:sw19"].authorities]
|
||||
assert names == ["Merton", "Wandsworth"] # largest first
|
||||
assert dict(reg["outcode:sw19"].authorities)["Wandsworth"] == 7
|
||||
|
||||
|
||||
def test_the_redirect_target_stays_a_single_authority():
|
||||
# parent_authority and authorities do different jobs: a 301 needs one
|
||||
# target, the page needs the truth.
|
||||
rows = (_town(26, "London", "Merton", start=300000)
|
||||
+ _town(7, "London", "Wandsworth", start=400000))
|
||||
for r in rows:
|
||||
r["postcode"] = "SW19 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["outcode:sw19"].parent_authority == "Merton"
|
||||
|
||||
|
||||
def test_a_stray_authority_below_the_share_threshold_is_not_named():
|
||||
# GIAS carries postcode errors — EN6 lists two Shropshire schools among
|
||||
# fourteen in Hertfordshire. Printing those as though real would be worse
|
||||
# than omitting them.
|
||||
rows = (_town(30, "Barnet", "Hertfordshire", start=300000)
|
||||
+ _town(1, "Barnet", "Shropshire", start=400000))
|
||||
for r in rows:
|
||||
r["postcode"] = "EN6 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert [n for n, _ in reg["outcode:en6"].authorities] == ["Hertfordshire"]
|
||||
|
||||
|
||||
def test_a_sentinel_authority_is_never_named():
|
||||
rows = (_town(20, "London", "Merton", start=300000)
|
||||
+ _town(6, "London", "Does not apply", start=400000))
|
||||
for r in rows:
|
||||
r["postcode"] = "SW19 1AA"
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert [n for n, _ in reg["outcode:sw19"].authorities] == ["Merton"]
|
||||
|
||||
|
||||
def test_a_place_always_names_at_least_one_authority():
|
||||
# Even when every authority is below the share threshold, the page has to
|
||||
# say where the place is.
|
||||
rows = []
|
||||
for i, la in enumerate(["A", "B", "C", "D", "E", "F", "G"]):
|
||||
rows += _town(1, "Fragmented", la, start=300000 + i * 100)
|
||||
reg = build_place_registry(_df(rows))
|
||||
place = reg.get("town:fragmented")
|
||||
assert place is not None
|
||||
assert len(place.authorities) == 1
|
||||
|
||||
|
||||
def test_every_qualifying_authority_is_named_with_no_cap():
|
||||
"""An earlier cut stopped at three, dropping the fourth silently.
|
||||
|
||||
That truncation bit exactly where the information matters most — a
|
||||
genuinely fragmented place — and nothing recorded it.
|
||||
"""
|
||||
rows = []
|
||||
for i, la in enumerate(["Hackney", "Lambeth", "Westminster", "Lewisham"]):
|
||||
rows += _town(3, "Fourway", la, start=300000 + i * 100)
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:fourway"].authorities) == 4
|
||||
|
||||
|
||||
def test_the_redirect_target_is_the_authority_named_first():
|
||||
"""They were computed separately — mode() against value_counts() — and on
|
||||
an exact tie pandas does not guarantee the two agree."""
|
||||
rows = (_town(26, "London", "Merton", start=300000)
|
||||
+ _town(7, "London", "Wandsworth", start=400000))
|
||||
for r in rows:
|
||||
r["postcode"] = "SW19 1AA"
|
||||
place = build_place_registry(_df(rows))["outcode:sw19"]
|
||||
assert place.parent_authority == place.authorities[0][0]
|
||||
|
||||
|
||||
def test_a_place_never_redirects_to_a_sentinel_authority():
|
||||
# Deriving the parent from `authorities` inherits its sentinel filter.
|
||||
rows = (_town(6, "Someplace", "Does not apply", start=300000)
|
||||
+ _town(5, "Someplace", "Essex", start=400000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["town:someplace"].parent_authority == "Essex"
|
||||
|
||||
|
||||
def test_spellings_of_one_place_are_merged_not_overwritten():
|
||||
"""GIAS spells the same place several ways, and they share a URL.
|
||||
|
||||
"London" (1,819 schools) and "LONDON" (12) both slugify to `london`.
|
||||
Grouping by raw value let the later group overwrite the earlier one, so
|
||||
the page could have shown twelve schools instead of 1,819 — silently, and
|
||||
depending on row order.
|
||||
"""
|
||||
rows = (_town(6, "Weston-super-Mare", "North Somerset", start=300000)
|
||||
+ _town(5, "Weston-Super-Mare", "North Somerset", start=400000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert len(reg["town:weston-super-mare"].urns) == 11
|
||||
|
||||
|
||||
def test_the_merged_place_takes_its_most_common_spelling():
|
||||
rows = (_town(9, "Newcastle-under-Lyme", "Staffordshire", start=300000)
|
||||
+ _town(5, "NEWCASTLE-UNDER-LYME", "Staffordshire", start=400000))
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["town:newcastle-under-lyme"].name == "Newcastle-under-Lyme"
|
||||
|
||||
|
||||
def test_a_phase_page_needs_results_not_merely_publishable_schools():
|
||||
"""/schools/kent/primary published with none of its five rows scored.
|
||||
|
||||
The threshold counted schools that were publishable — a result OR an
|
||||
Ofsted grade — while the page exists for its results column. Forty-four
|
||||
phase pages were majority-blank; one had no results at all.
|
||||
"""
|
||||
rows = _town(MIN_SCHOOLS, "Kent", "Kent")
|
||||
for r in rows:
|
||||
r["rwm_expected_pct"] = np.nan # Ofsted only, no results
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
assert "town:kent" in reg # the place still publishes
|
||||
assert not reg["town:kent"].publishes_phase("primary")
|
||||
|
||||
|
||||
def test_a_phase_page_publishes_once_enough_schools_carry_a_result():
|
||||
rows = _town(MIN_SCHOOLS, "Beccles", "Suffolk")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert reg["town:beccles"].publishes_phase("primary")
|
||||
|
||||
|
||||
def test_a_publishing_phase_page_still_lists_its_unscored_schools():
|
||||
"""The threshold gates whether the page exists; it does not filter rows.
|
||||
|
||||
A parent looking up a school by name has to find it whether or not it
|
||||
published results.
|
||||
"""
|
||||
scored = _town(MIN_SCHOOLS, "Beccles", "Suffolk", start=300000)
|
||||
unscored = _town(2, "Beccles", "Suffolk", start=400000)
|
||||
for r in unscored:
|
||||
r["rwm_expected_pct"] = np.nan
|
||||
reg = build_place_registry(_df(scored + unscored))
|
||||
|
||||
place = reg["town:beccles"]
|
||||
assert place.publishes_phase("primary")
|
||||
assert len(place.phase_urns["primary"]) == MIN_SCHOOLS + 2
|
||||
|
||||
|
||||
def test_the_secondary_threshold_counts_its_own_metric():
|
||||
# A town full of scored primaries must not thereby publish a secondary page.
|
||||
rows = _town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
reg = build_place_registry(_df(rows))
|
||||
assert not reg["town:brentwood"].publishes_phase("secondary")
|
||||
|
||||
|
||||
def test_an_outcode_publishes_no_phase_variants():
|
||||
"""There is no /schools/near/[outcode]/[phase] route, by design.
|
||||
|
||||
Nobody searches "primary schools in SW11", so the spec gives outcodes no
|
||||
phase variants. The registry computed them anyway, and the place page —
|
||||
which links whatever phases the registry reports — put two 404s on every
|
||||
outcode page in the site.
|
||||
|
||||
This is the single rule now: a kind with no phase route reports no phases,
|
||||
so neither the page nor the sitemap can offer one.
|
||||
"""
|
||||
rows = [{"urn": 500000 + i, "school_name": f"SW11 School {i}",
|
||||
"town": "London", "local_authority": "Wandsworth",
|
||||
"postcode": "SW11 1AA"} for i in range(MIN_SCHOOLS + 3)]
|
||||
reg = build_place_registry(_df(rows))
|
||||
|
||||
place = reg["outcode:sw11"]
|
||||
assert place.phase_urns == {}
|
||||
assert not place.publishes_phase("primary")
|
||||
assert not place.publishes_phase("secondary")
|
||||
|
||||
|
||||
def test_an_authority_still_publishes_phase_variants():
|
||||
"""Authorities keep theirs — "primary schools in Kent" is a real query,
|
||||
and /schools/authority/[la]/[phase] is the route that serves it."""
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Maidstone", "Kent")))
|
||||
assert reg["authority:kent"].publishes_phase("primary")
|
||||
|
||||
|
||||
# ── The reverse index: which published places contain a school ──────────────
|
||||
#
|
||||
# School pages link out to the location layer through this. It is the whole
|
||||
# point of the index: before it, ~27k school pages linked to nothing on the
|
||||
# site and stranded whatever authority they held.
|
||||
|
||||
def test_a_school_resolves_to_every_published_place_containing_it():
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
places = places_for_urn(build_place_index(reg), 100000)
|
||||
|
||||
kinds = {p.kind for p in places}
|
||||
assert "town" in kinds
|
||||
assert "authority" in kinds
|
||||
|
||||
|
||||
def test_a_school_in_an_unpublished_town_still_resolves_to_its_authority():
|
||||
# A town below the threshold has no page, so there is no link to offer —
|
||||
# but the authority above it clears the threshold on the same schools and
|
||||
# is where that reader should be sent.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS - 1, "Tinytown", "Essex")
|
||||
+ _town(MIN_SCHOOLS, "Brentwood", "Essex", start=200000)
|
||||
))
|
||||
places = places_for_urn(build_place_index(reg), 100000)
|
||||
|
||||
# The town is below the threshold, so it has no page and must not be
|
||||
# offered as a link. The authority above it does, and is the right target.
|
||||
assert all(p.slug != "tinytown" for p in places)
|
||||
assert "authority" in {p.kind for p in places}
|
||||
|
||||
|
||||
def test_an_unknown_urn_resolves_to_nothing_rather_than_raising():
|
||||
# A school page renders for any URN the API knows; the link module is not
|
||||
# entitled to take the page down when it has nothing to say.
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
assert places_for_urn(build_place_index(reg), 999999) == ()
|
||||
|
||||
|
||||
def test_the_index_is_consistent_with_the_registry_it_was_built_from():
|
||||
# The invariant that matters: a link module must never offer a place whose
|
||||
# page does not exist, and never omit one that does.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
+ _town(MIN_SCHOOLS, "Bedford", "Bedford", start=300000)
|
||||
))
|
||||
index = build_place_index(reg)
|
||||
for key, place in reg.items():
|
||||
for urn in place.urns:
|
||||
assert place in places_for_urn(index, urn), (
|
||||
f"{urn} is in {key} but the index does not say so")
|
||||
|
||||
|
||||
def test_the_index_holds_no_school_the_registry_does_not():
|
||||
# The reverse direction of the invariant above. An index entry for a URN
|
||||
# no published place contains would put a link on a page for a place that
|
||||
# does not list that school.
|
||||
reg = build_place_registry(_df(
|
||||
_town(MIN_SCHOOLS, "Brentwood", "Essex")
|
||||
+ _town(MIN_SCHOOLS - 1, "Tinytown", "Essex", start=400000)
|
||||
))
|
||||
index = build_place_index(reg)
|
||||
|
||||
for urn, places in index.items():
|
||||
for place in places:
|
||||
assert urn in place.urns
|
||||
assert place.key in reg
|
||||
|
||||
|
||||
def test_the_index_preserves_the_widest_first_order():
|
||||
# The breadcrumb reads authority then town, and takes this order as given.
|
||||
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
|
||||
kinds = [p.kind for p in places_for_urn(build_place_index(reg), 100000)]
|
||||
|
||||
assert kinds.index("authority") < kinds.index("town")
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Tests for the places API (spec 2026-08-21)."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
|
||||
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 50.0 + i}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_registry_lists_each_published_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
slugs = {(p["kind"], p["slug"]) for p in body["places"]}
|
||||
assert ("town", "brentwood") in slugs
|
||||
assert ("authority", "essex") in slugs
|
||||
assert ("outcode", "cm13") in slugs
|
||||
|
||||
|
||||
def test_registry_carries_a_count_per_place(client):
|
||||
body = client.get("/api/places").json()
|
||||
town = next(p for p in body["places"] if p["slug"] == "brentwood")
|
||||
assert town["count"] == 6
|
||||
|
||||
|
||||
def test_place_detail_returns_its_schools_alphabetically(client):
|
||||
"""A place page is read by someone looking for a school they can name.
|
||||
|
||||
Scanning for it is what the order should serve, so the list is A-Z.
|
||||
/api/rankings is where the league-table ordering lives.
|
||||
"""
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
assert body["place"]["name"] == "Brentwood"
|
||||
names = [s["school_name"] for s in body["schools"]]
|
||||
assert names == sorted(names, key=str.lower)
|
||||
|
||||
|
||||
def test_place_ordering_ignores_case(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
names = [s["school_name"] for s in body["schools"]]
|
||||
# A capitalised name must not sort ahead of every lowercase one.
|
||||
assert names == sorted(names, key=str.lower)
|
||||
|
||||
|
||||
def test_the_rankings_endpoint_still_ranks_by_metric(client):
|
||||
# Alphabetical is a place-page decision, not a site-wide one.
|
||||
body = client.get("/api/rankings?metric=rwm_expected_pct&phase=primary").json()
|
||||
scores = [r["rwm_expected_pct"] for r in body.get("rankings", [])
|
||||
if r.get("rwm_expected_pct") is not None]
|
||||
assert scores == sorted(scores, reverse=True)
|
||||
|
||||
|
||||
def test_place_detail_carries_the_local_average(client):
|
||||
body = client.get("/api/places/town/brentwood").json()
|
||||
# 50..55 inclusive
|
||||
assert body["averages"]["rwm_expected_pct"] == pytest.approx(52.5)
|
||||
|
||||
|
||||
def test_phase_filter_narrows_the_school_list(client):
|
||||
body = client.get("/api/places/town/brentwood?phase=secondary").json()
|
||||
assert body["schools"] == []
|
||||
|
||||
|
||||
def test_unknown_place_404s(client):
|
||||
assert client.get("/api/places/town/atlantis").status_code == 404
|
||||
|
||||
|
||||
def test_unknown_kind_404s(client):
|
||||
assert client.get("/api/places/planet/mars").status_code == 404
|
||||
|
||||
|
||||
def _straddling_df() -> pd.DataFrame:
|
||||
"""Eight schools in CM13: six in Essex, which has a page, and two in an
|
||||
authority too small to have one.
|
||||
|
||||
Two, not one: the registry ignores an authority holding a single school in
|
||||
a place, because GIAS carries occasional postcode errors."""
|
||||
df = _schools_df()
|
||||
extra = df.iloc[:2].copy()
|
||||
extra["urn"] = [200000, 200001]
|
||||
extra["school_name"] = ["Scilly School 0", "Scilly School 1"]
|
||||
extra["local_authority"] = "Isles Of Scilly"
|
||||
return pd.concat([df, extra], ignore_index=True)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def straddling_client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _straddling_df)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _straddling_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_an_outcode_reports_no_phases_because_it_has_no_phase_route(client):
|
||||
body = client.get("/api/places/outcode/cm13").json()
|
||||
assert body["place"]["phases"] == []
|
||||
|
||||
|
||||
def test_an_authority_reports_the_phases_it_publishes(client):
|
||||
body = client.get("/api/places/authority/essex").json()
|
||||
assert body["place"]["phases"] == ["primary"]
|
||||
|
||||
|
||||
def test_an_authority_without_a_page_is_named_but_carries_no_slug(straddling_client):
|
||||
"""Two English authorities — City of London and the Isles of Scilly — hold
|
||||
fewer than the five schools a page needs, so they have no page.
|
||||
|
||||
Naming them is still right: the page says where the place is. Linking them
|
||||
would not be. A null slug is what tells the page to print the name plainly
|
||||
rather than invent a URL that 404s.
|
||||
"""
|
||||
body = straddling_client.get("/api/places/outcode/cm13").json()
|
||||
by_name = {a["name"]: a for a in body["place"]["authorities"]}
|
||||
assert by_name["Essex"]["slug"] == "essex"
|
||||
assert by_name["Isles Of Scilly"]["slug"] is None
|
||||
|
||||
|
||||
def _attributed_df() -> pd.DataFrame:
|
||||
"""The same town, with the four attributes the place table now shows."""
|
||||
df = _schools_df()
|
||||
df["age_range"] = "4-11"
|
||||
df["religious_denomination"] = "Church of England"
|
||||
df["nursery_provision"] = True
|
||||
df["parliamentary_constituency"] = "Brentwood and Ongar"
|
||||
return df
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def attributed_client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _attributed_df)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _attributed_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_place_detail_carries_the_attributes_the_table_shows(attributed_client):
|
||||
"""age_range and religious_denomination ride in on SCHOOL_COLUMNS.
|
||||
|
||||
nursery_provision and parliamentary_constituency do not, and the place
|
||||
table needs all four — a column the response cannot fill is a column of
|
||||
dashes on ~3,900 pages.
|
||||
"""
|
||||
body = attributed_client.get("/api/places/town/brentwood").json()
|
||||
school = body["schools"][0]
|
||||
assert school["age_range"] == "4-11"
|
||||
assert school["religious_denomination"] == "Church of England"
|
||||
assert school["nursery_provision"] is True
|
||||
assert school["parliamentary_constituency"] == "Brentwood and Ongar"
|
||||
|
||||
|
||||
def test_place_detail_survives_a_mart_without_the_optional_columns(client):
|
||||
"""The base fixture has neither column, as an unrebuilt mart does not.
|
||||
|
||||
data_loader degrades those to NULL rather than failing the load, so the
|
||||
endpoint must not assume they are present.
|
||||
"""
|
||||
res = client.get("/api/places/town/brentwood")
|
||||
assert res.status_code == 200
|
||||
assert "nursery_provision" not in res.json()["schools"][0]
|
||||
@@ -0,0 +1,68 @@
|
||||
"""Publication must preserve the current dataset until every replacement is ready."""
|
||||
import asyncio
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
from backend import app as api, data_loader
|
||||
from backend.tests.test_sixth_form_flag import _schools_df
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(monkeypatch):
|
||||
old = _schools_df()
|
||||
monkeypatch.setattr(data_loader, '_df_cache', old)
|
||||
monkeypatch.setattr(data_loader, '_df_latest_cache', old)
|
||||
monkeypatch.setattr(api, '_place_registry', {'old': 'registry'})
|
||||
monkeypatch.setattr(api, '_place_index', {'old': 'index'})
|
||||
monkeypatch.setattr(api, '_place_index_source', api._place_registry)
|
||||
monkeypatch.setattr(api, '_sitemaps', {'old.xml': 'old sitemap'})
|
||||
monkeypatch.setattr(api, '_publication_lock', asyncio.Lock())
|
||||
monkeypatch.setattr(api.limiter, 'enabled', False)
|
||||
api.app.dependency_overrides[api.verify_admin_api_key] = lambda: True
|
||||
yield TestClient(api.app, raise_server_exceptions=False)
|
||||
api.app.dependency_overrides.clear()
|
||||
|
||||
|
||||
def state():
|
||||
return (data_loader._df_cache, data_loader._df_latest_cache, api._place_registry,
|
||||
api._place_index, api._place_index_source, api._sitemaps)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('failure', ['empty', 'database', 'sitemap', 'duplicate'])
|
||||
def test_failed_reload_preserves_every_published_object(client, monkeypatch, failure):
|
||||
before = state()
|
||||
df = _schools_df()
|
||||
if failure == 'empty':
|
||||
df = pd.DataFrame()
|
||||
if failure == 'duplicate':
|
||||
df = pd.concat([df, df.iloc[:1]], ignore_index=True)
|
||||
def load():
|
||||
if failure == 'database':
|
||||
raise RuntimeError('database unavailable')
|
||||
return df
|
||||
monkeypatch.setattr(api, 'load_school_data_as_dataframe', load)
|
||||
if failure == 'sitemap':
|
||||
monkeypatch.setattr(api, 'build_sitemaps', lambda *args: (_ for _ in ()).throw(RuntimeError('bad XML')))
|
||||
response = client.post('/api/admin/reload')
|
||||
assert response.status_code == 503
|
||||
assert all(a is b for a, b in zip(before, state()))
|
||||
|
||||
|
||||
def test_success_publishes_school_data_places_and_sitemaps(client, monkeypatch):
|
||||
df = _schools_df()
|
||||
df.loc[0, 'school_name'] = 'Replacement School'
|
||||
monkeypatch.setattr(api, 'load_school_data_as_dataframe', lambda: df)
|
||||
response = client.post('/api/admin/reload')
|
||||
assert response.status_code == 200
|
||||
assert data_loader.load_school_data() is df
|
||||
assert data_loader.load_latest_school_data().iloc[0].school_name == 'Replacement School'
|
||||
assert api._place_index_source is api._place_registry
|
||||
assert 'old.xml' not in api._sitemaps
|
||||
assert 'replacement-school' in api._sitemaps['schools-1.xml']
|
||||
|
||||
|
||||
def test_failed_sitemap_regeneration_keeps_existing_publication(client, monkeypatch):
|
||||
before = state()
|
||||
monkeypatch.setattr(api, 'build_sitemaps', lambda *args: (_ for _ in ()).throw(RuntimeError('bad XML')))
|
||||
assert client.post('/api/admin/regenerate-sitemap').status_code == 503
|
||||
assert all(a is b for a, b in zip(before, state()))
|
||||
@@ -0,0 +1,147 @@
|
||||
"""The rate-limit bucket must be the caller, not the proxy in front of them.
|
||||
|
||||
`get_remote_address` reads request.client.host. In staging and production the
|
||||
backend has no published ports and its only caller is the Next proxy, so that
|
||||
host is the Next container — one bucket for every browser user on the site.
|
||||
Measured before this fix: 70 concurrent requests, 60 served and 10 refused.
|
||||
"""
|
||||
|
||||
from starlette.datastructures import Headers
|
||||
|
||||
from backend.app import client_key
|
||||
|
||||
|
||||
class _Req:
|
||||
"""Enough of a Request for the key function: headers and a client host."""
|
||||
|
||||
def __init__(self, headers: dict, host: str = "10.0.0.9"):
|
||||
self.headers = Headers(headers)
|
||||
self.client = type("C", (), {"host": host})()
|
||||
self.scope = {"type": "http", "client": (host, 0),
|
||||
"headers": [(k.lower().encode(), v.encode())
|
||||
for k, v in headers.items()]}
|
||||
|
||||
|
||||
def test_cloudflare_header_wins():
|
||||
# Cloudflare sets CF-Connecting-IP and overwrites any client-supplied
|
||||
# value, so it is trustworthy in a way a parsed XFF chain is not.
|
||||
assert client_key(_Req({"cf-connecting-ip": "203.0.113.7"})) == "203.0.113.7"
|
||||
|
||||
|
||||
def test_forwarded_for_is_the_fallback_and_takes_the_first_entry():
|
||||
# Left-most is the original client; everything after it is proxies.
|
||||
assert client_key(
|
||||
_Req({"x-forwarded-for": "203.0.113.7, 10.0.0.2"})) == "203.0.113.7"
|
||||
|
||||
|
||||
def test_remote_address_is_the_last_resort():
|
||||
assert client_key(_Req({}, host="10.0.0.9")) == "10.0.0.9"
|
||||
|
||||
|
||||
def test_cloudflare_header_beats_forwarded_for():
|
||||
key = client_key(_Req({"cf-connecting-ip": "203.0.113.7",
|
||||
"x-forwarded-for": "198.51.100.1"}))
|
||||
assert key == "203.0.113.7"
|
||||
|
||||
|
||||
def test_two_callers_get_two_buckets():
|
||||
# The whole point: one user exhausting their limit must not refuse another.
|
||||
a = client_key(_Req({"cf-connecting-ip": "203.0.113.7"}))
|
||||
b = client_key(_Req({"cf-connecting-ip": "203.0.113.8"}))
|
||||
assert a != b
|
||||
|
||||
|
||||
def test_whitespace_is_stripped():
|
||||
# "a, b" split on comma leaves a leading space on every entry but the
|
||||
# first; an unstripped key silently creates a second bucket per client.
|
||||
assert client_key(_Req({"x-forwarded-for": " 203.0.113.7 ,10.0.0.2"})) \
|
||||
== "203.0.113.7"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The ceiling that header rotation cannot raise.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def api(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend.config import settings
|
||||
|
||||
monkeypatch.setattr(settings, "global_rate_limit_per_minute", 5)
|
||||
monkeypatch.setattr(app_module, "_global_window", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
|
||||
def _ceiling_req(path: str, host: str):
|
||||
"""Enough of a Request for exempt_from_ceiling: a path and a peer host."""
|
||||
return type("R", (), {
|
||||
"url": type("U", (), {"path": path})(),
|
||||
"client": type("C", (), {"host": host})(),
|
||||
})()
|
||||
|
||||
|
||||
def _get(client, path="/api/flags", cf=None):
|
||||
headers = {"cf-connecting-ip": cf} if cf else {}
|
||||
return client.get(path, headers=headers)
|
||||
|
||||
|
||||
def test_rotating_the_cloudflare_header_cannot_buy_unlimited_requests(api):
|
||||
"""The attack the per-client keying opened up.
|
||||
|
||||
client_key trusts CF-Connecting-IP, and nothing in this process can tell an
|
||||
edge-set header from an attacker-set one — that distinction can only be
|
||||
made at Cloudflare, with Authenticated Origin Pulls or an origin firewall.
|
||||
A caller reaching the origin directly can therefore mint a fresh
|
||||
rate-limit bucket per request and evade per-client limits entirely.
|
||||
|
||||
Per-client fairness is still the right default; this is the backstop that
|
||||
bounds what evading it can achieve. Without it, correct keying would be a
|
||||
net regression against abuse compared with the shared bucket it replaced.
|
||||
"""
|
||||
codes = [_get(api, cf=f"203.0.113.{i}").status_code for i in range(8)]
|
||||
assert codes.count(200) == 5
|
||||
assert codes.count(429) == 3
|
||||
|
||||
|
||||
def test_the_ceiling_says_which_limit_was_hit(api):
|
||||
# Distinguishable from slowapi's per-client 429, or an operator reading
|
||||
# logs cannot tell "one noisy client" from "the origin is saturated".
|
||||
for i in range(5):
|
||||
_get(api, cf=f"203.0.113.{i}")
|
||||
refused = _get(api, cf="203.0.113.99")
|
||||
assert refused.status_code == 429
|
||||
assert "capacity" in refused.json()["detail"].lower()
|
||||
assert refused.headers.get("retry-after")
|
||||
|
||||
|
||||
def test_traffic_below_the_ceiling_is_untouched(api):
|
||||
codes = [_get(api, cf=f"203.0.113.{i}").status_code for i in range(5)]
|
||||
assert codes == [200] * 5
|
||||
|
||||
|
||||
def test_the_container_healthcheck_is_exempt(api):
|
||||
"""The healthcheck runs `curl http://localhost:80/api/data-info` inside the
|
||||
container. If the ceiling could starve it, saturation would fail the
|
||||
healthcheck, restart the container, and turn a load spike into an outage
|
||||
loop — the ceiling has to protect the origin, not kill it.
|
||||
"""
|
||||
from backend.app import exempt_from_ceiling
|
||||
|
||||
assert exempt_from_ceiling(_ceiling_req("/api/data-info", "127.0.0.1"))
|
||||
assert exempt_from_ceiling(_ceiling_req("/api/data-info", "::1"))
|
||||
# Everyone else is counted.
|
||||
assert not exempt_from_ceiling(_ceiling_req("/api/data-info", "10.0.0.9"))
|
||||
|
||||
|
||||
def test_the_ceiling_ignores_non_api_paths():
|
||||
# Sitemaps and robots.txt are served by this app too, and a crawler
|
||||
# fetching them must not be refused because the API is busy.
|
||||
from backend.app import exempt_from_ceiling
|
||||
|
||||
assert exempt_from_ceiling(_ceiling_req("/sitemap.xml", "10.0.0.9"))
|
||||
assert exempt_from_ceiling(_ceiling_req("/robots.txt", "10.0.0.9"))
|
||||
@@ -56,6 +56,11 @@ def client(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
app_module, "get_supplementary_data", lambda db, urn: {}
|
||||
)
|
||||
# The place registry is a module-level cache, so without this the endpoint
|
||||
# answers from whatever registry an earlier test happened to leave behind
|
||||
# — and a `places == []` assertion is satisfied by a stale registry just
|
||||
# as well as by this fixture's own data, which makes it prove nothing.
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
@@ -69,3 +74,174 @@ def test_nan_gias_fields_serialize_as_null(client):
|
||||
assert info["capacity"] is None
|
||||
assert info["total_pupils"] is None
|
||||
assert info["school_name"] == "West London Performing Arts Academy"
|
||||
|
||||
|
||||
# ── Links out to the location layer ─────────────────────────────────────────
|
||||
#
|
||||
# School pages carried no link into the site at all: the only anchor on the
|
||||
# template pointed at the school's own website, so ~27k pages received
|
||||
# whatever authority the site had and sent it off-site. `places` is what the
|
||||
# link module and the breadcrumb are built from.
|
||||
|
||||
def test_places_is_present_even_when_the_school_belongs_to_none(client):
|
||||
# This fixture's single school cannot clear any publish threshold, so the
|
||||
# honest answer is an empty list. The key must still be there: a missing
|
||||
# key and "no places" are different things to the page rendering it.
|
||||
body = client.get("/api/schools/150275").json()
|
||||
assert body["places"] == []
|
||||
|
||||
|
||||
def test_places_names_only_pages_that_exist(monkeypatch):
|
||||
from backend import app as app_module
|
||||
from backend.places import MIN_SCHOOLS
|
||||
|
||||
def _df():
|
||||
return pd.DataFrame([
|
||||
{
|
||||
"urn": 100000 + i,
|
||||
"school_name": f"Brentwood School {i}",
|
||||
"town": "Brentwood",
|
||||
"local_authority": "Essex",
|
||||
"postcode": "CM15 8AA",
|
||||
"phase": "Primary",
|
||||
"year": 202425,
|
||||
"rwm_expected_pct": 60.0,
|
||||
"attainment_8_score": np.nan,
|
||||
"ofsted_grade": 2.0,
|
||||
"ofsted_date": None,
|
||||
}
|
||||
for i in range(MIN_SCHOOLS)
|
||||
])
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _df)
|
||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
client = TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
places = client.get("/api/schools/100000").json()["places"]
|
||||
assert places, "a school in a published town must offer links"
|
||||
|
||||
by_kind = {p["kind"]: p for p in places}
|
||||
assert by_kind["town"]["url"] == "/schools/brentwood"
|
||||
assert by_kind["authority"]["url"] == "/schools/authority/essex"
|
||||
|
||||
# Every entry carries what the link text needs, and a count, so the anchor
|
||||
# can say what it leads to rather than "click here".
|
||||
for place in places:
|
||||
assert place["name"]
|
||||
assert place["count"] >= 1
|
||||
assert place["url"].startswith("/schools/")
|
||||
|
||||
|
||||
def _brentwood_df(phase: str = "Primary", n: int = None):
|
||||
from backend.places import MIN_SCHOOLS
|
||||
n = n if n is not None else MIN_SCHOOLS
|
||||
return lambda: pd.DataFrame([
|
||||
{
|
||||
"urn": 100000 + i,
|
||||
"school_name": f"Brentwood School {i}",
|
||||
"town": "Brentwood", "local_authority": "Essex",
|
||||
"postcode": "CM15 8AA", "phase": phase, "year": 202425,
|
||||
"rwm_expected_pct": 60.0, "attainment_8_score": 50.0,
|
||||
"ofsted_grade": 2.0, "ofsted_date": None,
|
||||
}
|
||||
for i in range(n)
|
||||
])
|
||||
|
||||
|
||||
def _places_for(monkeypatch, df_factory, urn: int):
|
||||
from backend import app as app_module
|
||||
monkeypatch.setattr(app_module, "load_school_data", df_factory)
|
||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
client = TestClient(app_module.app, raise_server_exceptions=False)
|
||||
return client.get(f"/api/schools/{urn}").json()["places"]
|
||||
|
||||
|
||||
def test_a_place_offers_the_phase_page_this_school_appears_on(monkeypatch):
|
||||
# "primary schools in brentwood" is the query the phase pages exist for,
|
||||
# and ~950 of them were once reachable by nothing at all.
|
||||
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
|
||||
town = next(p for p in places if p["kind"] == "town")
|
||||
|
||||
assert town["phases"], "a primary school in a published primary town has a link"
|
||||
assert town["phases"][0]["url"] == "/schools/brentwood/primary"
|
||||
assert town["phases"][0]["count"] >= 1
|
||||
|
||||
|
||||
def test_an_all_through_school_offers_both_phase_pages(monkeypatch):
|
||||
# It genuinely appears on both, so there is no tie to break.
|
||||
places = _places_for(monkeypatch, _brentwood_df("All-through"), 100000)
|
||||
town = next(p for p in places if p["kind"] == "town")
|
||||
|
||||
assert {p["phase"] for p in town["phases"]} == {"primary", "secondary"}
|
||||
|
||||
|
||||
def test_outcodes_never_offer_a_phase_page(monkeypatch):
|
||||
# The registry gives outcodes no phase route — nobody searches "primary
|
||||
# schools in SW11" — and computing them anyway once put a link to a
|
||||
# nonexistent route on all 1,720 outcode pages.
|
||||
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
|
||||
outcode = next((p for p in places if p["kind"] == "outcode"), None)
|
||||
|
||||
if outcode is not None:
|
||||
assert outcode["phases"] == []
|
||||
|
||||
|
||||
def test_a_school_absent_from_the_phase_page_is_not_linked_to_it(monkeypatch):
|
||||
# The check is URN membership in the registry's own phase list, not a
|
||||
# re-derivation of the phase mapping. A secondary school must not be sent
|
||||
# to a primary phase page that does not list it.
|
||||
from backend.places import MIN_SCHOOLS
|
||||
|
||||
def df():
|
||||
rows = [
|
||||
{"urn": 100000 + i, "school_name": f"P{i}", "town": "Brentwood",
|
||||
"local_authority": "Essex", "postcode": "CM15 8AA",
|
||||
"phase": "Primary", "year": 202425, "rwm_expected_pct": 60.0,
|
||||
"attainment_8_score": np.nan, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None}
|
||||
for i in range(MIN_SCHOOLS)
|
||||
]
|
||||
rows.append({
|
||||
"urn": 900000, "school_name": "Lone Secondary", "town": "Brentwood",
|
||||
"local_authority": "Essex", "postcode": "CM15 8AA",
|
||||
"phase": "Secondary", "year": 202425, "rwm_expected_pct": np.nan,
|
||||
"attainment_8_score": 50.0, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
})
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
places = _places_for(monkeypatch, df, 900000)
|
||||
town = next(p for p in places if p["kind"] == "town")
|
||||
|
||||
# The town publishes a primary page, but this secondary school is not on
|
||||
# it, and there are too few secondaries for a secondary page.
|
||||
assert town["phases"] == []
|
||||
|
||||
|
||||
def test_the_place_index_rebuilds_when_the_registry_is_replaced(monkeypatch):
|
||||
"""The reverse index is cached; a stale one would put another dataset's
|
||||
places on a school page. Invalidation is an identity check against the
|
||||
registry rather than a second flag, so this asserts the check works."""
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
monkeypatch.setattr(app_module, "_place_index", None)
|
||||
monkeypatch.setattr(app_module, "_place_index_source", None)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _brentwood_df("Primary"))
|
||||
|
||||
first = app_module.get_place_index()
|
||||
assert 100000 in first
|
||||
|
||||
# Same registry object, so the index is reused rather than rebuilt.
|
||||
assert app_module.get_place_index() is first
|
||||
|
||||
# Drop the registry the way every test that touches place data does. The
|
||||
# index must follow it, not survive it.
|
||||
app_module._place_registry = None
|
||||
monkeypatch.setattr(app_module, "load_school_data",
|
||||
_brentwood_df("Primary", n=0))
|
||||
|
||||
rebuilt = app_module.get_place_index()
|
||||
assert rebuilt is not first
|
||||
assert 100000 not in rebuilt, "the index outlived the registry it came from"
|
||||
@@ -0,0 +1,112 @@
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
Every GIAS code must be accounted for, so a new DfE code fails here instead of
|
||||
silently vanishing from the filter.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
from backend.gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
from backend.school_groups import (
|
||||
FAITH_GROUPS,
|
||||
TYPE_GROUPS,
|
||||
UNOFFERED_TYPE_CODES,
|
||||
faith_groups_for,
|
||||
type_group_for,
|
||||
type_group_key,
|
||||
)
|
||||
|
||||
DBT_PROJECT = Path(__file__).resolve().parents[2] / "pipeline" / "transform" / "dbt_project.yml"
|
||||
|
||||
|
||||
def _non_england_codes() -> set[int]:
|
||||
return set(yaml.safe_load(DBT_PROJECT.read_text())["vars"]["non_england_school_type_codes"])
|
||||
|
||||
|
||||
def test_every_type_code_is_in_exactly_one_place():
|
||||
places = [codes for _, _, codes in TYPE_GROUPS] + [UNOFFERED_TYPE_CODES, _non_england_codes()]
|
||||
for code in SCHOOL_TYPE:
|
||||
homes = sum(code in p for p in places)
|
||||
assert homes == 1, f"type code {code} ({SCHOOL_TYPE[code]}) is in {homes} places"
|
||||
|
||||
|
||||
def test_every_religion_code_has_a_faith():
|
||||
for code, name in RELIGIOUS_CHARACTER.items():
|
||||
assert any(code in codes for _, _, codes in FAITH_GROUPS), f"{code} {name!r}"
|
||||
|
||||
|
||||
def test_type_groups_in_display_order():
|
||||
assert [k for k, _, _ in TYPE_GROUPS] == [
|
||||
"state", "independent", "special", "post16", "alternative"]
|
||||
|
||||
|
||||
def test_faiths_in_display_order():
|
||||
assert [k for k, _, _ in FAITH_GROUPS] == [
|
||||
"none", "church_of_england", "roman_catholic", "other_christian",
|
||||
"jewish", "muslim", "other_faith"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, group", [
|
||||
("Academy converter", "state"),
|
||||
("University technical college", "state"),
|
||||
("Voluntary aided school", "state"),
|
||||
("Local authority nursery school", "state"),
|
||||
("Other independent school", "independent"),
|
||||
("Other independent special school", "special"),
|
||||
("Special post 16 institution", "special"),
|
||||
("Further education", "post16"),
|
||||
("Pupil referral unit", "alternative"),
|
||||
("academy CONVERTER", "state"),
|
||||
])
|
||||
def test_type_group_by_name(name, group):
|
||||
assert type_group_for(name) == group
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value, key", [
|
||||
("state", "state"),
|
||||
("Special", "special"),
|
||||
# The two state groups that preceded "state", kept so their links still work.
|
||||
("academy", "state"),
|
||||
("Council", "state"),
|
||||
("Community school", None),
|
||||
("", None),
|
||||
])
|
||||
def test_type_group_key_resolves_keys_and_old_keys(value, key):
|
||||
assert type_group_key(value) == key
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", [
|
||||
"Higher education institutions", "Miscellaneous", "Unknown (9999)", "Academy", "", None, np.nan,
|
||||
])
|
||||
def test_unoffered_or_unknown_types_have_no_group(name):
|
||||
assert type_group_for(name) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, faiths", [
|
||||
("Roman Catholic/Church of England", ("church_of_england", "roman_catholic")),
|
||||
("Roman Catholic/Anglican", ("church_of_england", "roman_catholic")),
|
||||
("Church of England/Methodist", ("church_of_england", "other_christian")),
|
||||
("Church of England/Christian", ("church_of_england",)),
|
||||
("Catholic", ("roman_catholic",)),
|
||||
("Inter- / non- denominational", ("other_christian",)),
|
||||
("Orthodox Jewish", ("jewish",)),
|
||||
("Sunni Deobandi", ("muslim",)),
|
||||
("Hindu", ("other_faith",)),
|
||||
("Does not apply", ("none",)),
|
||||
("None", ("none",)),
|
||||
])
|
||||
def test_faiths_by_name(name, faiths):
|
||||
assert faith_groups_for(name) == faiths
|
||||
|
||||
|
||||
@pytest.mark.parametrize("missing", [None, np.nan, "", " "])
|
||||
def test_a_missing_religion_is_no_religious_character(missing):
|
||||
assert faith_groups_for(missing) == ("none",)
|
||||
|
||||
|
||||
def test_an_unknown_religion_has_no_faith():
|
||||
assert faith_groups_for("Unknown (77)") == ()
|
||||
@@ -0,0 +1,114 @@
|
||||
from types import SimpleNamespace
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
from backend import app as api, data_loader
|
||||
from backend.tests.test_sixth_form_flag import _schools_df
|
||||
|
||||
|
||||
def client_for(monkeypatch, search):
|
||||
client = SimpleNamespace(collections={'schools': SimpleNamespace(documents=SimpleNamespace(search=search))})
|
||||
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: client)
|
||||
|
||||
|
||||
def test_search_returns_matches_beyond_first_page(monkeypatch):
|
||||
pages = []
|
||||
def search(params):
|
||||
pages.append(params['page'])
|
||||
urns = range(100000, 100250) if params['page'] == 1 else [100999]
|
||||
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
|
||||
client_for(monkeypatch, search)
|
||||
result = data_loader.search_schools_typesense('academy')
|
||||
assert len(result) == 251
|
||||
assert result[-1] == 100999
|
||||
assert pages == [1, 2]
|
||||
|
||||
|
||||
def test_search_caps_broad_queries_at_a_bounded_number_of_pages(monkeypatch):
|
||||
requests = []
|
||||
|
||||
def search(params):
|
||||
requests.append(params)
|
||||
return {
|
||||
'found': 10_000,
|
||||
'hits': [
|
||||
{'document': {'urn': 100000 + params['page'] * 1000 + i}}
|
||||
for i in range(params['per_page'])
|
||||
],
|
||||
}
|
||||
|
||||
client_for(monkeypatch, search)
|
||||
result = data_loader.search_schools_typesense('school')
|
||||
|
||||
assert len(result) == data_loader.SEARCH_MAX_CANDIDATES
|
||||
assert len(requests) == data_loader.SEARCH_MAX_CANDIDATES // data_loader.SEARCH_PAGE_SIZE
|
||||
assert all(request['per_page'] == data_loader.SEARCH_PAGE_SIZE for request in requests)
|
||||
assert requests[-1]['page'] == len(requests)
|
||||
|
||||
|
||||
def test_search_uses_a_smaller_final_page_when_the_cap_is_not_a_page_multiple(monkeypatch):
|
||||
monkeypatch.setattr(data_loader, 'SEARCH_MAX_CANDIDATES', 251)
|
||||
requests = []
|
||||
|
||||
def search(params):
|
||||
requests.append(params)
|
||||
return {
|
||||
'found': 10_000,
|
||||
'hits': [{'document': {'urn': 100000 + len(requests) * 1000 + i}}
|
||||
for i in range(params['per_page'])],
|
||||
}
|
||||
|
||||
client_for(monkeypatch, search)
|
||||
result = data_loader.search_schools_typesense('school')
|
||||
|
||||
assert len(result) == 251
|
||||
assert [request['per_page'] for request in requests] == [250, 1]
|
||||
|
||||
|
||||
def test_later_page_failure_does_not_return_partial_results(monkeypatch):
|
||||
def search(params):
|
||||
if params['page'] == 2:
|
||||
raise RuntimeError('timeout')
|
||||
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in range(100000, 100250)]}
|
||||
client_for(monkeypatch, search)
|
||||
assert data_loader.search_schools_typesense('academy') is None
|
||||
|
||||
|
||||
def test_zero_matches_are_distinct_from_unavailable(monkeypatch):
|
||||
client_for(monkeypatch, lambda _: {'found': 0, 'hits': []})
|
||||
assert data_loader.search_schools_typesense('academy') == []
|
||||
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: None)
|
||||
assert data_loader.search_schools_typesense('academy') is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize('matches, expected', [([], []), (None, [100001])])
|
||||
def test_fallback_only_on_dependency_failure(monkeypatch, matches, expected):
|
||||
monkeypatch.setattr(api.limiter, 'enabled', False)
|
||||
monkeypatch.setattr(api, 'load_latest_school_data', _schools_df)
|
||||
monkeypatch.setattr(api, 'search_schools_typesense', lambda _: matches)
|
||||
response = TestClient(api.app).get('/api/schools?search=Alpha')
|
||||
assert response.status_code == 200
|
||||
assert [s['urn'] for s in response.json()['schools']] == expected
|
||||
|
||||
|
||||
def test_filtered_api_keeps_match_from_second_search_page(monkeypatch):
|
||||
monkeypatch.setattr(api.limiter, 'enabled', False)
|
||||
df = _schools_df()
|
||||
monkeypatch.setattr(api, 'load_latest_school_data', lambda: df)
|
||||
def search(params):
|
||||
urns = range(200000, 200250) if params['page'] == 1 else [100001]
|
||||
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
|
||||
client_for(monkeypatch, search)
|
||||
response = TestClient(api.app).get('/api/schools?search=Alpha&local_authority=Testshire')
|
||||
assert response.status_code == 200
|
||||
assert response.json()['total'] == 1
|
||||
assert response.json()['schools'][0]['urn'] == 100001
|
||||
|
||||
|
||||
def test_unavailable_dataset_is_not_a_missing_school_or_empty_search(monkeypatch):
|
||||
import pandas as pd
|
||||
monkeypatch.setattr(api.limiter, 'enabled', False)
|
||||
monkeypatch.setattr(api, 'load_school_data', lambda: pd.DataFrame())
|
||||
monkeypatch.setattr(api, 'load_latest_school_data', lambda: pd.DataFrame())
|
||||
client = TestClient(api.app)
|
||||
assert client.get('/api/schools/100001').status_code == 503
|
||||
assert client.get('/api/schools?search=school').status_code == 503
|
||||
@@ -59,9 +59,14 @@ def static_child(monkeypatch) -> str:
|
||||
def test_every_loc_uses_the_www_host(sitemaps):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
# Checked across every file, index included, not just one.
|
||||
#
|
||||
# A child can legitimately be empty — this fixture holds two schools and no
|
||||
# town clearing the threshold — so the presence check applies only to files
|
||||
# that carry URLs. The absence check applies to all of them.
|
||||
for name, xml in sitemaps.items():
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
assert "https://schoolcompare.co.uk" not in xml, name
|
||||
if "<loc>" in xml:
|
||||
assert "https://www.schoolcompare.co.uk" in xml, name
|
||||
|
||||
|
||||
def test_school_with_results_is_listed(schools_child):
|
||||
@@ -217,3 +222,85 @@ def test_school_with_no_results_in_any_year_is_still_omitted(monkeypatch):
|
||||
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
|
||||
|
||||
assert "/school/100002" not in app_module.build_sitemaps()["schools-1.xml"]
|
||||
|
||||
|
||||
def _places_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy",
|
||||
"phase": "Primary", "year": 202425, "ofsted_grade": 2.0,
|
||||
"ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA",
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": 100000 + i, "school_name": f"Brentwood School {i}",
|
||||
"rwm_expected_pct": 60.0}
|
||||
for i in range(6)
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def place_sitemaps(monkeypatch) -> dict:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _places_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return app_module.build_sitemaps()
|
||||
|
||||
|
||||
def test_place_children_are_listed_in_the_index(place_sitemaps):
|
||||
index = place_sitemaps["sitemap.xml"]
|
||||
assert "/sitemaps/places-1.xml" in index
|
||||
assert "/sitemaps/outcodes-1.xml" in index
|
||||
|
||||
|
||||
def test_town_and_authority_urls_use_their_own_namespaces(place_sitemaps):
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood</loc>" in xml
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/authority/essex</loc>" in xml
|
||||
|
||||
|
||||
def test_outcode_urls_live_in_their_own_child(place_sitemaps):
|
||||
assert "/schools/near/cm13" in place_sitemaps["outcodes-1.xml"]
|
||||
assert "/schools/near/cm13" not in place_sitemaps["places-1.xml"]
|
||||
|
||||
|
||||
def test_place_urls_carry_no_priority_or_changefreq(place_sitemaps):
|
||||
for name in ("places-1.xml", "outcodes-1.xml"):
|
||||
assert "<priority>" not in place_sitemaps[name]
|
||||
assert "<changefreq>" not in place_sitemaps[name]
|
||||
|
||||
|
||||
def test_phase_variants_are_submitted_where_the_phase_clears_the_threshold(place_sitemaps):
|
||||
# "primary schools in beccles" is the query shape the baseline showed, so
|
||||
# each variant is its own page and has to be submitted. Emitting only the
|
||||
# bare place URL left ~950 of them reachable by nothing.
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "<loc>https://www.schoolcompare.co.uk/schools/brentwood/primary</loc>" in xml
|
||||
|
||||
|
||||
def test_a_phase_below_its_own_threshold_is_not_submitted(place_sitemaps):
|
||||
# The fixture is six primaries and no secondaries.
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert "/schools/brentwood/secondary" not in xml
|
||||
|
||||
|
||||
def test_outcodes_get_no_phase_variants(place_sitemaps):
|
||||
# Nobody searches "primary schools in CM13"; the routes do not exist.
|
||||
xml = place_sitemaps["outcodes-1.xml"]
|
||||
assert "/primary" not in xml and "/secondary" not in xml
|
||||
|
||||
|
||||
def test_authority_phase_variants_are_submitted_in_their_own_namespace(place_sitemaps):
|
||||
"""302 of these were already in the sitemap, and every one 404'd.
|
||||
|
||||
The spec gives authorities a phase route; the plan built the bare
|
||||
authority route and dropped it. Nothing noticed because the sitemap was
|
||||
written from the registry, which was right, while the routes were written
|
||||
by hand. This test fails if the URL ever leaves the sitemap; the e2e
|
||||
journey fails if the route ever leaves the app.
|
||||
"""
|
||||
xml = place_sitemaps["places-1.xml"]
|
||||
assert ("<loc>https://www.schoolcompare.co.uk"
|
||||
"/schools/authority/essex/primary</loc>") in xml
|
||||
# And never in the town namespace, which is a different set of schools.
|
||||
assert "/schools/essex/primary" not in xml
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Tests for school autosuggest (spec 2026-08-26)."""
|
||||
|
||||
from backend import data_loader
|
||||
|
||||
|
||||
class _FakeDocs:
|
||||
def __init__(self, hits, explode=False):
|
||||
self._hits = hits
|
||||
self._explode = explode
|
||||
self.last_params = None
|
||||
|
||||
def search(self, params):
|
||||
self.last_params = params
|
||||
if self._explode:
|
||||
raise RuntimeError("typesense is down")
|
||||
return {"hits": [{"document": d} for d in self._hits]}
|
||||
|
||||
|
||||
class _FakeClient:
|
||||
def __init__(self, hits, explode=False):
|
||||
self.docs = _FakeDocs(hits, explode)
|
||||
self.collections = {"schools": type("C", (), {"documents": self.docs})()}
|
||||
|
||||
|
||||
_HIT = {
|
||||
"urn": 100010, "school_name": "Brecknock Primary School",
|
||||
"local_authority": "Camden", "postcode": "NW1 1AA",
|
||||
"phase": "Primary", "school_type": "Community school",
|
||||
}
|
||||
|
||||
|
||||
def _use(monkeypatch, client):
|
||||
monkeypatch.setattr(data_loader, "_get_typesense_client", lambda: client)
|
||||
|
||||
|
||||
def test_returns_the_fields_a_suggestion_needs(monkeypatch):
|
||||
# Local authority is not decoration: there are many schools called
|
||||
# "St Mary's", and a list without it cannot be chosen between.
|
||||
_use(monkeypatch, _FakeClient([_HIT]))
|
||||
out = data_loader.suggest_schools_typesense("breck")
|
||||
assert out == [{
|
||||
"urn": 100010, "school_name": "Brecknock Primary School",
|
||||
"local_authority": "Camden", "postcode": "NW1 1AA",
|
||||
"phase": "Primary", "school_type": "Community school",
|
||||
}]
|
||||
|
||||
|
||||
def test_a_missing_optional_field_becomes_an_empty_string(monkeypatch):
|
||||
# phase and school_type are optional in the Typesense schema. A missing
|
||||
# key must not KeyError in the keystroke path.
|
||||
_use(monkeypatch, _FakeClient([{"urn": 1, "school_name": "X",
|
||||
"local_authority": "Y", "postcode": "Z"}]))
|
||||
out = data_loader.suggest_schools_typesense("x")
|
||||
assert out[0]["phase"] == "" and out[0]["school_type"] == ""
|
||||
|
||||
|
||||
def test_typesense_unavailable_gives_no_suggestions_rather_than_raising(monkeypatch):
|
||||
_use(monkeypatch, None)
|
||||
assert data_loader.suggest_schools_typesense("anything") == []
|
||||
|
||||
|
||||
def test_a_typesense_error_gives_no_suggestions_rather_than_raising(monkeypatch):
|
||||
_use(monkeypatch, _FakeClient([], explode=True))
|
||||
assert data_loader.suggest_schools_typesense("anything") == []
|
||||
|
||||
|
||||
def test_the_limit_is_passed_through_and_clamped(monkeypatch):
|
||||
client = _FakeClient([])
|
||||
_use(monkeypatch, client)
|
||||
data_loader.suggest_schools_typesense("x", limit=500)
|
||||
assert client.docs.last_params["per_page"] == 20
|
||||
|
||||
|
||||
def _client(monkeypatch, rows, *, blow_up_dataframe=False):
|
||||
from fastapi.testclient import TestClient
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "suggest_schools_typesense",
|
||||
lambda q, limit=8: rows)
|
||||
if blow_up_dataframe:
|
||||
def _boom():
|
||||
raise AssertionError("the suggest path must not load the DataFrame")
|
||||
monkeypatch.setattr(app_module, "load_school_data", _boom)
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _boom)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def test_the_endpoint_returns_suggestions(monkeypatch):
|
||||
body = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck").json()
|
||||
assert body["suggestions"][0]["school_name"] == "Brecknock Primary School"
|
||||
|
||||
|
||||
def test_the_endpoint_never_touches_the_dataframe(monkeypatch):
|
||||
"""The whole reason this is not a mode of /api/schools.
|
||||
|
||||
That endpoint filters and sorts 25,000 rows of pandas per query, holding
|
||||
the GIL. Per keystroke, that is the cost this endpoint exists to avoid.
|
||||
"""
|
||||
res = _client(monkeypatch, [_HIT], blow_up_dataframe=True).get("/api/suggest?q=breck")
|
||||
assert res.status_code == 200
|
||||
assert res.json()["suggestions"]
|
||||
|
||||
|
||||
def test_a_one_character_query_returns_nothing_and_does_not_error(monkeypatch):
|
||||
# The keystroke path never errors on ordinary input.
|
||||
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=b")
|
||||
assert res.status_code == 200
|
||||
assert res.json() == {"suggestions": []}
|
||||
|
||||
|
||||
def test_a_blank_query_returns_nothing_and_does_not_error(monkeypatch):
|
||||
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=")
|
||||
assert res.status_code == 200
|
||||
assert res.json() == {"suggestions": []}
|
||||
|
||||
|
||||
def test_typesense_down_is_an_empty_list_not_a_500(monkeypatch):
|
||||
res = _client(monkeypatch, []).get("/api/suggest?q=breck")
|
||||
assert res.status_code == 200
|
||||
assert res.json() == {"suggestions": []}
|
||||
|
||||
|
||||
def test_the_response_is_cacheable(monkeypatch):
|
||||
# Prefix queries repeat enormously across users, and school names change
|
||||
# once a year. Without this the endpoint pays full price every keystroke.
|
||||
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck")
|
||||
assert "s-maxage" in res.headers.get("cache-control", "")
|
||||
assert res.headers.get("etag")
|
||||
|
||||
|
||||
def test_a_malformed_urn_does_not_raise(monkeypatch):
|
||||
"""The docstring promises "never raises"; the parsing loop sat outside the
|
||||
try, so int(None) or int("abc") would have turned a keystroke into a 500.
|
||||
|
||||
Typesense declares urn as int32, so this should be unreachable — but the
|
||||
contract is what the caller relies on, and a search index is a separate
|
||||
system that can be reindexed by something other than this code.
|
||||
"""
|
||||
_use(monkeypatch, _FakeClient([{"urn": None, "school_name": "X",
|
||||
"local_authority": "Y", "postcode": "Z"}]))
|
||||
assert data_loader.suggest_schools_typesense("x") == []
|
||||
|
||||
|
||||
def test_a_malformed_row_does_not_discard_the_good_ones(monkeypatch):
|
||||
# One bad document must not blank the whole dropdown.
|
||||
_use(monkeypatch, _FakeClient([
|
||||
{"urn": "not-a-number", "school_name": "Bad", "local_authority": "Y",
|
||||
"postcode": "Z"},
|
||||
_HIT,
|
||||
]))
|
||||
out = data_loader.suggest_schools_typesense("x")
|
||||
assert [r["urn"] for r in out] == [100010]
|
||||
|
||||
|
||||
def test_a_hit_with_no_document_does_not_raise(monkeypatch):
|
||||
_use(monkeypatch, _FakeClient([{}]))
|
||||
assert data_loader.suggest_schools_typesense("x") == []
|
||||
@@ -132,16 +132,23 @@ def test_one_query_per_table_and_latest_row_per_urn():
|
||||
"FactPupilCharacteristics": [],
|
||||
"FactDeprivation": [],
|
||||
"FactFinance": [],
|
||||
"FactKs4Destinations": [],
|
||||
"FactKs5Destinations": [],
|
||||
}
|
||||
session = _FakeSession(rows)
|
||||
out = get_supplementary_data_batch(session, [1, 2])
|
||||
|
||||
# Exactly one query per table — six total, regardless of two URNs.
|
||||
# Exactly one query per table — eight total, regardless of two URNs.
|
||||
assert sorted(session.queries) == [
|
||||
"FactAdmissionDistance", "FactAdmissions", "FactDeprivation",
|
||||
"FactFinance", "FactOfstedInspection", "FactPupilCharacteristics",
|
||||
"FactFinance", "FactKs4Destinations", "FactKs5Destinations",
|
||||
"FactOfstedInspection", "FactPupilCharacteristics",
|
||||
]
|
||||
|
||||
# A school with no destination rows gets null, not an empty shell — the
|
||||
# frontend renders the section from the block's presence.
|
||||
assert out[1]["destinations"] is None
|
||||
|
||||
# Latest Ofsted kept per URN
|
||||
assert out[1]["ofsted"]["overall_effectiveness"] == 2
|
||||
assert out[2]["ofsted"]["overall_effectiveness"] == 1
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
"""/api/schools school-type groups and faith filter, and their /api/filters lists."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
# urn -> (GIAS school type, GIAS religious character)
|
||||
SCHOOLS = {
|
||||
100001: ("Community school", "Does not apply"),
|
||||
100002: ("Voluntary aided school", "Roman Catholic"),
|
||||
100003: ("Academy converter", "Roman Catholic/Church of England"),
|
||||
100004: ("Community special school", None),
|
||||
100005: ("Academy special converter", "Church of England"),
|
||||
100006: ("Other independent school", "Jewish"),
|
||||
100007: ("Miscellaneous", ""),
|
||||
}
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Testshire", "address": "1 Test Street", "town": "Testtown",
|
||||
"postcode": "TS1 1AA", "age_range": "4-11", "has_sixth_form": None,
|
||||
"gender": "Mixed", "admissions_policy": None, "ofsted_grade": np.nan,
|
||||
"ofsted_date": None, "ofsted_framework": None, "latitude": 51.5,
|
||||
"longitude": -0.1, "year": 202425, "total_pupils": 300,
|
||||
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan, "phase": "Primary",
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": urn, "school_name": f"School {urn}",
|
||||
"school_type": t, "religious_denomination": r}
|
||||
for urn, (t, r) in SCHOOLS.items()
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def _urns(client, **params):
|
||||
resp = client.get("/api/schools", params={"page_size": 50, **params})
|
||||
assert resp.status_code == 200, resp.text
|
||||
return sorted(s["urn"] for s in resp.json()["schools"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key, urns", [
|
||||
("state", [100001, 100002, 100003]),
|
||||
("academy", [100001, 100002, 100003]),
|
||||
("council", [100001, 100002, 100003]),
|
||||
("special", [100004, 100005]),
|
||||
("independent", [100006]),
|
||||
("Special", [100004, 100005]),
|
||||
])
|
||||
def test_a_type_group_key_filters_to_its_group(client, key, urns):
|
||||
assert _urns(client, school_type=key) == urns
|
||||
|
||||
|
||||
def test_a_raw_type_label_still_filters_exactly(client):
|
||||
assert _urns(client, school_type="Community school") == [100001]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key, urns", [
|
||||
("roman_catholic", [100002, 100003]),
|
||||
("church_of_england", [100003, 100005]),
|
||||
("none", [100001, 100004, 100007]),
|
||||
("jewish", [100006]),
|
||||
("Roman_Catholic", [100002, 100003]),
|
||||
])
|
||||
def test_faith_filters_to_its_group_joint_schools_included(client, key, urns):
|
||||
assert _urns(client, faith=key) == urns
|
||||
|
||||
|
||||
def test_an_unknown_faith_returns_nothing(client):
|
||||
assert _urns(client, faith="nonsense") == []
|
||||
|
||||
|
||||
def test_type_and_faith_combine(client):
|
||||
assert _urns(client, school_type="special", faith="church_of_england") == [100005]
|
||||
|
||||
|
||||
def test_filters_lists_only_groups_present_in_order(client):
|
||||
body = client.get("/api/filters").json()
|
||||
assert body["school_type_groups"] == [
|
||||
{"value": "state", "label": "State school (free)"},
|
||||
{"value": "independent", "label": "Independent (fee-paying)"},
|
||||
{"value": "special", "label": "Special school (SEND)"},
|
||||
]
|
||||
assert body["faiths"] == [
|
||||
{"value": "none", "label": "No religious character"},
|
||||
{"value": "church_of_england", "label": "Church of England"},
|
||||
{"value": "roman_catholic", "label": "Roman Catholic"},
|
||||
{"value": "jewish", "label": "Jewish"},
|
||||
]
|
||||
# The raw list is still there for anything that reads it.
|
||||
assert "Community school" in body["school_types"]
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Payloads name each school's type group.
|
||||
|
||||
The school page and the search rows say "State school" or "Independent
|
||||
school" in the search filter's own terms, not GIAS's 34 establishment types,
|
||||
and an independent school gets a Fee-paying flag. A type in no group keeps a
|
||||
null group, and the page prints the register's own name for it. Search rows
|
||||
also carry nursery_provision, for their "Nursery class" flag.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "phase": "Primary", "year": 202425,
|
||||
"ofsted_grade": 2.0, "ofsted_date": None, "attainment_8_score": np.nan,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
|
||||
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
|
||||
"rwm_expected_pct": 60.0, "nursery_provision": "No Nursery Classes",
|
||||
}
|
||||
rows = [
|
||||
{**base, "urn": 100001, "school_name": "Alpha Academy",
|
||||
"school_type": "Academy converter", "nursery_provision": "Has Nursery Classes"},
|
||||
{**base, "urn": 100002, "school_name": "Beta Prep",
|
||||
"school_type": "Other independent school"},
|
||||
{**base, "urn": 100003, "school_name": "Gamma Unit",
|
||||
"school_type": "Secure units"},
|
||||
]
|
||||
# Enough schools in one town for it to have a place page.
|
||||
rows += [
|
||||
{**base, "urn": 100010 + i, "school_name": f"Delta Primary {i}",
|
||||
"school_type": "Community school"}
|
||||
for i in range(3)
|
||||
]
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def _by_urn(schools: list[dict]) -> dict[int, dict]:
|
||||
return {s["urn"]: s for s in schools}
|
||||
|
||||
|
||||
def test_the_list_names_each_type_group(client):
|
||||
resp = client.get("/api/schools?page_size=50")
|
||||
assert resp.status_code == 200, resp.text
|
||||
schools = _by_urn(resp.json()["schools"])
|
||||
assert schools[100001]["type_group"] == "state"
|
||||
assert schools[100002]["type_group"] == "independent"
|
||||
assert schools[100003]["type_group"] is None
|
||||
|
||||
|
||||
def test_the_list_carries_nursery_provision(client):
|
||||
schools = _by_urn(client.get("/api/schools?page_size=50").json()["schools"])
|
||||
assert schools[100001]["nursery_provision"] == "Has Nursery Classes"
|
||||
|
||||
|
||||
def test_the_school_page_names_its_type_group(client):
|
||||
resp = client.get("/api/schools/100002")
|
||||
assert resp.status_code == 200, resp.text
|
||||
assert resp.json()["school_info"]["type_group"] == "independent"
|
||||
|
||||
|
||||
def test_a_place_page_names_each_type_group(client):
|
||||
resp = client.get("/api/places/town/brentwood")
|
||||
assert resp.status_code == 200, resp.text
|
||||
schools = _by_urn(resp.json()["schools"])
|
||||
assert schools[100001]["type_group"] == "state"
|
||||
assert schools[100003]["type_group"] is None
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Cards, map popups and place rows label total_pupils "pupils".
|
||||
|
||||
fact_performance's total_pupils is the cohort a year's results were measured
|
||||
on. For a secondary that is the GCSE year group alone: Burntwood showed 245 in
|
||||
search against 1,462 on roll. The list and place payloads therefore carry the
|
||||
register's whole-school count, and nothing when the register has none.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Essex", "school_type": "Academy converter",
|
||||
"year": 202425, "ofsted_grade": 2.0, "ofsted_date": None,
|
||||
"town": "Brentwood", "postcode": "CM13 1AA", "status": "Open",
|
||||
"address": "1 Test Street", "latitude": 51.6, "longitude": 0.3,
|
||||
"gender": "Mixed", "rwm_expected_pct": np.nan, "attainment_8_score": 50.0,
|
||||
}
|
||||
rows = [
|
||||
# Secondary: results cohort 245, register 1,462.
|
||||
{**base, "urn": 100001, "school_name": "Alpha High", "phase": "Secondary",
|
||||
"total_pupils": 245, "gias_total_pupils": 1462},
|
||||
# Register count missing: no count, never the cohort.
|
||||
{**base, "urn": 100002, "school_name": "Beta High", "phase": "Secondary",
|
||||
"total_pupils": 180, "gias_total_pupils": np.nan},
|
||||
]
|
||||
# Enough schools in one town for it to have a place page.
|
||||
rows += [
|
||||
{**base, "urn": 100010 + i, "school_name": f"Gamma High {i}", "phase": "Secondary",
|
||||
"total_pupils": 200, "gias_total_pupils": 1000 + i}
|
||||
for i in range(5)
|
||||
]
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "_place_registry", None)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def _pupils(schools: list[dict]) -> dict[int, object]:
|
||||
return {s["urn"]: s.get("total_pupils") for s in schools}
|
||||
|
||||
|
||||
def test_the_list_carries_the_whole_school_count(client):
|
||||
resp = client.get("/api/schools?page_size=50")
|
||||
assert resp.status_code == 200, resp.text
|
||||
pupils = _pupils(resp.json()["schools"])
|
||||
assert pupils[100001] == 1462
|
||||
assert pupils[100002] is None
|
||||
|
||||
|
||||
def test_a_place_page_carries_the_whole_school_count(client):
|
||||
resp = client.get("/api/places/town/brentwood")
|
||||
assert resp.status_code == 200, resp.text
|
||||
pupils = _pupils(resp.json()["schools"])
|
||||
assert pupils[100001] == 1462
|
||||
assert pupils[100002] is None
|
||||
@@ -1,26 +0,0 @@
|
||||
"""
|
||||
Schema versioning for database migrations.
|
||||
|
||||
HOW TO USE:
|
||||
- Bump SCHEMA_VERSION when making changes to database models
|
||||
- This triggers an automatic full data reimport on next app startup
|
||||
|
||||
WHEN TO BUMP:
|
||||
- Adding/removing columns in models.py
|
||||
- Changing column types or constraints
|
||||
- Modifying CSV column mappings in schemas.py
|
||||
- Any change that requires fresh data import
|
||||
"""
|
||||
|
||||
# Current schema version - increment when models change
|
||||
SCHEMA_VERSION = 6
|
||||
|
||||
# Changelog for documentation
|
||||
SCHEMA_CHANGELOG = {
|
||||
1: "Initial schema with School and SchoolResult tables",
|
||||
2: "Added pupil absence fields (reading, maths, gps, writing, science)",
|
||||
3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools",
|
||||
4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)",
|
||||
5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables",
|
||||
6: "Removed the Ofsted Parent View feature: dropped fact_parent_view table and model",
|
||||
}
|
||||
@@ -1,134 +1,37 @@
|
||||
# SchoolCompare.co.uk - Project Context
|
||||
# SchoolCompare project context
|
||||
|
||||
## Overview
|
||||
## Maintained documentation
|
||||
|
||||
SchoolCompare is a web application for comparing UK primary school (KS2) performance data. It allows users to:
|
||||
- Search and browse schools by name, location (postcode), or local authority
|
||||
- Compare multiple schools side-by-side with charts and tables
|
||||
- View school rankings by various KS2 metrics
|
||||
- See historical performance trends across years
|
||||
Read [README.md](README.md), [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) and
|
||||
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the current implementation.
|
||||
[docs/LEGACY_CODE.md](docs/LEGACY_CODE.md) records obsolete paths and deliberate
|
||||
compatibility code. Historical design documents are not current setup instructions.
|
||||
|
||||
## Architecture
|
||||
## Architecture constraints
|
||||
|
||||
### Backend (Python/FastAPI)
|
||||
- **Framework**: FastAPI with uvicorn
|
||||
- **Database**: PostgreSQL with SQLAlchemy ORM
|
||||
- **Data Source**: UK Government "Compare School Performance" CSV downloads
|
||||
|
||||
Key files:
|
||||
- `backend/app.py` - Main FastAPI application, API routes
|
||||
- `backend/config.py` - Configuration via pydantic-settings (env vars, .env file)
|
||||
- `backend/database.py` - SQLAlchemy engine, session management
|
||||
- `backend/models.py` - Database models (School, SchoolResult)
|
||||
- `backend/data_loader.py` - Data queries, geocoding, legacy DataFrame compatibility
|
||||
- `backend/schemas.py` - Column mappings, metric definitions, LA code mappings
|
||||
|
||||
### Frontend (Vanilla JS)
|
||||
- Single-page application with hash-based routing
|
||||
- Chart.js for data visualization
|
||||
- No build step required
|
||||
|
||||
Key files:
|
||||
- `frontend/index.html` - Main HTML structure
|
||||
- `frontend/app.js` - All application logic, API calls, rendering
|
||||
- `frontend/styles.css` - Styling (CSS variables, responsive design)
|
||||
|
||||
### Database Schema
|
||||
|
||||
```
|
||||
schools school_results
|
||||
├── id (PK) ├── id (PK)
|
||||
├── urn (unique, indexed) ├── school_id (FK → schools.id)
|
||||
├── school_name ├── year (indexed)
|
||||
├── local_authority ├── rwm_expected_pct
|
||||
├── school_type ├── reading_expected_pct
|
||||
├── postcode ├── ... (all KS2 metrics)
|
||||
├── latitude, longitude └── unique(school_id, year)
|
||||
└── results → SchoolResult[]
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Environment variables (or `.env` file):
|
||||
- `DATABASE_URL` - PostgreSQL connection string (default: `postgresql://schoolcompare:schoolcompare@localhost:5432/schoolcompare`)
|
||||
- `HOST`, `PORT` - Server binding (default: `0.0.0.0:80`)
|
||||
- `ALLOWED_ORIGINS` - CORS origins
|
||||
|
||||
## Running Locally
|
||||
|
||||
1. Start PostgreSQL:
|
||||
```bash
|
||||
docker compose up -d db
|
||||
```
|
||||
|
||||
2. Run migration to import CSV data:
|
||||
```bash
|
||||
python scripts/migrate_csv_to_db.py --drop
|
||||
# Add --geocode to geocode postcodes (slower, adds lat/long)
|
||||
```
|
||||
|
||||
3. Start the app:
|
||||
```bash
|
||||
uvicorn backend.app:app --host 0.0.0.0 --port 8000
|
||||
```
|
||||
|
||||
## Docker Deployment
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
This starts:
|
||||
- `db` - PostgreSQL 16 with persistent volume
|
||||
- `app` - FastAPI application on port 80
|
||||
|
||||
## Data
|
||||
|
||||
- Source: UK Government Compare School Performance downloads
|
||||
- Location: `data/` directory with year folders (e.g., `2023-2024/england_ks2final.csv`)
|
||||
- The `scripts/download_data.py` can fetch data from the government website
|
||||
|
||||
## Key Features
|
||||
|
||||
- **Location Search**: Enter postcode to find nearby schools (uses postcodes.io API)
|
||||
- **Multi-school Comparison**: Select multiple schools, view metrics across years
|
||||
- **Rankings**: Top schools by any KS2 metric, filterable by local authority
|
||||
- **Variability Analysis**: Shows standard deviation of scores across years
|
||||
|
||||
## API Endpoints
|
||||
|
||||
- `GET /api/schools` - List/search schools (supports pagination, location search)
|
||||
- `GET /api/schools/{urn}` - School details with all yearly data
|
||||
- `GET /api/compare?urns=123,456` - Compare multiple schools
|
||||
- `GET /api/rankings` - School rankings by metric
|
||||
- `GET /api/filters` - Available filter options (LAs, types, years)
|
||||
- `GET /api/metrics` - Metric definitions (single source of truth)
|
||||
- `GET /api/data-info` - Database stats
|
||||
- Next.js serves the public UI. FastAPI serves school data from dbt-built `marts.*`.
|
||||
The backend does not create school tables or import CSVs at startup.
|
||||
- School coverage spans England and multiple phases, not only primary schools in
|
||||
Wandsworth and Merton.
|
||||
- `/api/*` belongs to the FastAPI proxy. Payload uses `/cms-api` and `/admin`.
|
||||
- Payload runs inside Next.js, with its own `payload` schema and persistent media.
|
||||
Keep CMS migrations independent of school-data transformations.
|
||||
- Public and Payload route groups have separate root layouts. Do not introduce
|
||||
`app/layout.tsx`. Keep site-wide metadata files at the `app/` root.
|
||||
- Builds must succeed with `DATABASE_URL` unset. Do not call `getCachedPayload()`
|
||||
at module scope or add DB-backed `generateStaticParams`.
|
||||
- After changing CMS fields/editors, run `npm run generate:importmap` and commit
|
||||
the generated import map. See `nextjs-app/docs/PUBLISHING.md`.
|
||||
- The backend and pipeline GIAS dictionary copies are generated together; preserve
|
||||
their parity. Tests enforce it.
|
||||
|
||||
## SDLC
|
||||
|
||||
Full details in `docs/DEPLOY.md`. The short version:
|
||||
Follow [docs/DEPLOY.md](docs/DEPLOY.md).
|
||||
|
||||
- **Never push to `main` directly.** Work on a feature branch and open a PR;
|
||||
branch protection requires the PR checks (typecheck, tests, builds, AI review)
|
||||
to pass before merge.
|
||||
- Merging to `main` deploys automatically **to staging only**: images are
|
||||
built once, deployed to the staging Portainer stack, and verified by the
|
||||
Playwright journeys in `e2e/`. Production is a second, manual approval:
|
||||
the "Promote to Production (manual)" workflow in Gitea Actions, run after
|
||||
testing the feature on staging. It refuses commits whose staging E2E gate
|
||||
isn't green. Never trigger it yourself — promotion is the human's call.
|
||||
- If you change user-facing behaviour, update or extend the `e2e/` journey
|
||||
tests in the same PR — they gate whether staging is fit for human testing
|
||||
and whether a commit is promotable.
|
||||
|
||||
## Recent Changes
|
||||
|
||||
- Added staging environment + automated staging→prod pipeline (Gitea Actions)
|
||||
- Migrated from CSV file storage to PostgreSQL database
|
||||
- Added location-based search using postcode geocoding
|
||||
- Added local authority filter to rankings
|
||||
- Improved frontend with featured schools, loading states, API caching
|
||||
|
||||
# Important
|
||||
- Do not attempt to start a local server to test the application, it does not work
|
||||
- Never push directly to `main`. Use a feature branch and a PR with passing checks.
|
||||
- Merges deploy staging only. Production promotion is a separate human decision;
|
||||
do not trigger the promotion workflow yourself.
|
||||
- Update E2E journeys in the same PR when changing user-facing behaviour.
|
||||
- Do not attempt to start a local server to test the application; use unit checks
|
||||
and the configured integration environment.
|
||||
@@ -16,7 +16,16 @@
|
||||
# ADMIN_API_KEY — Backend admin API key
|
||||
# TYPESENSE_API_KEY — Typesense admin API key
|
||||
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
|
||||
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
|
||||
# UNLEASH_API_TOKEN — Unleash *client* token, environment: development
|
||||
# PAYLOAD_SECRET — Payload CMS encryption secret. REQUIRED: long and
|
||||
# random, and DIFFERENT from production's. Sharing
|
||||
# it would let a staging session authenticate
|
||||
# against production.
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
|
||||
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
|
||||
# refuses to start without it, rather than falling
|
||||
# back to a generated one that changes on restart.
|
||||
# STAGING_DB_IP — macvlan IP for staging Postgres (default 10.0.1.190)
|
||||
# STAGING_FRONTEND_IP — macvlan IP for staging frontend (default 10.0.1.151)
|
||||
|
||||
@@ -55,6 +64,12 @@ services:
|
||||
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
|
||||
TYPESENSE_URL: http://typesense:8108
|
||||
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
|
||||
# Unset means every feature flag is False — the correct dark state for an
|
||||
# environment with no Unleash, not a failure.
|
||||
UNLEASH_URL: ${UNLEASH_URL:-}
|
||||
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
|
||||
volumes:
|
||||
- unleash_cache:/app/.unleash
|
||||
depends_on:
|
||||
sc_database:
|
||||
condition: service_healthy
|
||||
@@ -78,9 +93,20 @@ services:
|
||||
- FASTAPI_URL=http://backend:80/api
|
||||
- TYPESENSE_URL=http://typesense:8108
|
||||
- TYPESENSE_API_KEY=${TYPESENSE_SEARCH_KEY:-changeme}
|
||||
# Payload CMS runs inside this container, in the `payload` schema of the
|
||||
# staging database. Staging has its own stack, its own Postgres and its
|
||||
# own admin account — never production's.
|
||||
- DATABASE_URL=postgresql://${DB_USERNAME}:${DB_PASSWORD}@sc_database:5432/${DB_DATABASE_NAME}
|
||||
- PAYLOAD_SECRET=${PAYLOAD_SECRET:?set PAYLOAD_SECRET in the staging Portainer stack environment}
|
||||
volumes:
|
||||
# Portainer prefixes volume names with the stack name, so this is
|
||||
# automatically isolated from production's media.
|
||||
- payload_media:/app/media
|
||||
depends_on:
|
||||
backend:
|
||||
condition: service_healthy
|
||||
sc_database:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
backend: {}
|
||||
macvlan:
|
||||
@@ -116,7 +142,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:staging
|
||||
container_name: sc_staging_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8081:8080"
|
||||
environment:
|
||||
@@ -128,6 +170,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-staging-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
|
||||
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
|
||||
PG_HOST: sc_database
|
||||
PG_PORT: "5432"
|
||||
@@ -212,3 +256,5 @@ volumes:
|
||||
postgres_data:
|
||||
typesense_data:
|
||||
airflow_logs:
|
||||
unleash_cache:
|
||||
payload_media:
|
||||
@@ -0,0 +1,73 @@
|
||||
# Portainer Stack Definition for School Compare — UNLEASH (feature flags)
|
||||
#
|
||||
# Deploy as a *separate* Portainer stack ("schoolcompare-unleash"), alongside
|
||||
# the production and staging stacks. It deliberately belongs to neither: a
|
||||
# staging redeploy must not be able to disturb production's flag state, and a
|
||||
# production redeploy must not disturb staging's.
|
||||
#
|
||||
# One instance serves both environments. Open-source Unleash ships with
|
||||
# `development` and `production` environments and environment-scoped client
|
||||
# tokens, so the same flag holds independent state in each — which is what
|
||||
# lets a feature be on in staging, where the E2E journeys exercise it, while
|
||||
# production stays dark.
|
||||
#
|
||||
# Portainer environment variables (set in Portainer UI -> Stack -> Environment):
|
||||
# UNLEASH_DB_PASSWORD — PostgreSQL password for the Unleash database
|
||||
# UNLEASH_ADMIN_PASSWORD — initial admin password for the Unleash UI
|
||||
# UNLEASH_IP — macvlan IP for the Unleash server (default 10.0.1.152)
|
||||
|
||||
services:
|
||||
|
||||
# ── PostgreSQL (Unleash's own; nothing else uses it) ──────────────────
|
||||
unleash_db:
|
||||
container_name: sc_unleash_postgres
|
||||
image: postgres:16-alpine
|
||||
environment:
|
||||
POSTGRES_USER: unleash
|
||||
POSTGRES_PASSWORD: ${UNLEASH_DB_PASSWORD}
|
||||
POSTGRES_DB: unleash
|
||||
volumes:
|
||||
- unleash_postgres_data:/var/lib/postgresql/data
|
||||
networks:
|
||||
- unleash
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U unleash"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 10s
|
||||
restart: unless-stopped
|
||||
|
||||
# ── Unleash server (UI + client API on 4242) ──────────────────────────
|
||||
unleash:
|
||||
container_name: sc_unleash
|
||||
image: unleashorg/unleash-server:6
|
||||
environment:
|
||||
DATABASE_URL: postgres://unleash:${UNLEASH_DB_PASSWORD}@unleash_db:5432/unleash
|
||||
DATABASE_SSL: "false"
|
||||
INIT_ADMIN_API_TOKENS: ""
|
||||
UNLEASH_DEFAULT_ADMIN_PASSWORD: ${UNLEASH_ADMIN_PASSWORD}
|
||||
depends_on:
|
||||
unleash_db:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
unleash: {}
|
||||
macvlan:
|
||||
ipv4_address: ${UNLEASH_IP:-10.0.1.152}
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "wget -qO- http://localhost:4242/health || exit 1"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 30s
|
||||
restart: unless-stopped
|
||||
|
||||
networks:
|
||||
unleash:
|
||||
driver: bridge
|
||||
macvlan:
|
||||
external:
|
||||
name: macvlan
|
||||
|
||||
volumes:
|
||||
unleash_postgres_data:
|
||||
@@ -7,7 +7,15 @@
|
||||
# ADMIN_API_KEY — Backend admin API key
|
||||
# TYPESENSE_API_KEY — Typesense admin API key
|
||||
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
|
||||
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
|
||||
# UNLEASH_API_TOKEN — Unleash *client* token, environment: production
|
||||
# PAYLOAD_SECRET — Payload CMS encryption secret. REQUIRED: long and
|
||||
# random. Changing it invalidates every admin
|
||||
# session. Staging MUST use a different value.
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
|
||||
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
|
||||
# refuses to start without it, rather than falling
|
||||
# back to a generated one that changes on restart.
|
||||
|
||||
services:
|
||||
|
||||
@@ -44,6 +52,12 @@ services:
|
||||
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
|
||||
TYPESENSE_URL: http://typesense:8108
|
||||
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
|
||||
# Unset means every feature flag is False — the correct dark state for an
|
||||
# environment with no Unleash, not a failure.
|
||||
UNLEASH_URL: ${UNLEASH_URL:-}
|
||||
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
|
||||
volumes:
|
||||
- unleash_cache:/app/.unleash
|
||||
depends_on:
|
||||
sc_database:
|
||||
condition: service_healthy
|
||||
@@ -67,9 +81,21 @@ services:
|
||||
- FASTAPI_URL=http://backend:80/api
|
||||
- TYPESENSE_URL=http://typesense:8108
|
||||
- TYPESENSE_API_KEY=${TYPESENSE_SEARCH_KEY:-changeme}
|
||||
# Payload CMS runs inside this container. It reaches Postgres over the
|
||||
# `backend` network and keeps its tables in the `payload` schema, so no
|
||||
# pipeline operation on `public` can touch blog content.
|
||||
- DATABASE_URL=postgresql://${DB_USERNAME}:${DB_PASSWORD}@sc_database:5432/${DB_DATABASE_NAME}
|
||||
# Same :? form as AIRFLOW_ADMIN_PASSWORD: refuse to start rather than
|
||||
# boot with an empty secret and silently accept forged sessions.
|
||||
- PAYLOAD_SECRET=${PAYLOAD_SECRET:?set PAYLOAD_SECRET in the Portainer stack environment}
|
||||
volumes:
|
||||
# Blog images. Not reproducible from the pipeline — must be backed up.
|
||||
- payload_media:/app/media
|
||||
depends_on:
|
||||
backend:
|
||||
condition: service_healthy
|
||||
sc_database:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
backend: {}
|
||||
macvlan:
|
||||
@@ -105,7 +131,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:prod
|
||||
container_name: schoolcompare_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8080:8080"
|
||||
environment:
|
||||
@@ -117,6 +159,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
|
||||
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
|
||||
PG_HOST: sc_database
|
||||
PG_PORT: "5432"
|
||||
@@ -201,3 +245,5 @@ volumes:
|
||||
postgres_data:
|
||||
typesense_data:
|
||||
airflow_logs:
|
||||
unleash_cache:
|
||||
payload_media:
|
||||
+23
-1
@@ -36,6 +36,10 @@ services:
|
||||
ADMIN_API_KEY: ${ADMIN_API_KEY:-changeme}
|
||||
TYPESENSE_URL: http://typesense:8108
|
||||
TYPESENSE_API_KEY: ${TYPESENSE_API_KEY:-changeme}
|
||||
# Unset means every feature flag is False — the correct dark state for an
|
||||
# environment with no Unleash, not a failure.
|
||||
UNLEASH_URL: ${UNLEASH_URL:-}
|
||||
UNLEASH_API_TOKEN: ${UNLEASH_API_TOKEN:-}
|
||||
volumes:
|
||||
- ./data:/app/data:ro
|
||||
depends_on:
|
||||
@@ -101,7 +105,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:latest
|
||||
container_name: schoolcompare_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8080:8080"
|
||||
environment: &airflow-env
|
||||
@@ -113,6 +133,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "admin:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:-admin}
|
||||
PG_HOST: db
|
||||
PG_PORT: "5432"
|
||||
PG_USER: schoolcompare
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
# Architecture
|
||||
|
||||
This describes the implementation as reviewed on 2026-09-14. It distinguishes
|
||||
current behaviour from improvements still to be implemented.
|
||||
|
||||
## Request flow
|
||||
|
||||
```text
|
||||
Browser → Next.js public routes
|
||||
├─ /api/* proxy → FastAPI → cached DataFrames / PostgreSQL marts
|
||||
│ ├─ Typesense (search and suggestions)
|
||||
│ └─ postcodes.io (postcode lookup)
|
||||
└─ /admin, /cms-api, /blog → Payload → payload schema + media volume
|
||||
|
||||
Next.js server rendering → FastAPI directly through FASTAPI_URL
|
||||
```
|
||||
|
||||
`nextjs-app/lib/api.ts` contains typed fetch wrappers and revalidation defaults.
|
||||
The proxy is `nextjs-app/app/(frontend)/api/[...path]/route.ts`. Payload uses
|
||||
`/cms-api` so its routes do not collide with the FastAPI proxy. The proxy denies
|
||||
`/api/flags`; server-side rendering reads flags directly from FastAPI.
|
||||
|
||||
## Data ownership
|
||||
|
||||
| Layer | Owner and role |
|
||||
|---|---|
|
||||
| Source data | GIAS, DfE EES, Ofsted, finance, deprivation and council admission-distance sources |
|
||||
| `raw` | Singer taps and the PostgreSQL target configured in `pipeline/meltano.yml` |
|
||||
| Staging/intermediate/marts | dbt models in `pipeline/transform`; marts are materialized tables |
|
||||
| `marts.dim_school`, `marts.dim_location` | School identity and location, filtered to supported England establishments |
|
||||
| `marts.fact_*` | Performance and supplementary datasets; coverage and years vary |
|
||||
| Typesense `schools` alias | Search documents built by `pipeline/scripts/sync_typesense.py` |
|
||||
| `payload` | CMS collections and migrations in `nextjs-app/`; independent of dbt |
|
||||
| Media volume | Uploaded blog media; requires backup and cannot be regenerated from school datasets |
|
||||
|
||||
`backend/models.py` maps existing marts for reading. It does not create the school
|
||||
schema. There is no startup schema-version migration or CSV reimport. Payload's
|
||||
`nextjs-app/migrations/` is active and must not be confused with the removed
|
||||
legacy backend migration code.
|
||||
|
||||
Coordinates normally come from GIAS British National Grid coordinates transformed
|
||||
by PostGIS in `dim_location.sql`. `pipeline/scripts/geocode_postcodes.py` is a
|
||||
manual fallback utility, not a task wired into the current school-data DAG.
|
||||
Backend postcode searches also use postcodes.io; that lookup does not populate
|
||||
school coordinates in the database.
|
||||
|
||||
## Backend boundaries
|
||||
|
||||
- `app.py`: routes, middleware, search filtering, sitemap/place publication and response assembly.
|
||||
- `data_loader.py`: SQL loading, process-local DataFrame caches, Typesense calls,
|
||||
postcode lookups, supplementary queries and benchmark calculation.
|
||||
- `database.py`: synchronous SQLAlchemy engine and sessions.
|
||||
- `schemas.py`: metric definitions, column mappings and display metadata; despite
|
||||
its name this is not a collection of Pydantic API response models.
|
||||
- `places.py` and `localities.py`: place registry and curated locality information.
|
||||
- `flags.py`: Unleash-backed feature flags, disabled when no server is configured.
|
||||
- `gias_codes.py` / `ofsted_codes.py`: source-code translation and display rules.
|
||||
|
||||
Search starts from a cached latest-row-per-school snapshot. Detail pages read
|
||||
history from the full DataFrame and supplementary data from marts. Comparisons
|
||||
batch supplementary queries across selected URNs. Async routes still contain
|
||||
synchronous dependency calls; a fully asynchronous database layer is not present.
|
||||
|
||||
## Frontend boundaries
|
||||
|
||||
`app/(frontend)` owns the public root layout and pages. `app/(payload)` owns the
|
||||
CMS root layout. Do not add a shared `app/layout.tsx`: these groups deliberately
|
||||
have separate root layouts. Root metadata files remain in `app/`.
|
||||
|
||||
Server pages fetch initial data and pass it to client views. Client state uses
|
||||
React hooks, URL search parameters and the comparison context/localStorage.
|
||||
There is no SWR dependency. Leaflet maps are loaded through dynamic wrappers;
|
||||
Chart.js renders performance and comparison charts.
|
||||
|
||||
`components/school/` contains detail sections, with section decisions and data
|
||||
preparation in `lib/schoolSections.ts`. The nearby-schools section is selected in
|
||||
`backend/nearby_schools.py` — hard filters decide eligibility (phase, provision,
|
||||
selectivity, gender) and distance alone decides the order, capped per phase —
|
||||
and served on `/api/schools/{urn}`. Its rules are presentation logic,
|
||||
deliberately kept out of `marts.*` so they can be tuned by deploy rather than by
|
||||
pipeline run. `lib/types.ts` contains manually maintained
|
||||
API types. `payload-types.ts` and the Payload import map are generated artifacts.
|
||||
|
||||
## Publication and caching today
|
||||
|
||||
1. Airflow DAGs extract and validate source data, then run selected dbt builds.
|
||||
2. Relevant DAGs rebuild Typesense and swap the `schools` alias.
|
||||
3. They call `POST /api/admin/reload` with `X-API-Key`. It builds and validates
|
||||
replacement DataFrames, places, reverse membership and sitemaps off the request
|
||||
loop, then publishes them together. Failure returns 503 and preserves live data.
|
||||
4. A separate weekly sitemap DAG can regenerate the derived publication from the
|
||||
current DataFrame without clearing the live registry first.
|
||||
|
||||
GIAS is scheduled daily, Ofsted monthly, and annual datasets are manually
|
||||
triggered. The DAG definitions are authoritative for selectors and dependencies.
|
||||
|
||||
Caches exist in several independent layers: backend DataFrames and registries,
|
||||
backend HTTP Cache-Control/ETags, Next.js fetch/page revalidation, and browser or
|
||||
shared HTTP caches where configured. Place fetches request a one-week revalidation
|
||||
interval. HTTP ETags are computed after route execution, not before database work.
|
||||
|
||||
Typesense publication validates every import response and the final document
|
||||
count before switching aliases. A session-scoped PostgreSQL advisory lock
|
||||
serialises index reads/publication across DAGs. The previous collection remains
|
||||
available for rollback; old unaliased collections are pruned after success.
|
||||
Failed drafts are retained until a later successful cleanup, because an uncertain
|
||||
alias-update response must never cause deletion of a potentially live index.
|
||||
|
||||
The backend snapshot swap is process-local and assumes the current single-worker
|
||||
deployment. It is not an atomic transaction spanning PostgreSQL marts, Typesense
|
||||
and Next.js caches. Next.js caches are not explicitly purged by the pipeline.
|
||||
School search retrieves a relevance-ordered candidate prefix (currently capped at
|
||||
1,000 URNs) before applying API filters. This keeps scoped searches useful while
|
||||
putting a hard ceiling on Typesense round trips; only a dependency failure invokes
|
||||
substring fallback, not a valid empty match set.
|
||||
|
||||
## Deployment references
|
||||
|
||||
See [DEPLOY.md](DEPLOY.md). PR checks include frontend typechecking/tests, backend
|
||||
unit tests, image builds and AI review. Staging journeys run after merging.
|
||||
Staging runs are serialised across builds, deployment and E2E. Build-stamped
|
||||
frontend/backend identities are checked before and after journeys. Only then are
|
||||
the captured image digests marked verified. Promotion resolves and validates the
|
||||
complete verified image set before retagging production. See the runbook for
|
||||
first-rollout requirements and remaining integration checks.
|
||||
+159
-4
@@ -19,8 +19,9 @@ PR checks (.gitea/workflows/pr-checks.yml)
|
||||
▼
|
||||
Stage pipeline (.gitea/workflows/deploy.yml) — automatic
|
||||
1. build & push images → tags sha-<sha>, staging
|
||||
2. staging Portainer webhook → wait for staging health
|
||||
2. staging Portainer webhook → verify frontend/backend SHA + build ID
|
||||
3. Playwright E2E journeys against staging ← gate before human testing
|
||||
4. verify identity again; tag tested digests verified-<full-sha>
|
||||
▼
|
||||
Manual testing on staging (stx.schoolcompare.co.uk)
|
||||
│ Actions → "Promote to Production (manual)" ← approval #2
|
||||
@@ -28,14 +29,15 @@ Manual testing on staging (stx.schoolcompare.co.uk)
|
||||
Promote pipeline (.gitea/workflows/promote.yml) — manual dispatch
|
||||
1. resolve target sha (input, or latest main if empty)
|
||||
2. REFUSE unless that commit's "E2E Journeys against Staging" status is green
|
||||
3. retag sha-<sha> → :prod (same bytes — build once, promote the image)
|
||||
3. resolve verified-<full-sha> digests, validate labels, retag digests → :prod
|
||||
previous :prod saved as :prod-previous
|
||||
4. prod Portainer webhook → wait for prod health
|
||||
4. prod Portainer webhook → verify expected SHA + build ID
|
||||
```
|
||||
|
||||
Key principle: **build once, promote the exact image**. Production pins `:prod`,
|
||||
which only moves when a human runs the promote workflow — and the workflow
|
||||
only accepts commits that passed the staging E2E gate. Nothing tags `:latest`
|
||||
only accepts commits that passed the staging E2E gate and have a complete verified
|
||||
image set. Nothing tags `:latest`
|
||||
anymore.
|
||||
|
||||
## Branch & PR workflow
|
||||
@@ -98,6 +100,12 @@ fail the E2E gate. That's the point: staging absorbs the risk.
|
||||
pr-checks status checks (frontend, backend, builds, ai-review) to pass.
|
||||
5. **Bootstrap staging data via Airflow** (no prod dump — staging populates
|
||||
itself from source, exercising the pipeline image end-to-end):
|
||||
- Set `AIRFLOW_ADMIN_PASSWORD` in the stack environment first. The
|
||||
api-server refuses to start without it. Airflow's simple auth manager
|
||||
otherwise generates a password on first start and writes it to a file, so
|
||||
the login changes every time the container restarts; the stack writes that
|
||||
file itself from this variable instead. `AIRFLOW_ADMIN_USER` defaults to
|
||||
`admin`.
|
||||
- Open the staging Airflow UI (`http://<host>:8081`) and trigger, in order:
|
||||
`school_data_daily`, `school_data_monthly_ofsted`, then the manual-schedule
|
||||
`school_data_annual_ees` and `school_data_annual_idaci`.
|
||||
@@ -150,3 +158,150 @@ token Gitea Actions provides automatically (`secrets.GITEA_TOKEN` — no setup
|
||||
needed), and fails the check only when a finding is rated
|
||||
**severe** (would break prod, leak data, or corrupt data). Minor findings are
|
||||
informational and never block a merge.
|
||||
|
||||
## Rate limiting, and the Cloudflare gap
|
||||
|
||||
Two independent limits protect the API:
|
||||
|
||||
- **Per client**, via slowapi, keyed on `CF-Connecting-IP` (falling back to
|
||||
`X-Forwarded-For`, then the peer address). 60/minute by default;
|
||||
`/api/suggest` gets 120/minute because typing is bursty.
|
||||
- **Globally**, via `GlobalRateLimitMiddleware`: a fixed 60-second window over
|
||||
all `/api/` traffic, `GLOBAL_RATE_LIMIT_PER_MINUTE` (default 3000),
|
||||
independent of any client identity. Requests from `127.0.0.1` are exempt so
|
||||
the container healthcheck cannot be starved into a restart loop.
|
||||
|
||||
### Open: the origin must only accept Cloudflare
|
||||
|
||||
`CF-Connecting-IP` is only meaningful for requests that actually reached the
|
||||
origin through Cloudflare, and **the application cannot verify that they did**.
|
||||
Anything able to reach the origin directly can set that header freely and, by
|
||||
rotating it, mint a fresh rate-limit bucket per request — defeating per-client
|
||||
limits on every endpoint.
|
||||
|
||||
The global ceiling bounds the damage to total origin capacity. It does not fix
|
||||
the underlying gap, and nothing in the code can. Closing it needs one of:
|
||||
|
||||
- **Authenticated Origin Pulls** — Cloudflare presents a client certificate the
|
||||
origin requires, so non-Cloudflare traffic is refused at TLS.
|
||||
- **An origin firewall** restricted to Cloudflare's published IP ranges.
|
||||
|
||||
Until one is in place, treat per-client limits as protection against accidents
|
||||
and ordinary load, not against a determined caller.
|
||||
|
||||
## Feature flags (Unleash)
|
||||
|
||||
Flag state lives in a self-hosted Unleash instance, deployed as its own
|
||||
Portainer stack from `docker-compose.portainer.unleash.yml`. It is separate
|
||||
from the application stacks on purpose — redeploying staging must not be able
|
||||
to disturb production's flags.
|
||||
|
||||
The flags themselves are declared in `backend/flags.py`. Unleash holds the
|
||||
state; the registry holds the list. A flag in the UI that is not in the
|
||||
registry is orphaned and nothing reads it.
|
||||
|
||||
### First-time setup
|
||||
|
||||
1. Deploy the stack in Portainer. Set `UNLEASH_DB_PASSWORD`,
|
||||
`UNLEASH_ADMIN_PASSWORD` and (optionally) `UNLEASH_IP`.
|
||||
2. Log in to the UI at `http://<UNLEASH_IP>:4242` as `admin`.
|
||||
3. Create one **client** API token per environment:
|
||||
- `schoolcompare-staging`, environment **development**
|
||||
- `schoolcompare-prod`, environment **production**
|
||||
|
||||
Client tokens, not admin tokens — the backend only reads.
|
||||
4. Put each token in the matching Portainer stack's `UNLEASH_API_TOKEN`
|
||||
variable, and set `UNLEASH_URL` to `http://<UNLEASH_IP>:4242/api`.
|
||||
5. Redeploy the application stacks.
|
||||
|
||||
### Adding a flag to Unleash
|
||||
|
||||
**Unleash does not create flags by itself.** The SDK reads definitions from the
|
||||
server and never registers anything, and metrics for a flag the server has
|
||||
never heard of are discarded. So a flag declared in `backend/flags.py` will be
|
||||
evaluated on every request, stay `False` forever, and never appear in the UI
|
||||
until someone creates it there by hand.
|
||||
|
||||
For each flag in the registry, create one in Unleash with:
|
||||
|
||||
- **Name** — character for character what `backend/flags.py` declares.
|
||||
snake_case, no hyphens or spaces. A typo produces a flag that looks correct
|
||||
in the UI and is read by nothing.
|
||||
- **Type** — Release. No strategies, constraints or variants: these are plain
|
||||
on/off switches, by design.
|
||||
|
||||
### Turning a feature on
|
||||
|
||||
Toggle the flag in the environment matching the stack you mean: **development**
|
||||
for staging, **production** for prod. The token in each stack is scoped to one
|
||||
environment, so toggling the other one has no visible effect.
|
||||
|
||||
The SDK refreshes every 15 seconds, so the API reflects the change almost at
|
||||
once; the pages follow on their own schedule, below.
|
||||
|
||||
A flip reaches school pages within about five minutes and place pages within
|
||||
the hour. Next's ISR does the propagating — it revalidates a route at the
|
||||
*lowest* `revalidate` among that route's fetches, which is 300s for
|
||||
`/school/[slug]` and 3600s for the place pages. There is no webhook, and
|
||||
adding one would only be worth it if flips ever needed to be instant.
|
||||
|
||||
### When Unleash is unreachable
|
||||
|
||||
Every flag evaluates to `False` and the site serves as though nothing were
|
||||
switched on. That is deliberate — an unfinished feature staying hidden is the
|
||||
safe direction — but it means a *released* feature disappears if a backend
|
||||
container cold-starts with an empty cache while Unleash is down. The SDK's
|
||||
disk cache is on a named volume so restarts keep last-known state, and flags
|
||||
are removed from the code within 90 days (enforced by a test), which bounds
|
||||
how long any feature is exposed to this.
|
||||
|
||||
If `UNLEASH_URL` is unset, every flag is `False` and no connection is
|
||||
attempted. That is the correct behaviour for local development and CI, and it
|
||||
means the test suites need no flag server.
|
||||
|
||||
## Release identity and the P1 reliability gate
|
||||
|
||||
Every staging run creates a random build ID before building its three images.
|
||||
Each image carries the commit and build ID as labels. Frontend/backend images
|
||||
also contain a build-time JSON file; environment overrides cannot rewrite it.
|
||||
`/release.json` returns both identities with `Cache-Control: no-store`. It fails
|
||||
with 503 when either identity cannot be read. FastAPI's internal endpoint is
|
||||
`/api/release`.
|
||||
|
||||
The entire staging workflow shares one concurrency group, with cancellation
|
||||
disabled. This needs Gitea 1.26 or newer, where workflow concurrency is supported
|
||||
([release notes](https://blog.gitea.com/release-of-1.26.0/)); the configured server
|
||||
reported 1.27.3 during this change. Do not run the workflow on an older server
|
||||
that ignores the concurrency key. Manual deployments outside this workflow must
|
||||
also avoid changing staging during journeys.
|
||||
|
||||
The gate checks both identities before and after Playwright. It then validates
|
||||
labels on the captured build output digests and tags them `verified-<full-sha>`.
|
||||
The manual promotion script resolves all three verified tags to immutable digests
|
||||
and confirms one matching commit/build ID before moving any `:prod` tag. It polls
|
||||
production for that same identity using a locally saved release manifest.
|
||||
A registry error can still interrupt the three tag writes; the Portainer webhook
|
||||
only runs after successful promotion, and rerunning promotion resolves the full
|
||||
verified set again. There is no cross-registry atomic tag transaction.
|
||||
|
||||
**First rollout:** old green commits without verified tags/build identities are
|
||||
not promotable through this gate. Build and test a commit containing the new
|
||||
workflow first. The release route must be reachable through the configured
|
||||
`STAGING_BASE_URL`/`PROD_BASE_URL`; it deliberately avoids the public staging
|
||||
`/api` proxy limitation. No new deployment secret is required.
|
||||
|
||||
`scripts/ci/release.py` implements identity polling and digest verification.
|
||||
The poller identifies itself as `SchoolCompare-Release-Check/1.0`: the public
|
||||
staging proxy has returned HTTP 403 to Python's default urllib user agent even
|
||||
while the release endpoint was healthy. It logs changes in HTTP/connection
|
||||
failures or observed release identities, and includes the last observation in
|
||||
the timeout error. If verification fails, use that observation to distinguish
|
||||
proxy rejection (403), an unavailable release endpoint (503), and containers
|
||||
still reporting an older SHA/build ID. Check the configured base URL from the
|
||||
CI runner; a successful request from another machine does not establish runner
|
||||
connectivity. Do not bypass identity verification to unblock a deployment.
|
||||
Its mocked tests run in PR checks alongside backend and index-publication tests.
|
||||
The new Playwright journeys also check deployed identity and stale pagination.
|
||||
Local unit checks do not validate registry credentials, Portainer behaviour,
|
||||
proxy routing or a deployed image; those require the staging run. Production
|
||||
promotion remains a separate human action.
|
||||
@@ -0,0 +1,94 @@
|
||||
# Development and validation
|
||||
|
||||
## Prerequisites and environment boundaries
|
||||
|
||||
Use a feature branch. The deployed stack is the integration environment; do not
|
||||
assume a local server can run from a fresh checkout. This cleanup did not start
|
||||
local servers or provision databases. Unit tests use fixtures and mocks.
|
||||
|
||||
The current versions are not yet aligned:
|
||||
|
||||
| Component | Container | PR checks |
|
||||
|---|---|---|
|
||||
| Backend | Python 3.11 | Python 3.12 |
|
||||
| Frontend | Node 24 | Node 22 |
|
||||
| Pipeline | Python 3.13 | Pipeline image build |
|
||||
|
||||
Use the component's container version when reproducing deployment behaviour.
|
||||
The backend dependency pins predate Python 3.14; do not assume the system Python
|
||||
can install or run them. Version alignment is a separate maintenance task.
|
||||
|
||||
## Frontend checks
|
||||
|
||||
```sh
|
||||
cd nextjs-app
|
||||
npm ci
|
||||
npm run typecheck
|
||||
npm test -- --runInBand
|
||||
```
|
||||
|
||||
`npm run build` is the production build check. There is no `lint` script.
|
||||
Tests live in `__tests__/` and use Jest/React Testing Library. These checks do not
|
||||
prove that live PostgreSQL queries, Typesense or a deployed proxy work.
|
||||
|
||||
The frontend `.env.example` documents runtime variables. Browser traffic normally
|
||||
uses `/api`; `FASTAPI_URL` is an absolute server-side URL ending in `/api`.
|
||||
Payload additionally needs `DATABASE_URL` and `PAYLOAD_SECRET` when used at runtime.
|
||||
Never commit credentials or real `.env` files.
|
||||
|
||||
## Backend checks
|
||||
|
||||
From the repository root, using an available Python 3.11 or 3.12 interpreter:
|
||||
|
||||
```sh
|
||||
python3.11 -m venv /tmp/schoolcompare-backend-venv
|
||||
/tmp/schoolcompare-backend-venv/bin/python -m pip install -r requirements.txt pytest 'httpx<0.28' pyyaml
|
||||
/tmp/schoolcompare-backend-venv/bin/python -m pytest backend/tests pipeline/tests scripts/ci/tests -q
|
||||
```
|
||||
|
||||
Substitute `python3.12` if matching PR CI. The test dependencies above match the
|
||||
current workflow; they are not yet captured in a dedicated development lockfile.
|
||||
Backend configuration is defined in `backend/config.py`; `.env.example` documents
|
||||
commonly used values. `ALLOWED_ORIGINS` uses a JSON array, not a comma-separated string.
|
||||
|
||||
## Data and pipeline work
|
||||
|
||||
The app needs populated `marts.*` tables. A new Postgres instance alone is not a
|
||||
working school-data environment. Use the existing managed pipeline or an approved
|
||||
snapshot; the removed CSV importer cannot build the current schema.
|
||||
|
||||
The pipeline container includes Meltano, dbt/Postgres, Airflow and the custom taps.
|
||||
Airflow commands/selectors live in `pipeline/dags/`. Schema tests live in
|
||||
`pipeline/transform/tests/` and model YAML files. Run the relevant `dbt build`
|
||||
selector in an isolated data environment for model changes; it writes tables and
|
||||
is not a read-only smoke test. Prefer `python -m dbt.cli.main` as the DAGs do.
|
||||
|
||||
GIAS dictionaries are generated together by
|
||||
`pipeline/scripts/generate_gias_codes.py`. The backend and pipeline copies are
|
||||
intentional; `backend/tests/test_gias_codes.py` checks that they stay identical.
|
||||
|
||||
For Payload collection/editor changes, run `npm run generate:importmap` in
|
||||
`nextjs-app/` and include the generated map. Preserve CMS migrations and the
|
||||
separate `payload` schema. See [publishing](../nextjs-app/docs/PUBLISHING.md).
|
||||
|
||||
## End-to-end checks
|
||||
|
||||
Against an existing, authorised test environment:
|
||||
|
||||
```sh
|
||||
cd e2e
|
||||
npm ci
|
||||
npx playwright install chromium
|
||||
BASE_URL=https://your-test-environment.example npx playwright test
|
||||
```
|
||||
|
||||
The suite does not start a web server. CI installs Chromium with system dependencies
|
||||
and runs against staging. Use the configured staging target: `docs/DEPLOY.md`
|
||||
records the public staging proxy limitation. User-visible behaviour changes should
|
||||
update the corresponding journeys.
|
||||
|
||||
## Before requesting review
|
||||
|
||||
Run checks relevant to the change, inspect `git diff --check`, and report checks
|
||||
that could not run. Do not publish or promote as part of local validation.
|
||||
[DEPLOY.md](DEPLOY.md) documents the PR and human promotion gates.
|
||||
@@ -0,0 +1,90 @@
|
||||
# Legacy and unused-code inventory
|
||||
|
||||
Reviewed 2026-09-14. This inventory records source evidence, not production usage
|
||||
telemetry. A command with no repository caller may still be run manually or from
|
||||
an external scheduler. Historical specs and prototypes are not runtime imports.
|
||||
|
||||
## Method and scope
|
||||
|
||||
Searched backend imports, tests, CLI scripts, Airflow DAGs, Meltano configuration,
|
||||
Gitea workflows, Dockerfiles and documentation. For frontend candidates, inspected
|
||||
TypeScript imports, re-exports, literal dynamic imports and `require` calls,
|
||||
resolving relative and `@/` paths while excluding tests, dependencies and build
|
||||
output. Checked candidates again with text searches including tests.
|
||||
|
||||
Next.js route files, generated Payload import-map entries and plugin discovery
|
||||
are entry points even without ordinary imports. This is why a zero-import count
|
||||
alone is not sufficient grounds for deletion. Computed imports and external
|
||||
operators are outside this static audit.
|
||||
|
||||
## Removed in this cleanup
|
||||
|
||||
These names are recorded for Git-history lookup; they are no longer file links.
|
||||
|
||||
| Removed path or symbol | Evidence and replacement |
|
||||
|---|---|
|
||||
| `backend/migration.py` | Imported `School` and `SchoolResult`, which no longer exist in `backend/models.py`. Only the legacy CSV CLI imported it. Current tables are built by dbt. |
|
||||
| `backend/version.py` | Only the legacy importer consumed `SCHEMA_VERSION`. FastAPI lifespan does not perform version-triggered imports. This is unrelated to active Payload migrations. |
|
||||
| `scripts/migrate_csv_to_db.py` | Imported removed `init_db`/`set_db_schema_version` helpers and the obsolete models indirectly. No runtime, DAG or workflow calls it. Use the managed pipeline for current marts. |
|
||||
| `scripts/geocode_schools.py` | Imported the removed `School` ORM model. No pipeline/workflow calls it. Coordinates now come from GIAS/PostGIS; a separate mart-aware manual utility remains under `pipeline/scripts/`. |
|
||||
| `backend.data_loader.haversine_distance` | No callers. Search uses its inline vectorised NumPy calculation. |
|
||||
| `nextjs-app/lib/api.ts: fetcher` | No callers; SWR is not installed. Application fetches use the named API wrappers. |
|
||||
| `nextjs-app/lib/api.ts: kmToMiles` | No callers. `calculateDistance` remains because `CutoffMapPanel` uses it. |
|
||||
|
||||
The removed command files could not import successfully against the current
|
||||
backend. This cleanup does not run replacements, migrate data or modify databases.
|
||||
Their previous implementations remain recoverable from Git history.
|
||||
|
||||
## Unused candidates retained for a separate cleanup
|
||||
|
||||
| Candidate | Evidence | Recommended next step |
|
||||
|---|---|---|
|
||||
| `nextjs-app/components/LoadingSkeleton.tsx` and its CSS | No application or test imports found. | Remove together after confirming no planned use. |
|
||||
| `nextjs-app/components/Pagination.tsx` and its CSS | No application or test imports found; HomeView implements load-more behaviour. | Remove as a pair if numbered pagination will not return. |
|
||||
| `nextjs-app/components/SchoolCard.tsx` and its CSS | Imported by its own tests, not application code. HomeView uses SchoolRow/SecondarySchoolRow. | Decide whether to retire the card design; if removed, remove its dedicated tests as well. Passing tests do not establish runtime use. |
|
||||
| `backend/database.py: get_db`, `get_db_session` | No remaining callers after removing the importer. Current code creates SessionLocal directly. | Either adopt these helpers during session-lifecycle cleanup or remove them; do not rewrite active sessions in a documentation change. |
|
||||
| `backend/schemas.py: COLUMN_MAPPINGS`, `NULL_VALUES`, `LA_CODE_TO_NAME` | No remaining Python consumers found after importer removal. Other constants in this module are active. | Remove individual constants after checking external data utilities; retain the module. |
|
||||
| `backend/app.py: result_filters` keys `school_types`, `phases`, `genders`, `admissions_policies` | Since 2026-10-02 FilterBar offers these from `/api/filters`, because options scoped to the results left only the chosen value on offer. Only `local_authorities` is still read. | Stop computing the four keys in a focused API change; keep `local_authorities`. |
|
||||
| `backend/config.py: data_dir`, `max_page_size`, `rate_limit_burst` | No active consumers found. `default_page_size` appears only in a branch that expects None, although the route supplies a concrete default. | Reconcile settings with route validation in a focused API change. |
|
||||
|
||||
## Legacy/manual paths requiring operational verification
|
||||
|
||||
| Path | Status and reason to retain for now |
|
||||
|---|---|
|
||||
| FastAPI `/`, `/compare`, `/rankings`, `/favicon.svg`, `/robots.txt`, and conditional `/static` | Old frontend-serving routes reference a `frontend/` directory absent from the checkout and backend image. Next.js owns these public surfaces. Removal changes externally callable routes, so first check proxy/operator usage and define replacement responses. |
|
||||
| `scripts/fetch_real_data.py`, `scripts/download_data.py` | Historical standalone CSV utilities. The fetch script targets Wandsworth/Merton; neither is wired into the managed pipeline. Marked historical, retained pending confirmation of manual use. |
|
||||
| `pipeline/scripts/geocode_postcodes.py` | Mart-aware postcode fallback, not called by the current DAGs. Do not confuse it with the removed legacy ORM geocoder. Verify the target schema before manual use. |
|
||||
| `docker-compose.yml` | Uses unpublished `:latest` release tags and lacks frontend Payload DB/secret/media configuration. Retained as an old development topology, not recommended onboarding. |
|
||||
| `nextjs-app/docker-compose.yml` | Standalone legacy recipe with old backend port assumptions and no CMS persistence setup. Retained until its consumers are checked. |
|
||||
| `MIGRATION_SUMMARY.md`, `docs/superpowers/`, `mockups/` | Historical designs and prototypes. Retain as history; do not follow as current deployment instructions. |
|
||||
| `scripts/sql/drop_fact_parent_view.sql` | One-off maintenance SQL. Not an application entry point; repository call-site searches cannot establish whether it is still needed operationally. |
|
||||
|
||||
## Active code that can look obsolete
|
||||
|
||||
- `backend/data_loader.py` older-mart query fallbacks are covered by backend tests
|
||||
and support databases at different migration stages. Remove only after verifying
|
||||
the deployed schemas in every supported environment.
|
||||
- `backend/gias_codes.py` and `pipeline/scripts/gias_codes.py` are intentionally
|
||||
generated copies for separate runtime images. Their parity is tested.
|
||||
- `nextjs-app/migrations/`, `payload-types.ts` and the Payload import map are active
|
||||
CMS artifacts, not remnants of the removed school importer.
|
||||
- `get_available_years`, `get_available_local_authorities` and `get_schools_count`
|
||||
in `data_loader.py` are called through `get_data_info`, which serves the backend
|
||||
data-info endpoint. They are not dead functions.
|
||||
- `get_supplementary_data` is an intentional single-school wrapper around the
|
||||
batch implementation.
|
||||
- `pipeline/transform` models named `legacy` can be active data sources: annual
|
||||
DAG selectors explicitly include legacy KS2/KS4 lineage. Names alone do not
|
||||
establish obsolescence.
|
||||
|
||||
## Suggested next passes
|
||||
|
||||
1. Decide the fate of the three unused UI components and remove paired assets/tests.
|
||||
2. Consolidate backend session usage and remove abandoned settings/constants.
|
||||
3. Verify external consumers, then retire static-serving API routes and old compose recipes.
|
||||
4. Audit manual data utilities with pipeline operators before deleting them.
|
||||
5. Revisit compatibility fallbacks only after documenting supported schema versions.
|
||||
|
||||
Validation for this cleanup should include frontend typechecking/tests, Python
|
||||
syntax checks, reference searches and documentation link checks. Live database,
|
||||
external scheduler and deployed route usage require separate integration evidence.
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,911 @@
|
||||
# School Type Groups and a Faith Filter Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Replace the School type filter's 34 GIAS types with six parent-facing groups, and add a Faith filter with eight options.
|
||||
|
||||
**Architecture:** A new backend module `backend/school_groups.py` holds both groupings as GIAS code sets and exposes lookups by the translated *name* (the DataFrame the API filters only carries names). `/api/schools` accepts a group key for `school_type` and a new `faith` key; `/api/filters` gains `school_type_groups` and `faiths` as `{value, label}` lists. The frontend's FilterBar reads those lists for its School type select and a new Faith select.
|
||||
|
||||
**Tech Stack:** FastAPI + pandas (backend, pytest via uv), Next.js 15 / React (frontend, Jest + Testing Library), Playwright (E2E against staging).
|
||||
|
||||
**Spec:** `docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md`
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Type group keys, labels and order, verbatim: `academy` "State school: academy or free school", `council` "State school: council-run", `independent` "Independent (fee-paying)", `special` "Special school (SEND)", `post16` "Sixth form or college", `alternative` "Alternative provision".
|
||||
- Faith keys, labels and order, verbatim: `none` "No religious character", `church_of_england` "Church of England", `roman_catholic` "Roman Catholic", `other_christian` "Other Christian", `jewish` "Jewish", `muslim` "Muslim", `other_faith` "Other faith".
|
||||
- Select defaults, verbatim: "Any school type", "Any faith or none". Faith select `aria-label="Faith"`.
|
||||
- `/api/filters` keeps `school_types` (raw list) unchanged; the new keys are additions.
|
||||
- An old raw-label `school_type` value (e.g. `Community school`) still filters by that label exactly.
|
||||
- An unknown `faith` key returns no schools.
|
||||
- Do not change `backend/gias_codes.py` or `pipeline/scripts/gias_codes.py` (their parity is tested).
|
||||
- No copy may use an em dash (`__tests__/components/noEmDashCopy.test.ts`).
|
||||
- Backend tests run from the repo root with: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests -q`
|
||||
- Frontend checks run from `nextjs-app/`: `npx jest` and `npx tsc --noEmit`.
|
||||
- Do not start a local server. E2E journeys are verified against staging after merge.
|
||||
|
||||
## Review Focus
|
||||
|
||||
- A school whose `religious_denomination` is `None`/NaN or `""` (GIAS code 99) must count as "No religious character", not drop out of every faith option. Pinned in Task 1.
|
||||
- A type or religion label the dictionaries do not know (e.g. `"Unknown (9999)"`, or a test fixture's `"Academy"`) must map to no group without raising. Pinned in Task 1.
|
||||
- `/api/filters` returning no `school_type_groups`/`faiths` (API failed, or an older backend) must leave the selects out rather than render "Any…" alone. Pinned in Task 3.
|
||||
- An old bookmarked URL with `school_type=Community+school` must still return community schools and show a chip with its raw label. Pinned in Tasks 2 and 3.
|
||||
- Mixed case in a key (`?faith=Roman_Catholic`, `?school_type=Special`) must still match. Pinned in Task 2.
|
||||
|
||||
---
|
||||
|
||||
## File Structure
|
||||
|
||||
| File | Responsibility |
|
||||
|---|---|
|
||||
| `backend/school_groups.py` (new) | The two groupings as code sets, and name-based lookups. Nothing else. |
|
||||
| `backend/tests/test_school_groups.py` (new) | Coverage of every GIAS code, joint faiths, unknown/blank names. |
|
||||
| `backend/app.py` (modify) | `faith` param; group-key branch for `school_type`; two new `/api/filters` keys. |
|
||||
| `backend/tests/test_type_and_faith_filters.py` (new) | API behaviour of both filters and the new `/api/filters` keys. |
|
||||
| `nextjs-app/lib/types.ts` (modify) | `FilterOption`; optional `school_type_groups`/`faiths` on `Filters`; `faith` on `SchoolSearchParams`. |
|
||||
| `nextjs-app/app/(frontend)/page.tsx` (modify) | Read and forward `faith`. |
|
||||
| `nextjs-app/components/FilterBar.tsx` (modify) | Grouped School type options; Faith select; chips, counts, analytics. |
|
||||
| `nextjs-app/__tests__/components/FilterBarTypeFaith.test.tsx` (new) | Frontend behaviour of both selects. |
|
||||
| Existing FilterBar tests (modify) | Fixtures move from raw types to groups. |
|
||||
| `e2e/tests/journeys.spec.ts` (modify) | A journey for both filters. |
|
||||
| Spec (modify) | Architecture section: grouping is by name at filter time. |
|
||||
|
||||
---
|
||||
|
||||
### Task 1: The groupings module
|
||||
|
||||
**Files:**
|
||||
- Create: `backend/school_groups.py`
|
||||
- Test: `backend/tests/test_school_groups.py`
|
||||
- Modify: `docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md` (Architecture)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `backend.gias_codes.SCHOOL_TYPE: dict[int, str]`, `backend.gias_codes.RELIGIOUS_CHARACTER: dict[int, str]` (code 99 maps to `""`).
|
||||
- Produces:
|
||||
- `TYPE_GROUPS: tuple[tuple[str, str, frozenset[int]], ...]` (key, label, codes), in display order
|
||||
- `UNOFFERED_TYPE_CODES: frozenset[int]`
|
||||
- `FAITH_GROUPS: tuple[tuple[str, str, frozenset[int]], ...]`, in display order
|
||||
- `TYPE_GROUP_KEYS: frozenset[str]`, `FAITH_KEYS: frozenset[str]`
|
||||
- `type_group_for(name: object) -> str | None`
|
||||
- `faith_groups_for(name: object) -> tuple[str, ...]`
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
Create `backend/tests/test_school_groups.py`:
|
||||
|
||||
```python
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
Every GIAS code must be accounted for, so a new DfE code fails here instead of
|
||||
silently vanishing from the filter.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
from backend.gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
from backend.school_groups import (
|
||||
FAITH_GROUPS,
|
||||
TYPE_GROUPS,
|
||||
UNOFFERED_TYPE_CODES,
|
||||
faith_groups_for,
|
||||
type_group_for,
|
||||
)
|
||||
|
||||
DBT_PROJECT = Path(__file__).resolve().parents[2] / "pipeline" / "transform" / "dbt_project.yml"
|
||||
|
||||
|
||||
def _non_england_codes() -> set[int]:
|
||||
return set(yaml.safe_load(DBT_PROJECT.read_text())["vars"]["non_england_school_type_codes"])
|
||||
|
||||
|
||||
def test_every_type_code_is_in_exactly_one_place():
|
||||
places = [codes for _, _, codes in TYPE_GROUPS] + [UNOFFERED_TYPE_CODES, _non_england_codes()]
|
||||
for code in SCHOOL_TYPE:
|
||||
homes = sum(code in p for p in places)
|
||||
assert homes == 1, f"type code {code} ({SCHOOL_TYPE[code]}) is in {homes} places"
|
||||
|
||||
|
||||
def test_every_religion_code_has_a_faith():
|
||||
for code, name in RELIGIOUS_CHARACTER.items():
|
||||
assert any(code in codes for _, _, codes in FAITH_GROUPS), f"{code} {name!r}"
|
||||
|
||||
|
||||
def test_type_groups_in_display_order():
|
||||
assert [k for k, _, _ in TYPE_GROUPS] == [
|
||||
"academy", "council", "independent", "special", "post16", "alternative"]
|
||||
|
||||
|
||||
def test_faiths_in_display_order():
|
||||
assert [k for k, _, _ in FAITH_GROUPS] == [
|
||||
"none", "church_of_england", "roman_catholic", "other_christian",
|
||||
"jewish", "muslim", "other_faith"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, group", [
|
||||
("Academy converter", "academy"),
|
||||
("University technical college", "academy"),
|
||||
("Voluntary aided school", "council"),
|
||||
("Local authority nursery school", "council"),
|
||||
("Other independent school", "independent"),
|
||||
("Other independent special school", "special"),
|
||||
("Special post 16 institution", "special"),
|
||||
("Further education", "post16"),
|
||||
("Pupil referral unit", "alternative"),
|
||||
("academy CONVERTER", "academy"),
|
||||
])
|
||||
def test_type_group_by_name(name, group):
|
||||
assert type_group_for(name) == group
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", [
|
||||
"Higher education institutions", "Miscellaneous", "Unknown (9999)", "Academy", "", None, np.nan,
|
||||
])
|
||||
def test_unoffered_or_unknown_types_have_no_group(name):
|
||||
assert type_group_for(name) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, faiths", [
|
||||
("Roman Catholic/Church of England", ("church_of_england", "roman_catholic")),
|
||||
("Roman Catholic/Anglican", ("church_of_england", "roman_catholic")),
|
||||
("Church of England/Methodist", ("church_of_england", "other_christian")),
|
||||
("Church of England/Christian", ("church_of_england",)),
|
||||
("Catholic", ("roman_catholic",)),
|
||||
("Inter- / non- denominational", ("other_christian",)),
|
||||
("Orthodox Jewish", ("jewish",)),
|
||||
("Sunni Deobandi", ("muslim",)),
|
||||
("Hindu", ("other_faith",)),
|
||||
("Does not apply", ("none",)),
|
||||
("None", ("none",)),
|
||||
])
|
||||
def test_faiths_by_name(name, faiths):
|
||||
assert faith_groups_for(name) == faiths
|
||||
|
||||
|
||||
@pytest.mark.parametrize("missing", [None, np.nan, "", " "])
|
||||
def test_a_missing_religion_is_no_religious_character(missing):
|
||||
assert faith_groups_for(missing) == ("none",)
|
||||
|
||||
|
||||
def test_an_unknown_religion_has_no_faith():
|
||||
assert faith_groups_for("Unknown (77)") == ()
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run them to see them fail**
|
||||
|
||||
Run: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_school_groups.py -q`
|
||||
Expected: collection error, `ModuleNotFoundError: No module named 'backend.school_groups'`.
|
||||
|
||||
- [ ] **Step 3: Write the module**
|
||||
|
||||
Create `backend/school_groups.py`:
|
||||
|
||||
```python
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
The 34 GIAS establishment types describe governance and funding, which for a
|
||||
mainstream state school barely changes what a parent experiences. The search
|
||||
filter offers six groups a parent recognises instead, and a faith filter in
|
||||
place of the faith signal that "Voluntary aided" and "Voluntary controlled"
|
||||
only half carry. See
|
||||
docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md.
|
||||
|
||||
Groups are defined over GIAS codes and looked up by the translated name,
|
||||
because the DataFrame the API filters carries names only: the codes are
|
||||
replaced at load (data_loader.translate_gias_code_columns), the legacy-name
|
||||
mart fallback never had them, and test fixtures are written in names.
|
||||
"""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
|
||||
# (key, label, GIAS TypeOfEstablishment codes), in the order shown.
|
||||
TYPE_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Academy sponsor led, Academy converter, Free schools, University technical
|
||||
# college, Studio schools, City technology college. UTCs and studio schools
|
||||
# are legally academies, and a family considering one searches by name.
|
||||
("academy", "State school: academy or free school", frozenset({28, 34, 35, 40, 41, 6})),
|
||||
# Community, Voluntary aided, Voluntary controlled, Foundation, LA nursery.
|
||||
("council", "State school: council-run", frozenset({1, 2, 3, 5, 15})),
|
||||
("independent", "Independent (fee-paying)", frozenset({11})),
|
||||
# Every special type, independent ones included (usually funded by the
|
||||
# council through an EHCP, so SEND provision to a parent, not private
|
||||
# school), and Special post 16 institutions.
|
||||
("special", "Special school (SEND)", frozenset({7, 8, 10, 12, 32, 33, 36, 44})),
|
||||
# Further education, Sixth form centres, the 16-19 academies and free schools.
|
||||
("post16", "Sixth form or college", frozenset({18, 31, 39, 45, 46})),
|
||||
# Pupil referral units and AP academies and free schools. Last: parents do
|
||||
# not apply to these; the local authority places children there.
|
||||
("alternative", "Alternative provision", frozenset({14, 38, 42, 43})),
|
||||
)
|
||||
|
||||
# In no group, so reachable only under "Any school type": Secure units,
|
||||
# Miscellaneous, Higher education institutions, Online provider, Institution
|
||||
# funded by other government department, Academy secure 16 to 19.
|
||||
UNOFFERED_TYPE_CODES: frozenset[int] = frozenset({24, 27, 29, 49, 56, 57})
|
||||
|
||||
# (key, label, GIAS ReligiousCharacter codes), in the order shown. A joint
|
||||
# school is in every faith its label names; a generic "Christian" beside a
|
||||
# named church adds nothing.
|
||||
FAITH_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Does not apply, None, and 99 (a blank label).
|
||||
("none", "No religious character", frozenset({0, 6, 99})),
|
||||
("church_of_england", "Church of England",
|
||||
frozenset({2, 9, 10, 11, 12, 13, 19, 20, 30, 31, 32, 33, 34, 41, 48})),
|
||||
("roman_catholic", "Roman Catholic", frozenset({3, 11, 13, 35, 48})),
|
||||
# 28 "Inter- / non- denominational" is how GIAS files Christian schools
|
||||
# tied to no one church.
|
||||
("other_christian", "Other Christian",
|
||||
frozenset({4, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 22, 26, 28, 30, 33,
|
||||
37, 38, 39, 40, 41, 44, 45, 46, 47})),
|
||||
("jewish", "Jewish", frozenset({5, 36, 43})),
|
||||
("muslim", "Muslim", frozenset({7, 42, 49})),
|
||||
("other_faith", "Other faith", frozenset({21, 24, 25, 29})),
|
||||
)
|
||||
|
||||
TYPE_GROUP_KEYS: frozenset[str] = frozenset(k for k, _, _ in TYPE_GROUPS)
|
||||
FAITH_KEYS: frozenset[str] = frozenset(k for k, _, _ in FAITH_GROUPS)
|
||||
|
||||
|
||||
def _key(name: str) -> str:
|
||||
return name.strip().lower()
|
||||
|
||||
|
||||
_TYPE_GROUP_BY_NAME: dict[str, str] = {
|
||||
_key(SCHOOL_TYPE[code]): key
|
||||
for key, _, codes in TYPE_GROUPS
|
||||
for code in codes
|
||||
if code in SCHOOL_TYPE
|
||||
}
|
||||
|
||||
_FAITHS_BY_NAME: dict[str, tuple[str, ...]] = {}
|
||||
for _faith, _, _codes in FAITH_GROUPS:
|
||||
for _code in sorted(_codes):
|
||||
if _code in RELIGIOUS_CHARACTER:
|
||||
_name = _key(RELIGIOUS_CHARACTER[_code])
|
||||
_FAITHS_BY_NAME[_name] = _FAITHS_BY_NAME.get(_name, ()) + (_faith,)
|
||||
|
||||
|
||||
def type_group_for(name: object) -> Optional[str]:
|
||||
"""The type group of a GIAS establishment type name, or None."""
|
||||
if not isinstance(name, str):
|
||||
return None
|
||||
return _TYPE_GROUP_BY_NAME.get(_key(name))
|
||||
|
||||
|
||||
def faith_groups_for(name: object) -> tuple[str, ...]:
|
||||
"""The faith groups of a GIAS religious character name.
|
||||
|
||||
A missing or blank name is "No religious character". A name the
|
||||
dictionary does not know has no faith, so it matches no faith option.
|
||||
"""
|
||||
if not isinstance(name, str):
|
||||
return ("none",) if name is None or pd.isna(name) else ()
|
||||
return _FAITHS_BY_NAME.get(_key(name), ())
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run the tests to see them pass**
|
||||
|
||||
Run: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_school_groups.py -q`
|
||||
Expected: all pass.
|
||||
|
||||
- [ ] **Step 5: Correct the spec's architecture**
|
||||
|
||||
In the spec, replace the `### backend/data_loader.py` subsection with:
|
||||
|
||||
```markdown
|
||||
### Grouping by name, at filter time
|
||||
|
||||
The groups are defined over codes but looked up by the translated name
|
||||
(`type_group_for(name)`, `faith_groups_for(name)`), when `/api/schools` and
|
||||
`/api/filters` filter. The DataFrame those endpoints read carries names only:
|
||||
`translate_gias_code_columns` replaces the codes at load, the legacy-name
|
||||
mart fallback never had codes, and the API test fixtures are written in
|
||||
names. The names come from the same dictionaries, so the lookup is exact.
|
||||
`data_loader.py` is unchanged.
|
||||
```
|
||||
|
||||
and in `### backend/school_groups.py (new)` replace the `type_group_for(code)` / `faith_groups_for(code)` bullet with `type_group_for(name) -> str | None` and `faith_groups_for(name) -> tuple[str, ...]` (a missing or blank name gives `("none",)`).
|
||||
|
||||
- [ ] **Step 6: Commit**
|
||||
|
||||
```bash
|
||||
git add backend/school_groups.py backend/tests/test_school_groups.py docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md
|
||||
git commit -m "feat(api): group GIAS school types and religions for parents"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 2: The API filters
|
||||
|
||||
**Files:**
|
||||
- Modify: `backend/app.py` (imports near line 43; `get_schools` params ~738-747, sanitising ~755-759, secondary filters ~782-786, `school_type` filter ~886-889; `get_filter_options` ~1155-1180)
|
||||
- Test: `backend/tests/test_type_and_faith_filters.py`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: from Task 1, `TYPE_GROUPS`, `FAITH_GROUPS`, `TYPE_GROUP_KEYS`, `FAITH_KEYS`, `type_group_for`, `faith_groups_for`.
|
||||
- Produces:
|
||||
- `GET /api/schools?school_type=<group key | raw label>&faith=<faith key>`
|
||||
- `GET /api/filters` adds `"school_type_groups": [{"value": str, "label": str}]` and `"faiths": [{"value": str, "label": str}]`, in group order, groups with no school left out.
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
Create `backend/tests/test_type_and_faith_filters.py`:
|
||||
|
||||
```python
|
||||
"""/api/schools school-type groups and faith filter, and their /api/filters lists."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
# urn -> (GIAS school type, GIAS religious character)
|
||||
SCHOOLS = {
|
||||
100001: ("Community school", "Does not apply"),
|
||||
100002: ("Voluntary aided school", "Roman Catholic"),
|
||||
100003: ("Academy converter", "Roman Catholic/Church of England"),
|
||||
100004: ("Community special school", None),
|
||||
100005: ("Academy special converter", "Church of England"),
|
||||
100006: ("Other independent school", "Jewish"),
|
||||
100007: ("Miscellaneous", ""),
|
||||
}
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
base = {
|
||||
"local_authority": "Testshire", "address": "1 Test Street", "town": "Testtown",
|
||||
"postcode": "TS1 1AA", "age_range": "4-11", "has_sixth_form": None,
|
||||
"gender": "Mixed", "admissions_policy": None, "ofsted_grade": np.nan,
|
||||
"ofsted_date": None, "ofsted_framework": None, "latitude": 51.5,
|
||||
"longitude": -0.1, "year": 202425, "total_pupils": 300,
|
||||
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan, "phase": "Primary",
|
||||
}
|
||||
return pd.DataFrame([
|
||||
{**base, "urn": urn, "school_name": f"School {urn}",
|
||||
"school_type": t, "religious_denomination": r}
|
||||
for urn, (t, r) in SCHOOLS.items()
|
||||
])
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client(monkeypatch):
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_latest_school_data", _schools_df)
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return TestClient(app_module.app, raise_server_exceptions=False)
|
||||
|
||||
|
||||
def _urns(client, **params):
|
||||
resp = client.get("/api/schools", params={"page_size": 50, **params})
|
||||
assert resp.status_code == 200, resp.text
|
||||
return sorted(s["urn"] for s in resp.json()["schools"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key, urns", [
|
||||
("council", [100001, 100002]),
|
||||
("academy", [100003]),
|
||||
("special", [100004, 100005]),
|
||||
("independent", [100006]),
|
||||
("Special", [100004, 100005]),
|
||||
])
|
||||
def test_a_type_group_key_filters_to_its_group(client, key, urns):
|
||||
assert _urns(client, school_type=key) == urns
|
||||
|
||||
|
||||
def test_a_raw_type_label_still_filters_exactly(client):
|
||||
assert _urns(client, school_type="Community school") == [100001]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key, urns", [
|
||||
("roman_catholic", [100002, 100003]),
|
||||
("church_of_england", [100003, 100005]),
|
||||
("none", [100001, 100004, 100007]),
|
||||
("jewish", [100006]),
|
||||
("Roman_Catholic", [100002, 100003]),
|
||||
])
|
||||
def test_faith_filters_to_its_group_joint_schools_included(client, key, urns):
|
||||
assert _urns(client, faith=key) == urns
|
||||
|
||||
|
||||
def test_an_unknown_faith_returns_nothing(client):
|
||||
assert _urns(client, faith="nonsense") == []
|
||||
|
||||
|
||||
def test_type_and_faith_combine(client):
|
||||
assert _urns(client, school_type="special", faith="church_of_england") == [100005]
|
||||
|
||||
|
||||
def test_filters_lists_only_groups_present_in_order(client):
|
||||
body = client.get("/api/filters").json()
|
||||
assert body["school_type_groups"] == [
|
||||
{"value": "academy", "label": "State school: academy or free school"},
|
||||
{"value": "council", "label": "State school: council-run"},
|
||||
{"value": "independent", "label": "Independent (fee-paying)"},
|
||||
{"value": "special", "label": "Special school (SEND)"},
|
||||
]
|
||||
assert body["faiths"] == [
|
||||
{"value": "none", "label": "No religious character"},
|
||||
{"value": "church_of_england", "label": "Church of England"},
|
||||
{"value": "roman_catholic", "label": "Roman Catholic"},
|
||||
{"value": "jewish", "label": "Jewish"},
|
||||
]
|
||||
# The raw list is still there for anything that reads it.
|
||||
assert "Community school" in body["school_types"]
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run them to see them fail**
|
||||
|
||||
Run: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_type_and_faith_filters.py -q`
|
||||
Expected: the group-key, faith and `/api/filters` tests fail (`KeyError: 'school_type_groups'`, empty URN lists); `test_a_raw_type_label_still_filters_exactly` passes.
|
||||
|
||||
- [ ] **Step 3: Implement**
|
||||
|
||||
In `backend/app.py`, add beside the other local imports (after the `from .schemas import ...` line):
|
||||
|
||||
```python
|
||||
from .school_groups import (
|
||||
FAITH_GROUPS,
|
||||
FAITH_KEYS,
|
||||
TYPE_GROUP_KEYS,
|
||||
TYPE_GROUPS,
|
||||
faith_groups_for,
|
||||
type_group_for,
|
||||
)
|
||||
```
|
||||
|
||||
Add a helper just above `def sanitize_search_input`:
|
||||
|
||||
```python
|
||||
def _names_in_group(names: pd.Series, in_group) -> set:
|
||||
"""The distinct names in a column that a group predicate accepts.
|
||||
|
||||
Evaluated once per distinct name rather than per row, so a filter over
|
||||
every school costs a few dozen lookups.
|
||||
"""
|
||||
return {n for n in names.dropna().unique() if in_group(n)}
|
||||
```
|
||||
|
||||
In `get_schools`, add the parameter after `has_sixth_form`:
|
||||
|
||||
```python
|
||||
faith: Optional[str] = Query(None, description="Filter by faith group key", max_length=40),
|
||||
```
|
||||
|
||||
and after `phase = sanitize_search_input(phase)`:
|
||||
|
||||
```python
|
||||
faith = sanitize_search_input(faith)
|
||||
```
|
||||
|
||||
After the `has_sixth_form` filter block (the one ending `df_latest = df_latest[flag if has_sixth_form == "yes" else ~flag]`), add:
|
||||
|
||||
```python
|
||||
# Faith group (backend/school_groups.py). A joint school is in every faith
|
||||
# its label names; a missing religious character is "none". An unknown key
|
||||
# matches nothing rather than being ignored, so a typo cannot show all.
|
||||
if faith:
|
||||
faith_key = faith.lower()
|
||||
if faith_key in FAITH_KEYS and "religious_denomination" in df_latest.columns:
|
||||
column = df_latest["religious_denomination"]
|
||||
matches = column.isin(_names_in_group(column, lambda n: faith_key in faith_groups_for(n)))
|
||||
# _names_in_group skips missing names; a missing religious
|
||||
# character is "No religious character".
|
||||
if faith_key == "none":
|
||||
matches = matches | column.isna()
|
||||
df_latest = df_latest[matches]
|
||||
else:
|
||||
df_latest = df_latest.iloc[0:0]
|
||||
```
|
||||
|
||||
Replace the existing `school_type` filter:
|
||||
|
||||
```python
|
||||
if school_type:
|
||||
schools_df = schools_df[
|
||||
schools_df["school_type"].str.lower() == school_type.lower()
|
||||
]
|
||||
```
|
||||
|
||||
with:
|
||||
|
||||
```python
|
||||
# A type group key (backend/school_groups.py), or for an old link a raw
|
||||
# GIAS type label, matched exactly as before.
|
||||
if school_type:
|
||||
type_key = school_type.lower()
|
||||
if type_key in TYPE_GROUP_KEYS:
|
||||
column = schools_df["school_type"]
|
||||
schools_df = schools_df[
|
||||
column.isin(_names_in_group(column, lambda n: type_group_for(n) == type_key))
|
||||
]
|
||||
else:
|
||||
schools_df = schools_df[schools_df["school_type"].str.lower() == type_key]
|
||||
```
|
||||
|
||||
In `get_filter_options`, add to the early `if df.empty:` return dict:
|
||||
|
||||
```python
|
||||
"school_type_groups": [],
|
||||
"faiths": [],
|
||||
```
|
||||
|
||||
and before the final `return {`:
|
||||
|
||||
```python
|
||||
def offered(groups, present):
|
||||
return [{"value": key, "label": label} for key, label, _ in groups if key in present]
|
||||
|
||||
type_groups_present = (
|
||||
{type_group_for(n) for n in df["school_type"].dropna().unique()} - {None}
|
||||
if "school_type" in df.columns else set()
|
||||
)
|
||||
faiths_present = (
|
||||
{f for n in df["religious_denomination"].unique() for f in faith_groups_for(n)}
|
||||
if "religious_denomination" in df.columns else set()
|
||||
)
|
||||
```
|
||||
|
||||
and add to the returned dict:
|
||||
|
||||
```python
|
||||
"school_type_groups": offered(TYPE_GROUPS, type_groups_present),
|
||||
"faiths": offered(FAITH_GROUPS, faiths_present),
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run the new tests, then the whole backend suite**
|
||||
|
||||
Run: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests/test_type_and_faith_filters.py -q`
|
||||
Expected: all pass.
|
||||
|
||||
Run: `uv run --with-requirements requirements.txt --with pytest --with "httpx==0.27.0" python -m pytest backend/tests pipeline/tests scripts/ci/tests -q`
|
||||
Expected: all pass (the CI command; `pyyaml` comes from requirements or add `--with pyyaml` if collection fails on `import yaml`).
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add backend/app.py backend/tests/test_type_and_faith_filters.py
|
||||
git commit -m "feat(api): filter schools by type group and by faith"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 3: The frontend filters
|
||||
|
||||
**Files:**
|
||||
- Modify: `nextjs-app/lib/types.ts` (`Filters` ~487-494, `SchoolSearchParams` ~558-570)
|
||||
- Modify: `nextjs-app/app/(frontend)/page.tsx` (props type ~15-29, `hasSearchParams` ~79-88, `fetchSchools` call ~96-109)
|
||||
- Modify: `nextjs-app/components/FilterBar.tsx`
|
||||
- Test: `nextjs-app/__tests__/components/FilterBarTypeFaith.test.tsx` (new)
|
||||
- Modify tests: `__tests__/components/ResultsToolbar.test.tsx`, `FilterSheet.test.tsx`, `FilterSheetPending.test.tsx`, `FilterBarOptions.test.tsx`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `/api/filters` keys `school_type_groups`, `faiths` (Task 2); URL params `school_type` (group key) and `faith`.
|
||||
- Produces: `export interface FilterOption { value: string; label: string }`; `Filters.school_type_groups?: FilterOption[]`; `Filters.faiths?: FilterOption[]`; `SchoolSearchParams.faith?: string`. FilterBar selects named "School type" and "Faith".
|
||||
|
||||
- [ ] **Step 1: Write the failing tests**
|
||||
|
||||
Create `nextjs-app/__tests__/components/FilterBarTypeFaith.test.tsx`:
|
||||
|
||||
```tsx
|
||||
import { fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
import { track } from '@/lib/analytics';
|
||||
|
||||
/*
|
||||
* School type offers six groups a parent recognises, not GIAS's 34 types, and
|
||||
* a Faith filter sits beside it (spec 2026-10-02-school-type-groups-and-faith-
|
||||
* filter-design.md). Both lists come from /api/filters.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Wandsworth'], school_types: ['Community school'], years: [],
|
||||
phases: ['Primary', 'Secondary'], genders: [], admissions_policies: [],
|
||||
school_type_groups: [
|
||||
{ value: 'academy', label: 'State school: academy or free school' },
|
||||
{ value: 'council', label: 'State school: council-run' },
|
||||
{ value: 'special', label: 'Special school (SEND)' },
|
||||
],
|
||||
faiths: [
|
||||
{ value: 'none', label: 'No religious character' },
|
||||
{ value: 'roman_catholic', label: 'Roman Catholic' },
|
||||
],
|
||||
};
|
||||
|
||||
const pushedParams = () => new URLSearchParams(push.mock.calls.at(-1)![0].split('?')[1]);
|
||||
const openSheet = () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Filters/ }));
|
||||
return screen.getByRole('dialog', { name: 'Filters' });
|
||||
};
|
||||
const optionsOf = (scope: HTMLElement, name: string) =>
|
||||
within(within(scope).getByRole('combobox', { name })).getAllByRole('option').map((o) => o.textContent);
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
push.mockClear();
|
||||
jest.mocked(track).mockClear();
|
||||
});
|
||||
|
||||
describe('School type', () => {
|
||||
it('offers the groups, not the GIAS types, on desktop and in the sheet', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const groups = ['Any school type', 'State school: academy or free school',
|
||||
'State school: council-run', 'Special school (SEND)'];
|
||||
expect(optionsOf(screen.getByRole('group', { name: 'Filters' }), 'School type')).toEqual(groups);
|
||||
expect(optionsOf(openSheet(), 'School type')).toEqual(groups);
|
||||
});
|
||||
|
||||
it('puts the group key in the URL and names the chip by its label', () => {
|
||||
const view = render(<FilterBar filters={filters} />);
|
||||
fireEvent.change(within(openSheet()).getByRole('combobox', { name: 'School type' }),
|
||||
{ target: { value: 'special' } });
|
||||
expect(pushedParams().get('school_type')).toBe('special');
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=special');
|
||||
view.rerender(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Special school (SEND)' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('names an old raw-label link\'s chip by that label', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=Community+school');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Community school' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('is left out when the API sends no groups', () => {
|
||||
render(<FilterBar filters={{ ...filters, school_type_groups: undefined }} />);
|
||||
expect(screen.queryByRole('combobox', { name: 'School type' })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('Faith', () => {
|
||||
it('offers its options in the More filters panel and in the sheet', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /More filters/ }));
|
||||
const faiths = ['Any faith or none', 'No religious character', 'Roman Catholic'];
|
||||
expect(optionsOf(document.body, 'Faith')).toEqual(faiths);
|
||||
expect(optionsOf(openSheet(), 'Faith')).toEqual(faiths);
|
||||
});
|
||||
|
||||
it('shows as a chip, counts on both buttons, and clears with Clear all', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&faith=roman_catholic');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Roman Catholic' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'Filters, 1 applied' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: /More filters \(1\)/ })).toBeInTheDocument();
|
||||
fireEvent.click(within(screen.getByRole('group', { name: 'Applied filters' }))
|
||||
.getByRole('button', { name: 'Clear all' }));
|
||||
expect(pushedParams().get('faith')).toBeNull();
|
||||
expect(pushedParams().get('postcode')).toBe('SW196AR');
|
||||
});
|
||||
|
||||
it('goes into the search analytics event', () => {
|
||||
params = new URLSearchParams('faith=roman_catholic');
|
||||
render(<FilterBar filters={filters} />);
|
||||
const input = screen.getByRole('searchbox', { name: 'School name or postcode' });
|
||||
fireEvent.change(input, { target: { value: 'st marys' } });
|
||||
fireEvent.submit(input.closest('form')!);
|
||||
expect(track).toHaveBeenCalledWith('search_submitted',
|
||||
expect.objectContaining({ filters_active: 'faith=roman_catholic', filters_count: 1 }));
|
||||
});
|
||||
|
||||
it('is left out when the API sends no faiths', () => {
|
||||
render(<FilterBar filters={{ ...filters, faiths: [] }} />);
|
||||
expect(within(openSheet()).queryByRole('combobox', { name: 'Faith' })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run them to see them fail**
|
||||
|
||||
Run (from `nextjs-app/`): `npx jest __tests__/components/FilterBarTypeFaith.test.tsx`
|
||||
Expected: failures; School type still lists `Community school`, no `Faith` combobox. (`tsc` also flags `school_type_groups` on `Filters`, which Step 3 fixes.)
|
||||
|
||||
- [ ] **Step 3: Types and page plumbing**
|
||||
|
||||
In `nextjs-app/lib/types.ts`, above `export interface Filters`:
|
||||
|
||||
```ts
|
||||
/** A filter option whose URL value differs from what a parent reads. */
|
||||
export interface FilterOption {
|
||||
value: string;
|
||||
label: string;
|
||||
}
|
||||
```
|
||||
|
||||
and add to `Filters` (after `admissions_policies`):
|
||||
|
||||
```ts
|
||||
/** Parent-facing school type groups; the URL's school_type takes their value. */
|
||||
school_type_groups?: FilterOption[];
|
||||
faiths?: FilterOption[];
|
||||
```
|
||||
|
||||
and to `SchoolSearchParams` (after `has_sixth_form`):
|
||||
|
||||
```ts
|
||||
faith?: string;
|
||||
```
|
||||
|
||||
In `nextjs-app/app/(frontend)/page.tsx`: add `faith?: string;` to the `searchParams` type after `has_sixth_form?: string;`; add `params.faith ||` to `hasSearchParams` after `params.has_sixth_form`; add `faith: params.faith,` to the `fetchSchools({...})` call after `has_sixth_form: params.has_sixth_form,`.
|
||||
|
||||
- [ ] **Step 4: FilterBar**
|
||||
|
||||
In `nextjs-app/components/FilterBar.tsx`:
|
||||
|
||||
1. Add `"faith"` to `FILTER_KEYS` directly after `"school_type"`.
|
||||
2. After `const currentHasSixthForm = ...` add:
|
||||
|
||||
```tsx
|
||||
const currentFaith = searchParams.get("faith") || "";
|
||||
```
|
||||
|
||||
3. Add `currentFaith,` to the `activeDropdownFilters` array (after `currentLA,`). Faith lives behind More filters, so it counts there.
|
||||
4. In `handleSearchSubmit`'s `filters_active` list, after the `type=` line add:
|
||||
|
||||
```tsx
|
||||
currentFaith && `faith=${currentFaith}`,
|
||||
```
|
||||
|
||||
5. Add `currentFaith ||` to `hasActiveFilters` after `currentType ||`.
|
||||
6. Replace `const typeOptions = filters.school_types;` with:
|
||||
|
||||
```tsx
|
||||
// Six groups a parent recognises, not GIAS's 34 establishment types.
|
||||
const typeOptions = filters.school_type_groups ?? [];
|
||||
const faithOptions = filters.faiths ?? [];
|
||||
```
|
||||
|
||||
7. Add `faith: currentFaith,` to the `values` record after `school_type: currentType,`.
|
||||
8. In `labelFor`, before the `// Phase, gender and admissions values…` comment, add:
|
||||
|
||||
```tsx
|
||||
// An old link may carry a raw GIAS type, which reads as itself.
|
||||
const labelled = { school_type: typeOptions, faith: faithOptions }[
|
||||
key as "school_type" | "faith"
|
||||
];
|
||||
if (labelled) return labelled.find((o) => o.value === value)?.label ?? value;
|
||||
```
|
||||
|
||||
9. In `typeSelect`, replace the options map with:
|
||||
|
||||
```tsx
|
||||
{typeOptions.map((o) => (
|
||||
<option key={o.value} value={o.value}>
|
||||
{o.label}
|
||||
</option>
|
||||
))}
|
||||
```
|
||||
|
||||
10. After `laSelect`, add:
|
||||
|
||||
```tsx
|
||||
const faithSelect = (look: Look) => (
|
||||
<SelectShell wide>
|
||||
<select
|
||||
value={currentFaith}
|
||||
onChange={(e) => handleFilterChange("faith", e.target.value)}
|
||||
className={selectClass(look, currentFaith)}
|
||||
aria-label="Faith"
|
||||
disabled={isPending && look !== "sheet"}
|
||||
>
|
||||
<option value="">Any faith or none</option>
|
||||
{faithOptions.map((o) => (
|
||||
<option key={o.value} value={o.value}>
|
||||
{o.label}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</SelectShell>
|
||||
);
|
||||
```
|
||||
|
||||
11. In the desktop row, change `{typeSelect("pill")}` to `{typeOptions.length > 0 && typeSelect("pill")}`.
|
||||
12. In the More filters panel, after `{laSelect("panel")}` add `{faithOptions.length > 0 && faithSelect("panel")}`.
|
||||
13. In the sheet, replace `<SheetField label="School type">{typeSelect("sheet")}</SheetField>` with:
|
||||
|
||||
```tsx
|
||||
{typeOptions.length > 0 && (
|
||||
<SheetField label="School type">{typeSelect("sheet")}</SheetField>
|
||||
)}
|
||||
{faithOptions.length > 0 && (
|
||||
<SheetField label="Faith">{faithSelect("sheet")}</SheetField>
|
||||
)}
|
||||
```
|
||||
|
||||
- [ ] **Step 5: Move the existing tests' fixtures to groups**
|
||||
|
||||
These tests used raw types as values; give each fixture the groups and use keys:
|
||||
|
||||
- `ResultsToolbar.test.tsx`: add to `filters` `school_type_groups: [{ value: 'council', label: 'State school: council-run' }],`; in 'counts only what More filters hides' change `school_type=Community+school` to `school_type=council`.
|
||||
- `FilterSheet.test.tsx`: add the same `school_type_groups` to `filters`; change both `school_type=Community+school` to `school_type=council`.
|
||||
- `FilterSheetPending.test.tsx`: add the same `school_type_groups`; change the `{ target: { value: 'Community school' } }` to `{ target: { value: 'council' } }` and `toBe('Community school')` to `toBe('council')`.
|
||||
- `FilterBarOptions.test.tsx`: add to `filters` `school_type_groups: [{ value: 'academy', label: 'State school: academy or free school' }, { value: 'council', label: 'State school: council-run' }],`; change `school_type=Community+school` (two places) to `school_type=council`; change the expected School type options to `['Any school type', 'State school: academy or free school', 'State school: council-run']`; change `toBe('Community school')` to `toBe('council')`; rename the first test to 'offer every school type group, gender and admissions policy, whatever the results hold'; and in the header comment replace "School type, gender and admissions offer the full lists" with "School type groups, gender and admissions offer the full lists".
|
||||
|
||||
- [ ] **Step 6: Run the frontend checks**
|
||||
|
||||
Run (from `nextjs-app/`): `npx jest` then `npx tsc --noEmit`
|
||||
Expected: all suites pass; no type errors.
|
||||
|
||||
- [ ] **Step 7: Commit**
|
||||
|
||||
```bash
|
||||
git add nextjs-app/lib/types.ts "nextjs-app/app/(frontend)/page.tsx" nextjs-app/components/FilterBar.tsx nextjs-app/__tests__/components/FilterBarTypeFaith.test.tsx nextjs-app/__tests__/components/ResultsToolbar.test.tsx nextjs-app/__tests__/components/FilterSheet.test.tsx nextjs-app/__tests__/components/FilterSheetPending.test.tsx nextjs-app/__tests__/components/FilterBarOptions.test.tsx
|
||||
git commit -m "feat(search): school type groups and a faith filter"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 4: E2E journey, build, PR
|
||||
|
||||
**Files:**
|
||||
- Modify: `e2e/tests/journeys.spec.ts` (insert before `test('a phase outside primary/secondary filters to that phase, not to everything'`)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the "School type" and "Faith" selects (Task 3); `/api/schools?school_type=&faith=` (Task 2).
|
||||
|
||||
- [ ] **Step 1: Add the journey**
|
||||
|
||||
```ts
|
||||
/*
|
||||
* School type offers six groups a parent recognises, and Faith sits beside it.
|
||||
* Data-invariant: asserts what every returned school is, never how many.
|
||||
*/
|
||||
test('school type groups and the faith filter narrow to what they name', async ({ page }) => {
|
||||
await page.goto('/?search=school');
|
||||
const type = page.getByRole('combobox', { name: 'School type', exact: true });
|
||||
await expect(type).toBeVisible({ timeout: 15_000 });
|
||||
await type.selectOption({ label: 'Special school (SEND)' });
|
||||
await expect(page).toHaveURL(/[?&]school_type=special(&|$)/);
|
||||
|
||||
await page.getByRole('button', { name: /^More filters/ }).click();
|
||||
await page.getByRole('combobox', { name: 'Faith', exact: true }).selectOption({ label: 'Roman Catholic' });
|
||||
await expect(page).toHaveURL(/[?&]faith=roman_catholic(&|$)/);
|
||||
|
||||
const res = await page.request.get(
|
||||
'/api/schools?search=school&school_type=special&faith=roman_catholic&page_size=100');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
for (const s of (await res.json()).schools as { school_type: string; religious_denomination: string }[]) {
|
||||
expect(s.school_type).toMatch(/special/i);
|
||||
expect(s.religious_denomination).toMatch(/catholic/i);
|
||||
}
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Confirm it parses, and fails on today's staging**
|
||||
|
||||
Run (from `e2e/`): `npx playwright test --list | grep "school type groups"`
|
||||
Expected: the test is listed.
|
||||
|
||||
Run: `BASE_URL=https://stx.schoolcompare.co.uk npx playwright test -g "school type groups" --retries=0 --reporter=line`
|
||||
Expected: FAIL at `selectOption({ label: 'Special school (SEND)' })`, because staging still offers the raw types. It can only pass after merge (the staging E2E gate runs post-merge).
|
||||
|
||||
- [ ] **Step 3: Build without a database**
|
||||
|
||||
Run (from `nextjs-app/`): `env -u DATABASE_URL npm run build`
|
||||
Expected: build completes.
|
||||
|
||||
- [ ] **Step 4: Commit, push, open the PR**
|
||||
|
||||
```bash
|
||||
git add e2e/tests/journeys.spec.ts
|
||||
git commit -m "test(e2e): school type groups and the faith filter"
|
||||
git push -u origin feat/school-type-groups-faith
|
||||
```
|
||||
|
||||
Open the PR on Gitea against `main` (credential-helper basic auth, as for PRs #168/#169), listing: the six groups and seven faiths, the old-link compatibility, the backend-by-name decision, test counts, and that the new journey can only pass post-merge.
|
||||
@@ -0,0 +1,278 @@
|
||||
# W2: The Location Layer — Design
|
||||
|
||||
Date: 2026-08-21
|
||||
Status: awaiting review
|
||||
Supersedes: workstream W2 in `2026-08-20-seo-programme-design.md`
|
||||
|
||||
Scope note: this covers four page families in one spec. Splitting them — towns
|
||||
and authorities first, outcodes and localities after — was proposed and
|
||||
declined in favour of building the layer in one pass. The decomposition
|
||||
argument was that the curated locality seed needs human review and would hold
|
||||
up 783 pages of measured demand behind it; that risk is accepted here, and the
|
||||
implementation plan should sequence the seed early enough that review time
|
||||
does not become the critical path.
|
||||
|
||||
## Problem
|
||||
|
||||
Location intent is the largest unserved demand the site has. In the 16-month
|
||||
Search Console baseline it draws **874 impressions, one click, average
|
||||
position 49.5**. The site does not compete.
|
||||
|
||||
Unlike named-school queries — which the same baseline showed to be
|
||||
navigational and unwinnable, since a parent typing "audley junior school"
|
||||
wants that school's own website — location queries have no incumbent owner.
|
||||
Nobody owns "primary schools in Brentwood" the way a school owns its name.
|
||||
|
||||
The cause is structural: the site has no page about a place. Every competitor
|
||||
ranking above it does.
|
||||
|
||||
## What the demand actually looks like
|
||||
|
||||
Every location query in the baseline is **town or district level**. Not one is
|
||||
an administrative area:
|
||||
|
||||
| Query | Impressions | Position |
|
||||
|-------|-------------|----------|
|
||||
| colleges in solihull | 112 | 51.2 |
|
||||
| schools in ramsey | 64 | 42.5 |
|
||||
| schools in crosby | 57 | 47.7 |
|
||||
| primary schools in beccles | 44 | 40.9 |
|
||||
| private schools in battersea | 41 | 71.9 |
|
||||
| secondary schools in brentwood | 37 | 56.1 |
|
||||
| secondary schools in canary wharf | 30 | 35.9 |
|
||||
|
||||
Three patterns follow directly, and they drive the whole design.
|
||||
|
||||
**Towns, not authorities.** The superseded W2 put `/schools/[la]` first and
|
||||
towns second. The data inverts that. Brentwood appears four times in different
|
||||
phrasings; Beccles twice. Both are towns, not authorities.
|
||||
|
||||
**Phase is part of the query**, not a filter applied afterwards: "primary
|
||||
schools in beccles", "secondary schools in brentwood", "colleges in solihull".
|
||||
|
||||
**London is searched by district** — Battersea, Canary Wharf — and the GIAS
|
||||
`town` field cannot serve it at all.
|
||||
|
||||
## Measured sizing
|
||||
|
||||
Counted against the live corpus of 25,185 schools, not estimated.
|
||||
|
||||
| Family | Viable (≥5 schools) | Below threshold |
|
||||
|--------|--------------------|-----------------|
|
||||
| Towns | **783** | 907 → redirect to authority |
|
||||
| Outcodes | **1,760** | 305 |
|
||||
| Local authorities | 154 | — |
|
||||
| London localities | ~100–150 (curated) | — |
|
||||
|
||||
With phase variants — 783 town pages plus roughly 700 primary and 250
|
||||
secondary variants, 154 authorities across three variants, 1,760 outcodes and
|
||||
the curated localities — the total lands near **4,000 pages**. Phase variants
|
||||
need their own threshold: there are 17,426 primaries but only 4,456 secondaries nationally,
|
||||
so most towns will support a primary page and not a secondary one.
|
||||
|
||||
## Two design problems this spec exists to solve
|
||||
|
||||
### 1. Town and authority names collide, and neither contains the other
|
||||
|
||||
67 viable towns share a name with a local authority. The obvious fix — let the
|
||||
authority absorb the town, since it sounds like a superset — **does not work**:
|
||||
|
||||
| Place | Schools in the town | Schools in the authority |
|
||||
|-------|--------------------|-----------------------|
|
||||
| Bedford | 104 | 86 |
|
||||
| Birmingham | 520 | 518 |
|
||||
| Derby | 157 | 119 |
|
||||
| Doncaster | 152 | 145 |
|
||||
|
||||
The authority is the larger set in only 43 of the 67. Postal towns cross
|
||||
authority boundaries, so these are overlapping sets that happen to share a
|
||||
name. Publishing both into one namespace produces near-duplicate pages, which
|
||||
is the specific failure that sinks programmatic SEO.
|
||||
|
||||
**Resolution: two namespaces.**
|
||||
|
||||
```
|
||||
/schools/[place] towns and London localities
|
||||
/schools/[place]/primary
|
||||
/schools/[place]/secondary
|
||||
/schools/authority/[la] local authorities
|
||||
/schools/authority/[la]/primary
|
||||
/schools/authority/[la]/secondary
|
||||
/schools/near/[outcode]
|
||||
```
|
||||
|
||||
Outcodes carry no phase variants: nobody searches "primary schools in SW11",
|
||||
so the variants would be pages without demand.
|
||||
|
||||
Every collision disappears by construction. `/schools/[place]` keeps the clean
|
||||
URL for the pattern that carries the demand; authorities get a namespace whose
|
||||
purpose is genuinely different — admissions are authority-run, and the
|
||||
authority page is the one that can speak to catchment policy and LA averages.
|
||||
|
||||
A place page and an authority page of the same name must each say plainly
|
||||
which set of schools they cover, or they read as duplicates to a reader even
|
||||
when they differ in fact.
|
||||
|
||||
### 2. London has no locality field
|
||||
|
||||
`town` collapses **1,819 London schools into the single value "London"**. A
|
||||
page listing all of them is useless, and borough pages do not help because
|
||||
people search "Battersea", not "Wandsworth".
|
||||
|
||||
No single field solves it:
|
||||
|
||||
| Search term | `parliamentary_constituency` | postcodes.io `admin_ward` |
|
||||
|-------------|------------------------------|---------------------------|
|
||||
| Battersea | **Battersea** ✓ | Northcote / Wandsworth Town ✗ |
|
||||
| Canary Wharf | Poplar and Limehouse ✗ | **Canary Wharf** ✓ |
|
||||
| Vauxhall | Vauxhall and Camberwell Green ✗ | **Vauxhall** ✓ |
|
||||
|
||||
And neither covers Clapham, Shoreditch or Peckham, which are postal and
|
||||
colloquial rather than administrative.
|
||||
|
||||
**Resolution: a curated seed mapping locality to outcodes.**
|
||||
|
||||
```
|
||||
pipeline/transform/seeds/locality_outcodes.csv
|
||||
locality_slug,locality_name,outcodes,region
|
||||
battersea,Battersea,"SW11|SW8",London
|
||||
canary-wharf,Canary Wharf,"E14",London
|
||||
clapham,Clapham,"SW4|SW9",London
|
||||
```
|
||||
|
||||
This needs **no new ingestion** — the corpus already has postcodes. It puts
|
||||
the fuzzy, contested part of the problem in a reviewable file rather than in
|
||||
derived logic, which suits it: locality boundaries are a judgement, not a
|
||||
fact. The repo already uses dbt seeds for curated reference data
|
||||
(`la_code_names.csv`, `gias_code_names.csv`), so this follows an established
|
||||
pattern.
|
||||
|
||||
The seed generalises past London. Any colloquial place — Jesmond, Chorlton,
|
||||
Clifton — can be defined by its outcodes without a schema change.
|
||||
|
||||
**Constraint:** a locality slug may not collide with a viable town slug. The
|
||||
place registry enforces this and fails the build rather than silently
|
||||
shadowing a town.
|
||||
|
||||
## Architecture
|
||||
|
||||
### The place registry
|
||||
|
||||
One module owns the question "what places do we publish, and what is in each".
|
||||
Everything else reads from it: the pages, the sitemap, the internal links.
|
||||
|
||||
```
|
||||
backend/places.py
|
||||
|
||||
Place = { kind: "town"|"locality"|"authority"|"outcode",
|
||||
slug, name, urn_list, parent_authority | None }
|
||||
|
||||
build_place_registry(df) -> dict[str, Place]
|
||||
place_schools(slug, phase=None) -> list[School]
|
||||
```
|
||||
|
||||
Built once at startup from the same DataFrame the sitemap uses, and rebuilt by
|
||||
the existing `/api/admin/regenerate-sitemap` path after a pipeline run.
|
||||
Registry construction is where the threshold, the collision rules and the
|
||||
seed's uniqueness constraint are enforced — in one place, testable without a
|
||||
browser or a database.
|
||||
|
||||
### API
|
||||
|
||||
```
|
||||
GET /api/places the registry: slug, kind, name, count
|
||||
GET /api/places/{slug}?phase= aggregate + ranked schools for one place
|
||||
```
|
||||
|
||||
`/api/places` is what the sitemap and the internal-link modules enumerate.
|
||||
|
||||
### Routes
|
||||
|
||||
Next App Router, ISR with the same 7-day revalidate the school pages use.
|
||||
`generateStaticParams` gated behind an env flag, matching
|
||||
`PRERENDER_SCHOOLS`, because 3,900 more routes cannot be statically built in
|
||||
CI on every deploy.
|
||||
|
||||
## What each page must contain
|
||||
|
||||
A place page that is a name substituted into a template is the thing Google's
|
||||
helpful-content stance exists to demote. Each page carries computed local
|
||||
facts that exist nowhere else on the site:
|
||||
|
||||
- **H1** matching the query: "Primary schools in Brentwood"
|
||||
- **Counts framed usefully**: "29 schools, 4 rated Outstanding"
|
||||
- **A ranked table** of the top 20 on the phase's headline metric —
|
||||
`rwm_expected_pct` for primary, `attainment_8_score` for secondary, and for
|
||||
an unphased place page the metric matching whichever phase it holds more of
|
||||
- **The local average against the England average** — the one number a parent
|
||||
cannot get from a list
|
||||
- **Ofsted grade distribution** for the place
|
||||
- **A map**
|
||||
- **Links to neighbouring places** and to the parent authority
|
||||
- **An FAQ block**, feeding `FAQPage` structured data
|
||||
- **A link to every school page in scope** — this is what finally de-orphans
|
||||
the 23,000 school pages the original spec identified as near-orphans
|
||||
|
||||
## Thin-page controls
|
||||
|
||||
Three, and they are the difference between a location layer and index bloat:
|
||||
|
||||
1. **Five schools with current data minimum.** Below it, 301 to the parent
|
||||
authority. This drops 907 towns and 305 outcodes.
|
||||
2. **Per-phase thresholds.** A town with 30 primaries and 2 secondaries
|
||||
publishes a primary page and no secondary page.
|
||||
3. **No page without a local average.** If a place has too few schools with
|
||||
results to compute one, it has nothing to say that a list does not, and it
|
||||
falls back to the authority.
|
||||
|
||||
## Sitemap
|
||||
|
||||
Two new children in the existing index: `/sitemaps/places-{n}.xml` and
|
||||
`/sitemaps/outcodes-{n}.xml`. Per-family children are why the index was built
|
||||
in W1 — Search Console reports coverage per submitted sitemap, so indexation
|
||||
of the location layer is measurable separately from the school pages.
|
||||
|
||||
## Testing
|
||||
|
||||
Per `CLAUDE.md`, user-facing behaviour extends `e2e/tests/journeys.spec.ts` in
|
||||
the same PR.
|
||||
|
||||
**Unit (registry, no DB):** threshold enforcement; a sub-threshold town
|
||||
resolves to its authority; a locality slug colliding with a town fails the
|
||||
build; Bedford's town and authority pages hold different URN sets; per-phase
|
||||
thresholds.
|
||||
|
||||
**Backend:** `/api/places` shape; `/api/places/{slug}` aggregate correctness
|
||||
against a fixture; unknown slug 404s.
|
||||
|
||||
**e2e:** a known town, authority, locality and outcode page each render with
|
||||
the expected count; a below-threshold town 301s; every place page declares a
|
||||
canonical and appears in the sitemap; `/schools/bedford` and
|
||||
`/schools/authority/bedford` both resolve and state which set they cover.
|
||||
|
||||
## Risks
|
||||
|
||||
**Index bloat** is the failure mode of every programmatic SEO programme. The
|
||||
three controls above are the answer, and the per-family sitemap is how we find
|
||||
out early if they were not enough.
|
||||
|
||||
**Helpful-content exposure.** Templated location pages are exactly what
|
||||
Google's stance targets. The mitigation is that every page carries real
|
||||
computed local data — counts, distributions, local-versus-national comparison
|
||||
— rather than a name dropped into boilerplate. If indexation of the places
|
||||
sitemap stalls below roughly half, that is the signal to stop and rethink
|
||||
rather than to add more pages.
|
||||
|
||||
**Build cost.** ~4,000 additional ISR routes on top of 23,000 school pages.
|
||||
The env-flag gate on `generateStaticParams` keeps CI viable.
|
||||
|
||||
**Curation drift.** The locality seed is hand-maintained and will go stale as
|
||||
places change. It is small and reviewable, and a dbt test asserts every seed
|
||||
outcode matches at least one school so a typo fails the pipeline rather than
|
||||
publishing an empty page.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Catchment-area estimation. It is a strong driver for this cluster and
|
||||
`fact_admissions` carries the distances, but it is a modelling problem with
|
||||
real accuracy risk and deserves its own design.
|
||||
@@ -0,0 +1,294 @@
|
||||
# Feature Flags — Design
|
||||
|
||||
**Date:** 2026-08-23
|
||||
**Status:** approved for planning
|
||||
**First consumer:** the last-distance-offered feature (`admission_distance`)
|
||||
|
||||
## Goal
|
||||
|
||||
Let work merge to `main` and deploy to production without becoming visible,
|
||||
so that releasing a feature stops being the same event as deploying it.
|
||||
|
||||
The site has no way to do this today. A feature is either on `main` and live,
|
||||
or it is on a branch. That forces long-lived branches for anything not ready,
|
||||
and it makes every promotion to production an all-or-nothing decision about
|
||||
everything queued behind it.
|
||||
|
||||
This is a **ship-dark** capability, not a kill switch. Flags are expected to
|
||||
flip on the order of once a month, by a person, deliberately. Nothing here is
|
||||
designed for flipping something off in seconds under pressure, and nothing
|
||||
here does percentage rollouts, user targeting or A/B tests — the site has no
|
||||
user identity to target.
|
||||
|
||||
## Decision: Unleash
|
||||
|
||||
Flag state is held in a self-hosted [Unleash](https://www.getunleash.io/)
|
||||
instance (Apache-2.0), not in the repository.
|
||||
|
||||
A lighter option was considered and rejected by the project owner: a typed
|
||||
registry in each runtime with environment-variable overrides set in the
|
||||
Portainer stack files, which would have needed no new container and kept flag
|
||||
state in git. The argument for Unleash is that it provides a UI and an audit
|
||||
log without a deploy, and that flags are expected to become an ongoing
|
||||
operational tool rather than an occasional one.
|
||||
|
||||
Two consequences follow from choosing a service, and this design exists mostly
|
||||
to handle them:
|
||||
|
||||
1. **Flag state lives outside the repository.** `main` is no longer the whole
|
||||
truth about what is switched on. The registry in §2 exists to bound that.
|
||||
2. **A flag can change without a deploy**, so nothing else clears the caches
|
||||
that a deploy would have cleared. §4 establishes how long a flip takes to
|
||||
become visible, and why that is short enough to need no extra mechanism.
|
||||
|
||||
Also considered: Flagsmith (heavier — Django, Postgres and Redis), GrowthBook
|
||||
(requires MongoDB), and Flipt v2 (the closest conceptual fit, git-native, but
|
||||
now under the Fair Core Licence — source-available, not OSI open source).
|
||||
|
||||
## 1. Topology
|
||||
|
||||
A third Portainer stack, `docker-compose.portainer.unleash.yml`, holding
|
||||
`unleashorg/unleash-server` and its own PostgreSQL 16. It is on the macvlan so
|
||||
both application stacks can reach it, and it belongs to neither of them — a
|
||||
staging redeploy must not be able to disturb production's flag state, and vice
|
||||
versa.
|
||||
|
||||
One instance serves both environments. Open-source Unleash ships with
|
||||
`development` and `production` environments and environment-scoped client
|
||||
tokens, so the same flag holds independent state in each: staging's FastAPI
|
||||
carries a `development` token, production's carries a `production` one.
|
||||
|
||||
That property is what makes ship-dark testable. A feature can be **on in
|
||||
staging and off in production** for as long as it takes, which means the `e2e/`
|
||||
journeys exercise it against staging while production stays unchanged.
|
||||
|
||||
## 2. The registry
|
||||
|
||||
Unleash supplies flag *state* and the toggle UI. It does not supply the list of
|
||||
flags. `backend/flags.py` declares every flag the code knows about:
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class Flag:
|
||||
name: str # identical in the registry, in Unleash, and in JSON
|
||||
description: str # one line: what turning this on reveals
|
||||
added: date # for the staleness test in §8
|
||||
```
|
||||
|
||||
**Every flag defaults to `False`.** There is no per-flag default field, because
|
||||
a flag that defaults on is not a ship-dark flag — it is a kill switch, and this
|
||||
design does not offer one. A single unconditional default also means the
|
||||
fallback path has no branching to get wrong.
|
||||
|
||||
Three reasons the registry is not optional:
|
||||
|
||||
- The Unleash SDK evaluates an unknown flag to `False`. Without a registry that
|
||||
is an *undeclared* false — indistinguishable from a typo in a flag name.
|
||||
- `/api/flags` needs a key set to return when Unleash is unreachable. It cannot
|
||||
enumerate flags it has never heard of.
|
||||
- A flag present in the Unleash UI but absent from the registry is orphaned,
|
||||
and should be visibly so rather than quietly authoritative.
|
||||
|
||||
**Naming.** One string, used unchanged as the registry key, the Unleash flag
|
||||
name, and the JSON key in `/api/flags`. It is snake_case, matching the API's
|
||||
existing convention (`admission_distance`, `rwm_expected_pct`) and the mirrored
|
||||
types in `nextjs-app/lib/types.ts`. No case transformation anywhere, so there
|
||||
is no mapping layer to get wrong.
|
||||
|
||||
## 3. Read paths
|
||||
|
||||
### Backend
|
||||
|
||||
`backend/flags.py` wraps `UnleashClient` behind `is_enabled(name: str) -> bool`.
|
||||
|
||||
Fail-closed is the default rather than something added: the Python SDK
|
||||
evaluates every flag to `False` until it has synchronised with the server. An
|
||||
unfinished feature therefore stays hidden when Unleash is unreachable, which is
|
||||
the correct direction for ship-dark.
|
||||
|
||||
The SDK's fcache directory is mounted on a named volume so a container restart
|
||||
during an Unleash outage keeps last-known state rather than reverting a
|
||||
released feature to dark. The registry default remains `False`, so the worst
|
||||
case is a feature disappearing, never one appearing.
|
||||
|
||||
### Frontend
|
||||
|
||||
`nextjs-app/lib/flags.ts` exposes `getFlags(): Promise<Flags>`, a single
|
||||
server-side fetch of `/api/flags` returning a typed record. Server components
|
||||
only — no flag value reaches the browser bundle, and `package.json` gains no
|
||||
Unleash dependency. The Unleash client library stays entirely inside the
|
||||
service that already owns every other piece of data the frontend renders.
|
||||
|
||||
The cost, named plainly: a purely front-end flag must still be declared in a
|
||||
Python file. It is a flat data edit rather than programming, and the return is
|
||||
one list, so nobody has to ask which service knows about a given flag.
|
||||
|
||||
### `/api/flags` must not be publicly reachable
|
||||
|
||||
`nextjs-app/app/api/[...path]/route.ts` proxies **everything** under `/api/` to
|
||||
FastAPI. Left alone, `https://www.schoolcompare.co.uk/api/flags` would return
|
||||
`{"admission_distance": false, ...}` — publishing the name and state of every
|
||||
unreleased feature, which defeats the purpose of shipping dark.
|
||||
|
||||
The proxy therefore gains a denylist, and `flags` is on it: a request for a
|
||||
denied path returns 404 rather than being forwarded. Next's own `getFlags()` is
|
||||
unaffected because it calls `FASTAPI_URL` directly across the Docker network
|
||||
and never transits the public proxy.
|
||||
|
||||
This is a general hole rather than a flags-specific one — the proxy will
|
||||
forward any future internal endpoint too — so the denylist is written as a
|
||||
named constant with a comment saying what belongs on it.
|
||||
|
||||
## 4. Propagation
|
||||
|
||||
**Time-based revalidation is sufficient. There is no webhook.**
|
||||
|
||||
An earlier draft of this section specified two Unleash webhooks and a
|
||||
`revalidateTag('flags')` purge, on the premise that pages cache for seven days.
|
||||
That premise was wrong, and checking it removed the most complex part of the
|
||||
design.
|
||||
|
||||
Next uses the **lowest** `revalidate` among a route's fetches to set the
|
||||
revalidation frequency of the whole route — the segment-level
|
||||
`export const revalidate` does not override a lower value inside it. Measured
|
||||
against this codebase:
|
||||
|
||||
| Page family | Segment | Lowest fetch | Effective |
|
||||
|---|---|---|---|
|
||||
| `/school/[slug]` | 604800 | `fetchSchoolDetails` at 300 | **5 minutes** |
|
||||
| `/schools/*` | 604800 | `fetchNationalAverages` at 3600 | **1 hour** |
|
||||
|
||||
The Unleash SDK polls every 15 seconds, so a flip reaches school pages within
|
||||
about five minutes and place pages within the hour, unaided. Flags flip
|
||||
monthly, by hand, deliberately. That is fast enough.
|
||||
|
||||
What this removes: two webhook integrations, a `/api/revalidate-flags` route, a
|
||||
shared-secret-in-a-query-string scheme, an idempotency requirement against
|
||||
duplicate and out-of-order delivery, and a rule that every fetch in
|
||||
`nextjs-app/lib/` carry a cache tag. None of it has to be built, maintained, or
|
||||
kept correct as new fetches are added.
|
||||
|
||||
**If instant flips are ever wanted**, the webhook is the way to add them, and it
|
||||
is purely additive — nothing in this design has to change first.
|
||||
|
||||
### Two constraints this leaves behind
|
||||
|
||||
**Never flag content on a `force-static` page.** `app/admissions/page.tsx`
|
||||
declares `export const dynamic = 'force-static'`, so it is baked at build time
|
||||
and never revalidates. A flag gating anything on such a page would not take
|
||||
effect until the next deploy, silently. If a flag ever needs to reach one, that
|
||||
page must first move to ISR.
|
||||
|
||||
**A route-family flag still needs the sitemap rebuilt.** The sitemap is held in
|
||||
memory and rebuilt only at startup or via `POST /api/admin/regenerate-sitemap`.
|
||||
No flag in scope touches the sitemap (§6), so this is deferred with the route
|
||||
case rather than solved now — but a route flag must not ship without it, or the
|
||||
sitemap will advertise URLs that `notFound()`.
|
||||
|
||||
## 5. What "off" means, per surface
|
||||
|
||||
| Surface | Off |
|
||||
|---|---|
|
||||
| Route | `notFound()`, **and** absent from the sitemap, **and** absent from nav |
|
||||
| UI element | Not rendered; surrounding page byte-identical to today |
|
||||
| API field | Key **absent**, not `null` |
|
||||
| API endpoint | 404, not 403 |
|
||||
|
||||
The three parts of the route rule move together or not at all. Submitting URLs
|
||||
to Google that return 404 is the bug fixed in PR #124, and a flag is a new way
|
||||
to reintroduce it.
|
||||
|
||||
An API field is withheld **at the source**, never rendered-but-hidden. The
|
||||
precedent is already set in this codebase by commit `c9a1892`: `/api/schools/`
|
||||
is public and unauthenticated, so leaving a withheld field in the payload hands
|
||||
the record to anyone who opens the network tab.
|
||||
|
||||
## 6. First consumer: `admission_distance`
|
||||
|
||||
The last-distance-offered feature is merged to `main` and live on staging.
|
||||
Production has never received it: `/api/schools/100010` on production carries
|
||||
no `admission_distance` key, and no Distance section renders.
|
||||
|
||||
It needs **exactly one gate** — `backend/app.py:809`, where the field is
|
||||
attached to the school payload:
|
||||
|
||||
```python
|
||||
"admission_distance": (
|
||||
supplementary.get("admission_distance")
|
||||
if flags.is_enabled("admission_distance") else None
|
||||
),
|
||||
```
|
||||
|
||||
The frontend follows with no change. `DistanceSection` already returns `null`
|
||||
when `admission_distance?.distance_m == null`, and `PrimarySchoolSections`
|
||||
already conditions the admissions block on `(admissions || admissionDistance)`.
|
||||
The off-state is the commonest state on the site — only 57 local authorities
|
||||
publish cut-off distances at all — so it is well covered by construction.
|
||||
|
||||
The flag does not touch the sitemap: school pages exist either way.
|
||||
|
||||
Intended lifecycle: default off, so production receives the code dark on the
|
||||
next promotion; on in the `development` environment so staging keeps testing
|
||||
it; flipped on in `production` when the owner chooses.
|
||||
|
||||
**This flag exercises two of the three surfaces** in §5 — API field and UI
|
||||
element. No route case ships with it. The route rule is specified but unproven
|
||||
until a route-shaped flag exists, and should be treated as such.
|
||||
|
||||
## 7. Testing
|
||||
|
||||
**Backend unit.** The registry is well-formed; an unknown flag evaluates
|
||||
`False`; `/api/flags` returns every declared flag with its default when the
|
||||
SDK is unreachable; `admission_distance` is absent from the school payload when
|
||||
the flag is off and present when on.
|
||||
|
||||
**Frontend unit.** `getFlags()` returns declared defaults when `/api/flags`
|
||||
fails, rather than throwing and taking the page with it.
|
||||
|
||||
**E2E.** Journeys read `/api/flags` and gate flag-dependent assertions on it,
|
||||
matching the `test.skip` shape the suite already uses.
|
||||
|
||||
One trap to avoid, worth stating because the existing distance journeys walk
|
||||
straight into it: they already skip when no school has a published figure, so
|
||||
with the flag off they would skip silently and the suite would go green. The
|
||||
gate must be explicit — **if `/api/flags` reports `admission_distance` on, then
|
||||
a school with a cut-off must be found**, converting a silent skip into a real
|
||||
assertion.
|
||||
|
||||
## 8. Lifecycle
|
||||
|
||||
A flag is temporary scaffolding, and the failure mode of every flag system is
|
||||
accumulation.
|
||||
|
||||
The registry records the date each flag was added, and a backend test fails any
|
||||
flag older than **90 days**. Removing a flag means deleting the registry entry,
|
||||
the branches that read it, and the flag in the Unleash UI.
|
||||
|
||||
Unleash SDK usage metrics stay enabled, so the UI shows which flags are still
|
||||
being evaluated — the evidence needed to retire one safely.
|
||||
|
||||
## 9. Risks
|
||||
|
||||
**Production gains a homelab dependency.** If Unleash is unreachable when a
|
||||
production container cold-starts with an empty cache, every flag evaluates
|
||||
`False` and any feature currently switched on disappears. The fcache volume
|
||||
covers restarts; the 90-day lifecycle rule bounds how long any feature is
|
||||
exposed to this. It is a real regression risk and the reason flags must be
|
||||
retired rather than left on indefinitely.
|
||||
|
||||
**Flag state is not in git.** `main` no longer tells you what production is
|
||||
showing. The registry lists what *can* be flagged; only the Unleash UI says
|
||||
what *is*. This is inherent to the choice of a service.
|
||||
|
||||
**A large promotion backlog exists.** Production is running the
|
||||
pre-SEO-programme build — no place pages, and a sitemap still declaring the
|
||||
apex host. The first promotion after this work ships that entire backlog. The
|
||||
flag isolates the distance feature from it and nothing else.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Percentage rollouts, user targeting, A/B testing, and Unleash strategies
|
||||
beyond simple on/off. Flags are booleans.
|
||||
- Pipeline and dbt flags. Airflow and dbt are not flag consumers.
|
||||
- Client-side flag evaluation. Flags are server-side only.
|
||||
- Automatic flag removal. The staleness test reports; a person deletes.
|
||||
@@ -0,0 +1,294 @@
|
||||
# School Autosuggest — Design
|
||||
|
||||
**Date:** 2026-08-26
|
||||
**Status:** approved for planning
|
||||
**Depends on:** the feature-flag layer (PR #125, merged)
|
||||
|
||||
## Goal
|
||||
|
||||
Suggest schools by name as someone types in the site's main search box, so a
|
||||
parent who knows the school they want reaches it in one step instead of
|
||||
searching, scanning a result list, and clicking.
|
||||
|
||||
Scope is **schools only**. Places and postcodes were considered and excluded —
|
||||
see *Out of scope*.
|
||||
|
||||
## The finding that shapes everything
|
||||
|
||||
The site's rate limiter does not do what it looks like it does.
|
||||
|
||||
`limiter = Limiter(key_func=get_remote_address)` with `60/minute` reads
|
||||
`request.client.host`. In staging and production the backend has no published
|
||||
ports and sits on the internal `backend` network, so its only caller is the
|
||||
Next proxy — and `request.client.host` is therefore **the Next container**, for
|
||||
every browser user on the site.
|
||||
|
||||
Measured against staging: 70 concurrent requests to `/api/schools` returned
|
||||
**60 × 200 and 10 × 429**. One machine consumed the whole site's budget for
|
||||
that minute.
|
||||
|
||||
Autosuggest is the worst possible feature to build on that. One person typing
|
||||
"st marys primary" produces six to eight debounced requests; **eight concurrent
|
||||
searchers would 429 the site.** The compare modal's search-as-you-type already
|
||||
shares this bucket, so the exposure exists today — autosuggest makes it
|
||||
certain.
|
||||
|
||||
Fixing the keying is therefore part of this work, not a follow-up.
|
||||
|
||||
## 1. Rate-limit keying
|
||||
|
||||
Both environments sit behind Cloudflare (`server: cloudflare`, `cf-ray` present
|
||||
on staging and production). Cloudflare sets `CF-Connecting-IP` on every request
|
||||
to the origin and **overwrites any client-supplied value**, which makes it
|
||||
trustworthy in a way a parsed `X-Forwarded-For` chain is not.
|
||||
|
||||
```python
|
||||
def client_key(request: Request) -> str:
|
||||
"""Rate-limit bucket: the real caller, not the proxy in front of them."""
|
||||
cf = request.headers.get("cf-connecting-ip")
|
||||
if cf:
|
||||
return cf.strip()
|
||||
xff = request.headers.get("x-forwarded-for")
|
||||
if xff:
|
||||
return xff.split(",")[0].strip()
|
||||
return get_remote_address(request)
|
||||
```
|
||||
|
||||
`nextjs-app/app/api/[...path]/route.ts` already forwards every inbound header
|
||||
except `host` and `connection`, so `CF-Connecting-IP` reaches the backend with
|
||||
no proxy change.
|
||||
|
||||
**This header is trustworthy only for traffic that actually passed through
|
||||
Cloudflare, and nothing in the application can verify that it did.** An earlier
|
||||
draft of this section claimed Cloudflare "replaces the header, so a browser
|
||||
cannot forge it", and that only the `X-Forwarded-For` fallback was forgeable.
|
||||
That was wrong. Cloudflare does overwrite the header *on requests it handles* —
|
||||
but a caller reaching the origin directly sets whatever it likes, and this
|
||||
process cannot distinguish an edge-set header from an attacker-set one. Both
|
||||
headers are equally forgeable in that scenario.
|
||||
|
||||
The consequence is sharper than a weakened defence. An attacker rotating
|
||||
`CF-Connecting-IP` per request mints a fresh rate-limit bucket every time and
|
||||
evades per-client limits entirely — including on the DataFrame-heavy
|
||||
`/api/schools`. Against abuse that is *worse* than the shared bucket it
|
||||
replaced, which at least capped everyone at 60/minute together.
|
||||
|
||||
Two mitigations, and they are not interchangeable:
|
||||
|
||||
1. **The real fix is at Cloudflare** — Authenticated Origin Pulls, or an origin
|
||||
firewall that refuses connections not from Cloudflare's ranges. Only the
|
||||
edge can vouch for its own header. This is infrastructure work and is not
|
||||
part of this change; it is the thing that makes the header mean anything.
|
||||
2. **The ceiling in §1.1 bounds what evading the keying can achieve** while
|
||||
that remains open. It does not make the header trustworthy — it makes
|
||||
trusting it survivable.
|
||||
|
||||
The backend being unreachable from outside the Docker network is a real second
|
||||
layer, but it depends on the ingress path in front of the frontend, which this
|
||||
design does not control and should not assume.
|
||||
|
||||
### 1.1 The ceiling, which is back
|
||||
|
||||
The shared bucket was acting as an accidental global throttle on a
|
||||
single-process uvicorn backend that filters a 25,000-row DataFrame in-process.
|
||||
Correct per-user keying removes it: the origin becomes reachable at 60/min *per
|
||||
user* rather than 60/min in total, and — per above — at an unbounded rate by
|
||||
anyone willing to rotate a header.
|
||||
|
||||
An earlier draft dropped the in-app ceiling, arguing it belonged at Cloudflare.
|
||||
That argument assumed the keying was sound. It is not, so the ceiling is
|
||||
load-bearing rather than redundant, and it ships here:
|
||||
|
||||
`GlobalRateLimitMiddleware` counts all `/api/` requests in a fixed 60-second
|
||||
window against `global_rate_limit_per_minute` (3000), independent of any client
|
||||
identity, and refuses with a 429 that names capacity rather than the client —
|
||||
an operator has to be able to tell "one noisy client" from "the origin is
|
||||
saturated". It is registered last so it is outermost: a ceiling that applies
|
||||
after the expensive work has run is not a ceiling.
|
||||
|
||||
slowapi cannot express this. `default_limits` and `application_limits` are both
|
||||
evaluated with the same `key_func`, making them per-client rather than global,
|
||||
and `application_limits` only apply with `SlowAPIMiddleware` installed, which
|
||||
this app does not use. Hence the explicit middleware — about thirty lines, and
|
||||
obviously correct, which is what a backstop needs to be.
|
||||
|
||||
Requests from `127.0.0.1` are exempt. The container healthcheck runs
|
||||
`curl http://localhost:80/api/data-info` from inside the container, and
|
||||
starving it would fail the check, restart the container, and turn a load spike
|
||||
into an outage loop. The exemption keys on the peer address, never the `Host`
|
||||
header, which the caller sets.
|
||||
|
||||
3000/minute is an estimate, not a measurement, and worth revisiting against
|
||||
real traffic.
|
||||
|
||||
### Per-user limits
|
||||
|
||||
Per-user fairness and origin protection are different jobs, and this design now
|
||||
does both separately: the ceiling above for the origin, and per-route limits
|
||||
for fairness. Conflating them is what produced the original behaviour, where
|
||||
one bucket served the whole internet.
|
||||
|
||||
The existing 60/minute default is unchanged, and `/api/suggest` gets
|
||||
120/minute. Both are estimates rather than measurements, and are a starting
|
||||
point to revisit once the keying is correct enough for real per-user traffic to
|
||||
be visible — which it was not before, because everyone shared one bucket.
|
||||
|
||||
## 2. `GET /api/suggest`
|
||||
|
||||
A dedicated endpoint, not a mode of `/api/schools`.
|
||||
|
||||
The existing search path calls Typesense for URNs and then filters, ranks and
|
||||
sorts the full in-memory DataFrame — a pandas pass per keystroke, holding the
|
||||
GIL and blocking other requests in the same worker. Suggestions need none of
|
||||
it: `urn`, `school_name`, `phase`, `school_type`, `local_authority`,
|
||||
`postcode` and `ofsted_rating` are all already in the Typesense document
|
||||
(`pipeline/scripts/sync_typesense.py`).
|
||||
|
||||
```
|
||||
GET /api/suggest?q=<query>&limit=8
|
||||
→ 200 {"suggestions": [
|
||||
{"urn": 100010, "school_name": "Brecknock Primary School",
|
||||
"local_authority": "Camden", "postcode": "NW1 1AA",
|
||||
"phase": "Primary", "school_type": "Community school"}
|
||||
]}
|
||||
```
|
||||
|
||||
- **Under two characters** returns `{"suggestions": []}` with 200. The
|
||||
keystroke path never returns an error for ordinary input.
|
||||
- **Typesense unavailable** returns `{"suggestions": []}` with 200. There is
|
||||
deliberately **no DataFrame fallback**: the substring scan `/api/schools`
|
||||
falls back to is precisely the cost this endpoint exists to avoid, and a
|
||||
silent 25,000-row scan per keystroke is worse than no suggestions.
|
||||
- **`limit` is clamped** to 20. It is a public endpoint.
|
||||
- **Rate limit `120/minute`** per client, not the default 60. A 200 ms
|
||||
debounce tops out near 5 requests/second while someone is actively typing,
|
||||
but averages far below that across a real search; 120 leaves headroom for
|
||||
bursts while still bounding one client.
|
||||
- **Local authority is part of the payload, not decoration.** There are many
|
||||
schools called "St Mary's"; a suggestion list without the authority is
|
||||
unusable for exactly the queries autosuggest is meant to serve.
|
||||
|
||||
### Caching
|
||||
|
||||
`CACHE_RULES` gains `("/api/suggest", (60, 3600, 86400))`. Prefix queries
|
||||
repeat enormously across users and school names change once a year.
|
||||
|
||||
The client fetch must **not** use `cache: "no-store"`. The compare modal does,
|
||||
and copying that pattern would throw away both the browser cache and the ETag
|
||||
304s the existing `CacheAndETagMiddleware` already provides.
|
||||
|
||||
Both environments currently report `cf-cache-status: DYNAMIC` — Cloudflare
|
||||
ignores the `Cache-Control` the API already sends, because it does not cache
|
||||
dynamic paths by default. **A Cloudflare Cache Rule for `/api/suggest*` would
|
||||
let the edge absorb most of this traffic and never reach the origin.** That is
|
||||
a dashboard change, it is optional, and nothing here depends on it.
|
||||
|
||||
## 3. The combobox
|
||||
|
||||
This is an ARIA combobox, not a text input with a list underneath.
|
||||
|
||||
**Files.** `FilterBar.tsx` is already long. The work splits three ways:
|
||||
`hooks/useSchoolSuggest.ts` owns fetching, debouncing and cancellation;
|
||||
`components/SuggestList.tsx` owns rendering and ARIA; `FilterBar.tsx` wires
|
||||
them to the existing input and form.
|
||||
|
||||
**Fetching.** 200 ms debounce; minimum two characters; an `AbortController`
|
||||
cancels the superseded request on every keystroke. Cancellation is not an
|
||||
optimisation — without it, a slow response for `"st"` can land after the fast
|
||||
one for `"st marys"` and replace a correct list with a stale one.
|
||||
|
||||
**Suppressed during postcode entry.** The box takes a school name *or* a
|
||||
postcode, and `isValidPostcode` already distinguishes them. Suggestions do not
|
||||
appear once the value parses as a postcode.
|
||||
|
||||
**Keyboard.** `ArrowDown`/`ArrowUp` move the active option, `Escape` closes and
|
||||
keeps the typed text, `Tab` closes. `Enter` **with an option active** navigates
|
||||
to that school's page. `Enter` **with none active** submits the free-text
|
||||
search exactly as it does today — the existing behaviour is preserved, not
|
||||
replaced.
|
||||
|
||||
**ARIA.** `role="combobox"` with `aria-expanded` and `aria-controls` on the
|
||||
input, `aria-activedescendant` pointing at the active option, `role="listbox"`
|
||||
on the list and `role="option"` on each row.
|
||||
|
||||
**Both instances get it.** `HomeView` renders `FilterBar` twice — hero and
|
||||
sticky — from one component, so there is one implementation.
|
||||
|
||||
## 4. Behind a flag
|
||||
|
||||
Flag `school_autosuggest`, declared in `backend/flags.py`, default off.
|
||||
|
||||
This is the most-used control on the site and the first change to it in a
|
||||
while. `app/page.tsx` is an async server component, so it reads the flag and
|
||||
threads it to `FilterBar` through `HomeView` — two prop hops, explicit, no
|
||||
client-side flag read.
|
||||
|
||||
Off means the input behaves exactly as it does today: no listener, no fetch, no
|
||||
markup. Not a rendered-then-hidden dropdown.
|
||||
|
||||
The rate-limit keying is **not** flagged. It is a correctness fix that should
|
||||
apply whether or not autosuggest is on, and flagging it would mean shipping a
|
||||
known-wrong limiter into production deliberately.
|
||||
|
||||
## 5. Analytics
|
||||
|
||||
`search_submitted` already carries `via: 'input'`. Selecting a suggestion fires
|
||||
it with `via: 'suggestion'` plus the chosen `urn`, so the obvious question —
|
||||
does this actually help, or do people ignore it — has an answer in the data
|
||||
rather than an opinion.
|
||||
|
||||
## 6. Testing
|
||||
|
||||
**Backend.** `client_key` prefers `CF-Connecting-IP`, falls back through
|
||||
`X-Forwarded-For` to the remote address, and two different values get two
|
||||
different buckets. `/api/suggest` returns matches, returns empty below two
|
||||
characters, returns empty and 200 when Typesense is unavailable, and clamps
|
||||
`limit`. That it never touches the DataFrame is asserted by making
|
||||
`load_school_data` raise and requiring the endpoint to answer anyway.
|
||||
|
||||
**Frontend.** The hook debounces, aborts superseded requests, and drops a
|
||||
late-arriving response for a stale query. The list renders the ARIA
|
||||
attributes. Keyboard navigation moves the active option; `Enter` on an option
|
||||
navigates; `Enter` on none submits the search.
|
||||
|
||||
**E2E.** With the flag on, typing a known school name shows it and selecting it
|
||||
lands on that school's page. With the flag off, no combobox markup exists.
|
||||
Gated on the flag the same way the distance journeys are — read the observable
|
||||
effect, since `/api/flags` is denied to the public.
|
||||
|
||||
## 7. Risks
|
||||
|
||||
**Removing the accidental throttle.** Covered in §1. Correct per-user keying
|
||||
means the origin is reachable at 60/minute *per user* where it was 60/minute
|
||||
in total, and no in-app global cap replaces it — that job goes to Cloudflare,
|
||||
which is not done as part of this change. Until it is, a determined caller
|
||||
with many source addresses can put more load on a single-process origin than
|
||||
they can today. Against this site's traffic that is a theoretical risk rather
|
||||
than a live one, but it is a real one and it is the price of the fix.
|
||||
|
||||
**Cloudflare bypass — the open one.** If the origin is reachable without
|
||||
passing through Cloudflare, `CF-Connecting-IP` is attacker-controlled, and
|
||||
rotating it per request defeats per-client limits on every endpoint. The
|
||||
ceiling in §1.1 bounds the damage to the origin's total capacity; it does not
|
||||
restore per-client fairness under attack, and it cannot. Closing this properly
|
||||
means Authenticated Origin Pulls or an origin firewall restricted to
|
||||
Cloudflare's published ranges — infrastructure work, outside this change, and
|
||||
the single most valuable follow-up here.
|
||||
|
||||
**Typesense becomes user-visible.** Today a Typesense outage degrades search to
|
||||
a slow substring match. With autosuggest it also means the dropdown silently
|
||||
stops appearing. That is the correct failure — quiet, not broken — but it makes
|
||||
Typesense health worth monitoring in a way it was not before.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- **Place suggestions.** The 2,646 town, authority and outcode pages are a
|
||||
strong candidate and would route people onto the pages W2 built, but they
|
||||
live in the place registry rather than Typesense, so it is a second index and
|
||||
a ranking rule for comparing two kinds of result. Worth its own change.
|
||||
- **Postcode completion.** Would put postcodes.io in the keystroke path, with
|
||||
its own latency and rate limits.
|
||||
- **The compare modal.** It already has search-as-you-type. Converting it to
|
||||
this component is a reasonable follow-up, not part of this.
|
||||
- **Recent or popular searches.** No storage for either, and no evidence yet
|
||||
that they are wanted.
|
||||
@@ -0,0 +1,399 @@
|
||||
# Destination Measures — Design
|
||||
|
||||
**Date:** 2026-08-28
|
||||
**Status:** awaiting review
|
||||
**Scope:** secondary school detail pages only
|
||||
|
||||
## Goal
|
||||
|
||||
Say what happened to a school's leavers after they left. Two sections on the
|
||||
secondary template:
|
||||
|
||||
- **After Year 11** — every secondary, from the KS4 destination measures
|
||||
- **After the sixth form** — sixth-form schools only, from the 16-18 measures
|
||||
|
||||
This replaces the "Post-16 destination data coming soon" placeholder standing in
|
||||
`nextjs-app/components/school/SecondaryAdmissionsSection.tsx:117` since the exam
|
||||
phase taxonomy work, and fills the `ks5_destinations_pct` slot specified but
|
||||
never built in `2026-07-07-exam-phase-taxonomy-design.md:201`.
|
||||
|
||||
Mockup, with all three data states live:
|
||||
<https://claude.ai/code/artifact/5be149d6-252f-473c-9a4f-4c36b05161b0>
|
||||
|
||||
## The finding that shapes everything
|
||||
|
||||
**Suppression is per cell, and the cells sum to the cohort.**
|
||||
|
||||
DfE withholds a figure it considers disclosive by writing `c`. It does this at
|
||||
the level of an individual destination category, not the whole school, and it
|
||||
publishes the cohort total alongside. The categories form a clean partition. So
|
||||
where exactly one category is suppressed, subtracting the published ones from the
|
||||
cohort recovers it exactly.
|
||||
|
||||
Verified against three real schools in the 2022/23 file:
|
||||
|
||||
| School | URN | Withheld | Recovers to |
|
||||
|---|---|---|---|
|
||||
| North East Futures UTC | 145900 | School sixth form | **3 pupils** |
|
||||
| Whitley Bay High School | 108638 | Further education | **18 pupils** |
|
||||
| St Matthew's RC High School | 148389 | School sixth form | **4 pupils** |
|
||||
|
||||
Those are the precise numbers the `c` exists to hide, and in a random 400-school
|
||||
sample **22% of mainstream secondaries** have exactly one suppressed category in
|
||||
their disadvantaged group. This is the normal case, not an edge case.
|
||||
|
||||
Three rules follow, and everything else in this document is downstream of them.
|
||||
|
||||
**R1 — Never *publish* enough to derive a remainder.**
|
||||
|
||||
An earlier draft of this rule said "never *render* a derived remainder", and
|
||||
that was the defect code review caught in PR #137. Not drawing a number does
|
||||
nothing to stop it being computed: `GET /api/schools/{urn}` is public and
|
||||
unauthenticated, so anything in the payload is published whatever the UI
|
||||
chooses to draw. The rendering guards shipped; the payload still carried the
|
||||
cohort and every published category, and `cohort - sum(published)` returned
|
||||
Whitley Bay's withheld figure exactly.
|
||||
|
||||
The rule is therefore about the serialiser, and the UI guards are a second line
|
||||
of defence behind it. Two identities have to be closed:
|
||||
|
||||
- within a pupil group the categories sum to the cohort, so a group with
|
||||
exactly **one** suppressed category gives it away;
|
||||
- across groups, disadvantaged + other = all for every category, so a category
|
||||
suppressed in exactly **one** of the three gives itself away.
|
||||
|
||||
`_mask_for_disclosure` applies DfE's own answer — secondary suppression —
|
||||
withholding a companion cell until every row and every column hides either none
|
||||
or at least two. It iterates, because each new suppression can break the other
|
||||
identity, and terminates because cells are only ever added.
|
||||
|
||||
The companion must carry pupils. Suppressing a zero looks like secondary
|
||||
suppression and protects nothing: the residual still equals the original
|
||||
withheld figure.
|
||||
|
||||
Where no companion can do the job — a sparse cohort whose every other category
|
||||
is `not_applicable`, routine in special schools and alternative provision — the
|
||||
pupil group is **dropped from the payload entirely**. A first version simply
|
||||
returned at that point with the violation intact and no signal, which review
|
||||
caught: a disclosure-control pass that fails silently is worse than none,
|
||||
because everything downstream trusts it. The function now cannot terminate
|
||||
except in a state where `disclosure_invariant_holds()` is true, and an
|
||||
exhaustive test sweeps all 81 suppression patterns of a four-category group to
|
||||
prove it.
|
||||
|
||||
Measured cost on the 400-school sample: the all-pupils bar survives on **94%**
|
||||
of mainstream secondaries rather than 100%. That is the price of not
|
||||
republishing what DfE withheld.
|
||||
|
||||
**R2 — Never aggregate across a suppression boundary.** Summing published
|
||||
components to fill a gap is R1 with extra steps.
|
||||
|
||||
DfE's own aggregates (`Sustained education destination`, `Sustained education,
|
||||
employment & apprenticeships`) are ingested but **not served**. An aggregate
|
||||
spanning exactly one suppressed component names it, and nothing renders them
|
||||
today — an unused field that leaks is not a trade-off worth carrying. They can
|
||||
be re-added with their own guard if the fallback ladder is ever built.
|
||||
|
||||
**R3 — The three pupil groups are one disclosure surface, not three.**
|
||||
Disadvantaged and Not-known-to-be-disadvantaged partition All pupils, so
|
||||
rendering any *two* of them recovers the third. Where a category is suppressed in
|
||||
the disadvantaged group, it must therefore also be withheld from **all other
|
||||
pupils** — the all-pupils view is the primary one and keeps it.
|
||||
|
||||
This costs almost nothing, because DfE already applies the same masking: across
|
||||
the sample, 493 of 498 suppressed disadvantaged cells were suppressed in the
|
||||
other group too. The mart enforces the remaining 5, which fell on 2 schools of
|
||||
262. **The all-pupils bar is unaffected** — masking the whole page wherever the
|
||||
disadvantaged group is thin would remove the bar from 80% of schools, and is not
|
||||
what this rule says.
|
||||
|
||||
R1 and R2 both hold within a group and still leak across the switch, which is why
|
||||
R3 is stated separately.
|
||||
|
||||
### The convention that would break this quietly
|
||||
|
||||
`macros/safe_numeric.sql` coerces every EES sentinel — `z`, `c`, `x`, `q`, `u` —
|
||||
to `NULL`, deliberately and correctly for attainment, where "suppressed" and "no
|
||||
data" are equally unrenderable. Here they are not the same thing: one must print
|
||||
*withheld*, the other must print nothing at all, and the difference is what keeps
|
||||
R1 enforceable.
|
||||
|
||||
**`safe_numeric` must not be used on destination counts.** The staging model
|
||||
keeps the sentinel in a companion status column. This is the single most likely
|
||||
way for this feature to regress into a disclosure, so it gets its own dbt test.
|
||||
|
||||
## What is actually available
|
||||
|
||||
Measured against the EES public API (open, no key). Both datasets carry
|
||||
`geographicLevel: School` with `urn` on every location option, so the join to
|
||||
`dim_school` is direct.
|
||||
|
||||
| | KS4 | 16-18 |
|
||||
|---|---|---|
|
||||
| Dataset id | `019d4f41-22d1-71b2-a1a7-f3b91026815b` | `019d4e73-6440-7523-b60c-bfab1ad4a30d` |
|
||||
| Rows | 1,871,739 | 3,862,658 |
|
||||
| Institutions | 4,946 | 3,065 |
|
||||
| Time periods | 2009/10–2022/23 | 2016/17–2022/23 |
|
||||
|
||||
**Destination categories (KS4).** School sixth form · Sixth form college ·
|
||||
Further education · Other education destination · Sustained apprenticeships (with
|
||||
level breakdown) · Sustained employment destination · Not recorded as a sustained
|
||||
destination · Activity not captured. Plus the aggregates `Sustained education
|
||||
destination` and `Sustained education, employment & apprenticeships`.
|
||||
|
||||
**16-18 adds** UK higher education institution and FE split by level, which is
|
||||
what makes the post-16 section worth having.
|
||||
|
||||
**Breakdowns.** `Disadvantage Status` gives Disadvantaged / Not known to be
|
||||
disadvantaged / Total — exactly the three-way switch. Sex, ethnicity, FSM status,
|
||||
prior attainment and SEN provision also travel in the same table; we ingest none
|
||||
of them.
|
||||
|
||||
**Indicators.** Both counts and percentages, plus the cohort size. Bar widths use
|
||||
the counts — the published percentages do not sum to 100.
|
||||
|
||||
### Coverage, and what degrades
|
||||
|
||||
Random 400-school sample, 2022/23, mainstream secondaries (n=262):
|
||||
|
||||
| View | As published by DfE | After R1–R3 masking | Consequence |
|
||||
|---|---|---|---|
|
||||
| All pupils, all categories | 100% | **94%** | Bar works nearly everywhere |
|
||||
| Disadvantaged, headline rate | 95% | 95% | Gap panel works |
|
||||
| Disadvantaged, three grouped cards | 68% | 68% | Degrades card by card |
|
||||
| Disadvantaged, all six categories | 20% | **20%** | Bar unusable for this group |
|
||||
|
||||
The middle column is what the site actually serves. Masking costs the
|
||||
all-pupils bar on 6% of mainstream secondaries — those are schools where a
|
||||
category was suppressed in exactly one pupil group and no non-zero companion
|
||||
existed below the all-pupils row.
|
||||
|
||||
Special schools and alternative provision are far worse: 13% and 41% respectively
|
||||
have the whole cohort suppressed even for all pupils. The empty state is
|
||||
load-bearing, not defensive.
|
||||
|
||||
## The display
|
||||
|
||||
Question-led. Three cards over one bar, with the cards acting as a lens on the
|
||||
bar rather than a summary beside it — hovering a card dims the bar, table and
|
||||
England reference to the categories that card is built from. The full mockup is
|
||||
linked above; what matters for implementation:
|
||||
|
||||
**The headline is not the sustained rate.** That figure sits between 92% and 97%
|
||||
for nearly every school in England. The mix is what varies, so the mix leads.
|
||||
|
||||
**The grouping is ours, not DfE's.** "Academic route" = school sixth form +
|
||||
sixth-form college; "College" = FE and other colleges; "Work" = apprenticeship +
|
||||
employment. This is the most arguable thing on the page, so it lives in one place
|
||||
in `lib/destinations.ts`, is explained in a tooltip, and is reversible in one
|
||||
edit.
|
||||
|
||||
**The absence is hatched neutral, never a colour.** "Activity not captured" means
|
||||
no record in the sources DfE holds — it includes independent schools, moving
|
||||
abroad and private training. Colouring it as a bad outcome would be a factual
|
||||
error rendered in CSS. The hatch also fixes a real contrast problem: neutral
|
||||
against the employment blue failed CVD separation at ΔE 7.6, and texture is the
|
||||
secondary encoding that rescues it. Every other adjacent pair clears ΔE 10.9
|
||||
under protanopia.
|
||||
|
||||
**Colour tokens.** Education is one hue in three steps (school-like to
|
||||
college-like); apprenticeship and employment are separate hues. Six new tokens in
|
||||
`globals.css`, defined in both themes, per the existing token discipline.
|
||||
|
||||
**The disadvantage split rides the same control.** One visualisation serving
|
||||
three cohorts, with the England reference repointing to the matching national
|
||||
group. The gap statement stays visible below the bar whatever is selected,
|
||||
because a gap nobody clicks on is a gap nobody sees.
|
||||
|
||||
## Data model
|
||||
|
||||
### Extraction
|
||||
|
||||
A new `tap-uk-ees-destinations` extractor, separate from `tap-uk-ees`. The
|
||||
existing tap downloads a release ZIP and reads a CSV inside it; the destinations
|
||||
files are far larger than we need and the query API filters server-side, so this
|
||||
one POSTs to `/v1/data-sets/{id}/query` and pages through results.
|
||||
|
||||
With every dimension pinned — destination measures, disadvantage status, sex
|
||||
Total, characteristic topic Total — one year returns **252,610 rows** across all
|
||||
geographic levels. Three school-level years is comfortably tractable.
|
||||
|
||||
Pinning is mandatory, not an optimisation: leaving the characteristic dimensions
|
||||
unconstrained returned 45 rows where 9 were wanted, because every breakdown
|
||||
shares one table.
|
||||
|
||||
The tap emits the raw value as text. **It does not coerce `c`.**
|
||||
|
||||
### Staging
|
||||
|
||||
`stg_ees_ks4_destinations` / `stg_ees_ks5_destinations`. Each raw value becomes
|
||||
two columns:
|
||||
|
||||
```sql
|
||||
case when raw ~ '^-?[0-9]+(\.[0-9]+)?$' then raw::numeric end as pupils,
|
||||
case
|
||||
when raw ~ '^-?[0-9]+(\.[0-9]+)?$' then 'published'
|
||||
when lower(trim(raw)) = 'c' then 'suppressed'
|
||||
else 'not_applicable'
|
||||
end as status
|
||||
```
|
||||
|
||||
### Marts
|
||||
|
||||
`fact_ks4_destinations` and `fact_ks5_destinations`, **long format**:
|
||||
|
||||
```
|
||||
urn, year, pupil_group, destination_category, cohort_pupils, pupils, percentage, status
|
||||
```
|
||||
|
||||
This departs from the wide house pattern (`fact_ks4_performance` and friends) on
|
||||
purpose. `pupil_group` is a genuine third dimension; going wide would need three
|
||||
sets of every column, and R2 is far easier to test on rows than on columns.
|
||||
|
||||
Roughly 8 categories × 3 groups × 4,946 schools × 3 years ≈ 356k rows.
|
||||
|
||||
`fact_destination_national` carries the same grain for England, so the page's
|
||||
England reference repoints with the switch.
|
||||
|
||||
### dbt tests
|
||||
|
||||
- `assert_destinations_no_derived_remainder` — for every (urn, year,
|
||||
pupil_group) with exactly one suppressed category, assert no aggregate row
|
||||
exists that would let the residual be recovered. **This is the R1 guard.**
|
||||
- `assert_destinations_group_masking` — for every (urn, year, category), if the
|
||||
disadvantaged group carries `suppressed`, so does the other-pupils group.
|
||||
**This is the R3 guard**, applied in the mart so no consumer can reach an
|
||||
unmasked combination.
|
||||
- `assert_destination_status_null_agreement` — `pupils is null` wherever
|
||||
`status != 'published'`, and never null where it is.
|
||||
- `assert_destinations_join_dim_school` — no orphaned URNs, matching the
|
||||
existing `assert_no_orphaned_facts` pattern.
|
||||
|
||||
## API
|
||||
|
||||
`GET /api/schools/{urn}` gains a `destinations` block:
|
||||
|
||||
```json
|
||||
{
|
||||
"ks4": {
|
||||
"cohort_year": "2022/23",
|
||||
"published": "2026-04",
|
||||
"groups": {
|
||||
"all": { "cohort": 180, "categories": [ … ], "aggregates": { … } },
|
||||
"disadvantaged": { … },
|
||||
"other": { … }
|
||||
}
|
||||
},
|
||||
"ks5": { … }
|
||||
}
|
||||
```
|
||||
|
||||
Each category carries `pupils`, `percentage` and `status`. **The serialiser never
|
||||
emits a computed remainder**, and a backend test asserts that a group containing a
|
||||
suppressed category serialises no total that closes the gap.
|
||||
|
||||
`null` for the whole block where nothing is published — the frontend renders the
|
||||
empty state from its absence, not from a sentinel.
|
||||
|
||||
## Frontend
|
||||
|
||||
| File | Kind | Job |
|
||||
|---|---|---|
|
||||
| `lib/destinations.ts` | pure | Category list, the academic/college/work grouping, `canAggregate()` enforcing R2, percentage derivation from counts |
|
||||
| `components/school/DestinationsSection.tsx` | server | Section shell, renders **all pupils** into the HTML |
|
||||
| `components/school/DestinationsView.tsx` | client | Cohort switch, card↔bar linkage |
|
||||
| `components/school/Post16DestinationsSection.tsx` | server | Year 13 section, sixth-form schools only |
|
||||
| `app/globals.css` | tokens | Six destination colours, both themes |
|
||||
|
||||
Server-first matches the directory's existing discipline — every component in
|
||||
`components/school/` is a server component except `AdmissionsViewToggle`, which
|
||||
is the precedent this follows. All-pupils figures are in the HTML for crawlers
|
||||
and for no-JS; only the switch and the hover linkage need the client.
|
||||
|
||||
`lib/schoolSections.ts` gains `hasKs4Destinations` / `hasKs5Destinations` flags
|
||||
and the nav items, following the existing `computeSchoolFlags` pattern.
|
||||
|
||||
**Placement** on the secondary template: GCSE results → After Year 11 → After the
|
||||
sixth form → admissions. Destinations follow attainment because they answer "and
|
||||
then what happened".
|
||||
|
||||
**Dating.** The latest destination year is 2022/23, published April 2026, while
|
||||
the site's newest KS4 year is 2024/25. The section header states its own cohort
|
||||
year, or it reads as stale data next to the GCSE section above it.
|
||||
|
||||
## Edge states
|
||||
|
||||
| State | Frequency | Behaviour |
|
||||
|---|---|---|
|
||||
| Whole cohort suppressed | 13% of special, 41% of AP | Section renders the explanation, no chart |
|
||||
| Some categories withheld | 80% of disadvantaged views | Cards degrade individually; **no bar**; table marks withheld rows |
|
||||
| Disadvantaged group suppressed entirely | 5% | Switch drops to two options, gap panel not rendered |
|
||||
| No sixth form | — | Post-16 section not rendered at all — absence is correct, a "no data" placeholder would imply something is missing |
|
||||
| School too new | — | "First figures expected in 2026", not a bare no |
|
||||
|
||||
## Testing
|
||||
|
||||
Per CLAUDE.md, user-facing behaviour extends `e2e/` in the same PR.
|
||||
|
||||
**Unit** — `lib/destinations.ts` is where R1 and R2 live, so it carries the
|
||||
heaviest tests: `canAggregate()` refuses a group containing one suppressed cell,
|
||||
allows one spanning two, and the bar builder refuses to emit segments for any
|
||||
group with suppression. These are the tests that must fail loudly if someone
|
||||
later "fixes" a gap in the chart.
|
||||
|
||||
**dbt** — the three tests above.
|
||||
|
||||
**Backend** — the serialiser emits no closing total for a partially suppressed
|
||||
group.
|
||||
|
||||
**E2E** — a school with full data renders three cards and a bar; a school with a
|
||||
partially suppressed disadvantaged group renders the withheld state and **no bar
|
||||
element**; a suppressed school renders the explanation; a school with no sixth
|
||||
form renders no post-16 section.
|
||||
|
||||
Note the staging caveat: mart changes are inert until the Airflow pipeline runs,
|
||||
and the staging E2E gate runs post-merge.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- **Compare view and rankings.** The long mart shape supports both; neither is
|
||||
built here. Flagged because "% to a school sixth form" is a plausible rankings
|
||||
metric and the mart shape should not have to change to allow it.
|
||||
- **Ethnicity, sex, SEN and prior-attainment breakdowns.** Available in the same
|
||||
file, ingested deliberately not at all — each is a separate editorial decision
|
||||
about what a school page should assert.
|
||||
- **Longer term destinations** (3 and 5 years out) and **Progression to higher
|
||||
education** — separate publications, worth a later look for sixth forms.
|
||||
- **Primary schools.** No KS2 destination measures publication exists; DfE
|
||||
tracking starts at KS4. Naming the secondaries a primary's leavers go to needs
|
||||
the National Pupil Database, which is not publishable at that grain.
|
||||
|
||||
## Risks
|
||||
|
||||
**A later change reintroduces the disclosure.** The likeliest routes are
|
||||
applying `safe_numeric` to a destination column for consistency, adding a
|
||||
`coalesce` in a mart, or — as happened in review — enforcing a disclosure rule
|
||||
at the rendering layer instead of the publishing layer. Mitigation is the dbt
|
||||
tests plus `backend/tests/test_destinations_api.py`, which reconstructs the
|
||||
residual the way an attacker would and asserts it no longer resolves.
|
||||
|
||||
**The two-year lag reads as staleness.** Mitigated by dating the cohort in the
|
||||
section header rather than only in a tooltip.
|
||||
|
||||
**Sixth-form retention will be misread.** "41% went to a school sixth form" says
|
||||
nothing about *which* school. The published file reports destination type, never
|
||||
destination institution. Copy must never imply "stayed on here", and the tooltip
|
||||
should say so.
|
||||
|
||||
**Section length.** The secondary template is already long and this adds two
|
||||
sections. If it becomes a problem the post-16 section is the one to collapse
|
||||
behind a disclosure, not the Year 11 one.
|
||||
|
||||
## Open questions
|
||||
|
||||
1. Is the disadvantage split its own section or a sub-block inside the
|
||||
destinations section? Modelled as a sub-block; it is the most differentiating
|
||||
figure on the page and the most easily misread on a small cohort.
|
||||
2. Do we ingest the apprenticeship level breakdown (intermediate / advanced /
|
||||
higher) now, or collapse to one apprenticeship figure and revisit? Collapsed
|
||||
in this design.
|
||||
@@ -0,0 +1,368 @@
|
||||
# Giving schoolcompare a human author: an About page and a blog
|
||||
|
||||
**Date:** 2026-09-02
|
||||
**Status:** Design — awaiting review
|
||||
**Scope:** A named author for the site, an `/about` page, and a Payload-CMS-backed
|
||||
blog at `/blog`.
|
||||
|
||||
## Why
|
||||
|
||||
The site reads as synthetic. Not because of its tone, but because of three
|
||||
specific absences:
|
||||
|
||||
1. **Nobody is accountable for the numbers.** There is no author, no statement
|
||||
of why the site exists, and no one who can be wrong. The only human trace on
|
||||
the entire site is `contact@schoolcompare.co.uk` in the footer.
|
||||
2. **No visible judgement.** Every figure is presented as though it fell out of
|
||||
a machine. Hundreds of editorial decisions went into this codebase — which
|
||||
metrics to show, when a benchmark is invalid, what to suppress — and not one
|
||||
of them is visible to a reader. `isSpecialSchool()` silently drops the
|
||||
England comparison for special schools and PRUs because that comparison is
|
||||
meaningless; nowhere does the site *say* so.
|
||||
3. **The voice is institutional third person.** "schoolcompare brings it all
|
||||
into one place." "Built for parents, governors, journalists." That is
|
||||
brochure register, and it is precisely the register that machine-generated
|
||||
content defaults to.
|
||||
|
||||
There is a second, independent reason. The SEO programme
|
||||
(`2026-08-20-seo-programme-design.md`) defines eight workstreams and none of
|
||||
them address E-E-A-T or authorship. School performance data is YMYL territory;
|
||||
an anonymous site republishing DfE figures has no authorship signal at all. This
|
||||
work fills that hole, and the blog gives W6 (explainer content) somewhere to
|
||||
live.
|
||||
|
||||
### The failure mode to avoid
|
||||
|
||||
The standard fix — a stock photo and "Hi, I'm Tudor, and I'm passionate about
|
||||
education!" — reads as *more* synthetic than the current coldness. Manufactured
|
||||
warmth is a stronger machine-tell than plain institutional voice. Everything
|
||||
here has to be specific, occasionally awkward, and willing to be unflattering,
|
||||
or it makes the problem worse.
|
||||
|
||||
## Positioning
|
||||
|
||||
The author is **Tudor**: first name only, real photograph, no surname, no
|
||||
employer named.
|
||||
|
||||
The credibility claim is deliberately **not** educational expertise. The About
|
||||
page states plainly: *"I'm not an education expert."* Authority comes from two
|
||||
things that are actually true:
|
||||
|
||||
- **Experience.** A parent going through primary admissions in south-west London
|
||||
right now. Google's E-E-A-T leads with Experience, and lived experience of the
|
||||
thing is exactly what the DfE's own service lacks.
|
||||
- **Method.** Every number's provenance is stated, so a reader can check the
|
||||
site rather than trust it.
|
||||
|
||||
This is more durable than borrowed expertise: it cannot be undermined by someone
|
||||
noticing the author has no teaching qualification.
|
||||
|
||||
**Consequence for the design.** A `Person` entity with no surname is a weak
|
||||
search signal and cannot be corroborated off-site. The credibility load
|
||||
therefore shifts onto the methodology being visibly rigorous. That is a design
|
||||
constraint, not a caveat — it is why the About page carries a substantial
|
||||
"how this is built and where it can be wrong" section rather than a short bio.
|
||||
|
||||
### Voice rules
|
||||
|
||||
Applied to About and every post. Recorded here so the voice does not drift.
|
||||
|
||||
- First person singular. "I built", not "we provide".
|
||||
- Concrete over general. "when we were looking at schools in Wandsworth" beats
|
||||
any amount of stated warmth.
|
||||
- State limits before someone else finds them. Every post that presents a
|
||||
metric says what it does not show.
|
||||
- No mission statements, no "passionate about", no invented team.
|
||||
- No em dashes. One of the clearest tells of machine-written prose, which is
|
||||
the exact problem this work exists to fix.
|
||||
- Short sentences. The existing code comments in this repo are already written
|
||||
this way; the prose should match.
|
||||
|
||||
## Scope
|
||||
|
||||
**In:**
|
||||
|
||||
- `/about` — a coded page (not CMS-managed).
|
||||
- `/blog` and `/blog/[slug]` — Payload-backed, with an index and post pages.
|
||||
- Payload CMS installed into the existing Next application.
|
||||
- Footer and navigation links to both.
|
||||
- `Person`, `Organization`, `BlogPosting`, `BreadcrumbList` JSON-LD.
|
||||
- RSS feed and sitemap integration.
|
||||
- One first post, so the blog does not launch empty.
|
||||
|
||||
**Out (deliberately):**
|
||||
|
||||
- Rewriting existing homepage/how-it-works copy into first person. Worth doing,
|
||||
but it would double the review surface of this PR. Separate change.
|
||||
- In-product signed notes on school pages (the "distributed humanity" idea).
|
||||
Revisit once About and the blog exist.
|
||||
- Comments, newsletter, author accounts beyond one.
|
||||
- A team page. There is no team.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Topology
|
||||
|
||||
Payload 3 installs **into the existing Next application** and serves `/admin`
|
||||
from the same container. One image, one deploy, no new service. This is
|
||||
Payload 3's native model and it makes on-demand revalidation trivial, because
|
||||
the CMS hooks run in the same process as the Next cache.
|
||||
|
||||
Accepted costs: the public site's image now carries Payload, so a CMS security
|
||||
patch redeploys the whole site; and the image grows substantially.
|
||||
|
||||
### Two collisions that must be handled
|
||||
|
||||
**1. `/api` is already taken.** `app/api/[...path]/route.ts` is a catch-all that
|
||||
proxies `/api/*` to FastAPI at runtime. Payload's default API route is also
|
||||
`/api`. Left alone, these fight, and the failure is not clean — the catch-all
|
||||
would swallow Payload's admin API calls and forward them to FastAPI.
|
||||
|
||||
Payload's API route is therefore remapped:
|
||||
|
||||
```ts
|
||||
routes: { api: '/cms-api', admin: '/admin' }
|
||||
```
|
||||
|
||||
with its route group at `app/(payload)/cms-api/[...slug]/route.ts`. The
|
||||
`/cms-api` prefix must also be added to the FastAPI proxy's excluded-paths list
|
||||
as a defensive second line.
|
||||
|
||||
**2. `next.config.js` is CommonJS.** Payload's `withPayload()` wrapper is ESM
|
||||
only. The config must become `next.config.mjs`, converting `module.exports` to
|
||||
`export default` and wrapping the export. All existing content — the standalone
|
||||
output, `outputFileTracingIncludes`, the staging `X-Robots-Tag` header block,
|
||||
the CSP — carries over unchanged. This is mechanical but it touches the file
|
||||
that controls staging's noindex, so it needs care and an explicit test.
|
||||
|
||||
### Database
|
||||
|
||||
Payload uses the existing `sc_database` Postgres instance, in its **own
|
||||
`payload` schema**:
|
||||
|
||||
```ts
|
||||
db: postgresAdapter({
|
||||
pool: { connectionString: process.env.DATABASE_URL },
|
||||
schemaName: 'payload',
|
||||
})
|
||||
```
|
||||
|
||||
The frontend container is already on the `backend` Docker network, so it can
|
||||
reach `sc_database:5432` with no networking change. It needs a new
|
||||
`DATABASE_URL` environment variable.
|
||||
|
||||
Schema isolation is not cosmetic. `public` currently holds the application
|
||||
tables and Airflow's metadata, and `scripts/migrate_csv_to_db.py --drop` exists
|
||||
to drop and reimport. Blog content living in its own schema means no data
|
||||
pipeline operation can destroy it.
|
||||
|
||||
**Verified 2026-09-02** (this was an open question when the spec was written).
|
||||
`--drop` calls `run_full_migration()` in `backend/migration.py`, which drops
|
||||
exactly two tables by name:
|
||||
|
||||
```python
|
||||
ks2_tables = ["school_results", "schools"]
|
||||
for tname in ks2_tables:
|
||||
if tname in existing:
|
||||
Base.metadata.tables[tname].drop(bind=engine)
|
||||
```
|
||||
|
||||
There is no `Base.metadata.drop_all()` anywhere in `backend/`, and no
|
||||
`DROP SCHEMA`. The only other drop is `_apply_schema_drops()`, a single
|
||||
schema-qualified `DROP TABLE IF EXISTS marts.fact_parent_view CASCADE`.
|
||||
Nothing sets `search_path`, so the SQLAlchemy metadata resolves to `public`,
|
||||
and `inspector.get_table_names()` does not even enumerate other schemas.
|
||||
|
||||
So the guarantee is stronger than schema isolation alone: `--drop` targets two
|
||||
named tables that Payload does not have, and would not reach `posts`, `media`
|
||||
or `users` even if they shared a schema. The `payload` schema remains the right
|
||||
choice — it protects against a *future* broadening of that script rather than
|
||||
today's behaviour — but the safety claim rests on verified code, not on
|
||||
assumption.
|
||||
|
||||
Putting CMS tables in this instance is consistent with existing practice —
|
||||
Airflow already stores its metadata there.
|
||||
|
||||
### Migrations
|
||||
|
||||
Payload's Postgres adapter auto-pushes schema in development and requires
|
||||
explicit migrations in production. Use `prodMigrations`, which runs pending
|
||||
migrations during server initialisation:
|
||||
|
||||
```ts
|
||||
db: postgresAdapter({ /* ... */, prodMigrations: migrations })
|
||||
```
|
||||
|
||||
This is preferred over a one-shot init container (the `airflow-init` pattern)
|
||||
because the app is a single long-running process and there is no ordering
|
||||
problem to solve. Migration files are generated with `payload migrate:create`
|
||||
and committed, so schema changes travel through the same PR and staging gate as
|
||||
code.
|
||||
|
||||
### Media
|
||||
|
||||
Uploads go to a Docker named volume, consistent with `postgres_data`,
|
||||
`typesense_data` and `airflow_logs`.
|
||||
|
||||
- `staticDir` must be an **absolute** path in Payload 3: `/app/media`.
|
||||
- The container runs as `nextjs` (uid 1001). The Dockerfile must
|
||||
`mkdir -p /app/media && chown nextjs:nodejs /app/media` **before** the volume
|
||||
is mounted, or Docker will create the mountpoint root-owned and every upload
|
||||
will fail with EACCES.
|
||||
- `sharp` moves from `devDependencies` to `dependencies` — Payload needs it at
|
||||
runtime to generate `imageSizes`.
|
||||
- The volume must be added to the backup routine alongside Postgres. A blog
|
||||
post's images are not reproducible from the pipeline.
|
||||
|
||||
### Rendering
|
||||
|
||||
**Constraint:** CI builds the image with no database reachable. Blog pages
|
||||
therefore cannot use build-time `generateStaticParams` — that would either fail
|
||||
the build or bake in an empty post list.
|
||||
|
||||
Instead: ISR. Post and index pages declare a `revalidate` window and render on
|
||||
first request, with Payload `afterChange` / `afterDelete` hooks calling
|
||||
`revalidatePath('/blog')` and `revalidatePath('/blog/' + slug)` for immediate
|
||||
publication. Because Payload runs in the same process, the hook calls
|
||||
`revalidatePath` from `next/cache` directly — no webhook, no shared secret.
|
||||
|
||||
The ISR cache lives on container disk and is cleared by a redeploy. For a
|
||||
single container serving a handful of posts this is fine.
|
||||
|
||||
### Collections
|
||||
|
||||
- **`posts`** — `title`, `slug`, `publishedAt`, `excerpt`, `heroImage`
|
||||
(relation to `media`), `content` (Lexical rich text), `seo` group
|
||||
(`metaTitle`, `metaDescription`), `_status` (drafts enabled).
|
||||
- **`media`** — upload collection, `alt` required, `imageSizes` for thumbnail
|
||||
and hero widths, public read access.
|
||||
- **`users`** — Payload's auth collection. One account. Public creation
|
||||
disabled.
|
||||
|
||||
Drafts are enabled so posts can be written over several sittings and previewed
|
||||
before publication.
|
||||
|
||||
**Payload Blocks** are how posts embed live product components — a real trend
|
||||
chart or comparison table inside a post, rendered from live data rather than
|
||||
screenshotted. This is the main thing the CMS has to earn back against
|
||||
file-based MDX, and it directly serves the goal: showing judgement in context.
|
||||
Ship with one block (a callout/aside for "what this number doesn't tell you");
|
||||
add a live-chart block once a post needs it.
|
||||
|
||||
### Security
|
||||
|
||||
`/admin` is the first authenticated surface on this site. Public, hardened:
|
||||
|
||||
- `PAYLOAD_SECRET` — long, random, set in the Portainer stack environment, never
|
||||
committed. The same variable must exist in staging with a *different* value.
|
||||
- Strong unique password on the single admin account.
|
||||
- Login rate limiting via Payload's `maxLoginAttempts` / `lockTime`.
|
||||
- `X-Robots-Tag: noindex, nofollow` on `/admin/*` and `/cms-api/*`, and a
|
||||
`robots.ts` disallow. The admin panel must never be indexed.
|
||||
- Public user creation disabled; no open registration.
|
||||
- Verify the existing CSP `frame-ancestors` directive does not break the admin
|
||||
panel.
|
||||
|
||||
Residual risk, accepted: a future Payload authentication CVE is live against the
|
||||
public internet. Mitigation is prompt patching, which the staging→prod pipeline
|
||||
already supports. If this becomes uncomfortable, restricting `/admin` at the
|
||||
proxy to LAN/VPN is a one-line change later.
|
||||
|
||||
Staging note: staging runs the same image on `stx.`, so it gets its own admin
|
||||
panel and its own database. It must have its own `PAYLOAD_SECRET` and its own
|
||||
credentials — never production's.
|
||||
|
||||
## Deployment changes
|
||||
|
||||
- `nextjs-app/Dockerfile` — create and chown `/app/media`; ensure Payload's
|
||||
admin bundle and `sharp` survive standalone output file tracing.
|
||||
- `docker-compose.portainer.yml` and the staging equivalent — add
|
||||
`DATABASE_URL` and `PAYLOAD_SECRET` to the `frontend` service, add a
|
||||
`payload_media` volume mounted at `/app/media`, and add
|
||||
`depends_on: sc_database`.
|
||||
- Document both new environment variables in the compose header comment block,
|
||||
which is where this stack records its configuration.
|
||||
|
||||
## SEO
|
||||
|
||||
- `Person` (Tudor, with photo) and `Organization` JSON-LD on `/about`.
|
||||
- `BlogPosting` + `BreadcrumbList` on post pages, with `author` referencing the
|
||||
same `Person`.
|
||||
- Canonical URLs on `/blog` and every post.
|
||||
- Posts and `/about` added to the existing sitemap (`app/sitemap.xml/route.ts`
|
||||
and `app/sitemaps/[...parts]`). Post URLs come from Payload at request time.
|
||||
- RSS feed at `/blog/rss.xml`.
|
||||
- Footer links to both pages, under a new "About" column.
|
||||
|
||||
**Navigation is deliberately left alone.** `Navigation.tsx` renders a bottom tab
|
||||
bar on mobile that already carries four items (Search, Compare, Rankings,
|
||||
Admissions). A fifth tab makes each one cramped at 320px, and About and Blog are
|
||||
both lower-intent than any of the four. Both live in the footer; About
|
||||
additionally gets a byline link from every post, which is where a reader who
|
||||
cares actually asks the question. Revisit only if analytics show people hunting
|
||||
for it.
|
||||
|
||||
## Testing
|
||||
|
||||
Unit (Jest):
|
||||
|
||||
- Post rendering, including a post with no hero image and one with no excerpt.
|
||||
- Slug generation and collision handling.
|
||||
- JSON-LD shape for `BlogPosting` and `Person`.
|
||||
- The `next.config.mjs` conversion preserves the staging `X-Robots-Tag` rule —
|
||||
this guards the riskiest mechanical change in the plan.
|
||||
|
||||
E2E (Playwright, `e2e/`, required by CLAUDE.md for user-facing change):
|
||||
|
||||
- `/about` renders, shows the author name and photo, and is reachable from the
|
||||
footer and nav.
|
||||
- `/blog` lists at least one post; clicking through reaches the post.
|
||||
- A post page renders title, date, body and byline.
|
||||
- `/admin` responds with `noindex` and does not leak a stack trace when
|
||||
unauthenticated.
|
||||
|
||||
Note the known constraint: new journeys cannot be proven in PR checks, because
|
||||
the staging E2E gate runs post-merge.
|
||||
|
||||
## Risks
|
||||
|
||||
| Risk | Mitigation |
|
||||
|---|---|
|
||||
| `next.config.mjs` conversion silently drops the staging noindex header, making staging a crawlable duplicate | Unit test asserting the header rule; verify on staging before promotion |
|
||||
| Payload API route collides with the FastAPI `/api` proxy | Remap to `/cms-api`; add to the proxy's exclusion list |
|
||||
| Media volume mounts root-owned; all uploads fail with EACCES | `mkdir`+`chown` in the Dockerfile before the mount; test an upload on staging |
|
||||
| Build fails or bakes empty content because CI has no DB | No build-time DB access; ISR only |
|
||||
| A pipeline `--drop` destroys blog content | Separate `payload` schema; verify `--drop` blast radius before building |
|
||||
| Media volume not backed up; images unrecoverable | Add `payload_media` to the backup routine |
|
||||
| Payload auth CVE exposed publicly | Prompt patching; proxy restriction available as a fallback |
|
||||
| Blog launches empty or goes stale | Ship with one post; cadence is explicitly "a few times a year", so no cadence is promised anywhere on the page — no dates implying a schedule |
|
||||
|
||||
## Sequence
|
||||
|
||||
Each step is independently reviewable and mergeable.
|
||||
|
||||
1. **Payload foundation** — install, `next.config.mjs` conversion, `payload`
|
||||
schema, `/cms-api` remap, `users` collection, `/admin` hardening, compose and
|
||||
Dockerfile changes. No public-facing change yet. Verify on staging that the
|
||||
site is unchanged and `/admin` works.
|
||||
2. **`/about`** — coded page, photo, `Person`/`Organization` JSON-LD, footer and
|
||||
nav links, e2e journey. Independently valuable and does not depend on the
|
||||
blog.
|
||||
3. **Blog** — `posts` and `media` collections, `/blog` index and post pages, ISR
|
||||
plus revalidation hooks, RSS, sitemap, structured data, e2e journeys.
|
||||
4. **First post** — written in the admin panel, published through the normal
|
||||
flow, proving the whole path end to end.
|
||||
|
||||
Step 1 carries all the infrastructure risk and none of the visible benefit, so
|
||||
it should be verified on staging carefully before step 2 starts.
|
||||
|
||||
## Dependencies on Tudor
|
||||
|
||||
- **A photograph.** Blocks step 2. Nothing else in the plan is blocked by it.
|
||||
- **The first post's subject.** Blocks step 4 only. Suggested: what school
|
||||
performance data cannot tell you — it demonstrates judgement, is genuinely
|
||||
useful, and is the kind of thing an anonymous or machine-written site will not
|
||||
publish.
|
||||
- ~~Confirmation that `scripts/migrate_csv_to_db.py --drop` is schema-scoped.~~
|
||||
**Resolved 2026-09-02** — verified in `backend/migration.py`; see the
|
||||
Database section. No action needed.
|
||||
@@ -0,0 +1,432 @@
|
||||
# Other Schools Nearby — Design
|
||||
|
||||
**Date:** 2026-09-21, revised 2026-09-22
|
||||
**Status:** revised after staging review
|
||||
**Scope:** school detail pages, both phase templates
|
||||
|
||||
> **Revision, 2026-09-22.** The first build ranked by intake similarity and used
|
||||
> distance as a tiebreak. On staging a Catholic primary showed six Catholic
|
||||
> primaries, none of them close enough to be a real option, and omitted the
|
||||
> community school down the road. Distance now decides the order and nothing
|
||||
> else does; the tier system is gone. The reasoning is kept below rather than
|
||||
> quietly overwritten, because the mistake is the instructive part.
|
||||
|
||||
## Goal
|
||||
|
||||
Give a school detail page an answer to the question every reader arrives with
|
||||
after the results tables: *and what else is around here?*
|
||||
|
||||
Today a school page links outward to its place pages through
|
||||
`components/school/NearbyPlaces.tsx` and nowhere else. It never links to another
|
||||
school. This section adds that edge — up to six nearby schools of the same phase
|
||||
and a comparable intake, three at a time in a carousel, each a crawlable link and
|
||||
each addable to the comparison basket in one click.
|
||||
|
||||
Mockup, in both themes (drawn against the original tiered design, so its ledes
|
||||
and chip fallbacks are one revision behind the copy specified below):
|
||||
<https://claude.ai/artifact/168KdUMcfkUeGWW2FGjuec>
|
||||
|
||||
Source of the same page in the repo: `mockups/similar-schools-nearby.html`.
|
||||
|
||||
## The constraint that shapes everything
|
||||
|
||||
**A nearby school is not automatically a comparable school.**
|
||||
|
||||
The section's whole value is that a reader treats what it shows as a shortlist.
|
||||
That makes every card an implicit claim that the school is a realistic
|
||||
alternative, and there are three ways that claim goes wrong:
|
||||
|
||||
1. A **selective** school beside a non-selective one. Their intakes are
|
||||
different by construction, so putting their Attainment 8 figures side by side
|
||||
invites a conclusion the data cannot support.
|
||||
2. A **special school, PRU or AP** beside a mainstream school. This is the same
|
||||
error PR #70 fixed for the England benchmark, where Greenmead (URN 101099)
|
||||
rendered "0% — 62 below England".
|
||||
3. A **single-sex** school of the opposite sex. Not a weak match — not an option
|
||||
at all.
|
||||
|
||||
So the design separates two kinds of fact, and never confuses them:
|
||||
|
||||
- **Hard filters** encode the claims above. They decide eligibility, and are
|
||||
never relaxed at any distance, even if that means the section does not render.
|
||||
- **Shared characteristics** — gender, religious character, selectivity —
|
||||
describe how closely an intake resembles this school's. They are *reported on
|
||||
the card and never ranked on*, so the reader weighs them rather than having
|
||||
them weighed for them.
|
||||
|
||||
Everything below follows from that split. The revision at the top of this
|
||||
document is what happens when the second kind is treated as the first.
|
||||
|
||||
## Selection algorithm
|
||||
|
||||
A backend helper, `_nearby_schools_payload(urn)` in `backend/app.py`, modelled
|
||||
on the existing `_places_payload(urn)` and delegating to
|
||||
`backend/nearby_schools.select_nearby(frame, urn)`, which operates on the cached
|
||||
`load_latest_school_data()` frame — one row per URN, already carrying
|
||||
`latitude`, `longitude`, `phase`, `gender`, `religious_denomination`,
|
||||
`admissions_policy`, `school_type` and `status`.
|
||||
|
||||
### Hard filters
|
||||
|
||||
| Filter | Rule |
|
||||
|---|---|
|
||||
| Self | `urn` is excluded |
|
||||
| Status | GIAS status must be open |
|
||||
| Coordinates | both `latitude` and `longitude` present on both schools |
|
||||
| Phase | same phase group via the existing `PHASE_GROUPS` map |
|
||||
| Provision | special/PRU/AP match only each other |
|
||||
| Selectivity | selective matches selective; non-selective matches non-selective |
|
||||
| Gender | Boys never matches Girls; Mixed is compatible with both |
|
||||
|
||||
`PHASE_GROUPS` is reused rather than re-derived so an all-through school is
|
||||
offered correctly on both the primary and secondary sides, exactly as it already
|
||||
behaves in search.
|
||||
|
||||
The provision filter needs a backend counterpart to the frontend's
|
||||
`isSpecialSchool()` in `nextjs-app/lib/utils.ts:897`, reading the same GIAS
|
||||
establishment types through `backend/gias_codes.py`. The two must agree: a
|
||||
school the frontend treats as special for benchmarking but the backend treats as
|
||||
mainstream for matching would be dropped from its own England comparison and
|
||||
then offered as a peer to a mainstream school on the next page along.
|
||||
|
||||
**Up to six cards, three visible.** Three fit the row; the rest are reached with
|
||||
the carousel arrows. Two is the minimum that renders at all.
|
||||
|
||||
### Order: distance, and nothing else
|
||||
|
||||
The nearest eligible schools, closest first. Similarity does not enter the
|
||||
ranking at any point.
|
||||
|
||||
**Why not, having built it the other way first.** The original design ranked by
|
||||
tiers — same gender and faith within 3 miles, then same gender within 5, then
|
||||
anything within 10 — and used distance only to order the result. Two things
|
||||
followed, and both showed up on the first Catholic primary anyone looked at:
|
||||
|
||||
- A faith match at 2.9 miles outranked a community school at 0.3 miles. For a
|
||||
primary, whose catchment is routinely under a mile, the far school is not a
|
||||
weaker option; it is not an option.
|
||||
- Because the row filled from the best tier before widening, three Catholic
|
||||
schools within 3 miles were enough to fill all six slots with Catholic
|
||||
schools. The stopping rule that produced this had been added to prevent the
|
||||
*opposite* failure — padding a row with weak distant matches — and made this
|
||||
one certain.
|
||||
|
||||
The premise was backwards. **Distance is a constraint and intake is a
|
||||
preference.** A parent cannot act on a school outside their reach however well
|
||||
it matches, and they are perfectly capable of noticing a shared denomination
|
||||
for themselves if we show it to them. So similarity moved from the ranking to
|
||||
the card: `shared` reports what a school genuinely has in common, and the reader
|
||||
applies their own weighting.
|
||||
|
||||
The hard filters above were always where the defensibility lived. They are
|
||||
untouched.
|
||||
|
||||
### Reach: a sanity bound, not a target
|
||||
|
||||
| Phase | Reach |
|
||||
|---|---|
|
||||
| Primary, middle deemed primary, all-through | 2 miles |
|
||||
| Secondary, middle deemed secondary | 6 miles |
|
||||
| 16 plus | 10 miles |
|
||||
|
||||
Ordering by distance already handles density — a school in inner London fills
|
||||
all six slots inside a mile and never approaches the cap. The cap decides one
|
||||
thing: what happens where the area is sparse. It differs by phase because
|
||||
catchments do, and because people travel furthest for post-16.
|
||||
|
||||
**A primary with nothing inside two miles renders no section**, and that is the
|
||||
intended answer rather than a gap. The alternative is a section headed "nearby"
|
||||
listing a school four miles from a five-year-old.
|
||||
|
||||
**Past the sixth school, the rest are dropped without a count.** The section
|
||||
does not try to be the list: `NearbyPlaces` sits directly beneath and already
|
||||
leads to the place pages, which are built for browsing a full set and which the
|
||||
school page exists to feed.
|
||||
|
||||
**Fewer than two results renders nothing.** Not an empty state, not a single
|
||||
lonely card. The section is absent, the nav item is absent, and the page is
|
||||
unchanged from before it existed.
|
||||
|
||||
### Distance
|
||||
|
||||
Straight-line, from the vectorised haversine already used for postcode search at
|
||||
`backend/app.py:831`, computed over the ~27k-row frame in numpy. Reported to one
|
||||
decimal place in miles, consistent with the rest of the site.
|
||||
|
||||
Straight-line distance is not road distance and is not measured from the
|
||||
reader's home. The section says so in its disclosure rather than leaving the
|
||||
reader to assume otherwise.
|
||||
|
||||
## API
|
||||
|
||||
`/api/schools/{urn}` gains a `similar_schools` array. Each row:
|
||||
|
||||
| Field | Notes |
|
||||
|---|---|
|
||||
| `urn` | for the link and the compare basket |
|
||||
| `school_name` | link text |
|
||||
| `distance_miles` | one decimal place |
|
||||
| `school_type` | GIAS type, translated, for the card's meta line |
|
||||
| `age_range` | for the meta line |
|
||||
| `shared` | what this school genuinely shares with the subject; may be empty |
|
||||
| `metric_value` | the phase-appropriate headline figure, or null |
|
||||
| `metric_key` | `rwm_expected_pct` or `attainment_8_score` — see below |
|
||||
| `metric_year` | the year the figure is from |
|
||||
|
||||
The metric follows the subject school's phase side, not the neighbour's own
|
||||
phase, so a row of cards never mixes two scales. The secondary
|
||||
side uses `attainment_8_score`; the primary side uses `rwm_expected_pct`. Where
|
||||
the neighbour has no value for that key, the card reads "Not published" rather
|
||||
than falling back to the other key.
|
||||
|
||||
Which side a school takes is decided once, in
|
||||
`similar_schools.is_secondary_phase`, by membership of `PHASE_GROUPS["secondary"]`
|
||||
minus all-through — never by testing for the substring "secondary", which misses
|
||||
`16 plus` (GIAS phase 6) and hands a sixth-form college the primary bucket.
|
||||
All-through is the exception in the other direction: `PHASE_GROUPS` lists it on
|
||||
both sides, but it takes the primary metric.
|
||||
|
||||
This is usually the same thing as "the template the page renders", but not
|
||||
always. `computeSchoolFlags` decides the template with that same substring test,
|
||||
so a `16 plus` school renders `PrimarySchoolSections` while being matched —
|
||||
correctly — against secondaries. The section therefore takes its lede noun from
|
||||
the school's own phase rather than from its template, or it would print "Other
|
||||
primary schools near <sixth form college>" above a row of secondaries.
|
||||
|
||||
Up to six rows of roughly 130 bytes each. It rides in the existing detail payload
|
||||
rather than a new endpoint because the page already makes exactly one server
|
||||
fetch for its data, and `/school/[slug]` regenerates at most weekly
|
||||
(`revalidate = 604800`), so the per-request cost is paid once per school per
|
||||
week.
|
||||
|
||||
**The key is absent, not null, on a backend that does not have this code.** The
|
||||
frontend treats absent and empty identically, which is what allowed
|
||||
`NearbyPlaces` to ship without a lockstep deploy of the two images.
|
||||
|
||||
`shared` is computed on the backend, beside the data it is derived from, not
|
||||
re-derived on the frontend. Deriving it twice is how a card comes to claim
|
||||
something the selection never established. An empty list is a real answer and
|
||||
renders no chips: a bare card costs a school nothing but the likeness it does
|
||||
not have, since the order was already settled by distance.
|
||||
|
||||
## Frontend
|
||||
|
||||
### Components
|
||||
|
||||
`components/school/SimilarSchoolsSection.tsx` — a server component wrapped in
|
||||
the shared `Section` shell from `sectionShared.tsx`. It renders the heading,
|
||||
the lede, the card grid, the footer CTA and one caption line. Every
|
||||
card's title is an `<a>` to the school's canonical slug URL via `schoolUrl()`.
|
||||
|
||||
`components/school/AddToCompareButton.tsx` — calls `addSchool` from
|
||||
`ComparisonProvider` and reports the selection with a `from: 'similar_schools'`
|
||||
attribution, mirroring `addSchoolFromSearch` in `HomeView.tsx:442`.
|
||||
|
||||
`components/school/SimilarSchoolsCarousel.tsx` — the scroller and its arrows. It
|
||||
takes the server-rendered cards as `children` and the server-rendered heading and
|
||||
lede as a `header` prop, so those stay server components while the client
|
||||
component owns only the ref, the scroll handler and the arrows' disabled state.
|
||||
|
||||
The split matters: the links — the part with SEO value and the part that must
|
||||
work without JavaScript — are server-rendered into the initial HTML, and only
|
||||
the basket interaction and the arrows are hydrated.
|
||||
|
||||
### The carousel
|
||||
|
||||
**Every card is in the initial HTML.** The arrows scroll a list; they never swap
|
||||
a view. Six `<a>` elements are in the markup whether or not anything is
|
||||
hydrated, which is the whole reason the section exists — a paginated widget that
|
||||
mounts cards on click would put four of the six links beyond a crawler and
|
||||
beyond a reader with no JavaScript.
|
||||
|
||||
So the scroller is a plain overflowing `<ul>` with `scroll-snap-type: x
|
||||
mandatory`, and the arrows call `scrollBy` on it. With no JavaScript it
|
||||
degrades to a horizontally scrollable row that still works by touch and by
|
||||
trackpad. Three cards are visible at desktop width and two below 820px.
|
||||
|
||||
**Arrows appear only when there is somewhere to go** — that is, only when more
|
||||
than three schools were found. Each disables itself at its own end of the
|
||||
travel.
|
||||
|
||||
#### Below 640px the arrows go away
|
||||
|
||||
This follows [MOBILE.md](../../../MOBILE.md), which makes 360px the design
|
||||
floor and mobile the primary target at ≥55% of traffic.
|
||||
|
||||
Kept in the heading's flex row at 360px, the two arrow buttons take 96px from a
|
||||
328px card and crush the lede into a four-line column — measured, not guessed.
|
||||
And swiping already does what they do. So below 640px the header becomes a
|
||||
single column, the arrows are not rendered, one card shows at 86% width so the
|
||||
next one peeks, and the affordance is carried by the right-edge scroll-fade that
|
||||
MOBILE.md documents for exactly this case:
|
||||
|
||||
```css
|
||||
mask-image: linear-gradient(to right, #000 calc(100% - 28px), transparent);
|
||||
```
|
||||
|
||||
The fade lifts at the end of the travel, where there is nothing left to hint
|
||||
at. That means the at-end state must be computed whether or not an arrow exists
|
||||
to consume it — on mobile it drives the mask alone.
|
||||
|
||||
**Every interactive element clears 44×44px**, per MOBILE.md's iOS HIG check: the
|
||||
arrow buttons and the add-to-compare button are both 44px, up from the 40px they
|
||||
were first drawn at. A card title's own box is shorter than that, but its hit
|
||||
area is the whole card through the `::after` overlay, so it passes on the target
|
||||
that actually receives the tap.
|
||||
|
||||
**The edge test needs a tolerance, and this is not fussiness.** The scroller
|
||||
carries 2px of padding so focus rings are not clipped, and scroll-snap treats
|
||||
that padding as the first card's snap position: a scroller sitting at its start
|
||||
reports `scrollLeft` of 2, not 0. Sub-pixel rounding moves it again at other
|
||||
zoom levels. Testing `scrollLeft === 0` therefore leaves the back arrow live and
|
||||
pointing nowhere on first paint — confirmed in the mockup before it was fixed.
|
||||
Both ends compare against an 8px tolerance.
|
||||
|
||||
**Selecting a school must not move the row.** Adding to the basket re-renders
|
||||
the footer; the scroll offset lives in the DOM rather than in React state, so
|
||||
the carousel must not remount or reset on that render. A reader who ticks the
|
||||
fifth school and is thrown back to the first has been punished for using the
|
||||
feature.
|
||||
|
||||
### Placement and navigation
|
||||
|
||||
Rendered as the last section **inside** `SchoolDetailShell`, from both
|
||||
`PrimarySchoolSections` and `SecondarySchoolSections`. Inside, not after, because
|
||||
the sticky nav's scroll-spy locates sections with `document.getElementById` and
|
||||
can only reach a section that lives in the shell.
|
||||
|
||||
`NearbyPlaces` stays where it is, outside the shell, immediately below. The
|
||||
resulting order — this school, then similar schools, then the places containing
|
||||
them — narrows before it widens, which is the order a reader leaves a page in.
|
||||
|
||||
`buildNavItems` and `buildSecondaryNavItems` both gain
|
||||
`{ id: 'similar', label: 'Similar schools' }`, gated on the section rendering.
|
||||
The id must match the `Section` id or the scroll-spy silently breaks.
|
||||
|
||||
### The comparison CTA
|
||||
|
||||
A plain `<a href="/compare?urns=…">`, built from this school's URN plus the
|
||||
selected ones. `/compare` already parses `urns` from the query string
|
||||
(`app/(frontend)/compare/page.tsx:55`), so this needs no new compare plumbing.
|
||||
With nothing selected the CTA is disabled; the button also adds to the shared
|
||||
basket so the site-wide comparison state stays consistent with what the page
|
||||
shows.
|
||||
|
||||
## Copy, and what the section is allowed to claim
|
||||
|
||||
**The lede never claims an intake.** It reads "Other primary schools near X." —
|
||||
one sentence, no variants. The earlier version varied the wording by tier, which
|
||||
only existed to soften a claim the section should not have been making.
|
||||
|
||||
**The heading is "Other schools nearby", not "Similar schools nearby".** The
|
||||
hard filters do guarantee a comparable set — same phase, same selectivity,
|
||||
mainstream never beside special — but nothing ranks on likeness, so the heading
|
||||
does not say it does. The nav item reads "Nearby schools" and the section id is
|
||||
`nearby`.
|
||||
|
||||
**Chips state only what is shared, and may be absent entirely.** A card with
|
||||
nothing in common renders no chip row rather than falling back to a filler.
|
||||
Since chips no longer affect the order, an empty one costs that school nothing
|
||||
except a claim it cannot support — and a Catholic parent scanning the row still
|
||||
spots "Roman Catholic" on the card that carries it, and weighs it themselves.
|
||||
|
||||
**The neighbour's metric carries no valence colour.** Green and terracotta are
|
||||
reserved site-wide for comparison against the England average. Colouring a
|
||||
neighbour's figure against this school's would read as ranking the neighbours
|
||||
against each other, which is precisely the endorsement this section must not
|
||||
make. The figure sits in neutral ink above a plain "72% at this school"
|
||||
reference line, and the reader draws their own conclusion.
|
||||
|
||||
**A missing figure reads "Not published".** Never 0, never blank, never an
|
||||
em dash. This follows the same rule the rest of the detail page uses: a school
|
||||
with no published result has not scored zero.
|
||||
|
||||
**There is no "how these are chosen" disclosure.** The method is visible in what
|
||||
the section already shows — the phase in the lede, the shared characteristics on
|
||||
each card, the distance above each name — and a collapsed panel restating it
|
||||
earns less than the space it costs.
|
||||
|
||||
**One caption line survives, and only one:** that distances are straight-line
|
||||
from the school and not road distance. This is not a method note. A reader who
|
||||
sees "0.6 miles away" and takes it for the walk has been misled by us, and no
|
||||
other element on the card corrects that. The remaining notes — that listing is
|
||||
not a recommendation, that special schools only meet special schools — are
|
||||
statements the selection rules already keep true without being narrated.
|
||||
|
||||
## Degradation
|
||||
|
||||
| Condition | Behaviour |
|
||||
|---|---|
|
||||
| `similar_schools` absent (older backend image) | no section, no nav item |
|
||||
| fewer than 2 qualifying schools | no section, no nav item |
|
||||
| this school has no coordinates | no section |
|
||||
| the helper raises | returns `[]`; the page renders without the section |
|
||||
|
||||
The helper is wrapped so a failure inside it never 500s a page that is otherwise
|
||||
complete — the posture `get_supplementary_data` already takes for its own
|
||||
queries.
|
||||
|
||||
## Testing
|
||||
|
||||
**Backend**, in a new `backend/tests/test_similar_schools.py`, against a
|
||||
synthetic frame rather than live marts:
|
||||
|
||||
- a selective school never returns a non-selective one, and vice versa
|
||||
- a special school returns only special schools; a mainstream school returns none
|
||||
- a Boys school never returns a Girls school; Mixed matches both
|
||||
- closed schools and schools without coordinates are never returned
|
||||
- results are ordered by distance ascending, always
|
||||
- a faith match never outranks a closer school (the staging defect, pinned)
|
||||
- the nearest eligible school is always present
|
||||
- more than six qualifying schools returns the six nearest
|
||||
- reach is capped per phase, and a primary beyond two miles returns `[]`
|
||||
- an all-through school is offered on both phase sides
|
||||
- a `16 plus` school is matched against secondaries and colleges, never primaries
|
||||
- `is_secondary_phase` and `PHASE_GROUPS` agree on every GIAS phase value
|
||||
- fewer than two qualifying schools returns `[]`
|
||||
- distances match a hand-computed haversine for a known pair
|
||||
|
||||
**Frontend**, in `nextjs-app/__tests__`:
|
||||
|
||||
- the section renders nothing for absent, empty and single-row inputs
|
||||
- the lede never claims a similar intake
|
||||
- an empty `shared` renders no chips rather than a filler
|
||||
- a null metric renders "Not published"
|
||||
- the nav item appears only alongside the section
|
||||
- every card is in the DOM, including the ones scrolled out of view
|
||||
- arrows render only when more than three schools were found
|
||||
|
||||
jsdom has no layout, so `scrollWidth` and `clientWidth` are both 0 there and the
|
||||
arrows' disabled state cannot be meaningfully asserted in Jest. That behaviour is
|
||||
covered in the journey instead, against a real engine, rather than by a unit test
|
||||
that would pass on a measurement that does not exist.
|
||||
|
||||
**E2E**, added to the existing journeys in `e2e/tests` in the same PR, per the
|
||||
repository's rule on user-facing behaviour:
|
||||
|
||||
- the section renders on a known staging URN, with resolving links
|
||||
- where arrows are present, the back arrow starts disabled and the forward arrow
|
||||
moves the row
|
||||
- selecting a school does not reset the scroll position
|
||||
- add-to-compare reaches `/compare` with the expected `urns`
|
||||
- at 360, 390 and 430px: no horizontal overflow, every interactive element in the
|
||||
section clears 44×44px, and no arrows are rendered
|
||||
|
||||
The E2E gate runs after merge on this project, so these journeys are not
|
||||
provable in the PR checks; the PR is verified on the unit tests, and the
|
||||
journeys are confirmed on the post-merge staging run.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- A map of the nearby schools. The section is a list; the page already has a map.
|
||||
- Autoplay, dots, or an infinite loop on the carousel. It is a short list a
|
||||
reader scans deliberately, not a banner competing for attention, and a row
|
||||
that moves on its own is a row that moves while someone is reading it.
|
||||
- Statistical neighbours on deprivation, size or cohort profile. If plain
|
||||
distance proves too blunt, that is the trigger to move this computation into a
|
||||
dbt mart — `select_nearby` is a deliberate seam for exactly that swap.
|
||||
- Precomputing neighbours in `marts.*`. Rejected for now: a new mart is inert
|
||||
until Airflow runs, so the feature would ship dark, and every tuning change to
|
||||
the rules would become a pipeline round-trip instead of a deploy. The revision
|
||||
at the top of this document is the argument for keeping that loop short.
|
||||
- Any change to `/api/compare`, the compare page, or the comparison basket.
|
||||
@@ -0,0 +1,233 @@
|
||||
# School Type Groups and a Faith Filter — Design
|
||||
|
||||
**Date:** 2026-10-02
|
||||
**Status:** shipped in PR #170; revised 2026-10-02 (one state group)
|
||||
**Scope:** search filters (`/` results toolbar and phone filter sheet), `/api/schools`, `/api/filters`
|
||||
|
||||
## Goal
|
||||
|
||||
Replace the School type filter's 34 GIAS establishment types with five groups a
|
||||
parent recognises, and add a Faith filter.
|
||||
|
||||
> **Revision, 2026-10-02.** PR #170 shipped six groups, with state schools
|
||||
> split into "academy or free school" and "council-run". They are now one,
|
||||
> "State school (free)". The split was two near-halves of the same pool, so it
|
||||
> rarely narrowed anything, and it did not follow the difference a parent feels
|
||||
> most, admissions: voluntary aided and foundation schools set their own
|
||||
> admissions, as academies do, while community and voluntary controlled
|
||||
> schools have theirs set by the council. Faith, which voluntary aided mostly
|
||||
> meant, has its own filter. The old keys `academy` and `council` resolve to
|
||||
> `state`, so their links keep working.
|
||||
|
||||
Since PR #169 the School type select offers the full GIAS list rather than the
|
||||
types in the results. That fixed the trap where choosing a type left only that
|
||||
type on offer, but it exposed the list itself: "Academy converter", "Academy
|
||||
sponsor led", "Free schools", "Foundation school", "Voluntary controlled
|
||||
school" and 29 more. These describe governance and funding. For a mainstream
|
||||
state school they change almost nothing a parent experiences, and parents
|
||||
cannot be expected to know the differences.
|
||||
|
||||
What parents actually ask is: is it free, is it mainstream, is it for children
|
||||
with special needs, is it a sixth form or college, and is it a faith school.
|
||||
The first four are the type groups below. The fifth is the real meaning behind
|
||||
"Voluntary aided" and "Voluntary controlled", but many academies are faith
|
||||
schools too, so it gets its own filter rather than hiding inside type.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Changing what a school page or result row shows. They keep the precise GIAS
|
||||
type ("Academy sponsor led"), where it is information, not a choice.
|
||||
- Removing the "not offered" types from search results. They stay reachable
|
||||
under "Any school type"; whether parents should see them at all is a
|
||||
separate decision.
|
||||
- Reordering the type options by phase (for example "Sixth form or college"
|
||||
first for 16 plus). Fixed order is simpler.
|
||||
- The rankings page, which has no type filter.
|
||||
- Changing `result_filters`. Its `school_types` key is already unread
|
||||
(docs/LEGACY_CODE.md).
|
||||
|
||||
## Type groups
|
||||
|
||||
Order is the order shown. Codes are GIAS `TypeOfEstablishment` codes
|
||||
(`backend/gias_codes.py: SCHOOL_TYPE`). Counts are staging schools on
|
||||
2026-10-02.
|
||||
|
||||
| Key | Label | GIAS codes | Schools |
|
||||
|---|---|---|---|
|
||||
| `state` | State school (free) | 28 Academy sponsor led, 34 Academy converter, 35 Free schools, 40 University technical college, 41 Studio schools, 6 City technology college, 1 Community school, 2 Voluntary aided school, 3 Voluntary controlled school, 5 Foundation school, 15 Local authority nursery school | 20,502 |
|
||||
| `independent` | Independent (fee-paying) | 11 Other independent school | 1,585 |
|
||||
| `special` | Special school (SEND) | 7 Community special, 12 Foundation special, 44 Academy special converter, 33 Academy special sponsor led, 36 Free schools special, 8 Non-maintained special, 10 Other independent special, 32 Special post 16 institution | 2,227 |
|
||||
| `post16` | Sixth form or college | 18 Further education, 31 Sixth form centres, 45 Academy 16-19 converter, 46 Academy 16 to 19 sponsor led, 39 Free schools 16 to 19 | 302 |
|
||||
| `alternative` | Alternative provision | 14 Pupil referral unit, 42 Academy AP converter, 43 Academy AP sponsor led, 38 Free schools AP | 331 |
|
||||
|
||||
**Not offered** (no group, reachable only under "Any school type"): 29 Higher
|
||||
education institutions, 27 Miscellaneous, 24 Secure units, 49 Online provider,
|
||||
57 Academy secure 16 to 19, 56 Institution funded by other government
|
||||
department. 238 schools.
|
||||
|
||||
**Excluded upstream** (never in the marts): 25, 26, 30, 37, per
|
||||
`non_england_school_type_codes` in `pipeline/transform/dbt_project.yml`.
|
||||
|
||||
### Judgement calls
|
||||
|
||||
- **Independent special schools are Special, not Independent.** They are
|
||||
usually funded by the local authority through a child's EHCP; to a parent
|
||||
they are SEND provision, not private school.
|
||||
- **UTCs, studio schools and city technology colleges are state schools.** They
|
||||
are legally academies, they are few (66 together), and a family considering
|
||||
one searches for it by name.
|
||||
- **Academies and council-run schools are one group.** See the revision note
|
||||
under Goal.
|
||||
- **Special post 16 institutions are Special, not Sixth form or college.** The
|
||||
defining fact for a parent is the SEND provision.
|
||||
- **Alternative provision is last.** Parents do not apply to it; the local
|
||||
authority places children there.
|
||||
|
||||
## Faith groups
|
||||
|
||||
Codes are GIAS `ReligiousCharacter` codes
|
||||
(`backend/gias_codes.py: RELIGIOUS_CHARACTER`). A joint school belongs to every
|
||||
faith its label names, so "Roman Catholic/Church of England" matches both
|
||||
Church of England and Roman Catholic. A generic "Christian" alongside a named
|
||||
denomination adds nothing ("Church of England/Christian" is Church of England
|
||||
only).
|
||||
|
||||
| Key | Label | GIAS codes |
|
||||
|---|---|---|
|
||||
| `none` | No religious character | 0 Does not apply, 6 None, 99 (blank), and a missing code |
|
||||
| `church_of_england` | Church of England | 2, 31 Anglican, 34 Anglican/Church of England, 20 CofE/Christian, 32 Anglican/Christian, and the joint codes 9, 10, 11, 12, 13, 19, 30, 33, 41, 48 |
|
||||
| `roman_catholic` | Roman Catholic | 3, 35 Catholic, and the joint codes 11, 13, 48 |
|
||||
| `other_christian` | Other Christian | 4 Methodist, 8 Seventh Day Adventist, 14 Quaker, 15 Christian, 16 United Reformed Church, 17 Congregational Church, 18 Free Church, 22 Greek Orthodox, 26 Moravian, 28 Inter- / non- denominational, 37 Christian/Evangelical, 38 Christian Science, 39 Christian/Methodist, 40 Christian/non-denominational, 44 Plymouth Brethren Christian Church, 45 Protestant, 46 Protestant/Evangelical, 47 Reformed Baptist, and the joint codes 9, 10, 12, 19, 30, 33, 41 |
|
||||
| `jewish` | Jewish | 5 Jewish, 36 Charadi Jewish, 43 Orthodox Jewish |
|
||||
| `muslim` | Muslim | 7 Muslim, 42 Islam, 49 Sunni Deobandi |
|
||||
| `other_faith` | Other faith | 21 Sikh, 24 Buddhist, 25 Hindu, 29 Multi-faith |
|
||||
|
||||
The joint codes: 9 CofE/Methodist, 10 Methodist/CofE, 11 CofE/RC, 12 CofE/URC,
|
||||
13 RC/CofE, 19 CofE/Free Church, 30 CofE/Methodist/URC/Baptist,
|
||||
33 Anglican/Evangelical, 41 CofE/Evangelical, 48 RC/Anglican.
|
||||
|
||||
28 "Inter- / non- denominational" is filed as Other Christian: GIAS uses it for
|
||||
Christian schools that are not tied to one church.
|
||||
|
||||
## Architecture
|
||||
|
||||
Grouping lives in the backend, not dbt. The API already translates GIAS codes
|
||||
to names when it loads the marts (`backend/data_loader.py:
|
||||
translate_gias_code_columns`), and every filter is applied to that DataFrame.
|
||||
Grouping there needs no mart change, so there is no Airflow run between merge
|
||||
and staging showing it.
|
||||
|
||||
### `backend/school_groups.py` (new)
|
||||
|
||||
- `TYPE_GROUPS`: ordered `(key, label, frozenset[int])` per type group.
|
||||
- `UNOFFERED_TYPE_CODES`: the not-offered codes, so that "every code is
|
||||
accounted for" is testable.
|
||||
- `FAITH_GROUPS`: ordered `(key, label, frozenset[int])` per faith group.
|
||||
- `type_group_for(name) -> str | None` and
|
||||
`faith_groups_for(name) -> tuple[str, ...]` (a missing or blank name gives
|
||||
`("none",)`).
|
||||
|
||||
No dependence on the generated GIAS dictionaries beyond their codes, so the
|
||||
backend/pipeline dictionary parity test is untouched.
|
||||
|
||||
### Grouping by name, at filter time
|
||||
|
||||
The groups are defined over codes but looked up by the translated name
|
||||
(`type_group_for(name)`, `faith_groups_for(name)`), when `/api/schools` and
|
||||
`/api/filters` filter. The DataFrame those endpoints read carries names only:
|
||||
`translate_gias_code_columns` replaces the codes at load, the legacy-name
|
||||
mart fallback never had codes, and the API test fixtures are written in
|
||||
names. The names come from the same dictionaries, so the lookup is exact.
|
||||
`data_loader.py` is unchanged.
|
||||
|
||||
### `/api/schools`
|
||||
|
||||
Both filters work on the name columns at request time. `_names_in_group`
|
||||
collects the distinct `school_type` or `religious_denomination` names the
|
||||
group accepts, once per distinct name rather than per row, and the rows are
|
||||
kept with `isin`. No group column is stored.
|
||||
|
||||
- `school_type`: if the value is a type group key (any case), keep the rows
|
||||
whose `school_type` name `type_group_for` puts in that group. Otherwise
|
||||
filter on the raw label exactly as today, so an old
|
||||
`?school_type=Community+school` link keeps working.
|
||||
- `faith` (new, optional, `max_length=40`, sanitised like the others): keep the
|
||||
rows whose `religious_denomination` name `faith_groups_for` puts in that
|
||||
faith (any case); for `none`, rows with a missing name too. An unknown key
|
||||
returns no schools rather than being ignored, so a typo does not silently
|
||||
show everything.
|
||||
|
||||
### `/api/filters`
|
||||
|
||||
Two new keys:
|
||||
|
||||
- `school_type_groups`: `[{value, label}]` in `TYPE_GROUPS` order, only groups
|
||||
with at least one school.
|
||||
- `faiths`: `[{value, label}]` in `FAITH_GROUPS` order, same rule.
|
||||
|
||||
`school_types` stays as it is (the raw list), so nothing that reads it breaks.
|
||||
|
||||
### Frontend
|
||||
|
||||
- `lib/types.ts`: `Filters` gains optional `school_type_groups` and `faiths`
|
||||
(`{ value: string; label: string }[]`). Optional, so the empty fallbacks in
|
||||
`app/(frontend)/page.tsx` and `rankings/page.tsx` stay valid.
|
||||
- `app/(frontend)/page.tsx`: reads `faith` from the search params, counts it
|
||||
in `hasSearchParams`, and passes it to `fetchSchools`. `SchoolSearchParams`
|
||||
in `lib/types.ts` gains `faith`. HomeView's load-more already forwards every
|
||||
URL param. HomeView's `isSearchActive` is left as it is: like phase and
|
||||
gender, faith narrows a search rather than starting one.
|
||||
- `components/FilterBar.tsx`:
|
||||
- The School type select's options become `filters.school_type_groups`;
|
||||
"Any school type" stays first. With no groups (the API failed), the select
|
||||
is left out, as Gender and Admissions already are.
|
||||
- A new **Faith** select (`aria-label="Faith"`, "Any faith or none" first)
|
||||
from `filters.faiths`. On desktop it goes in the More filters panel, after
|
||||
Local authority. In the phone sheet it comes after School type.
|
||||
- `faith` joins `FILTER_KEYS` (chip and Filters count) and the More filters
|
||||
count, and `faith=` joins `filters_active` in the `search_submitted`
|
||||
analytics event.
|
||||
- Chip labels come from the option lists: "Special school (SEND)",
|
||||
"Roman Catholic". An old raw-label `school_type` shows its raw label.
|
||||
- `lib/utils.ts: isSpecialSchool` is unchanged. It reads a school's raw
|
||||
`school_type`, which still arrives.
|
||||
|
||||
## Testing
|
||||
|
||||
**Backend (pytest):**
|
||||
|
||||
- Every `SCHOOL_TYPE` code is in exactly one type group, in
|
||||
`UNOFFERED_TYPE_CODES`, or in `non_england_school_type_codes`. A new DfE
|
||||
code fails this test instead of silently vanishing from the filter.
|
||||
- Every `RELIGIOUS_CHARACTER` code maps to at least one faith group.
|
||||
- The joint codes map to each faith they name (11 and 48 → CofE and RC;
|
||||
9 → CofE and Other Christian) and 20 → CofE only.
|
||||
- `/api/schools?school_type=special` returns only special-group schools;
|
||||
`?school_type=Community+school` still filters by label.
|
||||
- `/api/schools?faith=roman_catholic` returns only matching schools, a joint
|
||||
school included; `?faith=nonsense` returns none.
|
||||
- `/api/filters` lists the groups in order and leaves out an empty one.
|
||||
- Run via uv, as the backend tests always are.
|
||||
|
||||
**Frontend (Jest):**
|
||||
|
||||
- School type offers the six group labels, not raw types.
|
||||
- Faith offers its options, in the panel and in the sheet.
|
||||
- A faith filter shows as a chip, counts on Filters and More filters, and
|
||||
clears with Clear all.
|
||||
- No groups or faiths in `filters` → the select is left out.
|
||||
|
||||
**E2E (`e2e/tests/journeys.spec.ts`):**
|
||||
|
||||
- A journey that picks "Special school (SEND)" and "Roman Catholic" from the
|
||||
selects, then checks through `/api/schools` with the same params that every
|
||||
returned school's `school_type` is a special type and its
|
||||
`religious_denomination` names Catholic. Data-invariant: it asserts the
|
||||
property, not a count.
|
||||
- It can only pass after merge; the staging E2E gate runs post-merge.
|
||||
|
||||
## Rollout
|
||||
|
||||
One PR; backend and frontend deploy together on merge. The frontend tolerates
|
||||
an API without the new keys (the selects are left out), so the order the
|
||||
containers update in does not matter.
|
||||
+1659
-21
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,41 @@
|
||||
import { test, expect, Route } from '@playwright/test';
|
||||
|
||||
test('the deployed frontend and backend report the tested build', async ({ request }) => {
|
||||
const response = await request.get('/release.json');
|
||||
expect(response.ok()).toBeTruthy();
|
||||
expect(response.headers()['cache-control']).toContain('no-store');
|
||||
const identity = await response.json();
|
||||
expect(identity.frontend).toEqual(identity.backend);
|
||||
expect(identity.frontend.sha).toMatch(/^[a-f0-9]{40}$/);
|
||||
expect(identity.frontend.build_id).toMatch(/^[a-f0-9]{32}$/);
|
||||
if (process.env.EXPECTED_SHA) expect(identity.frontend.sha).toBe(process.env.EXPECTED_SHA);
|
||||
if (process.env.EXPECTED_BUILD_ID) expect(identity.frontend.build_id).toBe(process.env.EXPECTED_BUILD_ID);
|
||||
});
|
||||
|
||||
test('changing search while loading another page does not append old results', async ({ page }) => {
|
||||
await page.goto('/?phase=primary');
|
||||
await expect(page.getByRole('button', { name: 'Load more schools' })).toBeVisible();
|
||||
let received!: (route: Route) => void;
|
||||
const pending = new Promise<Route>(resolve => { received = resolve; });
|
||||
await page.route('**/api/schools?**', async route => {
|
||||
if (new URL(route.request().url()).searchParams.get('page') === '2') {
|
||||
received(route);
|
||||
return;
|
||||
}
|
||||
await route.continue();
|
||||
});
|
||||
await page.getByRole('button', { name: 'Load more schools' }).click();
|
||||
const oldRequest = await pending;
|
||||
const search = page.getByPlaceholder('School name or postcode').first();
|
||||
await search.fill('secondary');
|
||||
await search.press('Enter');
|
||||
await page.waitForURL(/search=secondary/);
|
||||
// A cancelled fetch may prevent route fulfilment altogether; either way,
|
||||
// this deliberately late response must not become part of the new results.
|
||||
await oldRequest.fulfill({ json: {
|
||||
schools: [{ urn: 999998, school_name: 'P1 stale result sentinel', phase: 'Primary' }],
|
||||
total: 2, page: 2, page_size: 1, total_pages: 2,
|
||||
} }).catch(() => {});
|
||||
await expect(page.getByText('P1 stale result sentinel')).toHaveCount(0);
|
||||
await expect(page.getByRole('button', { name: 'Loading...' })).toHaveCount(0);
|
||||
});
|
||||
@@ -0,0 +1,373 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>Similar Schools Nearby</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=Manrope:wght@500;600;700&display=swap" rel="stylesheet">
|
||||
<style>
|
||||
/* Tokens copied verbatim from nextjs-app/app/(frontend)/globals.css so this
|
||||
mockup cannot drift from the shipped palette. Light values first, dark
|
||||
under prefers-color-scheme, both overridable by the theme switch. */
|
||||
:root {
|
||||
color-scheme: light;
|
||||
--bg-primary:#FAFAF8; --bg-secondary:#F5EFE6; --bg-card:#FFFFFF;
|
||||
--text-primary:#1C2731; --text-secondary:#4A5560; --text-muted:#5F6A75;
|
||||
--border:#E5E7EB; --border-strong:#D3D7DD;
|
||||
--brand:#0F766E; --brand-strong:#0C5F58; --brand-bg:rgba(15,118,110,.10); --brand-on:#FFFFFF;
|
||||
--action:#BE3C27; --action-strong:#A33320; --action-on:#FFFFFF;
|
||||
--sand:#F5EFE6;
|
||||
--font-display:Manrope,-apple-system,BlinkMacSystemFont,sans-serif;
|
||||
--font-ui:Inter,-apple-system,BlinkMacSystemFont,sans-serif;
|
||||
--radius-md:8px; --radius-lg:16px;
|
||||
--shadow:0 1px 2px rgba(28,39,49,.06),0 1px 3px rgba(28,39,49,.05);
|
||||
}
|
||||
@media (prefers-color-scheme: dark) {
|
||||
:root:not([data-theme="light"]) {
|
||||
color-scheme: dark;
|
||||
--bg-primary:#111A20; --bg-secondary:#16222A; --bg-card:#18242C;
|
||||
--text-primary:#E9EEF0; --text-secondary:#B4C2C7; --text-muted:#8B9AA1;
|
||||
--border:#26343D; --border-strong:#35454F;
|
||||
--brand:#5FC7BB; --brand-strong:#7BD6CC; --brand-bg:rgba(95,199,187,.14); --brand-on:#0A1418;
|
||||
--action:#F08A72; --action-strong:#F5A492; --action-on:#241009;
|
||||
--sand:#1B2730;
|
||||
--shadow:0 1px 2px rgba(0,0,0,.3),0 1px 3px rgba(0,0,0,.25);
|
||||
}
|
||||
}
|
||||
:root[data-theme="dark"] {
|
||||
color-scheme: dark;
|
||||
--bg-primary:#111A20; --bg-secondary:#16222A; --bg-card:#18242C;
|
||||
--text-primary:#E9EEF0; --text-secondary:#B4C2C7; --text-muted:#8B9AA1;
|
||||
--border:#26343D; --border-strong:#35454F;
|
||||
--brand:#5FC7BB; --brand-strong:#7BD6CC; --brand-bg:rgba(95,199,187,.14); --brand-on:#0A1418;
|
||||
--action:#F08A72; --action-strong:#F5A492; --action-on:#241009;
|
||||
--sand:#1B2730;
|
||||
--shadow:0 1px 2px rgba(0,0,0,.3),0 1px 3px rgba(0,0,0,.25);
|
||||
}
|
||||
* { box-sizing:border-box; }
|
||||
body {
|
||||
margin:0; padding:32px 16px 80px; background:var(--bg-primary);
|
||||
color:var(--text-primary); font:15px/1.55 var(--font-ui);
|
||||
-webkit-font-smoothing:antialiased;
|
||||
}
|
||||
.page { max-width:960px; margin:0 auto; }
|
||||
.page > header { margin-bottom:28px; display:flex; flex-wrap:wrap; gap:16px; align-items:flex-start; justify-content:space-between; }
|
||||
.page > header h1 { font:700 25px/1.25 var(--font-display); letter-spacing:-.6px; margin:0 0 6px; }
|
||||
.page > header p { margin:0; color:var(--text-muted); font-size:14px; max-width:60ch; }
|
||||
.theme-switch { border:1px solid var(--border-strong); background:var(--bg-card); color:var(--text-secondary); border-radius:999px; padding:8px 14px; font:500 13px var(--font-ui); cursor:pointer; min-height:44px; }
|
||||
.theme-switch:hover { border-color:var(--brand); color:var(--brand); }
|
||||
|
||||
/* ── The page context each variant is shown inside ─────────────────── */
|
||||
.variant { margin-bottom:40px; }
|
||||
.variant > .context { padding:0 4px 12px; }
|
||||
.variant .eyebrow { margin:0 0 4px; font-size:12px; letter-spacing:.04em; text-transform:uppercase; color:var(--text-muted); }
|
||||
.variant .context h2 { font:600 18px/1.35 var(--font-display); margin:0; color:var(--text-secondary); }
|
||||
.variant .note { margin:10px 4px 0; font-size:12.5px; color:var(--text-muted); }
|
||||
.variant .note b { color:var(--text-secondary); font-weight:600; }
|
||||
|
||||
/* ── The section itself — mirrors components/school/Section ────────── */
|
||||
.card {
|
||||
background:var(--bg-card); border:1px solid var(--border);
|
||||
border-radius:var(--radius-lg); padding:28px; box-shadow:var(--shadow);
|
||||
}
|
||||
.top { display:flex; align-items:flex-start; justify-content:space-between; gap:16px; }
|
||||
.top h2 { font:700 22px/1.25 var(--font-display); letter-spacing:-.4px; margin:0; }
|
||||
.lede { margin:8px 0 20px; color:var(--text-secondary); font-size:14.5px; max-width:64ch; }
|
||||
|
||||
/* ── Carousel ───────────────────────────────────────────────────────
|
||||
Every card is in the DOM and in the initial HTML — the arrows scroll a
|
||||
list, they do not swap a view. That keeps all six links crawlable and
|
||||
keeps the section usable with no JavaScript, where it degrades to a
|
||||
plain horizontally scrollable row. */
|
||||
.arrows { display:flex; gap:8px; flex:none; }
|
||||
.arrow {
|
||||
width:44px; height:44px; display:grid; place-items:center; cursor:pointer;
|
||||
border:1px solid var(--border-strong); border-radius:999px;
|
||||
background:var(--bg-card); color:var(--brand);
|
||||
}
|
||||
.arrow:hover:not(:disabled) { border-color:var(--brand); background:var(--brand-bg); }
|
||||
.arrow:disabled { opacity:.35; cursor:default; }
|
||||
.arrow:focus-visible { outline:2px solid var(--brand); outline-offset:2px; }
|
||||
.arrow svg { width:17px; height:17px; }
|
||||
|
||||
.scroller {
|
||||
display:grid; grid-auto-flow:column;
|
||||
grid-auto-columns:calc((100% - 28px) / 3);
|
||||
gap:14px; overflow-x:auto; scroll-snap-type:x mandatory;
|
||||
padding:2px; margin:-2px; /* room for focus rings */
|
||||
scrollbar-width:none; -ms-overflow-style:none;
|
||||
list-style:none;
|
||||
}
|
||||
.scroller::-webkit-scrollbar { display:none; }
|
||||
.scroller:focus-visible { outline:2px solid var(--brand); outline-offset:4px; border-radius:var(--radius-md); }
|
||||
@media (max-width:820px) { .scroller { grid-auto-columns:calc((100% - 14px) / 2); } }
|
||||
/* Touch widths: the arrows would squeeze the lede into a four-line column for a
|
||||
control that swiping already provides, so they go and the documented
|
||||
right-edge fade carries the affordance instead (MOBILE.md). The fade lifts at
|
||||
the end of the travel, where there is nothing more to hint at. */
|
||||
@media (max-width:640px) {
|
||||
.top { display:block; }
|
||||
.arrows { display:none; }
|
||||
.scroller { grid-auto-columns:86%; mask-image:linear-gradient(to right, #000 calc(100% - 28px), transparent); }
|
||||
.scroller[data-at-end=true] { mask-image:none; }
|
||||
.card { padding:20px; }
|
||||
}
|
||||
|
||||
.school {
|
||||
position:relative; display:flex; flex-direction:column; scroll-snap-align:start;
|
||||
border:1px solid var(--border); border-radius:var(--radius-md);
|
||||
padding:16px; background:var(--bg-card);
|
||||
}
|
||||
.school:has(.add[aria-pressed=true]) { border-color:var(--brand); background:var(--brand-bg); }
|
||||
.distance { display:flex; align-items:center; gap:5px; font-size:12px; color:var(--text-muted); margin:0 0 10px; }
|
||||
.distance svg { width:13px; height:13px; flex:none; }
|
||||
.school h3 { font:600 16px/1.35 var(--font-display); margin:0 0 6px; }
|
||||
/* The whole card is the link target; the button sits above it on z-index so
|
||||
it stays independently clickable. */
|
||||
.school h3 a { color:var(--text-primary); text-decoration:none; }
|
||||
.school h3 a::after { content:""; position:absolute; inset:0; border-radius:var(--radius-md); }
|
||||
.school:hover { border-color:var(--border-strong); }
|
||||
.school h3 a:hover { color:var(--brand); text-decoration:underline; }
|
||||
.school h3 a:focus-visible { outline:none; }
|
||||
.school:has(h3 a:focus-visible) { outline:2px solid var(--brand); outline-offset:2px; }
|
||||
.meta { margin:0 0 12px; font-size:12.5px; color:var(--text-muted); }
|
||||
.shared { display:flex; flex-wrap:wrap; gap:6px; margin:0 0 14px; padding:0; list-style:none; }
|
||||
.shared li { font-size:11.5px; line-height:1.4; padding:4px 8px; border-radius:999px; background:var(--brand-bg); color:var(--brand); border:1px solid transparent; }
|
||||
.shared li.loose { background:transparent; color:var(--text-muted); border-color:var(--border); }
|
||||
.metric { margin-top:auto; padding-top:13px; border-top:1px solid var(--border); }
|
||||
.value { font:700 26px/1.1 var(--font-display); letter-spacing:-.6px; margin:0; }
|
||||
.value.absent { font-size:15px; font-weight:600; color:var(--text-muted); letter-spacing:0; }
|
||||
.metric .label { margin:4px 0 0; font-size:12px; color:var(--text-secondary); }
|
||||
.metric .ref { margin:2px 0 0; font-size:12px; color:var(--text-muted); }
|
||||
.add {
|
||||
position:relative; z-index:1; margin-top:14px; width:100%; min-height:44px;
|
||||
font:500 13px var(--font-ui); cursor:pointer; border-radius:var(--radius-md);
|
||||
border:1px solid var(--border-strong); background:var(--bg-card); color:var(--brand);
|
||||
}
|
||||
.add:hover { border-color:var(--brand); background:var(--brand-bg); }
|
||||
.add[aria-pressed=true] { border-color:var(--brand); background:var(--brand-bg); font-weight:600; }
|
||||
.add:focus-visible { outline:2px solid var(--brand); outline-offset:2px; }
|
||||
|
||||
.footer {
|
||||
display:flex; flex-wrap:wrap; align-items:center; justify-content:space-between;
|
||||
gap:14px; margin-top:20px; padding-top:18px; border-top:1px solid var(--border);
|
||||
}
|
||||
.footer p { margin:0; font-size:12.5px; color:var(--text-muted); }
|
||||
.footer strong { display:block; font:600 14px var(--font-ui); color:var(--text-primary); }
|
||||
/* Coral: the one decisive action in this section, and there is only one. */
|
||||
.compare {
|
||||
min-height:44px; padding:0 20px; border-radius:var(--radius-md); cursor:pointer;
|
||||
font:600 14px var(--font-ui); background:var(--action); color:var(--action-on);
|
||||
border:1px solid var(--action); text-decoration:none; display:inline-flex; align-items:center; gap:8px;
|
||||
}
|
||||
.compare:hover { background:var(--action-strong); border-color:var(--action-strong); }
|
||||
.compare[aria-disabled=true] { opacity:.45; pointer-events:none; }
|
||||
.caption { margin:16px 0 0; font-size:11.5px; color:var(--text-muted); }
|
||||
.sr { position:absolute; width:1px; height:1px; padding:0; margin:-1px; overflow:hidden; clip:rect(0 0 0 0); white-space:nowrap; border:0; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="page">
|
||||
<header>
|
||||
<div>
|
||||
<h1>Similar schools nearby</h1>
|
||||
<p>A new section on the school detail page. Fictional schools and figures; shipped
|
||||
colour, type and section shell taken from <code>globals.css</code>.</p>
|
||||
</div>
|
||||
<button class="theme-switch" type="button" id="theme">Dark theme</button>
|
||||
</header>
|
||||
<div id="variants"></div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
const PIN = '<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><path d="M20 10c0 6-8 12-8 12s-8-6-8-12a8 8 0 0 1 16 0Z"/><circle cx="12" cy="10" r="3"/></svg>';
|
||||
const CHEV = (dir) => `<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><path d="${dir === 'prev' ? 'M15 18l-6-6 6-6' : 'M9 18l6-6-6-6'}"/></svg>`;
|
||||
|
||||
const variants = [
|
||||
{
|
||||
id: 'dense',
|
||||
eyebrow: 'Variant 1 · Dense urban primary — six matches, carousel active',
|
||||
context: 'Meadowbrook Primary School — Ages 4–11 · Mixed · No religious character · Community school',
|
||||
lede: 'Other primary schools near Meadowbrook Primary School, with a similar intake.',
|
||||
metric: 'Reading, writing & maths',
|
||||
caption: 'Meeting the expected standard at key stage 2, 2025.',
|
||||
thisValue: '72%',
|
||||
note: 'Fourteen schools cleared <b>tier 1</b> within three miles, so the section takes the six nearest and stops there. The arrows scroll a list that is entirely in the HTML — all six links are crawlable, and with JavaScript off the row still scrolls.',
|
||||
schools: [
|
||||
{ name:'Willow Lane Primary School', distance:'0.4', meta:'Community school · Ages 4–11', shared:['Mixed','No religious character'], value:'74%' },
|
||||
{ name:'Oakfield Primary School', distance:'0.6', meta:'Academy converter · Ages 3–11', shared:['Mixed','No religious character'], value:'69%' },
|
||||
{ name:'Brookside Primary School', distance:'0.9', meta:'Community school · Ages 4–11', shared:['Mixed','No religious character'], value:'Not published' },
|
||||
{ name:'Hollytree Primary School', distance:'1.3', meta:'Academy converter · Ages 4–11', shared:['Mixed','No religious character'], value:'81%' },
|
||||
{ name:'Marsh Green Primary School', distance:'1.8', meta:'Community school · Ages 3–11', shared:['Mixed','No religious character'], value:'64%' },
|
||||
{ name:'Kingsway Primary School', distance:'2.2', meta:'Foundation school · Ages 4–11', shared:['Mixed','No religious character'], value:'77%' },
|
||||
],
|
||||
},
|
||||
{
|
||||
id: 'secondary',
|
||||
eyebrow: 'Variant 2 · Secondary — four matches, mixed tiers',
|
||||
context: 'Meadowbrook High School — Ages 11–18 · Mixed · Non-selective · Academy',
|
||||
lede: 'Other secondary schools near Meadowbrook High School, with a similar intake.',
|
||||
metric: 'Attainment 8',
|
||||
caption: 'Average GCSE attainment score across eight qualifications, 2025.',
|
||||
thisValue: '51.2',
|
||||
note: 'Only two schools cleared tier 1, so the search widened to <b>tier 2</b> and found two more. It stops there rather than widening again to reach six — tiers relax to reach a usable set, never to fill the last slots. Selectivity never relaxes, so no grammar school can appear here.',
|
||||
schools: [
|
||||
{ name:'Rivermead High School', distance:'0.9', meta:'Academy converter · Ages 11–18', shared:['Mixed','Non-selective','No religious character'], value:'52.8' },
|
||||
{ name:'Oakfield Academy', distance:'1.7', meta:'Academy sponsor led · Ages 11–16', shared:['Mixed','Non-selective','No religious character'], value:'49.6' },
|
||||
{ name:'St Aidan’s Catholic High School', distance:'2.4', meta:'Voluntary aided · Ages 11–18', shared:['Mixed','Non-selective'], value:'53.4' },
|
||||
{ name:'Parkside Community School', distance:'3.8', meta:'Community school · Ages 11–16', shared:['Mixed','Non-selective'], value:'50.9' },
|
||||
],
|
||||
},
|
||||
{
|
||||
id: 'sparse',
|
||||
eyebrow: 'Variant 3 · Rural — two matches, no arrows',
|
||||
context: 'Little Ashby Church of England Primary School — Ages 4–11 · Mixed · Church of England · Voluntary controlled',
|
||||
lede: 'Other primary schools near Little Ashby Church of England Primary School.',
|
||||
metric: 'Reading, writing & maths',
|
||||
caption: 'Meeting the expected standard at key stage 2, 2025.',
|
||||
thisValue: '66%',
|
||||
note: 'Nothing matched on religious character within range. At <b>tier 3</b> the lede drops the phrase “with a similar intake” and the chips fall back to the plain phase. Two cards fit the row, so the arrows are not rendered at all. One school fewer and the section would not render either.',
|
||||
schools: [
|
||||
{ name:'Great Marden Primary School', distance:'4.2', meta:'Community school · Ages 4–11', shared:['Primary school'], loose:true, value:'71%' },
|
||||
{ name:'Ashby Vale Academy', distance:'7.8', meta:'Academy converter · Ages 4–11', shared:['Primary school'], loose:true, value:'58%' },
|
||||
],
|
||||
},
|
||||
];
|
||||
|
||||
const selected = Object.fromEntries(variants.map((v) => [v.id, new Set()]));
|
||||
|
||||
function card(v, s, i) {
|
||||
const on = selected[v.id].has(i);
|
||||
const absent = s.value === 'Not published';
|
||||
return `
|
||||
<li class="school">
|
||||
<p class="distance">${PIN}${s.distance} miles away</p>
|
||||
<h3><a href="#">${s.name}</a></h3>
|
||||
<p class="meta">${s.meta}</p>
|
||||
<ul class="shared">${s.shared.map((c) => `<li class="${s.loose ? 'loose' : ''}">${c}</li>`).join('')}</ul>
|
||||
<div class="metric">
|
||||
<p class="value ${absent ? 'absent' : ''}">${s.value}</p>
|
||||
<p class="label">${v.metric}</p>
|
||||
<p class="ref">${v.thisValue} at this school</p>
|
||||
</div>
|
||||
<button class="add" type="button" data-variant="${v.id}" data-index="${i}" aria-pressed="${on}">
|
||||
${on ? '✓ Added to compare' : '+ Add to compare'}<span class="sr"> — ${s.name}</span>
|
||||
</button>
|
||||
</li>`;
|
||||
}
|
||||
|
||||
function render() {
|
||||
document.getElementById('variants').innerHTML = variants.map((v) => {
|
||||
const count = selected[v.id].size;
|
||||
// Three fit the row, so anything more is what the arrows are for.
|
||||
const scrollable = v.schools.length > 3;
|
||||
return `
|
||||
<section class="variant">
|
||||
<div class="context">
|
||||
<p class="eyebrow">${v.eyebrow}</p>
|
||||
<h2>${v.context}</h2>
|
||||
</div>
|
||||
<div class="card">
|
||||
<div class="top">
|
||||
<div>
|
||||
<h2 id="h-${v.id}">Similar schools nearby</h2>
|
||||
<p class="lede">${v.lede}</p>
|
||||
</div>
|
||||
${scrollable ? `<div class="arrows">
|
||||
<button class="arrow" type="button" data-scroll="prev" data-variant="${v.id}" aria-label="Previous schools" aria-controls="sc-${v.id}">${CHEV('prev')}</button>
|
||||
<button class="arrow" type="button" data-scroll="next" data-variant="${v.id}" aria-label="More schools" aria-controls="sc-${v.id}">${CHEV('next')}</button>
|
||||
</div>` : ''}
|
||||
</div>
|
||||
|
||||
<ul class="scroller" id="sc-${v.id}" ${scrollable ? `tabindex="0" role="group" aria-labelledby="h-${v.id}"` : ''}>
|
||||
${v.schools.map((s, i) => card(v, s, i)).join('')}
|
||||
</ul>
|
||||
|
||||
<div class="footer">
|
||||
<p aria-live="polite">
|
||||
<strong>${count ? `${count} school${count === 1 ? '' : 's'} selected` : 'Compare side by side'}</strong>
|
||||
${count ? 'This school is included automatically.' : 'Add a school to compare it with this one.'}
|
||||
</p>
|
||||
<a class="compare" href="#" aria-disabled="${count ? 'false' : 'true'}">
|
||||
${count ? `Compare ${count + 1} schools` : 'Compare'} →
|
||||
</a>
|
||||
</div>
|
||||
<p class="caption">Distances are straight-line from this school, not road distance.
|
||||
${v.caption} Fictional schools and figures for this mockup.</p>
|
||||
</div>
|
||||
<p class="note">${v.note}</p>
|
||||
</section>`;
|
||||
}).join('');
|
||||
|
||||
variants.forEach((v) => {
|
||||
const scroller = document.getElementById(`sc-${v.id}`);
|
||||
if (scroller) syncArrows(v.id, scroller);
|
||||
});
|
||||
}
|
||||
|
||||
/** An arrow that scrolls nowhere is a dead control, so each end disables its own.
|
||||
*
|
||||
* EDGE is not paranoia. The scroller carries 2px of padding so focus rings are
|
||||
* not clipped, and scroll-snap treats that padding as the first card's snap
|
||||
* position — so a scroller sitting at its start reports scrollLeft 2, not 0.
|
||||
* Sub-pixel rounding at other zoom levels moves it again. Testing against an
|
||||
* exact 0 leaves the back arrow live at the start, pointing nowhere. */
|
||||
const EDGE = 8;
|
||||
|
||||
function syncArrows(id, scroller) {
|
||||
const max = scroller.scrollWidth - scroller.clientWidth;
|
||||
const atStart = scroller.scrollLeft <= EDGE;
|
||||
const atEnd = scroller.scrollLeft >= max - EDGE;
|
||||
|
||||
// Drives the mobile scroll-fade, so it is computed even where no arrow is
|
||||
// rendered to consume it.
|
||||
scroller.dataset.atEnd = String(atEnd);
|
||||
|
||||
const prev = document.querySelector(`.arrow[data-scroll="prev"][data-variant="${id}"]`);
|
||||
const next = document.querySelector(`.arrow[data-scroll="next"][data-variant="${id}"]`);
|
||||
if (!prev || !next) return;
|
||||
prev.disabled = atStart;
|
||||
next.disabled = atEnd;
|
||||
}
|
||||
|
||||
document.getElementById('variants').addEventListener('click', (event) => {
|
||||
const arrow = event.target.closest('.arrow');
|
||||
if (arrow) {
|
||||
const scroller = document.getElementById(`sc-${arrow.dataset.variant}`);
|
||||
// A page is what the reader can see, so the viewport is the step.
|
||||
scroller.scrollBy({ left: (arrow.dataset.scroll === 'next' ? 1 : -1) * scroller.clientWidth, behavior: 'smooth' });
|
||||
return;
|
||||
}
|
||||
const button = event.target.closest('.add');
|
||||
if (!button) return;
|
||||
const { variant, index } = button.dataset;
|
||||
const set = selected[variant];
|
||||
const i = Number(index);
|
||||
// Scroll position is DOM state, not React state; keep it across the re-render.
|
||||
const offset = document.getElementById(`sc-${variant}`).scrollLeft;
|
||||
set.has(i) ? set.delete(i) : set.add(i);
|
||||
render();
|
||||
const scroller = document.getElementById(`sc-${variant}`);
|
||||
scroller.scrollLeft = offset;
|
||||
syncArrows(variant, scroller);
|
||||
document.querySelector(`.add[data-variant="${variant}"][data-index="${index}"]`).focus();
|
||||
}, true);
|
||||
|
||||
document.getElementById('variants').addEventListener('scroll', (event) => {
|
||||
const scroller = event.target.closest('.scroller');
|
||||
if (scroller) syncArrows(scroller.id.replace('sc-', ''), scroller);
|
||||
}, true);
|
||||
|
||||
const themeButton = document.getElementById('theme');
|
||||
themeButton.addEventListener('click', () => {
|
||||
const dark = document.documentElement.dataset.theme === 'dark';
|
||||
document.documentElement.dataset.theme = dark ? 'light' : 'dark';
|
||||
themeButton.textContent = dark ? 'Dark theme' : 'Light theme';
|
||||
});
|
||||
|
||||
render();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
+12
-4
@@ -1,8 +1,16 @@
|
||||
# API Configuration
|
||||
NEXT_PUBLIC_API_URL=http://localhost:8000/api
|
||||
# Browser requests use the same-origin Next.js proxy.
|
||||
NEXT_PUBLIC_API_URL=/api
|
||||
|
||||
# Production API URL (for deployment)
|
||||
# NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api
|
||||
# Absolute URL for server-side fetching and the proxy; include /api.
|
||||
# In the managed container network this is http://backend:80/api (staging differs).
|
||||
FASTAPI_URL=http://localhost:8000/api
|
||||
|
||||
# Payload CMS runtime configuration. Use the managed environment's database;
|
||||
# Payload owns the payload schema, independently of the school marts.
|
||||
DATABASE_URL=postgresql://schoolcompare:CHANGE_THIS_PASSWORD@localhost:5432/schoolcompare
|
||||
# Generate a secret: python -c "import secrets; print(secrets.token_urlsafe(32))"
|
||||
# Use distinct secrets for staging and production.
|
||||
PAYLOAD_SECRET=CHANGE_THIS_TO_A_SECURE_RANDOM_SECRET
|
||||
|
||||
# Node Environment
|
||||
NODE_ENV=development
|
||||
@@ -39,3 +39,4 @@ yarn-error.log*
|
||||
# typescript
|
||||
*.tsbuildinfo
|
||||
next-env.d.ts
|
||||
|
||||
+12
-288
@@ -1,291 +1,15 @@
|
||||
# Deployment Guide
|
||||
# Frontend deployment
|
||||
|
||||
This guide covers deployment options for the SchoolCompare Next.js application.
|
||||
Next.js and Payload run in the same frontend container. The maintained deployment
|
||||
procedure is [docs/DEPLOY.md](../docs/DEPLOY.md), with the production and staging
|
||||
Portainer compose files at the repository root.
|
||||
|
||||
## Deployment Options
|
||||
The frontend Dockerfile builds a standalone Next.js image. Runtime configuration
|
||||
supplies `FASTAPI_URL`, `DATABASE_URL` and `PAYLOAD_SECRET`; uploaded CMS media is
|
||||
persisted in a volume. Promote the built image through the repository's Gitea
|
||||
workflow after human staging approval.
|
||||
|
||||
### Option 1: Vercel (Recommended for Next.js)
|
||||
|
||||
Vercel is the easiest and most optimized platform for Next.js applications.
|
||||
|
||||
#### Steps:
|
||||
|
||||
1. **Install Vercel CLI**:
|
||||
```bash
|
||||
npm install -g vercel
|
||||
```
|
||||
|
||||
2. **Login to Vercel**:
|
||||
```bash
|
||||
vercel login
|
||||
```
|
||||
|
||||
3. **Deploy**:
|
||||
```bash
|
||||
vercel --prod
|
||||
```
|
||||
|
||||
4. **Configure Environment Variables** in Vercel dashboard:
|
||||
- `NEXT_PUBLIC_API_URL`: Your FastAPI endpoint (e.g., `https://api.schoolcompare.co.uk/api`)
|
||||
- `FASTAPI_URL`: Same as above for server-side requests
|
||||
|
||||
#### Benefits:
|
||||
- Automatic HTTPS
|
||||
- Global CDN
|
||||
- Zero-config deployment
|
||||
- Automatic preview deployments
|
||||
- Built-in analytics
|
||||
|
||||
---
|
||||
|
||||
### Option 2: Docker (Self-hosted)
|
||||
|
||||
Deploy using Docker containers for full control.
|
||||
|
||||
#### Prerequisites:
|
||||
- Docker 20+
|
||||
- Docker Compose 2+
|
||||
|
||||
#### Steps:
|
||||
|
||||
1. **Build Docker Image**:
|
||||
```bash
|
||||
docker build -t schoolcompare-nextjs:latest .
|
||||
```
|
||||
|
||||
2. **Run with Docker Compose**:
|
||||
```bash
|
||||
# Create .env file with production variables
|
||||
echo "NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api" > .env
|
||||
echo "FASTAPI_URL=http://backend:8000/api" >> .env
|
||||
|
||||
# Start services
|
||||
docker-compose up -d
|
||||
```
|
||||
|
||||
3. **Verify Deployment**:
|
||||
```bash
|
||||
curl http://localhost:3000
|
||||
```
|
||||
|
||||
#### Environment Variables:
|
||||
- `NEXT_PUBLIC_API_URL`: Public API endpoint (client-side)
|
||||
- `FASTAPI_URL`: Internal API endpoint (server-side)
|
||||
- `NODE_ENV`: `production`
|
||||
|
||||
---
|
||||
|
||||
### Option 3: PM2 (Node.js Process Manager)
|
||||
|
||||
Deploy directly on a Node.js server using PM2.
|
||||
|
||||
#### Prerequisites:
|
||||
- Node.js 24+
|
||||
- PM2 (`npm install -g pm2`)
|
||||
|
||||
#### Steps:
|
||||
|
||||
1. **Build Application**:
|
||||
```bash
|
||||
npm run build
|
||||
```
|
||||
|
||||
2. **Create PM2 Ecosystem File** (`ecosystem.config.js`):
|
||||
```javascript
|
||||
module.exports = {
|
||||
apps: [{
|
||||
name: 'schoolcompare-nextjs',
|
||||
script: 'npm',
|
||||
args: 'start',
|
||||
cwd: '/path/to/nextjs-app',
|
||||
instances: 'max',
|
||||
exec_mode: 'cluster',
|
||||
env: {
|
||||
NODE_ENV: 'production',
|
||||
PORT: 3000,
|
||||
NEXT_PUBLIC_API_URL: 'https://api.schoolcompare.co.uk/api',
|
||||
FASTAPI_URL: 'http://localhost:8000/api',
|
||||
},
|
||||
}],
|
||||
};
|
||||
```
|
||||
|
||||
3. **Start with PM2**:
|
||||
```bash
|
||||
pm2 start ecosystem.config.js
|
||||
pm2 save
|
||||
pm2 startup
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Option 4: Nginx Reverse Proxy
|
||||
|
||||
Use Nginx as a reverse proxy in front of Next.js.
|
||||
|
||||
#### Nginx Configuration:
|
||||
|
||||
```nginx
|
||||
server {
|
||||
listen 80;
|
||||
server_name schoolcompare.co.uk;
|
||||
|
||||
# Redirect to HTTPS
|
||||
return 301 https://$server_name$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl http2;
|
||||
server_name schoolcompare.co.uk;
|
||||
|
||||
# SSL Configuration
|
||||
ssl_certificate /etc/ssl/certs/schoolcompare.crt;
|
||||
ssl_certificate_key /etc/ssl/private/schoolcompare.key;
|
||||
|
||||
# Security Headers
|
||||
# frame-ancestors replaces X-Frame-Options so the analytics subdomain
|
||||
# (Umami heatmap/recorder) can embed the site in an iframe.
|
||||
add_header Content-Security-Policy "frame-ancestors 'self' https://analytics.schoolcompare.co.uk" always;
|
||||
add_header X-Content-Type-Options "nosniff" always;
|
||||
add_header X-XSS-Protection "1; mode=block" always;
|
||||
|
||||
# Proxy to Next.js
|
||||
location / {
|
||||
proxy_pass http://localhost:3000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection 'upgrade';
|
||||
proxy_set_header Host $host;
|
||||
proxy_cache_bypass $http_upgrade;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
|
||||
# Proxy to FastAPI
|
||||
location /api/ {
|
||||
proxy_pass http://localhost:8000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
|
||||
# Cache static files
|
||||
location /_next/static/ {
|
||||
proxy_pass http://localhost:3000;
|
||||
add_header Cache-Control "public, max-age=31536000, immutable";
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Pre-Deployment Checklist
|
||||
|
||||
- [ ] Run `npm run build` successfully
|
||||
- [ ] Run `npm test` - all tests pass
|
||||
- [ ] Environment variables configured
|
||||
- [ ] FastAPI backend accessible
|
||||
- [ ] Database migrations applied
|
||||
- [ ] SSL certificates configured (production)
|
||||
- [ ] Domain DNS configured
|
||||
- [ ] Monitoring/logging set up
|
||||
- [ ] Backup strategy in place
|
||||
|
||||
---
|
||||
|
||||
## Post-Deployment Verification
|
||||
|
||||
1. **Health Check**:
|
||||
```bash
|
||||
curl https://schoolcompare.co.uk
|
||||
```
|
||||
|
||||
2. **Test Routes**:
|
||||
- Home: `https://schoolcompare.co.uk/`
|
||||
- School Page: `https://schoolcompare.co.uk/school/100001`
|
||||
- Compare: `https://schoolcompare.co.uk/compare`
|
||||
- Rankings: `https://schoolcompare.co.uk/rankings`
|
||||
|
||||
3. **Check SEO**:
|
||||
- Sitemap: `https://schoolcompare.co.uk/sitemap.xml`
|
||||
- Robots: `https://schoolcompare.co.uk/robots.txt`
|
||||
|
||||
4. **Performance Audit**:
|
||||
- Run Lighthouse in Chrome DevTools
|
||||
- Target scores: 90+ for Performance, Accessibility, Best Practices, SEO
|
||||
|
||||
---
|
||||
|
||||
## Monitoring
|
||||
|
||||
### Recommended Tools:
|
||||
- **Vercel Analytics** (if using Vercel)
|
||||
- **Sentry** for error tracking
|
||||
- **Google Analytics** for user analytics
|
||||
- **Uptime Robot** for uptime monitoring
|
||||
|
||||
### Health Check Endpoint:
|
||||
The application automatically serves health data at the root route.
|
||||
|
||||
---
|
||||
|
||||
## Rollback Procedure
|
||||
|
||||
### Vercel:
|
||||
```bash
|
||||
vercel rollback
|
||||
```
|
||||
|
||||
### Docker:
|
||||
```bash
|
||||
docker-compose down
|
||||
docker-compose up -d --force-recreate
|
||||
```
|
||||
|
||||
### PM2:
|
||||
```bash
|
||||
pm2 stop schoolcompare-nextjs
|
||||
# Restore previous build
|
||||
pm2 start schoolcompare-nextjs
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Issue: API requests failing
|
||||
- **Solution**: Check `NEXT_PUBLIC_API_URL` and `FASTAPI_URL` environment variables
|
||||
- **Verify**: FastAPI backend is accessible from Next.js container/server
|
||||
|
||||
### Issue: Build fails
|
||||
- **Solution**: Check Node.js version (requires 24+)
|
||||
- **Clear cache**: `rm -rf .next node_modules && npm install && npm run build`
|
||||
|
||||
### Issue: Slow page loads
|
||||
- **Solution**: Enable caching in API calls
|
||||
- **Check**: Network latency to FastAPI backend
|
||||
- **Verify**: CDN is serving static assets
|
||||
|
||||
---
|
||||
|
||||
## Security Considerations
|
||||
|
||||
- ✅ HTTPS enabled
|
||||
- ✅ Security headers configured (X-Frame-Options, CSP, etc.)
|
||||
- ✅ API keys in environment variables (never in code)
|
||||
- ✅ CORS properly configured
|
||||
- ✅ Rate limiting on API endpoints
|
||||
- ✅ Regular security updates
|
||||
- ✅ Dependency vulnerability scanning
|
||||
|
||||
---
|
||||
|
||||
## Support
|
||||
|
||||
For deployment issues, contact the DevOps team or refer to:
|
||||
- [Next.js Deployment Docs](https://nextjs.org/docs/deployment)
|
||||
- [Vercel Documentation](https://vercel.com/docs)
|
||||
- [Docker Documentation](https://docs.docker.com/)
|
||||
Earlier Vercel and standalone deployment recipes have been retired from this file
|
||||
because they do not describe the current CMS, persistence and promotion setup.
|
||||
See [development](../docs/DEVELOPMENT.md) for checks and
|
||||
[publishing](docs/PUBLISHING.md) for CMS operations.
|
||||
@@ -28,6 +28,10 @@ ENV NODE_ENV=production
|
||||
ARG FASTAPI_URL=http://backend:80/api
|
||||
ENV FASTAPI_URL=${FASTAPI_URL}
|
||||
|
||||
ARG BUILD_SHA=development
|
||||
ARG BUILD_ID=development
|
||||
RUN node -e 'require("fs").writeFileSync("build-info.json", JSON.stringify({sha:process.argv[1],build_id:process.argv[2]}))' "$BUILD_SHA" "$BUILD_ID"
|
||||
|
||||
# Build application
|
||||
RUN npm run build
|
||||
|
||||
@@ -53,6 +57,13 @@ COPY --from=builder /app/.next/static ./.next/static
|
||||
# a miss here is a silent 500 on /opengraph-image, not a build failure.
|
||||
COPY --from=builder /app/assets ./assets
|
||||
|
||||
# Payload writes uploads here, and the compose file mounts a named volume over
|
||||
# it. The directory must exist and be owned by the runtime user BEFORE the
|
||||
# mount: Docker seeds a fresh named volume from the image path, so a missing or
|
||||
# root-owned directory here makes every upload fail with EACCES at runtime,
|
||||
# long after the build passed. The chown below covers it.
|
||||
RUN mkdir -p /app/media
|
||||
|
||||
# Set correct permissions
|
||||
RUN chown -R nextjs:nodejs /app
|
||||
|
||||
@@ -63,6 +74,12 @@ USER nextjs
|
||||
EXPOSE 3000
|
||||
|
||||
# Set environment variables
|
||||
ARG BUILD_SHA=development
|
||||
ARG BUILD_ID=development
|
||||
LABEL io.schoolcompare.build-id=$BUILD_ID
|
||||
LABEL io.schoolcompare.commit=$BUILD_SHA
|
||||
COPY --from=builder /app/build-info.json ./build-info.json
|
||||
|
||||
ENV PORT=3000
|
||||
ENV HOSTNAME="0.0.0.0"
|
||||
|
||||
|
||||
+43
-141
@@ -1,156 +1,58 @@
|
||||
# SchoolCompare Next.js Application
|
||||
# SchoolCompare frontend and CMS
|
||||
|
||||
Modern Next.js application for comparing primary school KS2 performance across England.
|
||||
Next.js App Router with React, TypeScript, CSS Modules, Chart.js, Leaflet and
|
||||
Payload CMS. It serves school search, comparisons, rankings, school/place detail
|
||||
pages and editorial content across England.
|
||||
|
||||
## Features
|
||||
Start with the [repository overview](../README.md),
|
||||
[architecture](../docs/ARCHITECTURE.md) and [development checks](../docs/DEVELOPMENT.md).
|
||||
|
||||
- **Server-Side Rendering (SSR)**: Fast initial page loads with pre-rendered content
|
||||
- **Individual School Pages**: Dedicated pages for each school with full SEO optimization
|
||||
- **Side-by-Side Comparison**: Compare up to 5 schools simultaneously
|
||||
- **School Rankings**: Top-performing schools by various metrics
|
||||
- **Interactive Maps**: Leaflet integration for geographic visualization
|
||||
- **Performance Charts**: Chart.js visualizations for historical data
|
||||
- **Responsive Design**: Mobile-first approach with full responsive support
|
||||
- **SEO Optimized**: Dynamic sitemaps, meta tags, and structured data
|
||||
## Source map
|
||||
|
||||
## Tech Stack
|
||||
| Path | Purpose |
|
||||
|---|---|
|
||||
| `app/(frontend)/` | Public root layout, server pages and FastAPI proxy |
|
||||
| `app/(payload)/` | Payload root layout, `/admin` and `/cms-api` |
|
||||
| `app/robots.ts`, `app/opengraph-image.tsx`, root icons | Site-wide metadata endpoints |
|
||||
| `components/` | Client views and reusable display components |
|
||||
| `components/school/` | School detail sections |
|
||||
| `lib/api.ts`, `lib/types.ts` | Fetch wrappers and manual school API types |
|
||||
| `lib/schoolSections.ts`, `lib/compareLogic.ts` | Presentation decisions and data preparation |
|
||||
| `context/`, `hooks/` | Comparison state, suggestion state and responsive behaviour |
|
||||
| `collections/`, `blocks/`, `migrations/` | CMS schema and production migrations |
|
||||
| `__tests__/` | Jest and React Testing Library tests |
|
||||
|
||||
- **Framework**: Next.js 16 (App Router)
|
||||
- **Language**: TypeScript 5
|
||||
- **Styling**: CSS Modules + CSS Variables
|
||||
- **State Management**: React Context API + URL state
|
||||
- **Data Fetching**: SWR (client-side) + Next.js fetch (server-side)
|
||||
- **Charts**: Chart.js + react-chartjs-2
|
||||
- **Maps**: Leaflet + react-leaflet
|
||||
- **Testing**: Jest + React Testing Library
|
||||
- **Validation**: Zod
|
||||
Do not introduce a shared `app/layout.tsx`: public pages and Payload have separate
|
||||
root layouts. Keep root metadata files outside the route groups.
|
||||
|
||||
## Getting Started
|
||||
## Data and state
|
||||
|
||||
### Prerequisites
|
||||
Server pages fetch initial data directly from `FASTAPI_URL`. Browser fetches use
|
||||
`/api` by default, forwarded by `app/(frontend)/api/[...path]/route.ts`.
|
||||
`FASTAPI_URL` must include `/api`. See `.env.example` for CMS and API settings.
|
||||
|
||||
- Node.js 24+ (using nvm recommended)
|
||||
- FastAPI backend running on port 8000
|
||||
State uses React hooks/context, URL search parameters and localStorage for the
|
||||
comparison basket. SWR is not installed. Maps use dynamic Leaflet wrappers.
|
||||
Revalidation intervals are configured in fetch wrappers and pages; they vary by
|
||||
resource. Backend reloads do not automatically invalidate every Next.js cache.
|
||||
|
||||
### Installation
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
# Install dependencies
|
||||
npm install
|
||||
|
||||
# Copy environment variables
|
||||
cp .env.example .env.local
|
||||
|
||||
# Update .env.local with your configuration
|
||||
```
|
||||
|
||||
### Development
|
||||
|
||||
```bash
|
||||
# Start development server
|
||||
npm run dev
|
||||
|
||||
# Open http://localhost:3000
|
||||
```
|
||||
|
||||
### Building
|
||||
|
||||
```bash
|
||||
# Build for production
|
||||
```sh
|
||||
npm ci
|
||||
npm run typecheck
|
||||
npm test -- --runInBand
|
||||
npm run build
|
||||
|
||||
# Start production server
|
||||
npm start
|
||||
```
|
||||
|
||||
### Testing
|
||||
`test:watch` and `test:coverage` are also available. There is no `lint` script.
|
||||
A running application needs the backend/data environment described in the
|
||||
[development guide](../docs/DEVELOPMENT.md).
|
||||
|
||||
```bash
|
||||
# Run tests
|
||||
npm test
|
||||
After CMS field or editor changes, run `npm run generate:importmap`. Keep
|
||||
`payload-types.ts` generated from the CMS schema rather than editing it by hand.
|
||||
The build must work without a database connection; avoid module-scope CMS queries
|
||||
and DB-backed `generateStaticParams` functions.
|
||||
|
||||
# Run tests in watch mode
|
||||
npm run test:watch
|
||||
|
||||
# Run tests with coverage
|
||||
npm run test:coverage
|
||||
```
|
||||
|
||||
### Linting
|
||||
|
||||
```bash
|
||||
# Run ESLint
|
||||
npm run lint
|
||||
```
|
||||
|
||||
## Project Structure
|
||||
|
||||
```
|
||||
nextjs-app/
|
||||
├── app/ # App Router pages
|
||||
│ ├── layout.tsx # Root layout
|
||||
│ ├── page.tsx # Home page
|
||||
│ ├── compare/ # Compare page
|
||||
│ ├── rankings/ # Rankings page
|
||||
│ ├── school/[urn]/ # Individual school pages
|
||||
│ ├── sitemap.ts # Dynamic sitemap
|
||||
│ └── robots.ts # Robots.txt
|
||||
├── components/ # React components
|
||||
│ ├── SchoolCard.tsx # School card component
|
||||
│ ├── FilterBar.tsx # Search/filter controls
|
||||
│ ├── ComparisonView.tsx # Comparison interface
|
||||
│ ├── RankingsView.tsx # Rankings table
|
||||
│ └── ...
|
||||
├── lib/ # Utility libraries
|
||||
│ ├── api.ts # API client
|
||||
│ ├── types.ts # TypeScript types
|
||||
│ └── utils.ts # Helper functions
|
||||
├── hooks/ # Custom React hooks
|
||||
├── context/ # React Context providers
|
||||
├── styles/ # Global styles
|
||||
├── public/ # Static assets
|
||||
└── __tests__/ # Test files
|
||||
```
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description | Default |
|
||||
|----------|-------------|---------|
|
||||
| `NEXT_PUBLIC_API_URL` | Public API endpoint (client-side) | `http://localhost:8000/api` |
|
||||
| `FASTAPI_URL` | Server-side API endpoint | `http://localhost:8000/api` |
|
||||
| `NODE_ENV` | Environment mode | `development` |
|
||||
|
||||
## Performance Optimizations
|
||||
|
||||
- **Server-Side Rendering**: Initial HTML rendered on server
|
||||
- **Static Generation**: Where possible, pages are pre-generated
|
||||
- **Image Optimization**: Next.js Image component with AVIF/WebP support
|
||||
- **Code Splitting**: Automatic route-based code splitting
|
||||
- **Dynamic Imports**: Heavy components loaded on demand
|
||||
- **API Caching**: Configurable revalidation for data fetching
|
||||
- **Bundle Optimization**: Tree shaking and minification
|
||||
- **Compression**: Gzip compression enabled
|
||||
|
||||
## SEO Features
|
||||
|
||||
- **Dynamic Meta Tags**: Generated per page with Next.js Metadata API
|
||||
- **Open Graph**: Social media optimization
|
||||
- **JSON-LD**: Structured data for search engines
|
||||
- **Sitemap**: Auto-generated from database
|
||||
- **Robots.txt**: Search engine crawling rules
|
||||
- **Canonical URLs**: Duplicate content prevention
|
||||
|
||||
## Browser Support
|
||||
|
||||
- Chrome (latest)
|
||||
- Firefox (latest)
|
||||
- Safari (latest)
|
||||
- Edge (latest)
|
||||
|
||||
## License
|
||||
|
||||
Proprietary - SchoolCompare
|
||||
|
||||
## Support
|
||||
|
||||
For issues and questions, please contact the development team.
|
||||
See [publishing](docs/PUBLISHING.md) for CMS operations and
|
||||
[deployment](../docs/DEPLOY.md) for staging and production promotion.
|
||||
@@ -0,0 +1,31 @@
|
||||
/**
|
||||
* The /api/* proxy is public. Anything it forwards is on the internet.
|
||||
*
|
||||
* @jest-environment node
|
||||
*/
|
||||
// The docblock above is load-bearing. jest.config.js sets jsdom globally, and
|
||||
// NextRequest/NextResponse need the Web Fetch API globals that only the node
|
||||
// environment provides — under jsdom this suite fails on import, not on an
|
||||
// assertion.
|
||||
import { NextRequest } from 'next/server';
|
||||
import { GET } from '@/app/(frontend)/api/[...path]/route';
|
||||
|
||||
function request(path: string) {
|
||||
return new NextRequest(`http://localhost:3000/api/${path}`);
|
||||
}
|
||||
|
||||
describe('public API proxy', () => {
|
||||
it('refuses to forward internal-only paths', async () => {
|
||||
// /api/flags names every unreleased feature and its state. Forwarding it
|
||||
// publishes the thing shipping dark exists to keep quiet.
|
||||
const res = await GET(request('flags'), { params: Promise.resolve({ path: ['flags'] }) });
|
||||
expect(res.status).toBe(404);
|
||||
});
|
||||
|
||||
it('does not deny a path that merely starts with the same letters', async () => {
|
||||
// A prefix match would take /api/flagship down with /api/flags.
|
||||
const res = await GET(
|
||||
request('flagship'), { params: Promise.resolve({ path: ['flagship'] }) });
|
||||
expect(res.status).not.toBe(404);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,34 @@
|
||||
import { metadata } from '@/app/(frontend)/about/page';
|
||||
import { personJsonLd, organizationJsonLd } from '@/lib/jsonld';
|
||||
|
||||
describe('/about metadata', () => {
|
||||
it('canonicalises to the bare path', () => {
|
||||
expect(metadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/about');
|
||||
});
|
||||
});
|
||||
|
||||
describe('author structured data', () => {
|
||||
it('describes a Person with a first name and a photo', () => {
|
||||
const person = personJsonLd();
|
||||
expect(person['@type']).toBe('Person');
|
||||
expect(person.name).toBe('Tudor');
|
||||
expect(person.image).toBe('https://www.schoolcompare.co.uk/brand/tudor.jpg');
|
||||
expect(person.url).toBe('https://www.schoolcompare.co.uk/about');
|
||||
});
|
||||
|
||||
it('never publishes a surname or an employer', () => {
|
||||
// Author identity constraint: first name only. A surname here would be
|
||||
// the one place it leaks, since JSON-LD is machine-read and archived.
|
||||
const serialised = JSON.stringify(personJsonLd());
|
||||
expect(serialised).not.toMatch(/familyName|Sitaru/i);
|
||||
expect(serialised).not.toMatch(/worksFor|affiliation/i);
|
||||
});
|
||||
|
||||
it('describes the site as an Organization the Person authors for', () => {
|
||||
const org = organizationJsonLd();
|
||||
expect(org['@type']).toBe('Organization');
|
||||
expect(org.name).toBe('schoolcompare');
|
||||
expect(org.url).toBe('https://www.schoolcompare.co.uk');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,66 @@
|
||||
/**
|
||||
* The blog index imports getCachedPayload, which pulls in Payload — ESM-only,
|
||||
* and next/jest will not transform node_modules. Mocking that one module keeps
|
||||
* the page's metadata testable without loading the CMS; the mock is never
|
||||
* called, because `metadata` is a static export evaluated at import time.
|
||||
*/
|
||||
jest.mock('@/lib/payload', () => ({ getCachedPayload: jest.fn() }));
|
||||
|
||||
import { metadata } from '@/app/(frontend)/blog/page';
|
||||
import { blogPostingJsonLd, breadcrumbJsonLd } from '@/lib/jsonld';
|
||||
|
||||
const post = {
|
||||
title: 'What the data cannot tell you',
|
||||
slug: 'what-the-data-cannot-tell-you',
|
||||
excerpt: 'Results describe one year group on a handful of days.',
|
||||
publishedAt: '2026-09-15T00:00:00.000Z',
|
||||
};
|
||||
|
||||
describe('/blog metadata', () => {
|
||||
it('canonicalises to the bare path', () => {
|
||||
expect(metadata.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/blog');
|
||||
});
|
||||
});
|
||||
|
||||
describe('BlogPosting structured data', () => {
|
||||
it('names the same Person entity the about page declares', () => {
|
||||
// By @id, not by repeating the person: search engines must resolve every
|
||||
// post and the about page to one author entity, or the site has several.
|
||||
const ld = blogPostingJsonLd(post, { namedAuthor: true });
|
||||
expect(ld['@type']).toBe('BlogPosting');
|
||||
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk/about#tudor');
|
||||
expect(ld.publisher['@id']).toBe('https://www.schoolcompare.co.uk#organization');
|
||||
});
|
||||
|
||||
it('attributes to the organization when the about page is dark', () => {
|
||||
/*
|
||||
* The two flags are independent, so blog-on-about-off is a reachable
|
||||
* state. The Person entity lives at /about#tudor and that URL 404s while
|
||||
* the flag is dark, so claiming it would declare an author that resolves
|
||||
* to nothing — worse for the blog's credibility than having no named
|
||||
* author at all. Attribute to the publisher instead.
|
||||
*/
|
||||
const ld = blogPostingJsonLd(post, { namedAuthor: false });
|
||||
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk#organization');
|
||||
expect(JSON.stringify(ld)).not.toContain('/about');
|
||||
});
|
||||
|
||||
it('carries a self-referencing canonical url and the publish date', () => {
|
||||
const ld = blogPostingJsonLd(post, { namedAuthor: true });
|
||||
expect(ld.url).toBe(
|
||||
'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you',
|
||||
);
|
||||
expect(ld.datePublished).toBe('2026-09-15T00:00:00.000Z');
|
||||
});
|
||||
});
|
||||
|
||||
describe('breadcrumbs', () => {
|
||||
it('places the post under the blog index', () => {
|
||||
const ld = breadcrumbJsonLd(post);
|
||||
expect(ld.itemListElement[0].item).toBe('https://www.schoolcompare.co.uk/blog');
|
||||
expect(ld.itemListElement[1].item).toBe(
|
||||
'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you',
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,51 @@
|
||||
import { fireEvent, render, screen } from '@testing-library/react';
|
||||
import HomePage from '@/app/(frontend)/page';
|
||||
import SchoolPage from '@/app/(frontend)/school/[slug]/page';
|
||||
import ErrorPage from '@/app/(frontend)/error';
|
||||
import { APIFetchError, fetchSchools, fetchFilters, fetchSchoolDetails } from '@/lib/api';
|
||||
import { fetchPlace, fetchPlaces } from '@/lib/places';
|
||||
|
||||
jest.mock('@/lib/api', () => ({
|
||||
...jest.requireActual('@/lib/api'),
|
||||
fetchSchools: jest.fn(),
|
||||
fetchSchoolDetails: jest.fn(),
|
||||
fetchFilters: jest.fn(async () => ({})),
|
||||
fetchDataInfo: jest.fn(async () => null),
|
||||
fetchNationalAverages: jest.fn(async () => null),
|
||||
}));
|
||||
jest.mock('@/lib/flags', () => ({ getFlags: jest.fn(async () => ({})) }));
|
||||
jest.mock('next/navigation', () => ({
|
||||
notFound: () => { throw new Error('NEXT_NOT_FOUND'); },
|
||||
redirect: jest.fn(),
|
||||
}));
|
||||
const realFetch = global.fetch;
|
||||
afterEach(() => { global.fetch = realFetch; jest.clearAllMocks(); });
|
||||
|
||||
test('school outages propagate; only a real 404 becomes not found', async () => {
|
||||
const request = { params: Promise.resolve({ slug: '100001-school' }) };
|
||||
const outage = new APIFetchError('unavailable', 503);
|
||||
jest.mocked(fetchSchoolDetails).mockRejectedValueOnce(outage);
|
||||
await expect(SchoolPage(request)).rejects.toBe(outage);
|
||||
jest.mocked(fetchSchoolDetails).mockRejectedValueOnce(new APIFetchError('missing', 404));
|
||||
await expect(SchoolPage(request)).rejects.toThrow('NEXT_NOT_FOUND');
|
||||
});
|
||||
|
||||
test('homepage search failure is not returned as an empty successful page', async () => {
|
||||
jest.mocked(fetchSchools).mockRejectedValueOnce(new APIFetchError('unavailable', 503));
|
||||
await expect(HomePage({ searchParams: Promise.resolve({ search: 'school' }) })).rejects.toThrow('unavailable');
|
||||
});
|
||||
|
||||
test('a place is absent only on 404; other failures propagate', async () => {
|
||||
global.fetch = jest.fn().mockResolvedValue({ ok: false, status: 404 });
|
||||
await expect(fetchPlace('town', 'example')).resolves.toBeNull();
|
||||
jest.mocked(global.fetch).mockResolvedValue({ ok: false, status: 503 } as Response);
|
||||
await expect(fetchPlace('town', 'example')).rejects.toMatchObject({ status: 503 });
|
||||
await expect(fetchPlaces()).rejects.toMatchObject({ status: 503 });
|
||||
});
|
||||
|
||||
test('the error boundary offers a retry without showing an empty search', () => {
|
||||
const reset = jest.fn();
|
||||
render(<ErrorPage error={new Error('offline')} reset={reset} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Try again' }));
|
||||
expect(reset).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
@@ -1,7 +1,8 @@
|
||||
import { metadata as homeMetadata } from '@/app/page';
|
||||
import { metadata as rankingsMetadata } from '@/app/rankings/page';
|
||||
import { metadata as admissionsMetadata } from '@/app/admissions/page';
|
||||
import { generateMetadata as compareMetadata } from '@/app/compare/page';
|
||||
import { metadata as homeMetadata } from '@/app/(frontend)/page';
|
||||
import { metadata as rankingsMetadata } from '@/app/(frontend)/rankings/page';
|
||||
import { metadata as admissionsMetadata } from '@/app/(frontend)/admissions/page';
|
||||
import { generateMetadata as compareMetadata } from '@/app/(frontend)/compare/page';
|
||||
import { metadata as rootMetadata } from '@/app/(frontend)/layout';
|
||||
|
||||
describe('canonical URLs', () => {
|
||||
it('the homepage canonicalises to the bare root', () => {
|
||||
@@ -122,9 +123,47 @@ describe('C1 snippet copy', () => {
|
||||
|
||||
it('no C1 page claims a school count that will drift', () => {
|
||||
// The corpus moves with every data refresh; this repo has already shipped
|
||||
// one copy bug of that kind ("three schools" against MAX_SCHOOLS = 5).
|
||||
// one copy bug of that kind ("three schools" against a limit of five).
|
||||
for (const [, meta] of pages) {
|
||||
expect(meta.description as string).not.toMatch(/\b\d{2},\d{3}\b|\b\d{2},000\b/);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* The share card must be declared, not inherited.
|
||||
*
|
||||
* `app/opengraph-image.tsx` is a metadata file convention, and it does attach
|
||||
* to routes in the app root segment — `_not-found` gets an og:image from it.
|
||||
* It does NOT attach to the site's pages, which live in the `(frontend)`
|
||||
* route group whose own layout is a root layout. Staging served og:title,
|
||||
* og:description, og:url, og:site_name and og:type and no og:image at all,
|
||||
* so every link pasted into a chat rendered bare.
|
||||
*
|
||||
* The file stays at the app root, because /robots.txt and /icon.png depend on
|
||||
* it being there. The site's root layout points at the route it generates.
|
||||
*/
|
||||
describe('the share card', () => {
|
||||
it('declares an opengraph image on the site root layout', () => {
|
||||
// No og:image means every link pasted into a chat renders bare.
|
||||
const images = rootMetadata.openGraph?.images;
|
||||
expect(images).toBeTruthy();
|
||||
expect(JSON.stringify(images)).toContain('/opengraph-image');
|
||||
});
|
||||
|
||||
it('declares a twitter image too', () => {
|
||||
// twitter.card is summary_large_image. Claiming a large-image card and
|
||||
// supplying no image is worse than claiming a summary card.
|
||||
// Metadata['twitter'] is a union and `card` is not on every member, so
|
||||
// this reads the serialised shape rather than narrowing the type.
|
||||
const twitter = JSON.stringify(rootMetadata.twitter);
|
||||
expect(twitter).toContain('summary_large_image');
|
||||
expect(twitter).toContain('/opengraph-image');
|
||||
});
|
||||
|
||||
it('resolves the card to an absolute url via metadataBase', () => {
|
||||
// The e2e journey does `new URL(ogUrl)`, which throws on a relative path.
|
||||
expect(rootMetadata.metadataBase?.toString())
|
||||
.toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,66 @@
|
||||
/**
|
||||
* next.config.mjs carries the staging noindex rule. Breaking it turns
|
||||
* stx.schoolcompare.co.uk into a fully crawlable duplicate of production,
|
||||
* and nothing else in the suite would notice.
|
||||
*
|
||||
* The non-null assertions are deliberate: every key asserted here is optional
|
||||
* on NextConfig, and a missing one is precisely the regression under test, so
|
||||
* the assertion below should fail the test rather than the compile.
|
||||
*/
|
||||
import nextConfig from '@/next.config.mjs';
|
||||
|
||||
async function headerRules() {
|
||||
return nextConfig.headers!();
|
||||
}
|
||||
|
||||
describe('next.config.mjs', () => {
|
||||
it('keeps the staging host out of the index', async () => {
|
||||
const headers = await headerRules();
|
||||
const stagingRule = headers.find((rule) =>
|
||||
rule.has?.some(
|
||||
(cond) => cond.type === 'host' && cond.value === 'stx.schoolcompare.co.uk',
|
||||
),
|
||||
);
|
||||
expect(stagingRule).toBeDefined();
|
||||
expect(stagingRule!.headers).toContainEqual({
|
||||
key: 'X-Robots-Tag',
|
||||
value: 'noindex, nofollow',
|
||||
});
|
||||
});
|
||||
|
||||
it('still emits standalone output for the Docker runner', () => {
|
||||
expect(nextConfig.output).toBe('standalone');
|
||||
});
|
||||
|
||||
it('still traces the share-card fonts into the standalone bundle', () => {
|
||||
expect(nextConfig.outputFileTracingIncludes!['/opengraph-image']).toEqual([
|
||||
'./assets/**',
|
||||
]);
|
||||
});
|
||||
|
||||
it('still allows the analytics subdomain to frame the site', async () => {
|
||||
const headers = await headerRules();
|
||||
const csp = headers
|
||||
.flatMap((rule) => rule.headers)
|
||||
.find((header) => header.key === 'Content-Security-Policy');
|
||||
expect(csp).toBeDefined();
|
||||
expect(csp!.value).toContain('https://analytics.schoolcompare.co.uk');
|
||||
});
|
||||
});
|
||||
|
||||
describe('admin surface', () => {
|
||||
it('serves noindex on the admin panel and the CMS API', async () => {
|
||||
// robots.txt disallows these too, but a Disallow only blocks crawling — a
|
||||
// URL found from an external link can still be indexed without ever being
|
||||
// fetched. This header is what actually keeps them out.
|
||||
const headers = await headerRules();
|
||||
for (const source of ['/admin/:path*', '/cms-api/:path*']) {
|
||||
const rule = headers.find((entry) => entry.source === source);
|
||||
expect(rule).toBeDefined();
|
||||
expect(rule!.headers).toContainEqual({
|
||||
key: 'X-Robots-Tag',
|
||||
value: 'noindex, nofollow',
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,42 @@
|
||||
import { generateMetadata as placeMeta } from '@/app/(frontend)/schools/[place]/page';
|
||||
|
||||
jest.mock('@/lib/places', () => ({
|
||||
...jest.requireActual('@/lib/places'),
|
||||
fetchPlace: jest.fn(async (kind: string, slug: string) =>
|
||||
slug === 'atlantis' ? null : ({
|
||||
place: { kind, slug, name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex' },
|
||||
schools: [], averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
})),
|
||||
fetchPlaces: jest.fn(async () => []),
|
||||
}));
|
||||
|
||||
describe('place page metadata', () => {
|
||||
it('titles the page the way the place is searched', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect((m.title as { absolute: string }).absolute).toMatch(/schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('canonicalises to its own path on the www host', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'brentwood' }) });
|
||||
expect(m.alternates?.canonical)
|
||||
.toBe('https://www.schoolcompare.co.uk/schools/brentwood');
|
||||
});
|
||||
|
||||
it('opts out of the layout template, which would double the brand', () => {
|
||||
// The root layout appends '| schoolcompare' to a plain string title, and
|
||||
// these titles already carry it — every place page shipped reading
|
||||
// '... | schoolcompare | schoolcompare' until this was made absolute.
|
||||
return placeMeta({ params: Promise.resolve({ place: 'brentwood' }) })
|
||||
.then((m) => {
|
||||
expect(typeof m.title).toBe('object');
|
||||
expect((m.title as { absolute: string }).absolute)
|
||||
.not.toMatch(/schoolcompare.*schoolcompare/);
|
||||
});
|
||||
});
|
||||
|
||||
it('an unknown place gets a not-found title rather than inventing one', async () => {
|
||||
const m = await placeMeta({ params: Promise.resolve({ place: 'atlantis' }) });
|
||||
expect(m.title).toMatch(/not found/i);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,30 @@
|
||||
/** @jest-environment node */
|
||||
import { GET } from '@/app/(frontend)/release.json/route';
|
||||
import { readFile } from 'node:fs/promises';
|
||||
|
||||
jest.mock('node:fs/promises', () => ({ readFile: jest.fn() }));
|
||||
const realFetch = global.fetch;
|
||||
const identity = { sha: 'a'.repeat(40), build_id: 'b'.repeat(32) };
|
||||
beforeEach(() => {
|
||||
jest.mocked(readFile).mockResolvedValue(JSON.stringify(identity));
|
||||
global.fetch = jest.fn(async () => Response.json(identity));
|
||||
});
|
||||
afterEach(() => { global.fetch = realFetch; jest.resetAllMocks(); });
|
||||
|
||||
test('reports immutable file identity and backend identity without caching', async () => {
|
||||
const response = await GET();
|
||||
expect(response.status).toBe(200);
|
||||
expect(response.headers.get('Cache-Control')).toBe('no-store');
|
||||
expect(await response.json()).toEqual({ frontend: identity, backend: identity });
|
||||
expect(fetch).toHaveBeenCalledWith(expect.stringMatching(/\/api\/release$/), expect.objectContaining({ cache: 'no-store', signal: expect.anything() }));
|
||||
});
|
||||
|
||||
test('missing build metadata cannot pass the release gate', async () => {
|
||||
jest.mocked(readFile).mockRejectedValueOnce(new Error('missing file'));
|
||||
expect((await GET()).status).toBe(503);
|
||||
});
|
||||
|
||||
test('backend failure cannot pass the release gate', async () => {
|
||||
jest.mocked(fetch).mockResolvedValueOnce(new Response('', { status: 503 }));
|
||||
expect((await GET()).status).toBe(503);
|
||||
});
|
||||
@@ -0,0 +1,20 @@
|
||||
import robots from '@/app/robots';
|
||||
|
||||
describe('robots.txt', () => {
|
||||
it('disallows the admin panel and the CMS API', () => {
|
||||
const rules = robots().rules;
|
||||
const rule = Array.isArray(rules) ? rules[0] : rules;
|
||||
expect(rule.disallow).toEqual(
|
||||
expect.arrayContaining(['/api/', '/_next/', '/admin/', '/cms-api/']),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('sitemap discovery', () => {
|
||||
it('lists both the proxied school sitemap and the Next-owned content sitemap', () => {
|
||||
expect(robots().sitemap).toEqual([
|
||||
'https://www.schoolcompare.co.uk/sitemap.xml',
|
||||
'https://www.schoolcompare.co.uk/content-sitemap.xml',
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -104,7 +104,7 @@ describe('CompareAdmissions', () => {
|
||||
render(<CompareAdmissions schools={[grammar]} data={data} isSecondary={true} />);
|
||||
|
||||
expect(
|
||||
screen.getByText(/Entry is by entrance test — the school is selective/),
|
||||
screen.getByText(/Entry is by entrance test\. The school is selective/),
|
||||
).toBeInTheDocument();
|
||||
expect(screen.queryByText(/non-faith primaries/)).toBeNull();
|
||||
});
|
||||
|
||||
@@ -102,7 +102,7 @@ describe('CutoffMapPanel', () => {
|
||||
// Explanation is supporting text, not part of the bold verdict line.
|
||||
expect(result.querySelector('[class*="cutoffCheckHeadline"]')!.textContent)
|
||||
.not.toMatch(/measurement error/);
|
||||
expect(result).not.toHaveTextContent(/^\S+ away — inside/);
|
||||
expect(result).not.toHaveTextContent(/^\S+ away, inside/);
|
||||
});
|
||||
|
||||
it('surfaces a postcode the geocoder cannot find', async () => {
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { DestinationsSection } from '@/components/school/DestinationsSection';
|
||||
import type { DestinationPhase } from '@/lib/types';
|
||||
import type { DestinationCategory, DestinationStatus } from '@/lib/destinations';
|
||||
|
||||
const cell = (
|
||||
category: DestinationCategory,
|
||||
pupils: number | null,
|
||||
status: DestinationStatus = 'published',
|
||||
) => ({
|
||||
category, pupils,
|
||||
percentage: pupils === null ? null : (pupils / 180) * 100,
|
||||
status,
|
||||
});
|
||||
|
||||
const ALL_PUBLISHED = [
|
||||
cell('school_sixth_form', 75), cell('sixth_form_college', 21),
|
||||
cell('further_education', 55), cell('other_education', 6),
|
||||
cell('apprenticeship', 8), cell('employment', 6),
|
||||
cell('not_sustained', 5), cell('not_captured', 4),
|
||||
];
|
||||
|
||||
const fullPhase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: { all: { cohort: 180, categories: ALL_PUBLISHED } },
|
||||
};
|
||||
|
||||
const suppressedPhase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: {
|
||||
all: {
|
||||
cohort: 180,
|
||||
categories: [
|
||||
cell('school_sixth_form', 75), cell('sixth_form_college', null, 'suppressed'),
|
||||
cell('further_education', 55), cell('other_education', 6),
|
||||
cell('apprenticeship', 8), cell('employment', 6),
|
||||
cell('not_sustained', 5), cell('not_captured', 4),
|
||||
],
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
describe('DestinationsSection', () => {
|
||||
it('dates its own cohort so it is not read as stale next to the GCSE section', () => {
|
||||
render(<DestinationsSection destinations={fullPhase} />);
|
||||
expect(screen.getByText(/2022\/23/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('renders one bar segment per published category', () => {
|
||||
const { container } = render(<DestinationsSection destinations={fullPhase} />);
|
||||
expect(container.querySelectorAll('[data-destination-segment]')).toHaveLength(8);
|
||||
});
|
||||
|
||||
it('renders NO bar at all when a category is withheld', () => {
|
||||
const { container } = render(<DestinationsSection destinations={suppressedPhase} />);
|
||||
// R1: a bar with a gap in it publishes the withheld figure by its width.
|
||||
expect(container.querySelectorAll('[data-destination-segment]')).toHaveLength(0);
|
||||
expect(screen.getAllByText(/withheld/i).length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('never states the remainder for a partially suppressed group', () => {
|
||||
const { container } = render(<DestinationsSection destinations={suppressedPhase} />);
|
||||
// 180 cohort - 159 published = 21, the withheld figure. It must appear nowhere.
|
||||
expect(container.textContent).not.toMatch(/\b21\b/);
|
||||
});
|
||||
|
||||
it('shows a card value for a group whose components are all published', () => {
|
||||
render(<DestinationsSection destinations={fullPhase} />);
|
||||
// academic route = 75 + 21 = 96 of 180 = 53%
|
||||
expect(screen.getByText('53%')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('refuses a card value when one of its components is withheld', () => {
|
||||
render(<DestinationsSection destinations={suppressedPhase} />);
|
||||
// academic route needs sixth_form_college, which is suppressed.
|
||||
expect(screen.getByText(/not published/i)).toBeInTheDocument();
|
||||
expect(screen.queryByText('53%')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('never claims a pupil stayed at this school', () => {
|
||||
const { container } = render(<DestinationsSection destinations={fullPhase} />);
|
||||
// The published file reports destination TYPE, never destination institution.
|
||||
expect(container.textContent).not.toMatch(/stayed on (here|at this school)/i);
|
||||
});
|
||||
|
||||
it('renders nothing when no group carries categories', () => {
|
||||
const empty: DestinationPhase = { cohort_year: '2022/23', groups: {} };
|
||||
const { container } = render(<DestinationsSection destinations={empty} />);
|
||||
expect(container.firstChild).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the detail table keeps the three statuses apart', () => {
|
||||
// 'suppressed' and 'not_applicable' are different claims, and the mart, the
|
||||
// SQLAlchemy model and the serialiser all preserve the difference. The table
|
||||
// used to key its Share column off `percentage === null`, which is true for
|
||||
// both, so a category that simply does not apply was labelled "withheld" —
|
||||
// while the Pupils column beside it rendered blank.
|
||||
const mixedPhase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: {
|
||||
all: {
|
||||
cohort: 180,
|
||||
categories: [
|
||||
cell('school_sixth_form', 75),
|
||||
cell('sixth_form_college', null, 'suppressed'),
|
||||
cell('further_education', null, 'suppressed'),
|
||||
cell('apprenticeship', null, 'not_applicable'),
|
||||
],
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const rowFor = (container: HTMLElement, category: string) =>
|
||||
Array.from(container.querySelectorAll('tbody tr'))
|
||||
.find(tr => tr.textContent?.includes(category));
|
||||
|
||||
it('never labels a not-applicable category as withheld', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'Apprenticeship');
|
||||
expect(row).toBeTruthy();
|
||||
expect(row!.textContent).not.toMatch(/withheld/i);
|
||||
});
|
||||
|
||||
it('labels a genuinely suppressed category as withheld in both columns', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'Sixth-form college');
|
||||
expect(row).toBeTruthy();
|
||||
expect(row!.querySelectorAll('td')).toHaveLength(2);
|
||||
Array.from(row!.querySelectorAll('td')).forEach(td =>
|
||||
expect(td.textContent).toMatch(/withheld/i));
|
||||
});
|
||||
|
||||
it('the two columns of a row never disagree about what the row is', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
Array.from(container.querySelectorAll('tbody tr')).forEach(tr => {
|
||||
const cells = Array.from(tr.querySelectorAll('td'))
|
||||
.map(td => /withheld/i.test(td.textContent ?? ''));
|
||||
expect(new Set(cells).size).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
it('shows a published category its real figures', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'State-funded school sixth form');
|
||||
expect(row!.textContent).toMatch(/75/);
|
||||
expect(row!.textContent).toMatch(/42%/);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,129 @@
|
||||
import { fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
import type { ResultFilters } from '@/lib/types';
|
||||
|
||||
/*
|
||||
* A filter's options must not come from the results it is filtering, or
|
||||
* choosing one leaves only that one on offer: pick "Girls" and "Boys" is gone
|
||||
* until the filter is cleared. School type groups, gender and admissions offer the
|
||||
* full lists, as phase already did. Local authority stays scoped to the
|
||||
* results, so a postcode search offers the councils nearby rather than 153.
|
||||
*
|
||||
* Gender, sixth form and admissions show unless the phase chosen is a primary
|
||||
* one, so what the results happen to contain never decides which filters
|
||||
* there are.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Merton', 'Wandsworth'],
|
||||
school_types: ['Academy converter', 'Community school'], years: [],
|
||||
phases: ['Middle deemed primary', 'Nursery', 'Primary', 'Secondary', 'All-through'],
|
||||
genders: ['Boys', 'Girls', 'Mixed'],
|
||||
admissions_policies: ['Non-selective', 'Selective'],
|
||||
school_type_groups: [
|
||||
{ value: 'academy', label: 'State school: academy or free school' },
|
||||
{ value: 'council', label: 'State school: council-run' },
|
||||
],
|
||||
};
|
||||
|
||||
// What the results came back with once narrowed by the chosen filters.
|
||||
const narrowed: ResultFilters = {
|
||||
local_authorities: ['Wandsworth'], school_types: ['Community school'],
|
||||
phases: ['Secondary'], genders: ['Girls'], admissions_policies: ['Non-selective'],
|
||||
};
|
||||
|
||||
const pushedParams = () => new URLSearchParams(push.mock.calls.at(-1)![0].split('?')[1]);
|
||||
|
||||
const openSheet = () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Filters/ }));
|
||||
return screen.getByRole('dialog', { name: 'Filters' });
|
||||
};
|
||||
|
||||
const optionsOf = (sheet: HTMLElement, name: string) =>
|
||||
within(within(sheet).getByRole('combobox', { name }))
|
||||
.getAllByRole('option').map((o) => o.textContent);
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
push.mockClear();
|
||||
});
|
||||
|
||||
describe('filter options', () => {
|
||||
it('offer every school type group, gender and admissions policy, whatever the results hold', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=council&gender=girls');
|
||||
render(<FilterBar filters={filters} resultFilters={narrowed} />);
|
||||
const sheet = openSheet();
|
||||
expect(optionsOf(sheet, 'School type')).toEqual(['Any school type', 'State school: academy or free school', 'State school: council-run']);
|
||||
expect(optionsOf(sheet, 'Gender')).toEqual(['Boys, Girls & Mixed', 'Boys', 'Girls', 'Mixed']);
|
||||
expect(optionsOf(sheet, 'Admissions')).toEqual(['All admissions types', 'Non-selective', 'Selective']);
|
||||
});
|
||||
|
||||
it('keep local authority to the councils in the results', () => {
|
||||
render(<FilterBar filters={filters} resultFilters={narrowed} />);
|
||||
expect(optionsOf(openSheet(), 'Local authority')).toEqual(['All Local Authorities', 'Wandsworth']);
|
||||
});
|
||||
|
||||
it('offer the full school types in the desktop row too', () => {
|
||||
render(<FilterBar filters={filters} resultFilters={narrowed} />);
|
||||
const row = screen.getByRole('group', { name: 'Filters' });
|
||||
expect(within(within(row).getByRole('combobox', { name: 'School type' }))
|
||||
.getAllByRole('option')).toHaveLength(3);
|
||||
});
|
||||
});
|
||||
|
||||
describe('the secondary-only filters', () => {
|
||||
const secondaryOnly = ['Gender', 'Sixth form', 'Admissions'];
|
||||
const shown = (sheet: HTMLElement) =>
|
||||
secondaryOnly.filter((name) => within(sheet).queryByRole('combobox', { name }));
|
||||
|
||||
it('show with any phase, even when no secondary school is in the results', () => {
|
||||
const primariesOnly = { ...narrowed, phases: ['Primary'], genders: [], admissions_policies: [] };
|
||||
render(<FilterBar filters={filters} resultFilters={primariesOnly} />);
|
||||
expect(shown(openSheet())).toEqual(secondaryOnly);
|
||||
});
|
||||
|
||||
it.each(['primary', 'nursery', 'middle deemed primary', 'Middle-deemed Primary'])(
|
||||
'hide for the %s phase', (phase) => {
|
||||
params = new URLSearchParams(`postcode=SW196AR&radius=1&phase=${encodeURIComponent(phase)}`);
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(shown(openSheet())).toEqual([]);
|
||||
});
|
||||
|
||||
it('leave out a filter with no options, rather than show only its "any"', () => {
|
||||
render(<FilterBar filters={{ ...filters, genders: [], admissions_policies: [] }} />);
|
||||
expect(shown(openSheet())).toEqual(['Sixth form']);
|
||||
});
|
||||
|
||||
it.each(['secondary', 'all-through', 'middle deemed secondary', '16 plus'])('show for the %s phase', (phase) => {
|
||||
params = new URLSearchParams(`postcode=SW196AR&radius=1&phase=${phase}`);
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(shown(openSheet())).toEqual(secondaryOnly);
|
||||
});
|
||||
|
||||
it('are cleared by choosing a primary phase, rather than left applied and hidden', () => {
|
||||
params = new URLSearchParams(
|
||||
'postcode=SW196AR&radius=1&phase=secondary&gender=girls&has_sixth_form=yes&admissions_policy=selective&school_type=council');
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.change(within(openSheet()).getByRole('combobox', { name: 'Phase' }), { target: { value: 'primary' } });
|
||||
const next = pushedParams();
|
||||
expect(next.get('phase')).toBe('primary');
|
||||
for (const key of ['gender', 'has_sixth_form', 'admissions_policy']) expect(next.get(key)).toBeNull();
|
||||
expect(next.get('school_type')).toBe('council');
|
||||
});
|
||||
|
||||
it('are kept when the new phase still has them', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=secondary&gender=girls');
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.change(within(openSheet()).getByRole('combobox', { name: 'Phase' }), { target: { value: 'all-through' } });
|
||||
expect(pushedParams().get('gender')).toBe('girls');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,40 @@
|
||||
import { render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
|
||||
let searchParams = new URLSearchParams();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useRouter: () => ({ push: jest.fn(), replace: jest.fn(), prefetch: jest.fn() }),
|
||||
usePathname: () => '/',
|
||||
useSearchParams: () => searchParams,
|
||||
}));
|
||||
|
||||
const FILTERS = {
|
||||
local_authorities: [], school_types: [], years: [],
|
||||
phases: ['Primary', 'Secondary', 'All-through'],
|
||||
genders: [], admissions_policies: [],
|
||||
};
|
||||
|
||||
/**
|
||||
* The phase options must not come from the result set. The backend scopes its
|
||||
* result filters to the schools it returns, and it applies the phase filter
|
||||
* first — so with "secondary" chosen the scoped list holds only secondary-ish
|
||||
* phases, and switching to primary meant going back to "Any phase" first.
|
||||
*/
|
||||
describe('FilterBar phase options', () => {
|
||||
it('offers every phase while a phase filter narrows the results', () => {
|
||||
searchParams = new URLSearchParams('search=hampton&phase=secondary');
|
||||
render(
|
||||
<FilterBar
|
||||
filters={FILTERS}
|
||||
resultFilters={{
|
||||
local_authorities: [], school_types: [],
|
||||
phases: ['Secondary', 'All-through'],
|
||||
genders: [], admissions_policies: [],
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
const phase = screen.getByRole('combobox', { name: 'Phase' });
|
||||
expect(within(phase).getByRole('option', { name: 'Primary' })).toBeInTheDocument();
|
||||
expect(phase).toHaveValue('secondary');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,51 @@
|
||||
import { render, screen, waitFor } from '@testing-library/react';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
|
||||
const push = jest.fn();
|
||||
let searchParams = new URLSearchParams();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
usePathname: () => '/',
|
||||
useSearchParams: () => searchParams,
|
||||
}));
|
||||
|
||||
const FILTERS = {
|
||||
local_authorities: [], school_types: [], years: [], phases: [],
|
||||
genders: [], admissions_policies: [],
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
push.mockClear();
|
||||
searchParams = new URLSearchParams();
|
||||
});
|
||||
|
||||
describe('FilterBar default distance', () => {
|
||||
it('searches a new postcode within half a mile', async () => {
|
||||
render(<FilterBar filters={FILTERS} />);
|
||||
await userEvent.type(screen.getByPlaceholderText(/School name or postcode/i), 'SW19 6AR{Enter}');
|
||||
await waitFor(() => expect(push).toHaveBeenCalledWith(expect.stringContaining('radius=0.5')));
|
||||
});
|
||||
|
||||
it('shows half a mile when the URL carries a postcode but no radius', () => {
|
||||
searchParams = new URLSearchParams('postcode=SW196AR');
|
||||
render(<FilterBar filters={FILTERS} />);
|
||||
expect(screen.getByRole('combobox', { name: 'Distance' })).toHaveValue('0.5');
|
||||
});
|
||||
|
||||
it('keeps a distance the user already chose', async () => {
|
||||
searchParams = new URLSearchParams('postcode=SW196AR&radius=3');
|
||||
render(<FilterBar filters={FILTERS} />);
|
||||
expect(screen.getByRole('combobox', { name: 'Distance' })).toHaveValue('3');
|
||||
});
|
||||
});
|
||||
|
||||
describe('FilterBar distance options', () => {
|
||||
it('offers a quarter mile without making it the default', () => {
|
||||
searchParams = new URLSearchParams('postcode=SW196AR');
|
||||
render(<FilterBar filters={FILTERS} />);
|
||||
const distance = screen.getByRole('combobox', { name: 'Distance' });
|
||||
expect([...(distance as HTMLSelectElement).options].map(o => o.value)).toEqual(['0.25', '0.5', '1', '3', '5']);
|
||||
expect(distance).toHaveValue('0.5');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,110 @@
|
||||
import { render, screen, fireEvent, waitFor } from '@testing-library/react';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
|
||||
const push = jest.fn();
|
||||
let searchParams = new URLSearchParams();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
usePathname: () => '/',
|
||||
useSearchParams: () => searchParams,
|
||||
}));
|
||||
|
||||
const FILTERS = {
|
||||
local_authorities: [], school_types: [], years: [], phases: [],
|
||||
genders: [], admissions_policies: [],
|
||||
};
|
||||
|
||||
const realFetch = global.fetch;
|
||||
beforeEach(() => {
|
||||
global.fetch = jest.fn(async () => ({
|
||||
ok: true,
|
||||
json: async () => ({ suggestions: [{
|
||||
urn: 100010, school_name: 'Brecknock Primary School',
|
||||
local_authority: 'Camden', postcode: 'NW1 1AA',
|
||||
phase: 'Primary', school_type: 'Community school' }] }),
|
||||
})) as unknown as typeof fetch;
|
||||
push.mockClear();
|
||||
searchParams = new URLSearchParams();
|
||||
});
|
||||
afterEach(() => { global.fetch = realFetch; });
|
||||
|
||||
describe('FilterBar autosuggest', () => {
|
||||
it('is a combobox only when the flag is on', () => {
|
||||
const { rerender } = render(<FilterBar filters={FILTERS} autosuggest={false} />);
|
||||
expect(screen.queryByRole('combobox', { name: 'School name or postcode' })).not.toBeInTheDocument();
|
||||
rerender(<FilterBar filters={FILTERS} autosuggest />);
|
||||
expect(screen.getByRole('combobox', { name: 'School name or postcode' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('makes no request while the flag is off', async () => {
|
||||
// Off means off: no listener, no fetch, no markup.
|
||||
render(<FilterBar filters={FILTERS} autosuggest={false} />);
|
||||
await userEvent.type(screen.getByPlaceholderText(/School name or postcode/i),
|
||||
'brecknock');
|
||||
expect(global.fetch).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('shows suggestions and navigates when one is chosen', async () => {
|
||||
render(<FilterBar filters={FILTERS} autosuggest />);
|
||||
await userEvent.type(screen.getByRole('combobox', { name: 'School name or postcode' }), 'brecknock');
|
||||
const option = await screen.findByRole('option', { name: /Brecknock/ });
|
||||
await userEvent.click(option);
|
||||
expect(push).toHaveBeenCalledWith(
|
||||
expect.stringContaining('/school/100010'));
|
||||
});
|
||||
|
||||
it('suppresses suggestions once the value is a postcode', async () => {
|
||||
// The box takes a name OR a postcode; suggestions must get out of the way.
|
||||
//
|
||||
// fireEvent.change, not userEvent.type: typing sets "N", "NW", "NW1"... and
|
||||
// "NW1" is not a postcode, so a request for it is correct behaviour. Only
|
||||
// the settled value is the assertion, so set it in one go.
|
||||
render(<FilterBar filters={FILTERS} autosuggest />);
|
||||
fireEvent.change(screen.getByRole('combobox', { name: 'School name or postcode' }), { target: { value: 'NW1 1AA' } });
|
||||
await new Promise((r) => setTimeout(r, 300)); // past the 200ms debounce
|
||||
expect(global.fetch).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('Enter with no active option still submits the free-text search', async () => {
|
||||
// The existing behaviour is preserved, not replaced.
|
||||
render(<FilterBar filters={FILTERS} autosuggest />);
|
||||
const input = screen.getByRole('combobox', { name: 'School name or postcode' });
|
||||
await userEvent.type(input, 'brecknock{Enter}');
|
||||
// updateURL pushes inside startTransition, so the call is not synchronous.
|
||||
await waitFor(() => expect(push).toHaveBeenCalledWith(
|
||||
expect.stringContaining('search=brecknock')));
|
||||
});
|
||||
});
|
||||
|
||||
describe('FilterBar autosuggest does not reopen over results', () => {
|
||||
it('stays shut when the input arrives pre-filled from the URL', async () => {
|
||||
/*
|
||||
* The results-page bar renders with the search term already in the input.
|
||||
* Opening on that would drop the dropdown on top of the results the search
|
||||
* just produced — which is exactly what happened: the first result became
|
||||
* unclickable, because the list sat over it and swallowed the pointer.
|
||||
*
|
||||
* Suggestions answer typing, not the presence of a value.
|
||||
*/
|
||||
searchParams = new URLSearchParams('search=brecknock');
|
||||
render(<FilterBar filters={FILTERS} autosuggest />);
|
||||
|
||||
expect(screen.getByRole('combobox', { name: 'School name or postcode' })).toHaveValue('brecknock');
|
||||
await new Promise((r) => setTimeout(r, 300)); // past the 200ms debounce
|
||||
expect(global.fetch).not.toHaveBeenCalled();
|
||||
expect(screen.queryByRole('listbox')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('closes the dropdown when the search is submitted', async () => {
|
||||
render(<FilterBar filters={FILTERS} autosuggest />);
|
||||
const input = screen.getByRole('combobox', { name: 'School name or postcode' });
|
||||
|
||||
await userEvent.type(input, 'brecknock');
|
||||
expect(await screen.findByRole('listbox')).toBeInTheDocument();
|
||||
|
||||
await userEvent.type(input, '{Enter}');
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('listbox')).not.toBeInTheDocument());
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,137 @@
|
||||
import { fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
import { track } from '@/lib/analytics';
|
||||
|
||||
/*
|
||||
* School type offers six groups a parent recognises, not GIAS's 34 types, and
|
||||
* a Faith filter sits beside it (spec 2026-10-02-school-type-groups-and-faith-
|
||||
* filter-design.md). Both lists come from /api/filters.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Wandsworth'], school_types: ['Community school'], years: [],
|
||||
phases: ['Primary', 'Secondary'], genders: [], admissions_policies: [],
|
||||
school_type_groups: [
|
||||
{ value: 'academy', label: 'State school: academy or free school' },
|
||||
{ value: 'council', label: 'State school: council-run' },
|
||||
{ value: 'special', label: 'Special school (SEND)' },
|
||||
],
|
||||
faiths: [
|
||||
{ value: 'none', label: 'No religious character' },
|
||||
{ value: 'roman_catholic', label: 'Roman Catholic' },
|
||||
],
|
||||
};
|
||||
|
||||
const pushedParams = () => new URLSearchParams(push.mock.calls.at(-1)![0].split('?')[1]);
|
||||
const openSheet = () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Filters/ }));
|
||||
return screen.getByRole('dialog', { name: 'Filters' });
|
||||
};
|
||||
const optionsOf = (scope: HTMLElement, name: string) =>
|
||||
within(within(scope).getByRole('combobox', { name })).getAllByRole('option').map((o) => o.textContent);
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
push.mockClear();
|
||||
jest.mocked(track).mockClear();
|
||||
});
|
||||
|
||||
describe('School type', () => {
|
||||
it('offers the groups, not the GIAS types, on desktop and in the sheet', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const groups = ['Any school type', 'State school: academy or free school',
|
||||
'State school: council-run', 'Special school (SEND)'];
|
||||
expect(optionsOf(screen.getByRole('group', { name: 'Filters' }), 'School type')).toEqual(groups);
|
||||
expect(optionsOf(openSheet(), 'School type')).toEqual(groups);
|
||||
});
|
||||
|
||||
it('puts the group key in the URL and names the chip by its label', () => {
|
||||
const view = render(<FilterBar filters={filters} />);
|
||||
fireEvent.change(within(openSheet()).getByRole('combobox', { name: 'School type' }),
|
||||
{ target: { value: 'special' } });
|
||||
expect(pushedParams().get('school_type')).toBe('special');
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=special');
|
||||
view.rerender(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Special school (SEND)' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('names an old raw-label link\'s chip by that label', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=Community+school');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Community school' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('is left out when the API sends no groups', () => {
|
||||
render(<FilterBar filters={{ ...filters, school_type_groups: undefined }} />);
|
||||
expect(screen.queryByRole('combobox', { name: 'School type' })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('Faith', () => {
|
||||
it('offers its options in the More filters panel and in the sheet', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /More filters/ }));
|
||||
const faiths = ['Any faith or none', 'No religious character', 'Roman Catholic'];
|
||||
expect(optionsOf(document.body, 'Faith')).toEqual(faiths);
|
||||
expect(optionsOf(openSheet(), 'Faith')).toEqual(faiths);
|
||||
});
|
||||
|
||||
it('shows as a chip, counts on both buttons, and clears with Clear all', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&faith=roman_catholic');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Roman Catholic' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'Filters, 1 applied' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: /More filters \(1\)/ })).toBeInTheDocument();
|
||||
fireEvent.click(within(screen.getByRole('group', { name: 'Applied filters' }))
|
||||
.getByRole('button', { name: 'Clear all' }));
|
||||
expect(pushedParams().get('faith')).toBeNull();
|
||||
expect(pushedParams().get('postcode')).toBe('SW196AR');
|
||||
});
|
||||
|
||||
it('goes into the search analytics event', () => {
|
||||
params = new URLSearchParams('faith=roman_catholic');
|
||||
render(<FilterBar filters={filters} />);
|
||||
const input = screen.getByRole('searchbox', { name: 'School name or postcode' });
|
||||
fireEvent.change(input, { target: { value: 'st marys' } });
|
||||
fireEvent.submit(input.closest('form')!);
|
||||
expect(track).toHaveBeenCalledWith('search_submitted',
|
||||
expect.objectContaining({ filters_active: 'faith=roman_catholic', filters_count: 1 }));
|
||||
});
|
||||
|
||||
it('is left out when the API sends no faiths', () => {
|
||||
render(<FilterBar filters={{ ...filters, faiths: [] }} />);
|
||||
expect(within(openSheet()).queryByRole('combobox', { name: 'Faith' })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('a URL value the options do not spell the same way', () => {
|
||||
const row = () => screen.getByRole('group', { name: 'Filters' });
|
||||
|
||||
it('shows an old raw-label type in the select, and lets "Any" clear it', () => {
|
||||
params = new URLSearchParams('search=school&school_type=Community+school');
|
||||
render(<FilterBar filters={filters} />);
|
||||
const type = within(row()).getByRole('combobox', { name: 'School type' });
|
||||
expect(type).toHaveValue('Community school');
|
||||
expect(within(type).getByRole('option', { name: 'Community school' })).toBeInTheDocument();
|
||||
fireEvent.change(type, { target: { value: '' } });
|
||||
expect(pushedParams().get('school_type')).toBeNull();
|
||||
});
|
||||
|
||||
it('matches a key in another case, in the select and the chip', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=Special&faith=Roman_Catholic');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(within(row()).getByRole('combobox', { name: 'School type' })).toHaveValue('special');
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Special school (SEND)' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'Remove filter: Roman Catholic' })).toBeInTheDocument();
|
||||
expect(screen.getByRole('combobox', { name: 'Faith' })).toHaveValue('roman_catholic');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,157 @@
|
||||
import { fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
|
||||
/*
|
||||
* Phones filter through one "Filters" button and a bottom sheet holding every
|
||||
* filter, rather than a sideways-scrolling row whose later chips (phase, type)
|
||||
* sat off-screen beside a "More filters" panel that did not contain them.
|
||||
* Which markup shows at which width is CSS and invisible to jsdom; these pin
|
||||
* the behaviour and the accessible names.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Wandsworth', 'Merton'], school_types: ['Community school'], years: [],
|
||||
phases: ['Primary', 'Secondary'], genders: ['Girls', 'Mixed'], admissions_policies: [],
|
||||
school_type_groups: [{ value: 'council', label: 'State school: council-run' }],
|
||||
};
|
||||
|
||||
const pushedParams = () => new URLSearchParams(push.mock.calls.at(-1)![0].split('?')[1]);
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
push.mockClear();
|
||||
});
|
||||
|
||||
describe('the phone Filters button', () => {
|
||||
it('sits beside the folded search summary', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const summary = screen.getByRole('button', { name: /Edit search/ });
|
||||
expect(summary.parentElement).toContainElement(screen.getByRole('button', { name: 'Filters' }));
|
||||
});
|
||||
|
||||
it('counts every applied filter, phase and type included', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=primary&school_type=council');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Filters, 2 applied' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('is still offered before anything has been searched', () => {
|
||||
params = new URLSearchParams('local_authority=Wandsworth');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: 'Filters, 1 applied' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('stays out of the hero', () => {
|
||||
render(<FilterBar filters={filters} isHero />);
|
||||
expect(screen.queryByRole('button', { name: /^Filters/ })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the filter sheet', () => {
|
||||
const openSheet = () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Filters/ }));
|
||||
return screen.getByRole('dialog', { name: 'Filters' });
|
||||
};
|
||||
|
||||
it('holds every filter in one place', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const sheet = openSheet();
|
||||
expect(within(sheet).getByRole('radiogroup', { name: 'Distance' })).toBeInTheDocument();
|
||||
for (const name of ['Phase', 'School type', 'Local authority']) {
|
||||
expect(within(sheet).getByRole('combobox', { name })).toBeInTheDocument();
|
||||
}
|
||||
});
|
||||
|
||||
it('shows the secondary-only filters once they apply', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=secondary');
|
||||
render(<FilterBar filters={filters} />);
|
||||
const sheet = openSheet();
|
||||
for (const name of ['Gender', 'Sixth form']) {
|
||||
expect(within(sheet).getByRole('combobox', { name })).toBeInTheDocument();
|
||||
}
|
||||
});
|
||||
|
||||
it('offers distance only for a postcode search', () => {
|
||||
params = new URLSearchParams('search=southmead');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(within(openSheet()).queryByRole('radiogroup', { name: 'Distance' })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('changes the distance', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const distance = within(openSheet()).getByRole('radiogroup', { name: 'Distance' });
|
||||
expect(within(distance).getByRole('radio', { name: 'Within 1 mile' })).toBeChecked();
|
||||
fireEvent.click(within(distance).getByRole('radio', { name: 'Within 3 miles' }));
|
||||
expect(pushedParams().get('radius')).toBe('3');
|
||||
});
|
||||
|
||||
it('applies a change straight away and stays open for the next one', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const sheet = openSheet();
|
||||
fireEvent.change(within(sheet).getByRole('combobox', { name: 'Phase' }), { target: { value: 'primary' } });
|
||||
expect(pushedParams().get('phase')).toBe('primary');
|
||||
expect(screen.getByRole('dialog', { name: 'Filters' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('closes on the results button, which gives the count', () => {
|
||||
render(<FilterBar filters={filters} resultCount={12} />);
|
||||
fireEvent.click(within(openSheet()).getByRole('button', { name: 'Show 12 schools' }));
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('says so when nothing matches', () => {
|
||||
render(<FilterBar filters={filters} resultCount={0} />);
|
||||
expect(within(openSheet()).getByRole('button', { name: 'No schools match' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('clears the filters but keeps the search', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=primary&local_authority=Wandsworth');
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(within(openSheet()).getByRole('button', { name: 'Clear all' }));
|
||||
const next = pushedParams();
|
||||
expect(next.get('phase')).toBeNull();
|
||||
expect(next.get('local_authority')).toBeNull();
|
||||
expect(next.get('postcode')).toBe('SW196AR');
|
||||
expect(next.get('radius')).toBe('1');
|
||||
});
|
||||
});
|
||||
|
||||
describe('the applied-filter chips', () => {
|
||||
it('appear only when something is applied', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.queryByRole('group', { name: 'Applied filters' })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('name each filter by its label and remove it on tap', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=primary&gender=girls&has_sixth_form=no');
|
||||
render(<FilterBar filters={filters} />);
|
||||
const chips = screen.getByRole('group', { name: 'Applied filters' });
|
||||
for (const label of ['Primary', 'Girls', 'Without sixth form']) {
|
||||
expect(within(chips).getByRole('button', { name: `Remove filter: ${label}` })).toBeInTheDocument();
|
||||
}
|
||||
fireEvent.click(within(chips).getByRole('button', { name: 'Remove filter: Primary' }));
|
||||
const next = pushedParams();
|
||||
expect(next.get('phase')).toBeNull();
|
||||
expect(next.get('gender')).toBe('girls');
|
||||
expect(next.get('postcode')).toBe('SW196AR');
|
||||
});
|
||||
|
||||
it('carry a Clear all that keeps the search', () => {
|
||||
params = new URLSearchParams('search=southmead&school_type=council');
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(within(screen.getByRole('group', { name: 'Applied filters' }))
|
||||
.getByRole('button', { name: 'Clear all' }));
|
||||
const next = pushedParams();
|
||||
expect(next.get('school_type')).toBeNull();
|
||||
expect(next.get('search')).toBe('southmead');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,81 @@
|
||||
import { act, fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
|
||||
/*
|
||||
* While a filter change is navigating, the sheet's controls stay enabled: a
|
||||
* control disabled under the user's focus drops it to <body>, and a keyboard or
|
||||
* screen-reader user is thrown out of the sheet after every change. A second
|
||||
* change made before the first lands must build on the first, not on the URL
|
||||
* useSearchParams still reports.
|
||||
*/
|
||||
|
||||
// Every transition stays pending, as a slow server render would.
|
||||
jest.mock('react', () => ({
|
||||
...jest.requireActual('react'),
|
||||
useTransition: () => [true, (fn: () => void) => fn()],
|
||||
}));
|
||||
|
||||
const params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Wandsworth'], school_types: ['Community school'], years: [],
|
||||
phases: ['Primary', 'Secondary'], genders: [], admissions_policies: [],
|
||||
school_type_groups: [{ value: 'council', label: 'State school: council-run' }],
|
||||
};
|
||||
|
||||
const pushedParams = () => new URLSearchParams(push.mock.calls.at(-1)![0].split('?')[1]);
|
||||
|
||||
const openSheet = () => {
|
||||
fireEvent.click(screen.getByRole('button', { name: /^Filters/ }));
|
||||
return screen.getByRole('dialog', { name: 'Filters' });
|
||||
};
|
||||
|
||||
beforeEach(() => push.mockClear());
|
||||
|
||||
it('keeps the sheet usable, and says it is busy, while a change lands', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const sheet = openSheet();
|
||||
for (const name of ['Phase', 'School type', 'Local authority']) {
|
||||
expect(within(sheet).getByRole('combobox', { name })).toBeEnabled();
|
||||
}
|
||||
expect(within(sheet).getByRole('radio', { name: 'Within 3 miles' })).toBeEnabled();
|
||||
expect(sheet.querySelector('[aria-busy="true"]')).not.toBeNull();
|
||||
});
|
||||
|
||||
it('builds a second change on the first, not on the URL it has not reached', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const sheet = openSheet();
|
||||
fireEvent.change(within(sheet).getByRole('combobox', { name: 'Phase' }), { target: { value: 'primary' } });
|
||||
fireEvent.change(within(sheet).getByRole('combobox', { name: 'School type' }),
|
||||
{ target: { value: 'council' } });
|
||||
const next = pushedParams();
|
||||
expect(next.get('phase')).toBe('primary');
|
||||
expect(next.get('school_type')).toBe('council');
|
||||
expect(next.get('postcode')).toBe('SW196AR');
|
||||
});
|
||||
|
||||
describe('a screen that widens past phone width', () => {
|
||||
let listeners: ((e: { matches: boolean }) => void)[] = [];
|
||||
beforeEach(() => {
|
||||
listeners = [];
|
||||
window.matchMedia = jest.fn().mockImplementation((query: string) => ({
|
||||
matches: true, media: query,
|
||||
addEventListener: (_: string, l: (e: { matches: boolean }) => void) => listeners.push(l),
|
||||
removeEventListener: jest.fn(),
|
||||
}));
|
||||
});
|
||||
|
||||
it('closes the sheet, leaving the desktop row as the only filters', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
openSheet();
|
||||
act(() => listeners.forEach((l) => l({ matches: false })));
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,47 @@
|
||||
/**
|
||||
* The footer is the only navigational route to /about and /blog, so it is
|
||||
* where a dark flag would otherwise leave a link into a 404.
|
||||
*
|
||||
* Both props default to false. A caller that forgets to pass them hides the
|
||||
* links, which is the direction that cannot break a page — the same reasoning
|
||||
* as backend/flags.py's "every flag defaults to False".
|
||||
*/
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { Footer } from '@/components/Footer';
|
||||
|
||||
describe('footer feature links', () => {
|
||||
it('links to both when both flags are on', () => {
|
||||
render(<Footer aboutEnabled blogEnabled />);
|
||||
expect(screen.getByRole('link', { name: /who's behind this/i }))
|
||||
.toHaveAttribute('href', '/about');
|
||||
expect(screen.getByRole('link', { name: /^blog$/i }))
|
||||
.toHaveAttribute('href', '/blog');
|
||||
});
|
||||
|
||||
it('omits the about link when that flag is dark', () => {
|
||||
render(<Footer blogEnabled />);
|
||||
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
|
||||
expect(screen.getByRole('link', { name: /^blog$/i })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('omits the blog link when that flag is dark', () => {
|
||||
render(<Footer aboutEnabled />);
|
||||
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
|
||||
expect(screen.getByRole('link', { name: /who's behind this/i }))
|
||||
.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('drops the whole section when both are dark, not an empty heading', () => {
|
||||
// Shipping dark means the footer renders as it did before the feature
|
||||
// existed, not as a section with its contents removed.
|
||||
render(<Footer />);
|
||||
expect(screen.queryByRole('heading', { name: /^about$/i })).toBeNull();
|
||||
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
|
||||
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
|
||||
});
|
||||
|
||||
it('defaults to dark when a caller passes nothing', () => {
|
||||
render(<Footer />);
|
||||
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,86 @@
|
||||
import { act, fireEvent, render, screen } from '@testing-library/react';
|
||||
import { HomeView } from '@/components/HomeView';
|
||||
import { fetchSchools } from '@/lib/api';
|
||||
import { primaryFixture } from '../support/schoolFixtures';
|
||||
import type { SchoolsResponse, School } from '@/lib/types';
|
||||
|
||||
let params = new URLSearchParams('postcode=SW1A+1AA');
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push: jest.fn(), replace: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/context/ComparisonContext', () => ({
|
||||
useComparisonContext: () => ({ addSchool: jest.fn(), removeSchool: jest.fn(), selectedSchools: [] }),
|
||||
}));
|
||||
jest.mock('@/lib/api', () => ({
|
||||
fetchSchools: jest.fn(),
|
||||
fetchNationalAverages: jest.fn(async () => ({})),
|
||||
fetchLAaverages: jest.fn(async () => ({ secondary: { attainment_8_by_la: {} } })),
|
||||
}));
|
||||
// Renders only the List/Map switch HomeView hands it, which lives in its row.
|
||||
jest.mock('@/components/FilterBar', () => ({
|
||||
FilterBar: ({ viewSwitch }: { viewSwitch?: unknown }) => viewSwitch || null,
|
||||
}));
|
||||
jest.mock('@/components/SchoolRow', () => ({ SchoolRow: ({ school }: {school: School}) => <div>{school.school_name}</div> }));
|
||||
jest.mock('@/components/SchoolMap', () => ({ SchoolMap: ({ schools }: {schools: School[]}) => <div data-testid="map">{schools.map(s => s.school_name).join(',')}</div> }));
|
||||
|
||||
const filters = { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
|
||||
function response(name: string): SchoolsResponse {
|
||||
return { schools: [{ ...primaryFixture.schoolInfo, school_name: name }],
|
||||
total: 2, page: 1, page_size: 1, total_pages: 2 };
|
||||
}
|
||||
function deferred() {
|
||||
let resolve!: (value: SchoolsResponse) => void;
|
||||
let reject!: (error: Error) => void;
|
||||
const promise = new Promise<SchoolsResponse>((yes, no) => { resolve = yes; reject = no; });
|
||||
return { promise, resolve, reject };
|
||||
}
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW1A+1AA');
|
||||
jest.mocked(fetchSchools).mockReset();
|
||||
});
|
||||
|
||||
test('load-more results from an old search are discarded, even after returning to it', async () => {
|
||||
// Name searches: a postcode search opens on the map, which has no Load more.
|
||||
params = new URLSearchParams('search=abbey');
|
||||
const pending = deferred();
|
||||
jest.mocked(fetchSchools).mockReturnValueOnce(pending.promise);
|
||||
const view = render(<HomeView initialSchools={response('Initial A')} filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Load more schools' }));
|
||||
const signal = jest.mocked(fetchSchools).mock.calls[0][1]?.signal;
|
||||
params = new URLSearchParams('search=brecknock');
|
||||
view.rerender(<HomeView initialSchools={response('Initial B')} filters={filters} />);
|
||||
expect(signal?.aborted).toBe(true);
|
||||
params = new URLSearchParams('search=abbey');
|
||||
view.rerender(<HomeView initialSchools={response('Fresh A')} filters={filters} />);
|
||||
await act(async () => pending.resolve(response('Stale append')));
|
||||
expect(screen.queryByText('Stale append')).not.toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'Load more schools' })).toBeEnabled();
|
||||
});
|
||||
|
||||
test('an older map response cannot overwrite the current search', async () => {
|
||||
const first = deferred(), second = deferred();
|
||||
jest.mocked(fetchSchools).mockReturnValueOnce(first.promise).mockReturnValueOnce(second.promise);
|
||||
const initial = response('Initial A');
|
||||
const view = render(<HomeView initialSchools={initial} filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Map' }));
|
||||
params = new URLSearchParams('postcode=SW2+1AA');
|
||||
view.rerender(<HomeView initialSchools={response('Initial B')} filters={filters} />);
|
||||
await act(async () => second.resolve(response('Current map')));
|
||||
await act(async () => first.resolve(response('Stale map')));
|
||||
expect(screen.getByTestId('map')).toHaveTextContent('Current map');
|
||||
expect(screen.getByTestId('map')).not.toHaveTextContent('Stale map');
|
||||
});
|
||||
|
||||
test('failed map requests can be retried by reopening the map', async () => {
|
||||
const pending = deferred();
|
||||
jest.mocked(fetchSchools).mockReturnValueOnce(pending.promise).mockResolvedValue(response('Retry result'));
|
||||
render(<HomeView initialSchools={response('Initial')} filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Map' }));
|
||||
await act(async () => pending.reject(new Error('offline')));
|
||||
fireEvent.click(screen.getByRole('button', { name: 'List' }));
|
||||
await act(async () => fireEvent.click(screen.getByRole('button', { name: 'Map' })));
|
||||
expect(fetchSchools).toHaveBeenCalledTimes(2);
|
||||
expect(screen.getByTestId('map')).toHaveTextContent('Retry result');
|
||||
});
|
||||
@@ -0,0 +1,107 @@
|
||||
import { act, fireEvent, render } from '@testing-library/react';
|
||||
import LeafletMapInner from '@/components/LeafletMapInner';
|
||||
import { primaryFixture } from '../support/schoolFixtures';
|
||||
import type { School } from '@/lib/types';
|
||||
|
||||
/*
|
||||
* The results map's own logic, against real Leaflet in jsdom: which pin is
|
||||
* selected, whether the card opens, and what the card offers. jsdom lays out
|
||||
* nothing, so this pins behaviour, never positions.
|
||||
*/
|
||||
|
||||
const base = primaryFixture.schoolInfo;
|
||||
const a: School = { ...base, urn: 1, school_name: 'Southmead Primary School', latitude: 51.43, longitude: -0.21, distance: 0.2, rwm_expected_pct: 52 };
|
||||
const b: School = { ...base, urn: 2, school_name: 'Greenmead School', latitude: 51.431, longitude: -0.205, distance: 0.2,
|
||||
school_type: 'Community special school', rwm_expected_pct: 0, reading_expected_pct: 0, writing_expected_pct: 0, maths_expected_pct: 0 };
|
||||
const schools = [a, b];
|
||||
const centre: [number, number] = [51.43, -0.21];
|
||||
|
||||
function setWide(wide: boolean) {
|
||||
window.matchMedia = ((q: string) => ({
|
||||
matches: wide, media: q, addEventListener() {}, removeEventListener() {},
|
||||
})) as unknown as typeof window.matchMedia;
|
||||
}
|
||||
|
||||
function renderMap(props: Partial<React.ComponentProps<typeof LeafletMapInner>> = {}) {
|
||||
const all = {
|
||||
schools, center: centre, zoom: 13, referencePoint: centre, radiusMiles: 1,
|
||||
nationalAvgRwm: 62, ...props,
|
||||
};
|
||||
const view = render(<LeafletMapInner {...all} />);
|
||||
return { ...view, rerender: (next: Partial<typeof all>) => view.rerender(<LeafletMapInner {...all} {...next} />) };
|
||||
}
|
||||
|
||||
beforeEach(() => setWide(true));
|
||||
|
||||
it('draws a pin per school, the search location and the radius', () => {
|
||||
const { container } = renderMap();
|
||||
expect(container.querySelectorAll('.sc-pin')).toHaveLength(2);
|
||||
expect(container.querySelector('.sc-home')).not.toBeNull();
|
||||
expect(container.querySelector('.sc-radius-label')).toHaveTextContent('1 mile');
|
||||
});
|
||||
|
||||
it('reports a pin click, and marks and opens the selected school', () => {
|
||||
const onMarkerClick = jest.fn();
|
||||
const { container, rerender } = renderMap({ onMarkerClick });
|
||||
fireEvent.click(container.querySelectorAll('.sc-pin')[0]);
|
||||
expect(onMarkerClick).toHaveBeenCalledWith(a);
|
||||
|
||||
rerender({ onMarkerClick, selectedUrn: 1 });
|
||||
expect(container.querySelectorAll('.sc-pin--selected')).toHaveLength(1);
|
||||
expect(container.querySelector('.sc-popup')).toHaveTextContent('Southmead Primary School');
|
||||
expect(container.querySelector('.sc-popup')).toHaveTextContent('52% RWM -10 pts');
|
||||
});
|
||||
|
||||
it('keeps the special-school card free of a benchmark and a placeholder 0%', () => {
|
||||
const { container } = renderMap({ selectedUrn: 2 });
|
||||
const card = container.querySelector('.sc-popup')!;
|
||||
expect(card).toHaveTextContent('Greenmead School');
|
||||
expect(card).not.toHaveTextContent(/%|pts/);
|
||||
});
|
||||
|
||||
it('adds to compare from the card, and the card follows the basket', () => {
|
||||
const onAddToCompare = jest.fn();
|
||||
const { container, rerender } = renderMap({ onAddToCompare, selectedUrn: 1, compareUrns: [] });
|
||||
fireEvent.click(container.querySelector('[data-compare]')!);
|
||||
expect(onAddToCompare).toHaveBeenCalledWith(a);
|
||||
|
||||
rerender({ onAddToCompare, selectedUrn: 1, compareUrns: [1] });
|
||||
expect(container.querySelector('[data-compare]')).toHaveTextContent('✓ Comparing');
|
||||
fireEvent.click(container.querySelector('[data-compare]')!);
|
||||
expect(onAddToCompare).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('tells the page when the card is closed from the map, but not when replaced', () => {
|
||||
const onDeselect = jest.fn();
|
||||
const { container, rerender } = renderMap({ onDeselect, selectedUrn: 1 });
|
||||
rerender({ onDeselect, selectedUrn: 2 });
|
||||
expect(onDeselect).not.toHaveBeenCalled();
|
||||
|
||||
fireEvent.click(container.querySelector('.leaflet-popup-close-button')!);
|
||||
expect(onDeselect).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('keeps the selection when the results reload under it', () => {
|
||||
const onDeselect = jest.fn();
|
||||
const { container, rerender } = renderMap({ onDeselect, selectedUrn: 1 });
|
||||
act(() => rerender({ onDeselect, selectedUrn: 1, schools: [...schools] }));
|
||||
expect(onDeselect).not.toHaveBeenCalled();
|
||||
expect(container.querySelector('.sc-popup')).toHaveTextContent('Southmead Primary School');
|
||||
});
|
||||
|
||||
it('opens no card on a narrow screen, where the page shows a bottom sheet', () => {
|
||||
setWide(false);
|
||||
const { container } = renderMap({ selectedUrn: 1 });
|
||||
expect(container.querySelectorAll('.sc-pin--selected')).toHaveLength(1);
|
||||
expect(container.querySelector('.sc-popup')).toBeNull();
|
||||
});
|
||||
|
||||
it('puts the card back when the pins are rebuilt for a reason other than the schools', () => {
|
||||
const onDeselect = jest.fn();
|
||||
const { container, rerender } = renderMap({ onDeselect, selectedUrn: 1 });
|
||||
act(() => rerender({ onDeselect, selectedUrn: 1, radiusMiles: 3, referencePoint: [51.43, -0.21] }));
|
||||
expect(onDeselect).not.toHaveBeenCalled();
|
||||
expect(container.querySelector('.sc-radius-label')).toHaveTextContent('3 miles');
|
||||
expect(container.querySelector('.sc-popup')).toHaveTextContent('Southmead Primary School');
|
||||
expect(container.querySelectorAll('.sc-pin--selected')).toHaveLength(1);
|
||||
});
|
||||
@@ -0,0 +1,92 @@
|
||||
/**
|
||||
* The module that ends the stranding: before it, a school page's only anchor
|
||||
* pointed at the school's own website, so ~27k pages sent authority off-site
|
||||
* and none of it reached the location layer.
|
||||
*/
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
|
||||
|
||||
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
|
||||
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
|
||||
const cm15 = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
|
||||
|
||||
describe('NearbyPlaces', () => {
|
||||
it('links to every place the school belongs to', () => {
|
||||
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
|
||||
|
||||
expect(screen.getByRole('link', { name: /Brentwood/ }))
|
||||
.toHaveAttribute('href', '/schools/brentwood');
|
||||
expect(screen.getByRole('link', { name: /Essex/ }))
|
||||
.toHaveAttribute('href', '/schools/authority/essex');
|
||||
expect(screen.getByRole('link', { name: /CM15/ }))
|
||||
.toHaveAttribute('href', '/schools/near/cm15');
|
||||
});
|
||||
|
||||
it('says how many schools each link leads to', () => {
|
||||
// An anchor that states its destination's size is worth more to a reader
|
||||
// and to a crawler than "see more".
|
||||
render(<NearbyPlaces places={[brentwood]} />);
|
||||
expect(screen.getByRole('link', { name: /37 schools in Brentwood/ }))
|
||||
.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('renders nothing at all when the school has no published places', () => {
|
||||
// Not an empty heading. A school whose town and authority both fall below
|
||||
// the threshold has nowhere to point, and the page should look as it did
|
||||
// before the module existed.
|
||||
const { container } = render(<NearbyPlaces places={[]} />);
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
|
||||
it('puts the narrowest place first, which is the most useful link', () => {
|
||||
// The API orders widest-first for the breadcrumb; a reader on a school
|
||||
// page wants its town before its county.
|
||||
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
|
||||
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
|
||||
expect(hrefs.indexOf('/schools/brentwood'))
|
||||
.toBeLessThan(hrefs.indexOf('/schools/authority/essex'));
|
||||
});
|
||||
|
||||
it('handles a singular count without saying "1 schools"', () => {
|
||||
render(<NearbyPlaces places={[{ ...brentwood, count: 1 }]} />);
|
||||
expect(screen.getByRole('link', { name: /1 school in Brentwood/ }))
|
||||
.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('links the phase page the school appears on', () => {
|
||||
// "primary schools in brentwood" is the query these pages exist for.
|
||||
render(<NearbyPlaces places={[{
|
||||
...brentwood,
|
||||
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
|
||||
}]} />);
|
||||
|
||||
expect(screen.getByRole('link', { name: /22 primary schools in Brentwood/ }))
|
||||
.toHaveAttribute('href', '/schools/brentwood/primary');
|
||||
});
|
||||
|
||||
it('links both phase pages for an all-through school', () => {
|
||||
render(<NearbyPlaces places={[{
|
||||
...brentwood,
|
||||
phases: [
|
||||
{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' },
|
||||
{ phase: 'secondary', count: 9, url: '/schools/brentwood/secondary' },
|
||||
],
|
||||
}]} />);
|
||||
|
||||
expect(screen.getByRole('link', { name: /22 primary schools/ })).toBeInTheDocument();
|
||||
expect(screen.getByRole('link', { name: /9 secondary schools/ })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('keeps a phase link next to the place it belongs to', () => {
|
||||
// Grouping matters: "22 primary schools in Brentwood" directly after
|
||||
// "37 schools in Brentwood" reads as one place, not two unrelated links.
|
||||
render(<NearbyPlaces places={[essex, {
|
||||
...brentwood,
|
||||
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
|
||||
}]} />);
|
||||
|
||||
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
|
||||
expect(hrefs.indexOf('/schools/brentwood/primary'))
|
||||
.toBe(hrefs.indexOf('/schools/brentwood') + 1);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,192 @@
|
||||
/**
|
||||
* The section's job is to be honest about what it is showing. These tests pin
|
||||
* the ways it could mislead: rendering below the minimum, claiming a likeness
|
||||
* it does not rank on, showing a missing figure as a number, or hiding a card
|
||||
* behind an arrow where a crawler cannot reach it.
|
||||
*/
|
||||
|
||||
import { render, screen } from '@testing-library/react';
|
||||
|
||||
import {
|
||||
nearbyNoun,
|
||||
NearbySchoolsSection,
|
||||
shouldRenderNearby,
|
||||
} from '@/components/school/NearbySchoolsSection';
|
||||
import type { NearbySchool } from '@/lib/types';
|
||||
|
||||
jest.mock('@/components/school/AddToCompareButton', () => ({
|
||||
AddToCompareButton: ({ school }: { school: NearbySchool }) => (
|
||||
<button type="button">Add {school.school_name} to compare</button>
|
||||
),
|
||||
}));
|
||||
|
||||
jest.mock('@/components/school/NearbySchoolsCompareBar', () => ({
|
||||
NearbySchoolsCompareBar: ({ thisUrn }: { thisUrn: number }) => (
|
||||
<div data-testid="compare-bar">bar for {thisUrn}</div>
|
||||
),
|
||||
}));
|
||||
|
||||
function school(overrides: Partial<NearbySchool> = {}): NearbySchool {
|
||||
return {
|
||||
urn: 100002,
|
||||
school_name: 'Willow Lane Primary School',
|
||||
distance_miles: 0.6,
|
||||
school_type: 'Community school',
|
||||
age_range: '4-11',
|
||||
shared: ['Mixed', 'No religious character'],
|
||||
metric_value: 74,
|
||||
metric_key: 'rwm_expected_pct',
|
||||
metric_year: 202425,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function renderSection(nearby: NearbySchool[]) {
|
||||
return render(
|
||||
<NearbySchoolsSection
|
||||
urn={100001}
|
||||
schoolName="Meadowbrook Primary School"
|
||||
phase="Primary"
|
||||
thisMetricValue={72}
|
||||
nearby={nearby}
|
||||
/>,
|
||||
);
|
||||
}
|
||||
|
||||
describe('render gates', () => {
|
||||
it.each([
|
||||
['undefined', undefined],
|
||||
['null', null],
|
||||
['empty', []],
|
||||
['a single school', [school()]],
|
||||
])('renders nothing for %s', (_label, value) => {
|
||||
expect(shouldRenderNearby(value as NearbySchool[] | null | undefined)).toBe(false);
|
||||
});
|
||||
|
||||
it('renders for two or more schools', () => {
|
||||
expect(shouldRenderNearby([school(), school({ urn: 100003 })])).toBe(true);
|
||||
});
|
||||
|
||||
it('returns null rather than an empty shell below the minimum', () => {
|
||||
const { container } = renderSection([school()]);
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
});
|
||||
|
||||
describe('what the section claims', () => {
|
||||
it('never claims a similar intake, because it does not rank on one', () => {
|
||||
renderSection([school(), school({ urn: 100003, shared: [] })]);
|
||||
expect(screen.queryByText(/similar intake/i)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('is headed "Other schools nearby", not "similar"', () => {
|
||||
renderSection([school(), school({ urn: 100003 })]);
|
||||
expect(screen.getByRole('heading', { name: 'Other schools nearby' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('shows chips for what is shared', () => {
|
||||
renderSection([school({ shared: ['Mixed', 'Roman Catholic'] }), school({ urn: 100003 })]);
|
||||
expect(screen.getAllByText('Roman Catholic').length).toBe(1);
|
||||
});
|
||||
|
||||
it('shows no chips at all when nothing is shared, rather than inventing one', () => {
|
||||
const { container } = render(
|
||||
<NearbySchoolsSection
|
||||
urn={100001}
|
||||
schoolName="Meadowbrook Primary School"
|
||||
phase="Primary"
|
||||
thisMetricValue={72}
|
||||
nearby={[school({ shared: [] }), school({ urn: 100003, shared: [] })]}
|
||||
/>,
|
||||
);
|
||||
// The card still carries its distance, name, type and figure — just no
|
||||
// claim of likeness.
|
||||
expect(container.querySelectorAll('li ul').length).toBe(0);
|
||||
expect(screen.getAllByText(/miles away/).length).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
describe('what the lede calls the set', () => {
|
||||
it.each([
|
||||
['Primary', 'primary schools'],
|
||||
['Middle deemed primary', 'primary schools'],
|
||||
['Secondary', 'secondary schools'],
|
||||
['Middle deemed secondary', 'secondary schools'],
|
||||
['All-through', 'all-through schools'],
|
||||
// GIAS phase 6. Its candidates span the whole secondary group, so no
|
||||
// single noun fits and it takes the honest general one.
|
||||
['16 plus', 'schools and colleges'],
|
||||
['', 'schools'],
|
||||
[null, 'schools'],
|
||||
])('calls a %s school\'s neighbours "%s"', (phase, expected) => {
|
||||
expect(nearbyNoun(phase)).toBe(expected);
|
||||
});
|
||||
|
||||
it('never calls a sixth form college\'s neighbours primary schools', () => {
|
||||
render(
|
||||
<NearbySchoolsSection
|
||||
urn={100001}
|
||||
schoolName="Barnet Sixth Form College"
|
||||
phase="16 plus"
|
||||
thisMetricValue={null}
|
||||
nearby={[school(), school({ urn: 100003 })]}
|
||||
/>,
|
||||
);
|
||||
expect(screen.getByText(/Other schools and colleges near Barnet Sixth Form College/)).toBeInTheDocument();
|
||||
expect(screen.queryByText(/primary schools/)).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('cards', () => {
|
||||
it('links each school to its canonical slug', () => {
|
||||
renderSection([school(), school({ urn: 100003, school_name: 'Oakfield Primary School' })]);
|
||||
const link = screen.getByRole('link', { name: /Willow Lane Primary School/ });
|
||||
expect(link).toHaveAttribute('href', '/school/100002-willow-lane-primary-school');
|
||||
});
|
||||
|
||||
it('shows the distance and the shared characteristics', () => {
|
||||
renderSection([school(), school({ urn: 100003 })]);
|
||||
expect(screen.getAllByText('0.6 miles away').length).toBeGreaterThan(0);
|
||||
expect(screen.getAllByText('Mixed').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('renders a missing figure as "Not published", never as a number', () => {
|
||||
renderSection([school({ metric_value: null }), school({ urn: 100003 })]);
|
||||
expect(screen.getByText('Not published')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('anchors each figure against this school', () => {
|
||||
renderSection([school(), school({ urn: 100003 })]);
|
||||
expect(screen.getAllByText('72% at this school').length).toBe(2);
|
||||
});
|
||||
|
||||
it('offers the compare bar once, for this school', () => {
|
||||
renderSection([school(), school({ urn: 100003 })]);
|
||||
expect(screen.getByTestId('compare-bar')).toHaveTextContent('bar for 100001');
|
||||
});
|
||||
|
||||
it('keeps every card in the DOM, including the ones scrolled out of view', () => {
|
||||
const six = Array.from({ length: 6 }, (_, n) =>
|
||||
school({ urn: 100002 + n, school_name: `Peer ${n} School` }),
|
||||
);
|
||||
renderSection(six);
|
||||
expect(screen.getAllByRole('link', { name: /Peer \d School/ })).toHaveLength(6);
|
||||
});
|
||||
|
||||
it('offers no arrows when three cards fit the row', () => {
|
||||
renderSection([school(), school({ urn: 100003 }), school({ urn: 100004 })]);
|
||||
expect(screen.queryByRole('button', { name: /More schools/ })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('offers arrows once there is a fourth school', () => {
|
||||
renderSection(Array.from({ length: 4 }, (_, n) => school({ urn: 100002 + n })));
|
||||
expect(screen.getByRole('button', { name: /More schools/ })).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: /Previous schools/ })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('says distances are straight-line, and offers no method panel', () => {
|
||||
const { container } = renderSection([school(), school({ urn: 100003 })]);
|
||||
expect(screen.getByText(/straight-line from this school/i)).toBeInTheDocument();
|
||||
expect(container.querySelector('details')).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,491 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { PlaceView } from '@/components/places/PlaceView';
|
||||
import type { PlaceDetail } from '@/lib/places';
|
||||
|
||||
const detail: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 29,
|
||||
parent_authority: 'Essex', phases: ['primary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', rwm_expected_pct: 82,
|
||||
ofsted_grade: 1, phase: 'Primary' } as never,
|
||||
{ urn: 2, school_name: 'Beta Primary', rwm_expected_pct: 44,
|
||||
ofsted_grade: 3, phase: 'Primary' } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
describe('PlaceView', () => {
|
||||
it('leads with an H1 that matches how the place is searched', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('heading', { level: 1 }))
|
||||
.toHaveTextContent(/primary schools in brentwood/i);
|
||||
});
|
||||
|
||||
it('states the count so the page says something before the table', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText(/29 schools/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('compares the local average against England, which a list cannot', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('63');
|
||||
expect(screen.getByTestId('local-vs-england')).toHaveTextContent('61');
|
||||
});
|
||||
|
||||
it('links every school in scope, which is what de-orphans them', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getAllByRole('link', { name: /Primary$/ })).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('links to the parent authority so the place sits in a hierarchy', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /Essex/i }))
|
||||
.toHaveAttribute('href', '/schools/authority/essex');
|
||||
});
|
||||
|
||||
it('shows the Ofsted distribution, not just a count of Outstanding', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByTestId('ofsted-distribution')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('links to neighbouring places so the page is not a dead end', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[{ kind: 'town', slug: 'romford', name: 'Romford', count: 40 }]} />);
|
||||
expect(screen.getByRole('link', { name: /Romford/ }))
|
||||
.toHaveAttribute('href', '/schools/romford');
|
||||
});
|
||||
|
||||
it('says nothing about an average it does not have', () => {
|
||||
render(<PlaceView detail={{ ...detail, averages:
|
||||
{ rwm_expected_pct: null, attainment_8_score: null } }}
|
||||
phase="primary" englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByTestId('local-vs-england')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView structured data', () => {
|
||||
function jsonLd() {
|
||||
const { container } = render(<PlaceView detail={detail} phase="primary"
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const el = container.querySelector('script[type="application/ld+json"]');
|
||||
return JSON.parse(el!.textContent!);
|
||||
}
|
||||
|
||||
it('declares the page as a ranked list, not prose', () => {
|
||||
const types = jsonLd()['@graph'].map((n: { '@type': string }) => n['@type']);
|
||||
expect(types).toContain('ItemList');
|
||||
expect(types).toContain('BreadcrumbList');
|
||||
});
|
||||
|
||||
it('gives every listed school an absolute URL on the canonical host', () => {
|
||||
const list = jsonLd()['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
|
||||
expect(list.itemListElement).toHaveLength(2);
|
||||
for (const item of list.itemListElement) {
|
||||
expect(item.url).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\/school\//);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView phase variants', () => {
|
||||
it('links the phase variants that exist', () => {
|
||||
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /Primary schools in Brentwood/i }))
|
||||
.toHaveAttribute('href', '/schools/brentwood/primary');
|
||||
});
|
||||
|
||||
it('links no variant for a phase below its own threshold', () => {
|
||||
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByRole('link', { name: /Secondary schools in Brentwood/i }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('does not link sideways from a variant page to itself', () => {
|
||||
render(<PlaceView detail={detail} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.queryByRole('link', { name: /Primary schools in Brentwood/i }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView presentation', () => {
|
||||
// /schools/brentwood shipped with 8 of 27 rows blank: an unphased page shows
|
||||
// one primary-only measure for a list that also holds secondaries.
|
||||
const mixed: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 4,
|
||||
parent_authority: 'Essex', phases: ['primary', 'secondary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null } as never,
|
||||
{ urn: 2, school_name: 'Beta High', phase: 'Secondary',
|
||||
rwm_expected_pct: null, attainment_8_score: 47 } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: 45 },
|
||||
};
|
||||
|
||||
it('gives each phase its own table rather than one column of blanks', () => {
|
||||
render(<PlaceView detail={mixed} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('heading', { name: /^Primary schools/ })).toBeInTheDocument();
|
||||
expect(screen.getByRole('heading', { name: /^Secondary schools/ })).toBeInTheDocument();
|
||||
expect(screen.getByText('82%')).toBeInTheDocument();
|
||||
expect(screen.getByText('47')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('names the measure in plain words, not jargon', () => {
|
||||
// The first cut said "RWM expected", which appears nowhere else on the site.
|
||||
render(<PlaceView detail={mixed} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByText('Reading, writing & maths')).toBeInTheDocument();
|
||||
expect(screen.getByText('Attainment 8')).toBeInTheDocument();
|
||||
expect(screen.queryByText(/RWM expected/i)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('says a missing result is unpublished rather than showing a bare dash', () => {
|
||||
const noResult: PlaceDetail = {
|
||||
...mixed,
|
||||
schools: [{ urn: 3, school_name: 'New Primary', phase: 'Primary',
|
||||
rwm_expected_pct: null, attainment_8_score: null } as never],
|
||||
};
|
||||
render(<PlaceView detail={noResult} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByText('Not published')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('styles school links to the site convention rather than browser default', () => {
|
||||
const { container } = render(<PlaceView detail={mixed} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
const link = container.querySelector('a[href^="/school/"]');
|
||||
expect(link?.className).toBeTruthy();
|
||||
});
|
||||
|
||||
it('a phased page shows one table and no phase headings', () => {
|
||||
render(<PlaceView detail={mixed} phase="primary" englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.queryByRole('heading', { name: /^Secondary schools/ }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView table alignment', () => {
|
||||
const aligned: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 2,
|
||||
parent_authority: 'Essex', phases: ['primary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
it('aligns the measure heading and its values with the same class', () => {
|
||||
// They were aligned by two different selectors whose specificity did not
|
||||
// match: `.table th:last-child` (0,2,1) won and went right, while `.num`
|
||||
// (0,1,0) lost to `.table td` (0,1,1) and stayed left. Sharing one class
|
||||
// is what makes them impossible to drift apart.
|
||||
const { container } = render(<PlaceView detail={aligned} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
const th = container.querySelectorAll('th')[1];
|
||||
const td = container.querySelectorAll('tbody td')[1];
|
||||
expect(th.className).toBeTruthy();
|
||||
expect(td.className).toBe(th.className);
|
||||
});
|
||||
|
||||
it('leaves the school-name column unclassed so it takes the spare width', () => {
|
||||
const { container } = render(<PlaceView detail={aligned} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(container.querySelectorAll('th')[0].className).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView authorities', () => {
|
||||
const straddling: PlaceDetail = {
|
||||
place: { kind: 'outcode', slug: 'sw19', name: 'SW19', count: 33,
|
||||
parent_authority: 'Merton', phases: ['primary'],
|
||||
authorities: [
|
||||
{ name: 'Merton', slug: 'merton', count: 26 },
|
||||
{ name: 'Wandsworth', slug: 'wandsworth', count: 7 },
|
||||
] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
it('names every authority the place straddles, not just the largest', () => {
|
||||
// SW19 is mostly Merton but partly Wandsworth. Naming one asserts
|
||||
// something false about a quarter of outcodes.
|
||||
render(<PlaceView detail={straddling} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: 'Merton' }))
|
||||
.toHaveAttribute('href', '/schools/authority/merton');
|
||||
expect(screen.getByRole('link', { name: 'Wandsworth' }))
|
||||
.toHaveAttribute('href', '/schools/authority/wandsworth');
|
||||
});
|
||||
|
||||
it('joins them readably rather than as a bare list', () => {
|
||||
// Asserted on the summary line's whole text: a loose /and/ matcher also
|
||||
// hits "Wandsworth".
|
||||
const { container } = render(<PlaceView detail={straddling}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const summary = container.querySelector('header p');
|
||||
expect(summary?.textContent).toContain('Merton and Wandsworth');
|
||||
});
|
||||
|
||||
it('falls back to the single parent when the field is absent', () => {
|
||||
// A cached API response predating the authorities field must not blank
|
||||
// the line entirely.
|
||||
const legacy = { ...straddling,
|
||||
place: { ...straddling.place, authorities: undefined } };
|
||||
render(<PlaceView detail={legacy} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: 'Merton' })).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView list ordering', () => {
|
||||
const detail3: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 2,
|
||||
parent_authority: 'Essex', phases: ['primary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 40, attainment_8_score: null } as never,
|
||||
{ urn: 2, school_name: 'Beta Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 90, attainment_8_score: null } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 65, attainment_8_score: null },
|
||||
};
|
||||
|
||||
it('renders schools in the order the API sent them, not by score', () => {
|
||||
// The API sorts alphabetically now; the component must not re-sort.
|
||||
render(<PlaceView detail={detail3} englandAverage={61} neighbours={[]} />);
|
||||
const links = screen.getAllByRole('link', { name: /Primary$/ });
|
||||
expect(links.map((l) => l.textContent))
|
||||
.toEqual(['Alpha Primary', 'Beta Primary']);
|
||||
});
|
||||
|
||||
it('declares the list as ascending rather than implying a ranking', () => {
|
||||
// An ItemList carrying `position` reads as a ranking unless it says
|
||||
// otherwise, and the table is A-Z.
|
||||
const { container } = render(<PlaceView detail={detail3} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
const ld = JSON.parse(
|
||||
container.querySelector('script[type="application/ld+json"]')!.textContent!);
|
||||
const list = ld['@graph'].find((n: { '@type': string }) => n['@type'] === 'ItemList');
|
||||
expect(list.itemListOrder).toBe('https://schema.org/ItemListOrderAscending');
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView phase links', () => {
|
||||
const authority: PlaceDetail = {
|
||||
place: { kind: 'authority', slug: 'barnet', name: 'Barnet', count: 156,
|
||||
parent_authority: null, phases: ['primary', 'secondary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
it('keeps an authority phase link in the authority namespace', () => {
|
||||
// The link was built as `/schools/${slug}/${phase}` for every kind, so an
|
||||
// authority page pointed into the town namespace. For 87 of 151
|
||||
// authorities that 404'd; for the other 64 it silently landed on the town
|
||||
// page of the same name — a different set of schools, and exactly the
|
||||
// duplicate the two namespaces exist to prevent. Barnet is one of the 64.
|
||||
render(<PlaceView detail={authority} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /^Primary schools in Barnet$/ }))
|
||||
.toHaveAttribute('href', '/schools/authority/barnet/primary');
|
||||
expect(screen.getByRole('link', { name: /^Secondary schools in Barnet$/ }))
|
||||
.toHaveAttribute('href', '/schools/authority/barnet/secondary');
|
||||
});
|
||||
|
||||
it('still uses the bare namespace for a town', () => {
|
||||
render(<PlaceView detail={detail} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: /^Primary schools in Brentwood$/ }))
|
||||
.toHaveAttribute('href', '/schools/brentwood/primary');
|
||||
});
|
||||
|
||||
it('offers no phase link when the place publishes none', () => {
|
||||
// Outcodes are the case: no phase route exists for them, so the registry
|
||||
// reports no phases and the nav does not render.
|
||||
const outcode = { ...detail,
|
||||
place: { ...detail.place, kind: 'outcode', slug: 'cm13', name: 'CM13',
|
||||
phases: [] } };
|
||||
render(<PlaceView detail={outcode} englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.queryByRole('navigation', { name: 'By phase' }))
|
||||
.not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView unlinkable authorities', () => {
|
||||
const withUnpublished: PlaceDetail = {
|
||||
place: { kind: 'outcode', slug: 'tr21', name: 'TR21', count: 8,
|
||||
parent_authority: 'Cornwall', phases: [],
|
||||
authorities: [
|
||||
{ name: 'Cornwall', slug: 'cornwall', count: 6 },
|
||||
{ name: 'Isles Of Scilly', slug: null, count: 2 },
|
||||
] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: null },
|
||||
};
|
||||
|
||||
it('names an authority with no page without linking it', () => {
|
||||
// City of London and the Isles of Scilly hold fewer schools than a page
|
||||
// needs. Saying where the place is stays right; linking there would 404.
|
||||
const { container } = render(<PlaceView detail={withUnpublished}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
expect(screen.getByRole('link', { name: 'Cornwall' })).toBeInTheDocument();
|
||||
expect(screen.queryByRole('link', { name: 'Isles Of Scilly' }))
|
||||
.not.toBeInTheDocument();
|
||||
expect(container.querySelector('header p')?.textContent)
|
||||
.toContain('Isles Of Scilly');
|
||||
});
|
||||
});
|
||||
|
||||
describe('PlaceView school attributes', () => {
|
||||
/*
|
||||
* The table shipped with one column of scores, which answers "how did they
|
||||
* do" and nothing about whether the school is one a family could use. Age
|
||||
* range, faith, nursery and constituency are the four facts a parent
|
||||
* filters on before they look at a number at all.
|
||||
*/
|
||||
const withAttributes: PlaceDetail = {
|
||||
place: { kind: 'town', slug: 'chelmsford', name: 'Chelmsford', count: 3,
|
||||
parent_authority: 'Essex', phases: ['primary', 'secondary'] },
|
||||
schools: [
|
||||
{ urn: 1, school_name: 'Alpha Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 82, attainment_8_score: null,
|
||||
age_range: '4-11', religious_denomination: 'Church of England',
|
||||
nursery_provision: 'Has Nursery Classes',
|
||||
parliamentary_constituency: 'Chelmsford' } as never,
|
||||
{ urn: 2, school_name: 'Beta High', phase: 'Secondary',
|
||||
rwm_expected_pct: null, attainment_8_score: 47,
|
||||
age_range: '11-16', religious_denomination: 'Does not apply',
|
||||
nursery_provision: 'No Nursery Classes',
|
||||
parliamentary_constituency: 'Witham' } as never,
|
||||
],
|
||||
averages: { rwm_expected_pct: 63, attainment_8_score: 45 },
|
||||
};
|
||||
|
||||
function headings(container: HTMLElement, table = 0): string[] {
|
||||
return Array.from(container.querySelectorAll('table')[table]
|
||||
.querySelectorAll('thead th')).map((th) => th.textContent ?? '');
|
||||
}
|
||||
|
||||
it('heads a primary table with all four attributes', () => {
|
||||
const { container } = render(<PlaceView detail={withAttributes}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
expect(headings(container)).toEqual([
|
||||
'School', 'Reading, writing & maths',
|
||||
'Ages', 'Religious character', 'Nursery', 'Constituency',
|
||||
]);
|
||||
});
|
||||
|
||||
it('omits nursery from a secondary table, where it does not apply', () => {
|
||||
const { container } = render(<PlaceView detail={withAttributes}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
expect(headings(container, 1)).toEqual([
|
||||
'School', 'Attainment 8', 'Ages', 'Religious character', 'Constituency',
|
||||
]);
|
||||
});
|
||||
|
||||
it('keeps the measure beside the school name, where a phone can see it', () => {
|
||||
// Six columns overflow a phone and .tableWrap turns that into a swipe.
|
||||
// With the measure last, the one number the page exists for is the one
|
||||
// scrolled off the screen.
|
||||
const { container } = render(<PlaceView detail={withAttributes}
|
||||
phase="primary" englandAverage={61} neighbours={[]} />);
|
||||
expect(headings(container)[1]).toBe('Reading, writing & maths');
|
||||
});
|
||||
|
||||
it('shows the age range without repeating the column heading', () => {
|
||||
render(<PlaceView detail={withAttributes} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText('4–11')).toBeInTheDocument();
|
||||
expect(screen.queryByText('Ages 4–11')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('names the faith of a faith school', () => {
|
||||
render(<PlaceView detail={withAttributes} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText('Church of England')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('reads "Does not apply" as no religious character, not as a value', () => {
|
||||
// GIAS spells the absence of a faith as "Does not apply", which is a
|
||||
// database answer rather than an English one. The school page already
|
||||
// suppresses it; the two must not disagree about the same school.
|
||||
const { container } = render(<PlaceView detail={withAttributes}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const secondary = container.querySelectorAll('table')[1]
|
||||
.querySelectorAll('tbody td');
|
||||
expect(secondary[3].textContent).toBe('—');
|
||||
expect(screen.queryByText(/Does not apply/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('marks a nursery as such and a school without one as not', () => {
|
||||
const { container } = render(<PlaceView detail={withAttributes}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const cells = container.querySelectorAll('table')[0]
|
||||
.querySelectorAll('tbody td');
|
||||
expect(cells[4].textContent).toBe('Yes');
|
||||
});
|
||||
|
||||
it('names the constituency of each school', () => {
|
||||
render(<PlaceView detail={withAttributes} englandAverage={61}
|
||||
neighbours={[]} />);
|
||||
expect(screen.getByText('Chelmsford', { selector: 'td' })).toBeInTheDocument();
|
||||
expect(screen.getByText('Witham', { selector: 'td' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('dashes an attribute the data does not carry', () => {
|
||||
// nursery_provision and parliamentary_constituency are absent from marts
|
||||
// the pipeline has not rebuilt, and the API degrades them to null rather
|
||||
// than failing. A row must survive that.
|
||||
const bare: PlaceDetail = {
|
||||
...withAttributes,
|
||||
schools: [{ urn: 3, school_name: 'Gamma Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 70 } as never],
|
||||
};
|
||||
const { container } = render(<PlaceView detail={bare} phase="primary"
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const cells = Array.from(container.querySelectorAll('tbody td'))
|
||||
.map((td) => td.textContent);
|
||||
expect(cells.slice(2)).toEqual(['—', '—', '—', '—']);
|
||||
});
|
||||
|
||||
it('reads "Not applicable" as no nursery, not as a yes', () => {
|
||||
// GIAS sends text. Tested for truthiness, every value was a "Yes".
|
||||
const notApplicable: PlaceDetail = {
|
||||
...withAttributes,
|
||||
schools: [{ urn: 5, school_name: 'Epsilon Primary', phase: 'Primary',
|
||||
rwm_expected_pct: 70, nursery_provision: 'Not applicable' } as never],
|
||||
};
|
||||
const { container } = render(<PlaceView detail={notApplicable} phase="primary"
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
expect(container.querySelector('tbody')!.textContent).not.toContain('Yes');
|
||||
});
|
||||
|
||||
it('gives an all-through school its nursery under primary only', () => {
|
||||
// All-through schools render in both groups. Nursery belongs to the
|
||||
// primary reading of the same school, not the secondary one.
|
||||
const allThrough: PlaceDetail = {
|
||||
...withAttributes,
|
||||
schools: [{ urn: 4, school_name: 'Delta Academy', phase: 'All-through',
|
||||
rwm_expected_pct: 66, attainment_8_score: 51,
|
||||
age_range: '4-18', religious_denomination: 'None',
|
||||
nursery_provision: 'Has Nursery Classes',
|
||||
parliamentary_constituency: 'Chelmsford' } as never],
|
||||
};
|
||||
const { container } = render(<PlaceView detail={allThrough}
|
||||
englandAverage={61} neighbours={[]} />);
|
||||
const tables = container.querySelectorAll('table');
|
||||
expect(tables[0].textContent).toContain('Yes');
|
||||
expect(tables[1].textContent).not.toContain('Yes');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,44 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { Post16DestinationsSection } from '@/components/school/Post16DestinationsSection';
|
||||
import type { DestinationPhase } from '@/lib/types';
|
||||
|
||||
const phase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: {
|
||||
all: {
|
||||
cohort: 96,
|
||||
categories: [
|
||||
{ category: 'higher_education', pupils: 56, percentage: 58.3, status: 'published' },
|
||||
{ category: 'further_education', pupils: 12, percentage: 12.5, status: 'published' },
|
||||
{ category: 'apprenticeship', pupils: 9, percentage: 9.4, status: 'published' },
|
||||
{ category: 'employment', pupils: 13, percentage: 13.5, status: 'published' },
|
||||
{ category: 'not_sustained', pupils: 6, percentage: 6.3, status: 'published' },
|
||||
],
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
describe('Post16DestinationsSection', () => {
|
||||
it('names the Year 13 cohort, not Year 11', () => {
|
||||
const { container } = render(<Post16DestinationsSection destinations={phase} />);
|
||||
expect(container.textContent).toMatch(/Year 13/);
|
||||
expect(container.textContent).not.toMatch(/Year 11/);
|
||||
});
|
||||
|
||||
it('reports higher education destinations', () => {
|
||||
render(<Post16DestinationsSection destinations={phase} />);
|
||||
expect(screen.getByText(/UK higher education/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('uses its own anchor so the nav does not collide with After Year 11', () => {
|
||||
const { container } = render(<Post16DestinationsSection destinations={phase} />);
|
||||
expect(container.querySelector('#post16-destinations')).toBeTruthy();
|
||||
expect(container.querySelector('#destinations')).toBeNull();
|
||||
});
|
||||
|
||||
it('renders nothing when no group carries categories', () => {
|
||||
const empty: DestinationPhase = { cohort_year: '2022/23', groups: {} };
|
||||
const { container } = render(<Post16DestinationsSection destinations={empty} />);
|
||||
expect(container.firstChild).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,195 @@
|
||||
import { act, fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { HomeView } from '@/components/HomeView';
|
||||
import { fetchSchools, fetchNationalAverages } from '@/lib/api';
|
||||
import { primaryFixture } from '../support/schoolFixtures';
|
||||
import type { School, SchoolsResponse } from '@/lib/types';
|
||||
|
||||
/*
|
||||
* The map view: the list beside the map (mockup B), where every postcode
|
||||
* search opens. The map itself is Leaflet and mocked here; what is pinned is what
|
||||
* HomeView hands it and the list it draws beside it.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push: jest.fn(), replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/context/ComparisonContext', () => ({
|
||||
useComparisonContext: () => ({ addSchool: jest.fn(), removeSchool: jest.fn(), selectedSchools: [] }),
|
||||
}));
|
||||
jest.mock('@/lib/api', () => ({
|
||||
fetchSchools: jest.fn(),
|
||||
fetchNationalAverages: jest.fn(),
|
||||
fetchLAaverages: jest.fn(async () => ({ secondary: { attainment_8_by_la: {} } })),
|
||||
}));
|
||||
// Renders only the List/Map switch HomeView hands it, which lives in its row.
|
||||
jest.mock('@/components/FilterBar', () => ({
|
||||
FilterBar: ({ viewSwitch }: { viewSwitch?: unknown }) => viewSwitch || null,
|
||||
}));
|
||||
jest.mock('@/components/SchoolMap', () => ({
|
||||
SchoolMap: ({ selectedUrn, radiusMiles, onMarkerClick, schools }: {
|
||||
selectedUrn: number | null; radiusMiles?: number;
|
||||
onMarkerClick: (s: School) => void; schools: School[];
|
||||
}) => (
|
||||
<div data-testid="map" data-selected={selectedUrn ?? ''} data-radius={radiusMiles}>
|
||||
<button type="button" onClick={() => onMarkerClick(schools[1])}>pin</button>
|
||||
</div>
|
||||
),
|
||||
}));
|
||||
|
||||
const base = primaryFixture.schoolInfo;
|
||||
const southmead: School = {
|
||||
...base, urn: 2, school_name: 'Southmead Primary School', distance: 0.2,
|
||||
school_type: 'Community school', rwm_expected_pct: 52, total_pupils: 269,
|
||||
};
|
||||
const greenmead: School = {
|
||||
...base, urn: 3, school_name: 'Greenmead School', distance: 0.2,
|
||||
school_type: 'Community special school', rwm_expected_pct: 0,
|
||||
reading_expected_pct: 0, writing_expected_pct: 0, maths_expected_pct: 0, total_pupils: 62,
|
||||
};
|
||||
const ourLady: School = {
|
||||
...base, urn: 1, school_name: 'Our Lady Queen of Heaven RC School', distance: 0,
|
||||
school_type: 'Voluntary aided school', rwm_expected_pct: 70, total_pupils: 224,
|
||||
};
|
||||
|
||||
function results(): SchoolsResponse {
|
||||
return {
|
||||
schools: [ourLady, southmead, greenmead], total: 3, page: 1, page_size: 25, total_pages: 1,
|
||||
location_info: { postcode: 'SW196AR', radius: 1.60934, coordinates: [51.42, -0.21] },
|
||||
} as SchoolsResponse;
|
||||
}
|
||||
const filters = { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
jest.mocked(fetchSchools).mockReset().mockResolvedValue(results());
|
||||
jest.mocked(fetchNationalAverages).mockResolvedValue({ primary: { rwm_expected_pct: 62 } } as never);
|
||||
setWide(true);
|
||||
});
|
||||
|
||||
/** Desktop unless a test says otherwise: the list pane is shown from 769px. */
|
||||
function setWide(wide: boolean) {
|
||||
window.matchMedia = ((q: string) => ({
|
||||
matches: wide, media: q, addEventListener() {}, removeEventListener() {},
|
||||
})) as unknown as typeof window.matchMedia;
|
||||
}
|
||||
|
||||
async function renderMap() {
|
||||
const view = render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
return view;
|
||||
}
|
||||
|
||||
it('opens a postcode search on the map', async () => {
|
||||
await renderMap();
|
||||
expect(screen.getByTestId('map')).toHaveAttribute('data-radius', '1');
|
||||
expect(screen.getByRole('button', { name: 'Map' })).toHaveAttribute('aria-pressed', 'true');
|
||||
});
|
||||
|
||||
it('lists a name search, which has no map', async () => {
|
||||
params = new URLSearchParams('search=southmead');
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
expect(screen.queryByTestId('map')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('puts the count and the sort in the list beside the map, once', async () => {
|
||||
await renderMap();
|
||||
// Short beside the map, so it shares one line with the sort.
|
||||
expect(screen.getAllByRole('heading', { name: '3 schools within 1 mile' })).toHaveLength(1);
|
||||
expect(screen.getAllByRole('combobox')).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('selects the pin from the card, and the card from the pin', async () => {
|
||||
const { container } = await renderMap();
|
||||
const card = (urn: number) => container.querySelector(`[data-urn="${urn}"]`) as HTMLElement;
|
||||
|
||||
fireEvent.click(within(card(1)).getByText(/pupils/));
|
||||
expect(screen.getByTestId('map')).toHaveAttribute('data-selected', '1');
|
||||
expect(card(1).className).toMatch(/mapRowSelected/);
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'pin' }));
|
||||
expect(card(2).className).toMatch(/mapRowSelected/);
|
||||
expect(card(1).className).not.toMatch(/mapRowSelected/);
|
||||
});
|
||||
|
||||
it('clicking a card\'s link or button does not also select it', async () => {
|
||||
const { container } = await renderMap();
|
||||
const card = container.querySelector('[data-urn="1"]') as HTMLElement;
|
||||
fireEvent.click(within(card).getByRole('button', { name: '+ Compare' }));
|
||||
expect(screen.getByTestId('map')).toHaveAttribute('data-selected', '');
|
||||
});
|
||||
|
||||
it('shows the England comparison for mainstream schools only, and never a placeholder 0%', async () => {
|
||||
const { container } = await renderMap();
|
||||
const card = (urn: number) => container.querySelector(`[data-urn="${urn}"]`) as HTMLElement;
|
||||
expect(card(2)).toHaveTextContent('52%Reading, Writing & Maths-10 pts vs national');
|
||||
expect(card(2)).toHaveTextContent('269pupils');
|
||||
expect(card(3)).toHaveTextContent('62pupils');
|
||||
expect(card(3)).not.toHaveTextContent(/%|pts/);
|
||||
});
|
||||
|
||||
it('opens the map after a hero search, and keeps the reader\'s choice after that', async () => {
|
||||
// Landing page, then a hero search: the same instance gets new props.
|
||||
params = new URLSearchParams('');
|
||||
const empty = { schools: [], total: 0, page: 1, page_size: 25, total_pages: 0 } as SchoolsResponse;
|
||||
const view = render(<HomeView initialSchools={empty} filters={filters} />);
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
view.rerender(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
expect(screen.getByTestId('map')).toBeInTheDocument();
|
||||
|
||||
// Chosen: a later search keeps the list.
|
||||
fireEvent.click(screen.getByRole('button', { name: 'List' }));
|
||||
params = new URLSearchParams('postcode=SW170AA&radius=1');
|
||||
view.rerender(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
expect(screen.queryByTestId('map')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('builds no list cards on a phone, where the pane is hidden', async () => {
|
||||
setWide(false);
|
||||
const { container } = await renderMap();
|
||||
expect(screen.getByTestId('map')).toBeInTheDocument();
|
||||
expect(container.querySelectorAll('[data-urn]')).toHaveLength(0);
|
||||
// The count stays: it is the pane's heading, shown above the map.
|
||||
expect(screen.getByRole('heading', { name: /3 schools within/ })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('draws the list view\'s own row beside the map, with the same content', async () => {
|
||||
const { container } = await renderMap();
|
||||
const beside = container.querySelector('[data-urn="2"] > [class~="row"]')!.textContent;
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'List' }));
|
||||
const row = screen.getByRole('link', { name: 'Southmead Primary School' }).closest('[class~="row"]')!;
|
||||
expect(row.parentElement?.className).toMatch(/schoolList/);
|
||||
expect(row.textContent).toBe(beside);
|
||||
});
|
||||
|
||||
it('lets a keyboard pick a pin from the list, with a real button', async () => {
|
||||
await renderMap();
|
||||
const show = screen.getByRole('button', { name: 'Show Southmead Primary School on the map' });
|
||||
expect(show).toHaveAttribute('aria-pressed', 'false');
|
||||
fireEvent.click(show);
|
||||
expect(screen.getByTestId('map')).toHaveAttribute('data-selected', '2');
|
||||
expect(show).toHaveAttribute('aria-pressed', 'true');
|
||||
});
|
||||
|
||||
it('keeps the postcode in the heading in list view, where there is room', async () => {
|
||||
await renderMap();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'List' }));
|
||||
expect(screen.getByRole('heading', { name: '3 schools within 1 mile of SW196AR' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('draws and names a quarter-mile search as 0.25, not rounded to 0.3', async () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=0.25');
|
||||
const quarter = { ...results(), location_info: { postcode: 'SW196AR', radius: 0.25 * 1.60934, coordinates: [51.42, -0.21] } } as SchoolsResponse;
|
||||
render(<HomeView initialSchools={quarter} filters={filters} />);
|
||||
await act(async () => {});
|
||||
expect(screen.getByTestId('map')).toHaveAttribute('data-radius', '0.25');
|
||||
expect(screen.getByRole('heading', { name: '3 schools within 0.25 miles' })).toBeInTheDocument();
|
||||
fireEvent.click(screen.getByRole('button', { name: 'List' }));
|
||||
expect(screen.getByRole('heading', { name: '3 schools within 0.25 miles of SW196AR' })).toBeInTheDocument();
|
||||
});
|
||||
@@ -0,0 +1,198 @@
|
||||
import { act, fireEvent, render, screen, within } from '@testing-library/react';
|
||||
import { HomeView } from '@/components/HomeView';
|
||||
import { FilterBar } from '@/components/FilterBar';
|
||||
import { fetchSchools } from '@/lib/api';
|
||||
import { track } from '@/lib/analytics';
|
||||
import { primaryFixture } from '../support/schoolFixtures';
|
||||
import type { SchoolsResponse } from '@/lib/types';
|
||||
|
||||
/*
|
||||
* The results toolbar (option B of the 2026-09-30 results-controls mockups):
|
||||
* search, filters and the List/Map switch pinned under the header, with a
|
||||
* floating List/Map button standing in for the switch on phones. Layout is CSS
|
||||
* and not visible to jsdom; these pin the behaviour and the accessible names
|
||||
* the E2E journeys rely on.
|
||||
*/
|
||||
|
||||
let params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
const push = jest.fn();
|
||||
jest.mock('next/navigation', () => ({
|
||||
useSearchParams: () => params,
|
||||
usePathname: () => '/',
|
||||
useRouter: () => ({ push, replace: jest.fn(), prefetch: jest.fn() }),
|
||||
}));
|
||||
jest.mock('@/context/ComparisonContext', () => ({
|
||||
useComparisonContext: () => ({ addSchool: jest.fn(), removeSchool: jest.fn(), selectedSchools: [] }),
|
||||
}));
|
||||
jest.mock('@/lib/api', () => ({
|
||||
fetchSchools: jest.fn(),
|
||||
fetchNationalAverages: jest.fn(async () => ({})),
|
||||
fetchLAaverages: jest.fn(async () => ({ secondary: { attainment_8_by_la: {} } })),
|
||||
}));
|
||||
jest.mock('@/lib/analytics', () => ({ track: jest.fn() }));
|
||||
jest.mock('@/components/SchoolMap', () => ({ SchoolMap: () => <div data-testid="map" /> }));
|
||||
|
||||
const filters = {
|
||||
local_authorities: ['Wandsworth'], school_types: ['Community school'], years: [],
|
||||
phases: ['Primary', 'Secondary'], genders: [], admissions_policies: [],
|
||||
school_type_groups: [{ value: 'council', label: 'State school: council-run' }],
|
||||
};
|
||||
|
||||
function results(): SchoolsResponse {
|
||||
return { schools: [{ ...primaryFixture.schoolInfo, school_name: 'Southmead Primary School' }],
|
||||
total: 1, page: 1, page_size: 25, total_pages: 1 };
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1');
|
||||
push.mockClear();
|
||||
jest.mocked(track).mockClear();
|
||||
jest.mocked(fetchSchools).mockReset().mockResolvedValue(results());
|
||||
});
|
||||
|
||||
describe('the List/Map switch', () => {
|
||||
it('lives in the toolbar with the filters and says which view is on', () => {
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
const view = screen.getByRole('group', { name: 'Results view' });
|
||||
expect(view.closest('div[class*="resultsToolbar"]')).not.toBeNull();
|
||||
// A postcode search opens on the map.
|
||||
expect(screen.getByRole('button', { name: 'Map' })).toHaveAttribute('aria-pressed', 'true');
|
||||
expect(screen.getByRole('button', { name: 'List' })).toHaveAttribute('aria-pressed', 'false');
|
||||
});
|
||||
|
||||
it('has a floating twin that flips between map and list', async () => {
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Show list' }));
|
||||
expect(screen.queryByTestId('map')).not.toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'List' })).toHaveAttribute('aria-pressed', 'true');
|
||||
expect(track).toHaveBeenCalledWith('results_view_changed', { view: 'list', via: 'floating' });
|
||||
|
||||
await act(async () => fireEvent.click(screen.getByRole('button', { name: 'Show map' })));
|
||||
expect(screen.getByTestId('map')).toBeInTheDocument();
|
||||
expect(screen.getByRole('button', { name: 'Show list' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('does not track a click on the view already showing', async () => {
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
await act(async () => {});
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Map' }));
|
||||
expect(track).not.toHaveBeenCalledWith('results_view_changed', expect.anything());
|
||||
});
|
||||
|
||||
it('is absent from a name search, which has no map', () => {
|
||||
params = new URLSearchParams('search=southmead');
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
expect(screen.queryByRole('group', { name: 'Results view' })).not.toBeInTheDocument();
|
||||
expect(screen.queryByRole('button', { name: 'Show map' })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the toolbar filters', () => {
|
||||
it('keeps distance, phase and school type in the row, not behind More filters', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const row = screen.getByRole('group', { name: 'Filters' });
|
||||
for (const name of ['Distance', 'Phase', 'School type']) {
|
||||
expect(row).toContainElement(screen.getByRole('combobox', { name }));
|
||||
}
|
||||
expect(screen.getByRole('combobox', { name: 'Distance' })).toHaveDisplayValue('Within 1 mile');
|
||||
expect(screen.queryByRole('combobox', { name: 'Local authority' })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('counts only what More filters hides', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&school_type=council&local_authority=Wandsworth');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.getByRole('button', { name: /More filters \(1\)/ })).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the phone chips line', () => {
|
||||
it('drops its "more this way" fade when nothing is left to scroll', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=1&phase=primary');
|
||||
render(<FilterBar filters={filters} />);
|
||||
// jsdom lays nothing out, so the line reads as not overflowing at all.
|
||||
const line = screen.getByRole('group', { name: 'Applied filters' }).parentElement!;
|
||||
expect(line.className).toMatch(/controlsAtEnd/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('the phone filter sheet', () => {
|
||||
it('offers the results total', () => {
|
||||
render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Filters' }));
|
||||
expect(screen.getByRole('button', { name: 'Show 1 school' })).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the folded search', () => {
|
||||
it('summarises the search and unfolds on tap', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
const summary = screen.getByRole('button', { name: 'Edit search: SW196AR, within 1 mile' });
|
||||
fireEvent.click(summary);
|
||||
expect(screen.queryByRole('button', { name: /Edit search/ })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('folds again once the edited search is submitted', () => {
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /Edit search/ }));
|
||||
const input = screen.getByRole('searchbox', { name: 'School name or postcode' });
|
||||
fireEvent.change(input, { target: { value: 'SW19 1AA' } });
|
||||
fireEvent.submit(input.closest('form')!);
|
||||
expect(screen.getByRole('button', { name: /Edit search/ })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('refolds and shows the new text when the search changes some other way', () => {
|
||||
const view = render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /Edit search/ }));
|
||||
fireEvent.change(screen.getByRole('searchbox', { name: 'School name or postcode' }),
|
||||
{ target: { value: 'half-typed' } });
|
||||
|
||||
// Back button: the URL changes under the component, nothing is submitted.
|
||||
params = new URLSearchParams('postcode=SW170AA&radius=3');
|
||||
view.rerender(<FilterBar filters={filters} />);
|
||||
|
||||
expect(screen.getByRole('button', { name: 'Edit search: SW170AA, within 3 miles' }))
|
||||
.toBeInTheDocument();
|
||||
expect(screen.getByRole('searchbox', { name: 'School name or postcode' }))
|
||||
.toHaveValue('SW170AA');
|
||||
});
|
||||
|
||||
it('never appears in the hero, or before anything has been searched', () => {
|
||||
const { unmount } = render(<FilterBar filters={filters} isHero />);
|
||||
expect(screen.queryByRole('button', { name: /Edit search/ })).not.toBeInTheDocument();
|
||||
unmount();
|
||||
params = new URLSearchParams('local_authority=Wandsworth');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(screen.queryByRole('button', { name: /Edit search/ })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the results list', () => {
|
||||
it('repeats no applied filters above the results; the filter bar shows them', () => {
|
||||
params = new URLSearchParams('search=southmead&school_type=council&local_authority=Wandsworth');
|
||||
const { container } = render(<HomeView initialSchools={results()} filters={filters} />);
|
||||
expect(container.querySelector('[class*="activeFilters"]')).toBeNull();
|
||||
expect(screen.queryByText('Search: southmead')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the desktop Clear all', () => {
|
||||
const row = () => screen.getByRole('group', { name: 'Filters' });
|
||||
|
||||
it('removes the filters and keeps the search, rather than going home', () => {
|
||||
params = new URLSearchParams('postcode=SW196AR&radius=3&phase=primary&school_type=council&local_authority=Wandsworth');
|
||||
render(<FilterBar filters={filters} />);
|
||||
fireEvent.click(within(row()).getByRole('button', { name: 'Clear all' }));
|
||||
const pushed = push.mock.calls.at(-1)![0] as string;
|
||||
const next = new URLSearchParams(pushed.split('?')[1]);
|
||||
expect(next.get('postcode')).toBe('SW196AR');
|
||||
expect(next.get('radius')).toBe('3');
|
||||
for (const key of ['phase', 'school_type', 'local_authority']) expect(next.get(key)).toBeNull();
|
||||
});
|
||||
|
||||
it('is not offered when only a search is applied', () => {
|
||||
params = new URLSearchParams('search=southmead');
|
||||
render(<FilterBar filters={filters} />);
|
||||
expect(within(row()).queryByRole('button', { name: /^Clear/ })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,38 @@
|
||||
/**
|
||||
* The trail has to be written by something, and it has to be written on every
|
||||
* route — not only the ones that happen to track an event.
|
||||
*/
|
||||
import { render } from '@testing-library/react';
|
||||
|
||||
const recordVisitedPath = jest.fn();
|
||||
let pathname = '/schools/brentwood';
|
||||
|
||||
jest.mock('next/navigation', () => ({ usePathname: () => pathname }));
|
||||
jest.mock('@/lib/analytics', () => ({
|
||||
recordVisitedPath: (p: string) => recordVisitedPath(p),
|
||||
}));
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-var-requires
|
||||
const { RouteTrail } = require('@/components/RouteTrail');
|
||||
|
||||
describe('RouteTrail', () => {
|
||||
beforeEach(() => recordVisitedPath.mockClear());
|
||||
|
||||
it('records the page it is mounted on', () => {
|
||||
render(<RouteTrail />);
|
||||
expect(recordVisitedPath).toHaveBeenCalledWith('/schools/brentwood');
|
||||
});
|
||||
|
||||
it('records each new route as the user moves through the app', () => {
|
||||
const { rerender } = render(<RouteTrail />);
|
||||
pathname = '/school/115429-brentwood-school';
|
||||
rerender(<RouteTrail />);
|
||||
expect(recordVisitedPath).toHaveBeenLastCalledWith(
|
||||
'/school/115429-brentwood-school');
|
||||
});
|
||||
|
||||
it('renders nothing, so it can sit anywhere in the layout', () => {
|
||||
const { container } = render(<RouteTrail />);
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,54 @@
|
||||
/**
|
||||
* SchoolRow (primary search results): line 2 prints the religious character
|
||||
* only when the school has one. The register's "None" was printed as a chip.
|
||||
*/
|
||||
|
||||
import '@testing-library/jest-dom';
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { SchoolRow } from '@/components/SchoolRow';
|
||||
import type { School } from '@/lib/types';
|
||||
|
||||
const base = {
|
||||
urn: 100001,
|
||||
school_name: 'Alpha Primary School',
|
||||
local_authority: 'Testshire',
|
||||
school_type: 'Free schools',
|
||||
phase: 'Primary',
|
||||
gender: 'Mixed',
|
||||
age_range: '4-11',
|
||||
rwm_expected_pct: 70,
|
||||
} as unknown as School;
|
||||
|
||||
describe('SchoolRow religious character', () => {
|
||||
it.each(['None', 'Does not apply', ''])(
|
||||
'prints nothing when the register says %p',
|
||||
(religious_denomination) => {
|
||||
render(<SchoolRow school={{ ...base, religious_denomination }} />);
|
||||
expect(screen.queryByText('None')).not.toBeInTheDocument();
|
||||
expect(screen.queryByText('Does not apply')).not.toBeInTheDocument();
|
||||
},
|
||||
);
|
||||
|
||||
it('prints a religious character the school has', () => {
|
||||
render(<SchoolRow school={{ ...base, religious_denomination: 'Church of England' }} />);
|
||||
expect(screen.getByText('Church of England')).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('SchoolRow shares the school page flags', () => {
|
||||
it("prints the type in the search filter's terms", () => {
|
||||
render(<SchoolRow school={{ ...base, type_group: 'state' }} />);
|
||||
expect(screen.getByText('State school')).toBeInTheDocument();
|
||||
expect(screen.queryByText('Free schools')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('flags a nursery class', () => {
|
||||
render(<SchoolRow school={{ ...base, nursery_provision: 'Has Nursery Classes' }} />);
|
||||
expect(screen.getByText('Nursery class')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("flags a boys' school", () => {
|
||||
render(<SchoolRow school={{ ...base, gender: 'Boys' }} />);
|
||||
expect(screen.getByText("Boys' school")).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -65,3 +65,54 @@ describe('SecondarySchoolRow proposed-to-close tag', () => {
|
||||
expect(screen.queryByText(/Proposed to close/)).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('SecondarySchoolRow admissions tag', () => {
|
||||
it('tags a selective school', () => {
|
||||
render(<SecondarySchoolRow school={{ ...base, admissions_policy: 'Selective' }} />);
|
||||
expect(screen.getByText('Selective')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('does not tag a non-selective school as selective', () => {
|
||||
// "Non-selective" contains "selective": a substring test tagged every
|
||||
// comprehensive (Burntwood, Graveney) as Selective.
|
||||
render(<SecondarySchoolRow school={{ ...base, admissions_policy: 'Non-selective' }} />);
|
||||
expect(screen.queryByText('Selective')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it.each(['None', 'Does not apply'])(
|
||||
'gives no faith tag when the religious character is %p',
|
||||
(religious_denomination) => {
|
||||
render(
|
||||
<SecondarySchoolRow
|
||||
school={{ ...base, admissions_policy: 'Not applicable', religious_denomination }}
|
||||
/>,
|
||||
);
|
||||
expect(screen.queryByText(religious_denomination)).not.toBeInTheDocument();
|
||||
expect(screen.queryByText('Faith priority')).not.toBeInTheDocument();
|
||||
},
|
||||
);
|
||||
|
||||
it('tags the religious character the register records, not "Faith priority"', () => {
|
||||
render(
|
||||
<SecondarySchoolRow
|
||||
school={{ ...base, admissions_policy: 'Not applicable', religious_denomination: 'Church of England' }}
|
||||
/>,
|
||||
);
|
||||
expect(screen.getByText('Church of England')).toBeInTheDocument();
|
||||
expect(screen.queryByText('Faith priority')).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe('SecondarySchoolRow shares the school page flags', () => {
|
||||
it("prints the type in the search filter's terms", () => {
|
||||
render(<SecondarySchoolRow school={{ ...base, type_group: 'state' }} />);
|
||||
expect(screen.getByText('State school')).toBeInTheDocument();
|
||||
expect(screen.queryByText('Academy')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("flags a girls' school and fees", () => {
|
||||
render(<SecondarySchoolRow school={{ ...base, type_group: 'independent', gender: 'Girls' }} />);
|
||||
expect(screen.getByText("Girls' school")).toBeInTheDocument();
|
||||
expect(screen.getByText('Fee-paying')).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,50 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { SuggestList, suggestOptionId } from '@/components/SuggestList';
|
||||
|
||||
const ROWS = [
|
||||
{ urn: 1, school_name: "St Mary's Primary", local_authority: 'Camden',
|
||||
postcode: 'NW1 1AA', phase: 'Primary', school_type: 'Voluntary aided school' },
|
||||
{ urn: 2, school_name: "St Mary's Primary", local_authority: 'Barnet',
|
||||
postcode: 'EN5 2AA', phase: 'Primary', school_type: 'Community school' },
|
||||
];
|
||||
|
||||
describe('SuggestList', () => {
|
||||
it('is a listbox of options', () => {
|
||||
render(<SuggestList id="s" suggestions={ROWS} activeIndex={-1}
|
||||
onPick={() => {}} onHover={() => {}} />);
|
||||
expect(screen.getByRole('listbox')).toBeInTheDocument();
|
||||
expect(screen.getAllByRole('option')).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('shows the local authority, which is what tells two schools apart', () => {
|
||||
// Both rows are "St Mary's Primary". Without the authority the list is
|
||||
// unusable for exactly the query autosuggest exists to serve.
|
||||
render(<SuggestList id="s" suggestions={ROWS} activeIndex={-1}
|
||||
onPick={() => {}} onHover={() => {}} />);
|
||||
expect(screen.getByText('Camden')).toBeInTheDocument();
|
||||
expect(screen.getByText('Barnet')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('marks only the active option selected', () => {
|
||||
render(<SuggestList id="s" suggestions={ROWS} activeIndex={1}
|
||||
onPick={() => {}} onHover={() => {}} />);
|
||||
const options = screen.getAllByRole('option');
|
||||
expect(options[0]).toHaveAttribute('aria-selected', 'false');
|
||||
expect(options[1]).toHaveAttribute('aria-selected', 'true');
|
||||
});
|
||||
|
||||
it('gives each option the id the input will point at', () => {
|
||||
// aria-activedescendant on the input has to name a real element id, or
|
||||
// a screen reader announces nothing as the user arrows through.
|
||||
render(<SuggestList id="s" suggestions={ROWS} activeIndex={0}
|
||||
onPick={() => {}} onHover={() => {}} />);
|
||||
expect(screen.getAllByRole('option')[0]).toHaveAttribute(
|
||||
'id', suggestOptionId('s', 0));
|
||||
});
|
||||
|
||||
it('renders nothing when there is nothing to suggest', () => {
|
||||
const { container } = render(<SuggestList id="s" suggestions={[]}
|
||||
activeIndex={-1} onPick={() => {}} onHover={() => {}} />);
|
||||
expect(container).toBeEmptyDOMElement();
|
||||
});
|
||||
});
|
||||
Loaded 100 of 298 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user