Compare commits

..
Author SHA1 Message Date
TudorandClaude Opus 5 b2a3f32c62 fix(cms): regenerate the import map so the Content field renders
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m12s
PR Checks / Backend Smoke (pull_request) Successful in 8s
PR Checks / Build Backend (no push) (pull_request) Successful in 11s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m9s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 10s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 1m4s
Creating a post in the admin panel showed no Content editor, and saving
failed validation on the field the writer was never shown.

The admin panel does not import field components. The server hands the
client a path per field, and resolves it through the generated map at
app/(payload)/admin/importMap.js. A richText field's path is
@payloadcms/richtext-lexical/rsc#RscEntryLexicalField. The committed map
held one entry, @payloadcms/next/rsc#CollectionCards, generated before
the blog collections existed and never re-run. A path missing from the
map is not an error the panel reports: the field simply does not render,
while required is still enforced server-side on save.

next build does not regenerate the map, so the stale copy shipped in the
image and the editor was equally broken on staging and production.

Regenerated with payload generate:importmap, which adds the lexical RSC
field, cell and diff components, BlocksFeatureClient for the Callout
block, and the default toolbar features.

Two things stop it drifting again. There was no script to run, so
package.json gets generate:importmap. And a test asserts the map carries
an entry for each thing the config asks for, in the source-reading style
of the other payload suites; against the old map all five fail.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01DXnXQKnPpZBBP61fBQiFkq
2026-09-02 20:43:06 +01:00
71 changed files with 2098 additions and 2987 deletions

No files matched your search

+5 -13
View File
@@ -20,7 +20,7 @@ PORT=80
# ============================================================================= # =============================================================================
# CORS # CORS
# ============================================================================= # =============================================================================
# JSON array of allowed origins (pydantic-settings format) # Comma-separated list of allowed origins
# In production, only include your actual domain # In production, only include your actual domain
ALLOWED_ORIGINS=["https://schoolcompare.co.uk"] ALLOWED_ORIGINS=["https://schoolcompare.co.uk"]
@@ -33,21 +33,13 @@ ADMIN_API_KEY=CHANGE_THIS_TO_A_SECURE_RANDOM_KEY
# Rate limiting (requests per minute per IP) # Rate limiting (requests per minute per IP)
RATE_LIMIT_PER_MINUTE=60 RATE_LIMIT_PER_MINUTE=60
GLOBAL_RATE_LIMIT_PER_MINUTE=3000 RATE_LIMIT_BURST=10
# Maximum request body size in bytes (default 1MB) # Maximum request body size in bytes (default 1MB)
MAX_REQUEST_SIZE=1048576 MAX_REQUEST_SIZE=1048576
# ============================================================================= # =============================================================================
# SEARCH AND OPTIONAL FEATURE FLAGS # API
# ============================================================================= # =============================================================================
TYPESENSE_URL=http://localhost:8108 DEFAULT_PAGE_SIZE=50
TYPESENSE_API_KEY=CHANGE_THIS_TO_YOUR_TYPESENSE_KEY MAX_PAGE_SIZE=100
# Empty URL disables Unleash-backed flags. Match the managed environment when used.
UNLEASH_URL=
UNLEASH_API_TOKEN=
# Page-size limits are currently declared by route Query parameters.
# DEFAULT_PAGE_SIZE, MAX_PAGE_SIZE and RATE_LIMIT_BURST are not reliable tuning
# controls in the current routes; see docs/LEGACY_CODE.md.
+17 -73
View File
@@ -5,11 +5,6 @@ on:
branches: branches:
- main - main
# Serialise the entire build/deploy/test cycle: no other run can move staging tags.
concurrency:
group: staging-release
cancel-in-progress: false
env: env:
REGISTRY: privaterepo.sitaru.org REGISTRY: privaterepo.sitaru.org
BACKEND_IMAGE_NAME: ${{ gitea.repository }}-backend BACKEND_IMAGE_NAME: ${{ gitea.repository }}-backend
@@ -17,18 +12,7 @@ env:
PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline PIPELINE_IMAGE_NAME: ${{ gitea.repository }}-pipeline
jobs: jobs:
prepare:
runs-on: ubuntu-latest
outputs:
build_id: ${{ steps.identity.outputs.build_id }}
steps:
- id: identity
run: python3 -c 'import uuid; print("build_id=" + uuid.uuid4().hex)' >> "$GITHUB_OUTPUT"
build-backend: build-backend:
needs: [prepare]
outputs:
digest: ${{ steps.build.outputs.digest }}
name: Build Backend (FastAPI) name: Build Backend (FastAPI)
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
@@ -62,24 +46,17 @@ jobs:
type=raw,value=staging type=raw,value=staging
- name: Build and push Backend Docker image - name: Build and push Backend Docker image
id: build
uses: docker/build-push-action@v5 uses: docker/build-push-action@v5
with: with:
context: . context: .
file: ./Dockerfile file: ./Dockerfile
push: true push: true
build-args: |
BUILD_SHA=${{ gitea.sha }}
BUILD_ID=${{ needs.prepare.outputs.build_id }}
tags: ${{ steps.meta-backend.outputs.tags }} tags: ${{ steps.meta-backend.outputs.tags }}
labels: ${{ steps.meta-backend.outputs.labels }} labels: ${{ steps.meta-backend.outputs.labels }}
cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache
cache-to: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache,mode=max cache-to: type=registry,ref=${{ env.REGISTRY }}/${{ env.BACKEND_IMAGE_NAME }}:buildcache,mode=max
build-frontend: build-frontend:
needs: [prepare]
outputs:
digest: ${{ steps.build.outputs.digest }}
name: Build Frontend (Next.js) name: Build Frontend (Next.js)
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
@@ -113,23 +90,18 @@ jobs:
type=raw,value=staging type=raw,value=staging
- name: Build and push Frontend Docker image - name: Build and push Frontend Docker image
id: build
uses: docker/build-push-action@v5 uses: docker/build-push-action@v5
with: with:
context: ./nextjs-app context: ./nextjs-app
file: ./nextjs-app/Dockerfile file: ./nextjs-app/Dockerfile
push: true push: true
build-args: |
BUILD_SHA=${{ gitea.sha }}
BUILD_ID=${{ needs.prepare.outputs.build_id }}
tags: ${{ steps.meta-frontend.outputs.tags }} tags: ${{ steps.meta-frontend.outputs.tags }}
labels: ${{ steps.meta-frontend.outputs.labels }} labels: ${{ steps.meta-frontend.outputs.labels }}
build-args: |
FASTAPI_URL=http://backend:80/api
# Cache disabled due to registry size limits # Cache disabled due to registry size limits
build-pipeline: build-pipeline:
needs: [prepare]
outputs:
digest: ${{ steps.build.outputs.digest }}
name: Build Pipeline (Meltano + dbt + Airflow) name: Build Pipeline (Meltano + dbt + Airflow)
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
@@ -163,15 +135,11 @@ jobs:
type=raw,value=staging type=raw,value=staging
- name: Build and push Pipeline Docker image - name: Build and push Pipeline Docker image
id: build
uses: docker/build-push-action@v5 uses: docker/build-push-action@v5
with: with:
context: ./pipeline context: ./pipeline
file: ./pipeline/Dockerfile file: ./pipeline/Dockerfile
push: true push: true
build-args: |
BUILD_SHA=${{ gitea.sha }}
BUILD_ID=${{ needs.prepare.outputs.build_id }}
tags: ${{ steps.meta-pipeline.outputs.tags }} tags: ${{ steps.meta-pipeline.outputs.tags }}
labels: ${{ steps.meta-pipeline.outputs.labels }} labels: ${{ steps.meta-pipeline.outputs.labels }}
cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.PIPELINE_IMAGE_NAME }}:buildcache cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.PIPELINE_IMAGE_NAME }}:buildcache
@@ -180,23 +148,30 @@ jobs:
deploy-staging: deploy-staging:
name: Deploy to Staging name: Deploy to Staging
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: [prepare, build-backend, build-frontend, build-pipeline] needs: [build-backend, build-frontend, build-pipeline]
steps: steps:
- name: Trigger staging stack update - name: Trigger staging stack update
run: curl -fsSk -X POST "${{ secrets.PORTAINER_STAGING_WEBHOOK }}" run: curl -fsSk -X POST "${{ secrets.PORTAINER_STAGING_WEBHOOK }}"
- uses: actions/checkout@v4 - name: Wait for staging to become healthy
- name: Verify deployed release identity run: |
run: python3 scripts/ci/release.py wait echo "Polling ${STAGING_BASE_URL} for up to 5 minutes..."
for i in $(seq 1 60); do
if curl -fsS -o /dev/null --max-time 10 "${STAGING_BASE_URL}/"; then
echo "Staging is up (attempt $i)"
exit 0
fi
sleep 5
done
echo "Staging did not become healthy in time" >&2
exit 1
env: env:
BASE_URL: ${{ secrets.STAGING_BASE_URL }} STAGING_BASE_URL: ${{ secrets.STAGING_BASE_URL }}
EXPECTED_SHA: ${{ gitea.sha }}
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
e2e-staging: e2e-staging:
name: E2E Journeys against Staging name: E2E Journeys against Staging
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: [prepare, deploy-staging, build-backend, build-frontend, build-pipeline] needs: [deploy-staging]
steps: steps:
- name: Checkout repository - name: Checkout repository
uses: actions/checkout@v4 uses: actions/checkout@v4
@@ -212,42 +187,11 @@ jobs:
npm ci npm ci
npx playwright install --with-deps chromium npx playwright install --with-deps chromium
- name: Verify release before journeys
run: python3 scripts/ci/release.py wait --timeout 10
env:
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
EXPECTED_SHA: ${{ gitea.sha }}
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
- name: Run E2E journeys - name: Run E2E journeys
working-directory: e2e working-directory: e2e
run: npx playwright test run: npx playwright test
env: env:
BASE_URL: ${{ secrets.STAGING_BASE_URL }} BASE_URL: ${{ secrets.STAGING_BASE_URL }}
EXPECTED_SHA: ${{ gitea.sha }}
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
- name: Verify release after journeys
run: python3 scripts/ci/release.py wait --timeout 10
env:
BASE_URL: ${{ secrets.STAGING_BASE_URL }}
EXPECTED_SHA: ${{ gitea.sha }}
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
- uses: docker/setup-buildx-action@v3
- uses: docker/login-action@v3
with:
registry: ${{ env.REGISTRY }}
username: ${{ gitea.actor }}
password: ${{ secrets.REGISTRY_TOKEN }}
- name: Mark tested image digests as verified
run: python3 scripts/ci/release.py verify
env:
EXPECTED_SHA: ${{ gitea.sha }}
EXPECTED_BUILD_ID: ${{ needs.prepare.outputs.build_id }}
BACKEND_DIGEST: ${{ needs.build-backend.outputs.digest }}
FRONTEND_DIGEST: ${{ needs.build-frontend.outputs.digest }}
PIPELINE_DIGEST: ${{ needs.build-pipeline.outputs.digest }}
# Production deployment is a second, manual approval: see promote.yml # Production deployment is a second, manual approval: see promote.yml
# ("Promote to Production (manual)") and docs/DEPLOY.md. # ("Promote to Production (manual)") and docs/DEPLOY.md.
+2 -2
View File
@@ -68,13 +68,13 @@ jobs:
python-version: "3.12" python-version: "3.12"
- name: Install dependencies - name: Install dependencies
run: pip install -r requirements.txt pytest "httpx<0.28" pyyaml run: pip install -r requirements.txt pytest "httpx<0.28"
- name: Import smoke test - name: Import smoke test
run: python -c "from backend.app import app; print('backend imports OK')" run: python -c "from backend.app import app; print('backend imports OK')"
- name: Backend unit tests - name: Backend unit tests
run: python -m pytest backend/tests pipeline/tests scripts/ci/tests -q run: python -m pytest backend/tests -q
build-backend: build-backend:
name: Build Backend (no push) name: Build Backend (no push)
+25 -7
View File
@@ -97,15 +97,33 @@ jobs:
username: ${{ gitea.actor }} username: ${{ gitea.actor }}
password: ${{ secrets.REGISTRY_TOKEN }} password: ${{ secrets.REGISTRY_TOKEN }}
- name: Resolve verified digests and promote the complete image set - name: Retag approved images as prod (keeping rollback pointer)
run: python3 scripts/ci/release.py promote --output release.json run: |
env: SHORT_SHA="${{ steps.resolve.outputs.short }}"
EXPECTED_SHA: ${{ steps.resolve.outputs.full }} for IMAGE in \
"${REGISTRY}/${BACKEND_IMAGE_NAME}" \
"${REGISTRY}/${FRONTEND_IMAGE_NAME}" \
"${REGISTRY}/${PIPELINE_IMAGE_NAME}"; do
# Keep a rollback pointer before moving :prod
docker buildx imagetools create -t "${IMAGE}:prod-previous" "${IMAGE}:prod" || true
docker buildx imagetools create -t "${IMAGE}:prod" "${IMAGE}:${SHORT_SHA}"
echo "Promoted ${IMAGE}:${SHORT_SHA} -> :prod"
done
- name: Trigger production stack update - name: Trigger production stack update
run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}" run: curl -fsSk -X POST "${{ secrets.PORTAINER_PROD_WEBHOOK }}"
- name: Verify production release identity - name: Wait for production to become healthy
run: python3 scripts/ci/release.py wait --release release.json run: |
echo "Polling ${PROD_BASE_URL} for up to 5 minutes..."
for i in $(seq 1 60); do
if curl -fsS -o /dev/null --max-time 10 "${PROD_BASE_URL}/"; then
echo "Production is up (attempt $i)"
exit 0
fi
sleep 5
done
echo "Production did not become healthy in time" >&2
exit 1
env: env:
BASE_URL: ${{ secrets.PROD_BASE_URL }} PROD_BASE_URL: ${{ secrets.PROD_BASE_URL }}
+187 -13
View File
@@ -1,17 +1,191 @@
# Docker deployment # Docker Deployment Guide
The maintained deployment runbook is [docs/DEPLOY.md](docs/DEPLOY.md). ## Quick Start
- Production: `docker-compose.portainer.yml`, using `:prod` images. Deploy the complete SchoolCompare stack (PostgreSQL + FastAPI + Next.js) with one command:
- Staging: `docker-compose.portainer.staging.yml`, using `:staging` images.
- Builds and deployment: `.gitea/workflows/deploy.yml`.
- Human-approved production promotion: `.gitea/workflows/promote.yml`.
The generic `docker-compose.yml` is not a supported one-command onboarding path: ```bash
it still uses `:latest` tags that the release workflow no longer publishes and docker-compose up -d
lacks the full current CMS setup. Review the [legacy inventory](docs/LEGACY_CODE.md) ```
before using old compose examples. Starting an empty database does not populate
school marts.
For architecture, configuration and test commands, see This will start:
[ARCHITECTURE.md](docs/ARCHITECTURE.md) and [DEVELOPMENT.md](docs/DEVELOPMENT.md). - **PostgreSQL** on port 5432 (database)
- **FastAPI** on port 8000 (backend API)
- **Next.js** on port 3000 (frontend)
## Service Details
### PostgreSQL Database
- **Port**: 5432
- **Container**: `schoolcompare_db`
- **Credentials**:
- User: `schoolcompare`
- Password: `schoolcompare`
- Database: `schoolcompare`
- **Volume**: `postgres_data` (persistent storage)
### FastAPI Backend
- **Port**: 8000 → 80 (container)
- **Container**: `schoolcompare_backend`
- **Built from**: Root `Dockerfile`
- **API Endpoint**: http://localhost:8000/api
- **Health Check**: http://localhost:8000/api/data-info
### Next.js Frontend
- **Port**: 3000
- **Container**: `schoolcompare_nextjs`
- **Built from**: `nextjs-app/Dockerfile`
- **URL**: http://localhost:3000
- **Connects to**: Backend via internal network
## Commands
### Start all services
```bash
docker-compose up -d
```
### View logs
```bash
# All services
docker-compose logs -f
# Specific service
docker-compose logs -f nextjs
docker-compose logs -f backend
docker-compose logs -f db
```
### Check status
```bash
docker-compose ps
```
### Stop all services
```bash
docker-compose down
```
### Rebuild after code changes
```bash
# Rebuild and restart specific service
docker-compose up -d --build nextjs
# Rebuild all services
docker-compose up -d --build
```
### Clean restart (remove volumes)
```bash
docker-compose down -v
docker-compose up -d
```
## Initial Database Setup
After first start, you may need to initialize the database:
```bash
# Enter the backend container
docker exec -it schoolcompare_backend bash
# Run migrations or data loading
python -m backend.data_loader
```
## Accessing Services
Once running:
- **Frontend**: http://localhost:3000
- **Backend API**: http://localhost:8000/api
- **API Docs**: http://localhost:8000/docs (Swagger UI)
- **Database**: localhost:5432 (use any PostgreSQL client)
## Environment Variables
Create a `.env` file in the root directory to customize:
```env
# Database
POSTGRES_USER=schoolcompare
POSTGRES_PASSWORD=your_secure_password
POSTGRES_DB=schoolcompare
# Backend
DATABASE_URL=postgresql://schoolcompare:your_secure_password@db:5432/schoolcompare
# Frontend (for client-side access)
NEXT_PUBLIC_API_URL=http://localhost:8000/api
```
Then run:
```bash
docker-compose up -d
```
## Troubleshooting
### Backend not connecting to database
```bash
# Check database health
docker-compose ps
# View backend logs
docker-compose logs backend
# Restart backend
docker-compose restart backend
```
### Frontend not connecting to backend
```bash
# Check backend health
curl http://localhost:8000/api/data-info
# Check Next.js environment variables
docker exec schoolcompare_nextjs env | grep API
```
### Port already in use
```bash
# Change ports in docker-compose.yml
# For example, change "3000:3000" to "3001:3000"
```
### Rebuild from scratch
```bash
docker-compose down -v
docker system prune -a
docker-compose up -d --build
```
## Production Deployment
For production, update the following:
1. **Use secure passwords** in `.env` file
2. **Configure reverse proxy** (Nginx) in front of Next.js
3. **Enable HTTPS** with SSL certificates
4. **Set production environment variables**:
```env
NODE_ENV=production
POSTGRES_PASSWORD=<strong-password>
```
5. **Backup database** regularly:
```bash
docker exec schoolcompare_db pg_dump -U schoolcompare schoolcompare > backup.sql
```
## Network Architecture
```
Internet
↓
Next.js (port 3000) ← User browsers
↓ (internal network)
FastAPI (port 8000) ← API calls
↓ (internal network)
PostgreSQL (port 5432) ← Data queries
```
All services communicate via the `schoolcompare-network` Docker network.
-6
View File
@@ -24,12 +24,6 @@ RUN pip install --no-cache-dir -r requirements.txt
COPY backend/ ./backend/ COPY backend/ ./backend/
COPY scripts/ ./scripts/ COPY scripts/ ./scripts/
ARG BUILD_SHA=development
ARG BUILD_ID=development
LABEL io.schoolcompare.build-id=$BUILD_ID
LABEL io.schoolcompare.commit=$BUILD_SHA
RUN python -c 'import json,sys; open("backend/build-info.json", "w").write(json.dumps({"sha":sys.argv[1],"build_id":sys.argv[2]}))' "$BUILD_SHA" "$BUILD_ID"
# Expose the application port # Expose the application port
EXPOSE 80 EXPOSE 80
-2
View File
@@ -1,5 +1,3 @@
> Historical migration record, retained for context. Setup and architecture claims below may be obsolete. Use [README.md](README.md), [architecture](docs/ARCHITECTURE.md) and [deployment](docs/DEPLOY.md) for current guidance.
# SchoolCompare: Vanilla JS → Next.js Migration Summary # SchoolCompare: Vanilla JS → Next.js Migration Summary
## Overview ## Overview
+199 -52
View File
@@ -1,67 +1,214 @@
# SchoolCompare # Primary School Compass 🧒📚
SchoolCompare compares schools across England: primary (KS2), secondary (KS4), A modern web application for comparing **primary school (KS2)** performance data in **Wandsworth and Merton** over the last 5 years. Built with FastAPI and vanilla JavaScript with Chart.js visualizations.
all-through and post-16 provision, with coverage depending on the source dataset.
It provides school search, postcode maps, comparisons, rankings, place pages,
Ofsted information, admissions and destination measures. Editorial content lives
in a Payload CMS blog.
## Start here ![Python](https://img.shields.io/badge/Python-3.9+-blue)
![FastAPI](https://img.shields.io/badge/FastAPI-0.109-green)
![License](https://img.shields.io/badge/License-MIT-yellow)
- [Architecture and data flow](docs/ARCHITECTURE.md) ## Features
- [Development and validation](docs/DEVELOPMENT.md)
- [Deployment and promotion](docs/DEPLOY.md)
- [Legacy and unused-code inventory](docs/LEGACY_CODE.md)
- [Frontend conventions](nextjs-app/README.md)
- [CMS publishing](nextjs-app/docs/PUBLISHING.md)
## Repository map - 📊 **Interactive Charts** - Visualize KS2 performance trends over time
- 🔍 **Smart Search** - Find primary schools by name in Wandsworth & Merton
- ⚖️ **Side-by-Side Comparison** - Compare up to 5 schools simultaneously
- 🏆 **Rankings** - View top-performing primary schools by various KS2 metrics
- 📱 **Responsive Design** - Works beautifully on desktop and mobile
| Path | Responsibility | ## Key Metrics (KS2)
|---|---|
| `backend/` | FastAPI routes, cached school data, read-only SQLAlchemy mappings, feature flags |
| `nextjs-app/` | Next.js App Router, React UI, Payload CMS, frontend tests |
| `pipeline/plugins/extractors/` | Custom Singer taps for GIAS, EES, Ofsted and other datasets |
| `pipeline/transform/` | dbt staging/intermediate models, marts, seeds and data tests |
| `pipeline/dags/` | Airflow extraction, transformation and publication workflows |
| `pipeline/scripts/` | Search indexing, code generation and operational diagnostics |
| `e2e/` | Playwright journeys against a running environment |
| `.gitea/workflows/` | PR checks, staging deployment and manual production promotion |
| `scripts/` | CI review tooling and historical data utilities; see the legacy inventory |
| `docs/superpowers/`, `mockups/` | Design history and prototypes, not application entry points |
## Runtime The application tracks these Key Stage 2 performance indicators:
The public site is **Next.js**, not the FastAPI root page. Browser `/api/*` | Metric | Description |
requests pass through a Next.js route handler to FastAPI. Server-rendered pages |--------|-------------|
call FastAPI directly using `FASTAPI_URL`, including its `/api` suffix. | **Reading Progress** | Progress in reading from KS1 to KS2 |
| **Writing Progress** | Progress in writing from KS1 to KS2 |
| **Maths Progress** | Progress in maths from KS1 to KS2 |
| **Reading Expected %** | Percentage meeting expected standard in reading |
| **Writing Expected %** | Percentage meeting expected standard in writing |
| **Maths Expected %** | Percentage meeting expected standard in maths |
| **Reading, Writing & Maths Combined %** | Percentage meeting expected standard in all three subjects |
PostgreSQL/PostGIS stores school data. Meltano/Singer extracts source data; ## Quick Start
dbt builds `marts.*`; FastAPI reads those tables. Typesense serves text search
and autocomplete. Payload runs inside Next.js and owns a separate `payload`
database schema and uploaded media.
There is **no automatic CSV import or sample dataset on startup**. A working ### 1. Clone and Setup
school-data environment needs populated marts from the pipeline or an approved
database snapshot. See [development](docs/DEVELOPMENT.md) before choosing a setup.
## Validation ```bash
cd school_results
```sh # Create virtual environment
cd nextjs-app python -m venv venv
npm ci source venv/bin/activate # On Windows: venv\Scripts\activate
npm run typecheck
npm test -- --runInBand # Install dependencies
pip install -r requirements.txt
``` ```
Backend checks, pipeline validation, runtime versions and E2E requirements are ### 2. Run the Application
listed in [DEVELOPMENT.md](docs/DEVELOPMENT.md). No `npm run lint` script is
currently defined.
## Deployment ```bash
# Start the server
python -m uvicorn backend.app:app --reload --port 8000
```
Then open http://localhost:8000 in your browser.
The app will run with **sample data** by default, showing **110 primary schools** (66 in Wandsworth, 44 in Merton) with 5 years of KS2 performance data.
### 3. (Optional) Use Real Data
To use real UK school performance data:
1. Visit [Compare School Performance - Download Data](https://www.compare-school-performance.service.gov.uk/download-data)
2. Download **Key Stage 2** data for the years you want (2019-2024)
- Select "Key Stage 2" as the data type
3. Place the CSV files in the `data/` folder
4. Restart the server - it will automatically load and filter to Wandsworth & Merton schools
**Note:** The app only displays schools in Wandsworth and Merton. Data from other areas will be filtered out.
See the helper script for more details:
```bash
python scripts/download_data.py
```
## Project Structure
```
school_results/
├── backend/
│ └── app.py # FastAPI application with all API endpoints
├── frontend/
│ ├── index.html # Main HTML page
│ ├── styles.css # Styling (warm, editorial design)
│ └── app.js # Frontend JavaScript
├── data/
│ └── .gitkeep # Place CSV data files here
├── scripts/
│ └── download_data.py # Helper for downloading/processing data
├── requirements.txt # Python dependencies
└── README.md
```
## API Endpoints
| Endpoint | Description |
|----------|-------------|
| `GET /api/schools` | List schools with optional search/filter |
| `GET /api/schools/{urn}` | Get detailed data for a specific school |
| `GET /api/compare?urns=...` | Compare multiple schools |
| `GET /api/rankings` | Get school rankings by metric |
| `GET /api/filters` | Get available filter options |
| `GET /api/metrics` | Get available performance metrics |
### Example API Usage
```bash
# Search for schools
curl "http://localhost:8000/api/schools?search=academy"
# Get school details
curl "http://localhost:8000/api/schools/100001"
# Compare schools
curl "http://localhost:8000/api/compare?urns=100001,100002,100003"
# Get rankings
curl "http://localhost:8000/api/rankings?metric=rwm_expected_pct&year=2024"
```
## Data Format
If using your own CSV data, ensure it includes these columns (or similar):
| Column | Type | Description |
|--------|------|-------------|
| URN | Integer | Unique Reference Number |
| SCHNAME | String | School name |
| LA | String | Local Authority (must be Wandsworth or Merton) |
| READPROG | Float | Reading progress score |
| WRITPROG | Float | Writing progress score |
| MATPROG | Float | Maths progress score |
| PTRWM_EXP | Float | % meeting expected standard in reading, writing & maths |
| PTREAD_EXP | Float | % meeting expected standard in reading |
| PTWRIT_EXP | Float | % meeting expected standard in writing |
| PTMAT_EXP | Float | % meeting expected standard in maths |
The application normalizes column names automatically and filters to only show Wandsworth and Merton schools.
## Technology Stack
- **Backend**: FastAPI (Python) - High-performance async API framework
- **Frontend**: Vanilla JavaScript with Chart.js
- **Styling**: Custom CSS with CSS variables for theming
- **Data**: Pandas for CSV processing
## Design Philosophy
The UI features a warm, editorial design inspired by quality publications:
- **Typography**: DM Sans for body text, Playfair Display for headings
- **Color Palette**: Warm cream background with coral and teal accents
- **Interactions**: Smooth animations and hover effects
- **Charts**: Clean, readable data visualizations
## Development
```bash
# Run with auto-reload
python -m uvicorn backend.app:app --reload --port 8000
# Or run directly
python backend/app.py
```
## Coverage
This application is specifically designed for:
- **School Phase**: Primary schools only (Key Stage 2)
- **Geographic Area**: Wandsworth and Merton (London boroughs)
- **Time Period**: Last 5 years of data (2020-2024)
Note: 2021 data shows as unavailable because SATs were cancelled due to COVID-19.
## Data Source
Data is sourced from the UK Government's [Compare School Performance](https://www.compare-school-performance.service.gov.uk/) service, which provides official school performance data for England.
**Important**: When using real data, please comply with the [terms of use](https://www.compare-school-performance.service.gov.uk/download-data) and data protection regulations.
## Scheduled Jobs
### Geocoding Schools (Cron Job)
School postcodes are geocoded by a scheduled job, not on-demand. This improves performance and reduces API calls.
**Setup the cron job** (runs weekly on Sunday at 2am):
```bash
# Edit crontab
crontab -e
# Add this line (adjust paths as needed):
0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1
```
**Manual run:**
```bash
# Geocode only schools missing coordinates
python scripts/geocode_schools.py
# Force re-geocode all schools
python scripts/geocode_schools.py --force
```
## License
MIT License - feel free to use this project for educational purposes.
---
Built with ❤️ for Wandsworth & Merton families
Work on a feature branch and open a PR. Merging to `main` builds images and
deploys staging. Production promotion is a separate, human-triggered Gitea
workflow. Use [DEPLOY.md](docs/DEPLOY.md) and the Portainer compose files as the
operational references. The generic compose examples still reference `:latest`,
which the current release workflow does not publish.
+34 -147
View File
@@ -26,8 +26,7 @@ from starlette.middleware.base import BaseHTTPMiddleware
import asyncio import asyncio
from .config import settings from .config import settings
from .data_loader import ( from .data_loader import (
build_latest_school_data, clear_cache,
load_school_data_as_dataframe,
compute_benchmarks, compute_benchmarks,
load_school_data, load_school_data,
load_latest_school_data, load_latest_school_data,
@@ -39,7 +38,7 @@ from .data_loader import (
) )
from .data_loader import get_data_info as get_db_info from .data_loader import get_data_info as get_db_info
from . import flags from . import flags
from .places import build_place_index, build_place_registry, places_for_urn from .places import build_place_registry
from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS from .schemas import METRIC_DEFINITIONS, RANKING_COLUMNS, SCHOOL_COLUMNS
from .utils import clean_for_json, convert_to_native from .utils import clean_for_json, convert_to_native
@@ -66,10 +65,6 @@ _sitemaps: dict[str, str] | None = None
# Built from the same DataFrame the sitemap uses, so places and sitemap can # Built from the same DataFrame the sitemap uses, so places and sitemap can
# never describe different corpora. Reset by the same admin endpoint. # never describe different corpora. Reset by the same admin endpoint.
_place_registry: dict | None = None _place_registry: dict | None = None
# Cached beside the registry, and invalidated by identity against it — see
# get_place_index. Never cleared independently.
_place_index: dict | None = None
_place_index_source: dict | None = None
VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode") VALID_PLACE_KINDS = ("town", "locality", "authority", "outcode")
@@ -193,24 +188,6 @@ def get_place_registry() -> dict:
return _place_registry return _place_registry
def get_place_index() -> dict:
"""URN → its published places, cached against the registry it came from.
Invalidation is an identity check rather than a second flag to remember to
clear. Anything that drops `_place_registry` — the tests all do — gets a
fresh registry object here, which no longer matches the one the index was
built from, so the index rebuilds with it. A separate `_place_index = None`
would be one more thing to forget, and a stale reverse index is exactly the
bug that would put links to another dataset's places on a school page.
"""
global _place_index, _place_index_source
registry = get_place_registry()
if _place_index is None or _place_index_source is not registry:
_place_index = build_place_index(registry)
_place_index_source = registry
return _place_index
def _urlset(rows: list[str]) -> str: def _urlset(rows: list[str]) -> str:
return "\n".join([ return "\n".join([
'<?xml version="1.0" encoding="UTF-8"?>', '<?xml version="1.0" encoding="UTF-8"?>',
@@ -234,46 +211,7 @@ def _place_url(place) -> str:
return f"/schools/{place.slug}" return f"/schools/{place.slug}"
def _places_payload(urn: int) -> list[dict]: def _place_sitemap_rows(kinds: tuple[str, ...]) -> list[str]:
"""The published places containing this school, as the school page needs
them: a name to write in the link, a count so the anchor can say what it
leads to, and the canonical path.
`phases` carries the phase variants this school actually appears on, which
is usually one and is two for an all-through school — it is listed on both
pages, so there is no tie to break.
Membership is read straight from the registry's own `phase_urns` rather
than re-derived from the school's phase string. The registry is the one
place that decides which phases a place publishes and who is on them;
computing it a second time here is how a page comes to link a school to a
phase page that does not list it, or to a route that does not exist. That
is also why outcodes need no special case: they carry empty `phase_urns`,
so they report no phase links on their own.
"""
payload = []
for place in places_for_urn(get_place_index(), int(urn)):
phases = [
{
"phase": phase,
"count": len(phase_urns),
"url": f"{_place_url(place)}/{phase}",
}
for phase, phase_urns in sorted(place.phase_urns.items())
if int(urn) in phase_urns
]
payload.append({
"kind": place.kind,
"slug": place.slug,
"name": place.name,
"count": len(place.urns),
"url": _place_url(place),
"phases": phases,
})
return payload
def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
"""A <url> per place, plus a phase variant wherever that phase clears the """A <url> per place, plus a phase variant wherever that phase clears the
threshold on its own. threshold on its own.
@@ -283,9 +221,7 @@ def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
linked from the place page either. linked from the place page either.
""" """
rows: list[str] = [] rows: list[str] = []
if registry is None: for p in sorted(get_place_registry().values(), key=lambda p: (p.kind, p.slug)):
registry = get_place_registry()
for p in sorted(registry.values(), key=lambda p: (p.kind, p.slug)):
if p.kind not in kinds: if p.kind not in kinds:
continue continue
rows.append(_url_element(BASE_URL + _place_url(p))) rows.append(_url_element(BASE_URL + _place_url(p)))
@@ -299,10 +235,9 @@ def _place_sitemap_rows(kinds: tuple[str, ...], registry=None) -> list[str]:
return rows return rows
def build_sitemaps(df=None, registry=None) -> dict[str, str]: def build_sitemaps() -> dict[str, str]:
"""Build the sitemap index and every child, keyed by name.""" """Build the sitemap index and every child, keyed by name."""
if df is None: df = load_school_data()
df = load_school_data()
children: dict[str, str] = { children: dict[str, str] = {
"static.xml": _urlset( "static.xml": _urlset(
@@ -322,7 +257,7 @@ def build_sitemaps(df=None, registry=None) -> dict[str, str]:
# measured apart from the school pages'. # measured apart from the school pages'.
for label, kinds in (("places", ("town", "locality", "authority")), for label, kinds in (("places", ("town", "locality", "authority")),
("outcodes", ("outcode",))): ("outcodes", ("outcode",))):
rows = _place_sitemap_rows(kinds, registry) rows = _place_sitemap_rows(kinds)
chunks = [rows[i:i + SITEMAP_CHUNK_SIZE] chunks = [rows[i:i + SITEMAP_CHUNK_SIZE]
for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]] for i in range(0, len(rows), SITEMAP_CHUNK_SIZE)] or [[]]
for n, chunk in enumerate(chunks, start=1): for n, chunk in enumerate(chunks, start=1):
@@ -704,15 +639,6 @@ async def get_config():
} }
@app.get("/api/release")
async def release_identity():
import json
from pathlib import Path
path = Path(__file__).with_name("build-info.json")
identity = json.loads(path.read_text()) if path.exists() else {"sha": "development", "build_id": "development"}
return JSONResponse(identity, headers={"Cache-Control": "no-store"})
@app.get("/api/schools") @app.get("/api/schools")
@limiter.limit(f"{settings.rate_limit_per_minute}/minute") @limiter.limit(f"{settings.rate_limit_per_minute}/minute")
async def get_schools( async def get_schools(
@@ -749,7 +675,7 @@ async def get_schools(
df_latest = load_latest_school_data() df_latest = load_latest_school_data()
if df_latest.empty: if df_latest.empty:
raise HTTPException(status_code=503, detail="School data temporarily unavailable") return {"schools": [], "total": 0, "page": page, "page_size": 0}
# Use configured default if not specified # Use configured default if not specified
if page_size is None: if page_size is None:
@@ -848,8 +774,8 @@ async def get_schools(
# Apply filters # Apply filters
if search: if search:
ts_urns = await asyncio.to_thread(search_schools_typesense, search) ts_urns = search_schools_typesense(search)
if ts_urns is not None: if ts_urns:
urn_order = {urn: i for i, urn in enumerate(ts_urns)} urn_order = {urn: i for i, urn in enumerate(ts_urns)}
schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy() schools_df = schools_df[schools_df["urn"].isin(set(ts_urns))].copy()
schools_df["_ts_rank"] = schools_df["urn"].map(urn_order) schools_df["_ts_rank"] = schools_df["urn"].map(urn_order)
@@ -857,9 +783,9 @@ async def get_schools(
else: else:
# Fallback: Typesense unavailable, use substring match # Fallback: Typesense unavailable, use substring match
search_lower = search.lower() search_lower = search.lower()
mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False, regex=False) mask = schools_df["school_name"].str.lower().str.contains(search_lower, na=False)
if "address" in schools_df.columns: if "address" in schools_df.columns:
mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False, regex=False) mask = mask | schools_df["address"].str.lower().str.contains(search_lower, na=False)
schools_df = schools_df[mask] schools_df = schools_df[mask]
if local_authority: if local_authority:
@@ -918,7 +844,7 @@ async def get_school_details(request: Request, urn: int):
df = load_school_data() df = load_school_data()
if df.empty: if df.empty:
raise HTTPException(status_code=503, detail="School data temporarily unavailable") raise HTTPException(status_code=404, detail="No data available")
school_data = df[df["urn"] == urn] school_data = df[df["urn"] == urn]
@@ -976,13 +902,6 @@ async def get_school_details(request: Request, urn: int):
return { return {
"school_info": school_info, "school_info": school_info,
# Where this school sits in the location layer, for the page's link
# module and breadcrumb. Derived from the same registry the place
# pages and the sitemap use, so a link is never offered for a page
# that does not exist. Empty is a valid answer: a school whose town
# and authority both fall below the publish threshold has nowhere to
# point, and the page renders without the module.
"places": _places_payload(urn),
"yearly_data": clean_for_json(school_data), "yearly_data": clean_for_json(school_data),
# Supplementary data (null if not yet populated by Kestra) # Supplementary data (null if not yet populated by Kestra)
"ofsted": supplementary.get("ofsted"), "ofsted": supplementary.get("ofsted"),
@@ -1555,51 +1474,20 @@ async def get_data_info(request: Request):
} }
_publication_lock = asyncio.Lock()
def _prepare_publication(df):
if df.empty:
raise ValueError("Refusing to publish an empty school dataset")
if not {"urn", "year", "school_name"}.issubset(df.columns):
raise ValueError("School dataset is missing required columns")
if df["urn"].isna().any() or df.duplicated(["urn", "year"]).any():
raise ValueError("School dataset has missing URNs or duplicate school years")
latest = build_latest_school_data(df)
registry = build_place_registry(df)
index = build_place_index(registry)
sitemaps = build_sitemaps(df, registry)
return df, latest, registry, index, sitemaps
def _publish(prepared):
# Called on the event loop with no await: routes cannot observe half a swap.
# The application currently runs one worker; replicas require coordination.
from . import data_loader
global _place_registry, _place_index, _place_index_source, _sitemaps
df, latest, registry, index, sitemaps = prepared
data_loader._df_cache = df
data_loader._df_latest_cache = latest
_place_registry = registry
_place_index = index
_place_index_source = registry
_sitemaps = sitemaps
@app.post("/api/admin/reload") @app.post("/api/admin/reload")
@limiter.limit("5/minute") @limiter.limit("5/minute")
async def reload_data(request: Request, _: bool = Depends(verify_admin_api_key)): async def reload_data(
"""Validate a complete replacement before publishing it; retain data on failure.""" request: Request,
async with _publication_lock: _: bool = Depends(verify_admin_api_key)
try: ):
df = await asyncio.to_thread(load_school_data_as_dataframe) """
prepared = await asyncio.to_thread(_prepare_publication, df) Admin endpoint to force data reload (useful after data updates).
except Exception as exc: Requires X-API-Key header with valid admin API key.
import logging """
logging.getLogger(__name__).exception("Dataset reload failed") clear_cache()
raise HTTPException(status_code=503, detail="Dataset reload failed; previous data retained") from exc await asyncio.to_thread(load_school_data)
_publish(prepared) await asyncio.to_thread(load_latest_school_data)
return {"status": "reloaded", "schools": len(prepared[1])} return {"status": "reloaded"}
@@ -1651,16 +1539,15 @@ async def regenerate_sitemap(
request: Request, request: Request,
_: bool = Depends(verify_admin_api_key), _: bool = Depends(verify_admin_api_key),
): ):
"""Rebuild derived publication data without clearing the live registry.""" """Rebuild and cache the sitemap from current school data. Called by Airflow after data updates."""
async with _publication_lock: global _sitemaps, _place_registry
try: # Places and sitemap are rebuilt together — they read the same marts, and
prepared = await asyncio.to_thread(_prepare_publication, load_school_data()) # letting them drift apart would submit URLs for places that no longer
except Exception as exc: # exist.
raise HTTPException(status_code=503, detail="Sitemap rebuild failed; previous data retained") from exc _place_registry = None
_publish(prepared) _sitemaps = build_sitemaps()
n = sum(x.count("<url>") for x in prepared[4].values()) n = sum(x.count("<url>") for x in _sitemaps.values())
return {"status": "ok", "urls": n, "sitemaps": len(prepared[4])} return {"status": "ok", "urls": n, "sitemaps": len(_sitemaps)}
# Mount static files directly (must be after all routes to avoid catching API calls) # Mount static files directly (must be after all routes to avoid catching API calls)
+24 -55
View File
@@ -84,58 +84,21 @@ def _get_typesense_client():
return None return None
SEARCH_PAGE_SIZE = 250 def search_schools_typesense(query: str, limit: int = 250) -> List[int]:
# Search results are filtered again by the API (authority, phase, postcode, """Search Typesense. Returns URNs in relevance order, or [] if unavailable."""
# etc.), so one page is too small for scoped searches. Keep the candidate set
# bounded, though: a broad query must not turn into an unbounded sequence of
# Typesense requests. Four pages is enough to preserve useful scoped matches
# while putting a hard ceiling on latency and upstream load.
SEARCH_MAX_CANDIDATES = 1_000
def search_schools_typesense(query: str) -> Optional[List[int]]:
"""Return a bounded set of matching URNs in relevance order.
``None`` means Typesense is unavailable; ``[]`` is a valid zero-match
result. The API applies its remaining filters after this search, so the
first few pages are fetched rather than only the first page. Once the
candidate ceiling is reached, the relevance-ordered prefix is returned on
purpose; fetching every match would make common or adversarial queries
unbounded.
"""
client = _get_typesense_client() client = _get_typesense_client()
if client is None: if client is None:
return None return []
urns: list[int] = []
fetched = 0
try: try:
page = 1 result = client.collections["schools"].documents.search({
while fetched < SEARCH_MAX_CANDIDATES: "q": query,
page_size = min(SEARCH_PAGE_SIZE, SEARCH_MAX_CANDIDATES - fetched) "query_by": "school_name,local_authority,postcode",
result = client.collections["schools"].documents.search({ "per_page": min(limit, 250),
"q": query, "typo_tokens_threshold": 1,
"query_by": "school_name,local_authority,postcode", })
"per_page": page_size, return [int(h["document"]["urn"]) for h in result.get("hits", [])]
"page": page,
"typo_tokens_threshold": 1,
})
hits = result.get("hits", [])
urns.extend(int(h["document"]["urn"]) for h in hits)
fetched += len(hits)
if fetched >= result.get("found", fetched):
return list(dict.fromkeys(urns))
if not hits:
raise ValueError("Search pagination ended before all matches arrived")
page += 1
logging.getLogger(__name__).info(
"Typesense search capped at %d candidates for query %r",
SEARCH_MAX_CANDIDATES,
query,
)
return list(dict.fromkeys(urns))
except Exception: except Exception:
logging.getLogger(__name__).exception("School search unavailable") return []
return None
# The most a public endpoint will return in one response. # The most a public endpoint will return in one response.
@@ -225,6 +188,16 @@ def geocode_single_postcode(postcode: str) -> Optional[Tuple[float, float]]:
return None return None
def haversine_distance(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
"""Calculate great-circle distance between two points (miles)."""
from math import radians, cos, sin, asin, sqrt
lat1, lon1, lat2, lon2 = map(radians, [lat1, lon1, lat2, lon2])
dlat = lat2 - lat1
dlon = lon2 - lon1
a = sin(dlat / 2) ** 2 + cos(lat1) * cos(lat2) * sin(dlon / 2) ** 2
return 2 * asin(sqrt(a)) * 3956
# ============================================================================= # =============================================================================
# MAIN DATA LOAD — joins dim_school + dim_location + fact_performance # MAIN DATA LOAD — joins dim_school + dim_location + fact_performance
# fact_performance is a merged KS2+KS4 table (one row per URN per year). # fact_performance is a merged KS2+KS4 table (one row per URN per year).
@@ -539,12 +512,7 @@ def load_latest_school_data() -> pd.DataFrame:
if _df_latest_cache is not None: if _df_latest_cache is not None:
return _df_latest_cache return _df_latest_cache
_df_latest_cache = build_latest_school_data(load_school_data()) df = load_school_data()
return _df_latest_cache
def build_latest_school_data(df: pd.DataFrame) -> pd.DataFrame:
"""Build a replacement snapshot without mutating the published caches."""
if df.empty: if df.empty:
return df return df
@@ -577,7 +545,8 @@ def build_latest_school_data(df: pd.DataFrame) -> pd.DataFrame:
df_latest = pd.concat([df_latest, df_no_perf], ignore_index=True) df_latest = pd.concat([df_latest, df_no_perf], ignore_index=True)
print(f"Latest-snapshot cache built: {len(df_latest)} schools") print(f"Latest-snapshot cache built: {len(df_latest)} schools")
return df_latest _df_latest_cache = df_latest
return _df_latest_cache
def clear_cache(): def clear_cache():
-17
View File
@@ -57,23 +57,6 @@ REGISTRY: dict[str, Flag] = {
), ),
added=date(2026, 8, 26), added=date(2026, 8, 26),
), ),
Flag(
name="about_page",
description=(
"The /about page, its footer link, its sitemap entry, and the "
"named-author byline on every blog post."
),
added=date(2026, 9, 8),
),
Flag(
name="blog",
description=(
"The /blog index, post pages, the RSS feed, their footer link "
"and their sitemap entries. Not /admin: posts must be "
"writable before the blog is readable."
),
added=date(2026, 9, 8),
),
) )
} }
+512
View File
@@ -0,0 +1,512 @@
"""
Database migration logic for importing CSV data.
Used by both CLI script and automatic startup migration.
"""
import re
from pathlib import Path
from typing import Dict, Optional
import numpy as np
import pandas as pd
import requests
from .config import settings
from .database import Base, engine, get_db_session
from .models import School, SchoolResult
from .schemas import (
COLUMN_MAPPINGS,
LA_CODE_TO_NAME,
NULL_VALUES,
SCHOOL_TYPE_MAP,
)
def parse_numeric(value) -> Optional[float]:
"""Parse a numeric value, handling special cases."""
if pd.isna(value):
return None
if isinstance(value, (int, float)):
return float(value) if not np.isnan(value) else None
str_val = str(value).strip().upper()
if str_val in NULL_VALUES or str_val == "":
return None
# Remove percentage signs if present
str_val = str_val.replace("%", "")
try:
return float(str_val)
except ValueError:
return None
def extract_year_from_folder(folder_name: str) -> Optional[int]:
"""Extract year from folder name like '2023-2024'."""
match = re.search(r"(\d{4})-(\d{4})", folder_name)
if match:
return int(match.group(2))
match = re.search(r"(\d{4})", folder_name)
if match:
return int(match.group(1))
return None
def geocode_postcodes_bulk(postcodes: list) -> Dict[str, tuple]:
"""
Geocode postcodes in bulk using postcodes.io API.
Returns dict of postcode -> (latitude, longitude).
"""
results = {}
valid_postcodes = [
p.strip().upper()
for p in postcodes
if p and isinstance(p, str) and len(p.strip()) >= 5
]
valid_postcodes = list(set(valid_postcodes))
if not valid_postcodes:
return results
batch_size = 100
total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size
for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)):
batch = valid_postcodes[batch_start : batch_start + batch_size]
print(
f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)..."
)
try:
response = requests.post(
"https://api.postcodes.io/postcodes",
json={"postcodes": batch},
timeout=30,
)
if response.status_code == 200:
data = response.json()
for item in data.get("result", []):
if item and item.get("result"):
pc = item["query"].upper()
lat = item["result"].get("latitude")
lon = item["result"].get("longitude")
if lat and lon:
results[pc] = (lat, lon)
except Exception as e:
print(f" Warning: Geocoding batch failed: {e}")
return results
def load_csv_data(data_dir: Path) -> pd.DataFrame:
"""Load all CSV data from data directory."""
all_data = []
for folder in sorted(data_dir.iterdir()):
if not folder.is_dir():
continue
year = extract_year_from_folder(folder.name)
if not year:
continue
# Specifically look for the KS2 results file
ks2_file = folder / "england_ks2final.csv"
if not ks2_file.exists():
continue
csv_file = ks2_file
print(f" Loading {csv_file.name} (year {year})...")
try:
df = pd.read_csv(csv_file, encoding="latin-1", low_memory=False)
except Exception as e:
print(f" Error loading {csv_file}: {e}")
continue
# Rename columns
df.rename(columns=COLUMN_MAPPINGS, inplace=True)
df["year"] = year
# Handle local authority name
la_name_cols = ["LANAME", "LA (name)", "LA_NAME", "LA NAME"]
la_name_col = next((c for c in la_name_cols if c in df.columns), None)
if la_name_col and la_name_col != "local_authority":
df["local_authority"] = df[la_name_col]
elif "LEA" in df.columns:
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
df["local_authority"] = (
df["local_authority_code"]
.map(LA_CODE_TO_NAME)
.fillna(df["LEA"].astype(str))
)
# Store LEA code
if "LEA" in df.columns:
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
# Map school type
if "school_type_code" in df.columns:
df["school_type"] = (
df["school_type_code"]
.map(SCHOOL_TYPE_MAP)
.fillna(df["school_type_code"])
)
# Create combined address
addr_parts = ["address1", "address2", "town", "postcode"]
for col in addr_parts:
if col not in df.columns:
df[col] = None
df["address"] = df.apply(
lambda r: ", ".join(
str(v)
for v in [
r.get("address1"),
r.get("address2"),
r.get("town"),
r.get("postcode"),
]
if pd.notna(v) and str(v).strip()
),
axis=1,
)
all_data.append(df)
print(f" Loaded {len(df)} records")
if all_data:
result = pd.concat(all_data, ignore_index=True)
print(f"\nTotal records loaded: {len(result)}")
print(f"Unique schools: {result['urn'].nunique()}")
print(f"Years: {sorted(result['year'].unique())}")
return result
return pd.DataFrame()
def migrate_data(df: pd.DataFrame, geocode: bool = False, geocode_cache: dict = None):
"""Migrate DataFrame data to database."""
if geocode_cache is None:
geocode_cache = {}
# Clean URN column - convert to integer, drop invalid values
df = df.copy()
df["urn"] = pd.to_numeric(df["urn"], errors="coerce")
df = df.dropna(subset=["urn"])
df["urn"] = df["urn"].astype(int)
# Group by URN to get unique schools (use latest year's data)
school_data = (
df.sort_values("year", ascending=False).groupby("urn").first().reset_index()
)
print(f"\nMigrating {len(school_data)} unique schools...")
# Geocode postcodes that aren't already in the cache
geocoded = dict(geocode_cache) # start with preserved coordinates
if geocode and "postcode" in df.columns:
cached_postcodes = {
str(row.get("postcode", "")).strip().upper()
for _, row in school_data.iterrows()
if int(float(str(row.get("urn", 0) or 0))) in geocode_cache
}
postcodes_needed = [
p for p in df["postcode"].dropna().unique()
if str(p).strip().upper() not in cached_postcodes
]
if postcodes_needed:
print(f"\nGeocoding {len(postcodes_needed)} postcodes ({len(geocode_cache)} restored from cache)...")
fresh = geocode_postcodes_bulk(postcodes_needed)
geocoded.update(fresh)
print(f" Successfully geocoded {len(fresh)} new postcodes")
else:
print(f"\nAll {len(geocode_cache)} postcodes restored from cache, skipping geocoding.")
with get_db_session() as db:
# Create schools
urn_to_school_id = {}
schools_created = 0
for _, row in school_data.iterrows():
# Safely parse URN - handle None, NaN, whitespace, and invalid values
urn_val = row.get("urn")
urn = None
if pd.notna(urn_val):
try:
urn_str = str(urn_val).strip()
if urn_str:
urn = int(float(urn_str)) # Handle "12345.0" format
except (ValueError, TypeError):
pass
if not urn:
continue
# Skip if we've already added this URN (handles duplicates in source data)
if urn in urn_to_school_id:
continue
# Get geocoding data
postcode = row.get("postcode")
lat, lon = None, None
if postcode and pd.notna(postcode):
coords = geocoded.get(str(postcode).strip().upper())
if coords:
lat, lon = coords
# Safely parse local_authority_code
la_code = None
la_code_val = row.get("local_authority_code")
if pd.notna(la_code_val):
try:
la_code_str = str(la_code_val).strip()
if la_code_str:
la_code = int(float(la_code_str))
except (ValueError, TypeError):
pass
school = School(
urn=urn,
school_name=row.get("school_name")
if pd.notna(row.get("school_name"))
else "Unknown",
local_authority=row.get("local_authority")
if pd.notna(row.get("local_authority"))
else None,
local_authority_code=la_code,
school_type=row.get("school_type")
if pd.notna(row.get("school_type"))
else None,
school_type_code=row.get("school_type_code")
if pd.notna(row.get("school_type_code"))
else None,
religious_denomination=row.get("religious_denomination")
if pd.notna(row.get("religious_denomination"))
else None,
age_range=row.get("age_range")
if pd.notna(row.get("age_range"))
else None,
address1=row.get("address1") if pd.notna(row.get("address1")) else None,
address2=row.get("address2") if pd.notna(row.get("address2")) else None,
town=row.get("town") if pd.notna(row.get("town")) else None,
postcode=row.get("postcode") if pd.notna(row.get("postcode")) else None,
latitude=lat,
longitude=lon,
)
db.add(school)
db.flush() # Get the ID
urn_to_school_id[urn] = school.id
schools_created += 1
if schools_created % 1000 == 0:
print(f" Created {schools_created} schools...")
print(f" Created {schools_created} schools")
# Create results
print(f"\nMigrating {len(df)} yearly results...")
results_created = 0
for _, row in df.iterrows():
# Safely parse URN
urn_val = row.get("urn")
urn = None
if pd.notna(urn_val):
try:
urn_str = str(urn_val).strip()
if urn_str:
urn = int(float(urn_str))
except (ValueError, TypeError):
pass
if not urn or urn not in urn_to_school_id:
continue
school_id = urn_to_school_id[urn]
# Safely parse year
year_val = row.get("year")
year = None
if pd.notna(year_val):
try:
year = int(float(str(year_val).strip()))
except (ValueError, TypeError):
pass
if not year:
continue
result = SchoolResult(
school_id=school_id,
year=year,
total_pupils=parse_numeric(row.get("total_pupils")),
eligible_pupils=parse_numeric(row.get("eligible_pupils")),
# Expected Standard
rwm_expected_pct=parse_numeric(row.get("rwm_expected_pct")),
reading_expected_pct=parse_numeric(row.get("reading_expected_pct")),
writing_expected_pct=parse_numeric(row.get("writing_expected_pct")),
maths_expected_pct=parse_numeric(row.get("maths_expected_pct")),
gps_expected_pct=parse_numeric(row.get("gps_expected_pct")),
science_expected_pct=parse_numeric(row.get("science_expected_pct")),
# Higher Standard
rwm_high_pct=parse_numeric(row.get("rwm_high_pct")),
reading_high_pct=parse_numeric(row.get("reading_high_pct")),
writing_high_pct=parse_numeric(row.get("writing_high_pct")),
maths_high_pct=parse_numeric(row.get("maths_high_pct")),
gps_high_pct=parse_numeric(row.get("gps_high_pct")),
# Progress
reading_progress=parse_numeric(row.get("reading_progress")),
writing_progress=parse_numeric(row.get("writing_progress")),
maths_progress=parse_numeric(row.get("maths_progress")),
# Averages
reading_avg_score=parse_numeric(row.get("reading_avg_score")),
maths_avg_score=parse_numeric(row.get("maths_avg_score")),
gps_avg_score=parse_numeric(row.get("gps_avg_score")),
# Context
disadvantaged_pct=parse_numeric(row.get("disadvantaged_pct")),
eal_pct=parse_numeric(row.get("eal_pct")),
sen_support_pct=parse_numeric(row.get("sen_support_pct")),
sen_ehcp_pct=parse_numeric(row.get("sen_ehcp_pct")),
stability_pct=parse_numeric(row.get("stability_pct")),
# Absence
reading_absence_pct=parse_numeric(row.get("reading_absence_pct")),
gps_absence_pct=parse_numeric(row.get("gps_absence_pct")),
maths_absence_pct=parse_numeric(row.get("maths_absence_pct")),
writing_absence_pct=parse_numeric(row.get("writing_absence_pct")),
science_absence_pct=parse_numeric(row.get("science_absence_pct")),
# Gender
rwm_expected_boys_pct=parse_numeric(row.get("rwm_expected_boys_pct")),
rwm_expected_girls_pct=parse_numeric(row.get("rwm_expected_girls_pct")),
rwm_high_boys_pct=parse_numeric(row.get("rwm_high_boys_pct")),
rwm_high_girls_pct=parse_numeric(row.get("rwm_high_girls_pct")),
# Disadvantaged
rwm_expected_disadvantaged_pct=parse_numeric(
row.get("rwm_expected_disadvantaged_pct")
),
rwm_expected_non_disadvantaged_pct=parse_numeric(
row.get("rwm_expected_non_disadvantaged_pct")
),
disadvantaged_gap=parse_numeric(row.get("disadvantaged_gap")),
# 3-Year
rwm_expected_3yr_pct=parse_numeric(row.get("rwm_expected_3yr_pct")),
reading_avg_3yr=parse_numeric(row.get("reading_avg_3yr")),
maths_avg_3yr=parse_numeric(row.get("maths_avg_3yr")),
)
db.add(result)
results_created += 1
if results_created % 10000 == 0:
print(f" Created {results_created} results...")
db.flush()
print(f" Created {results_created} results")
# Commit all changes
db.commit()
print("\nMigration complete!")
def _apply_schema_alterations():
"""
Add new columns to existing tables using ALTER TABLE … ADD COLUMN IF NOT EXISTS.
Safe to run on every migration — no-ops if the column already exists.
Add entries here whenever models.py gains new columns on an existing table.
"""
alterations = [
# v4: Ofsted Report Card columns
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS framework VARCHAR(20)",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_safeguarding_met BOOLEAN",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_inclusion INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_curriculum_teaching INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_achievement INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_attendance_behaviour INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_personal_development INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_leadership_governance INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_early_years INTEGER",
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_sixth_form INTEGER",
]
from sqlalchemy import text as sa_text
with engine.connect() as conn:
for stmt in alterations:
try:
conn.execute(sa_text(stmt))
except Exception as e:
print(f" Warning: alteration skipped ({e})")
conn.commit()
def _apply_schema_drops():
"""
Drop tables retired from the schema. Idempotent (DROP … IF EXISTS), so it's
safe to run on every migration. Add entries here when a model is removed.
"""
drops = [
# v6: Ofsted Parent View feature removed
"DROP TABLE IF EXISTS marts.fact_parent_view CASCADE",
]
from sqlalchemy import text as sa_text
with engine.connect() as conn:
for stmt in drops:
try:
conn.execute(sa_text(stmt))
except Exception as e:
print(f" Warning: drop skipped ({e})")
conn.commit()
def run_full_migration(geocode: bool = False) -> bool:
"""
Run a complete migration: drop all tables and reimport from CSV.
Returns True if successful, False if no data found.
Raises exception on error.
"""
# Preserve existing geocoding so a reimport doesn't throw away coordinates
# that took a long time to compute.
geocode_cache: dict[int, tuple[float, float]] = {}
inspector = __import__("sqlalchemy").inspect(engine)
if "schools" in inspector.get_table_names():
try:
with get_db_session() as db:
rows = db.execute(
__import__("sqlalchemy").text(
"SELECT urn, latitude, longitude FROM schools "
"WHERE latitude IS NOT NULL AND longitude IS NOT NULL"
)
).fetchall()
geocode_cache = {r.urn: (r.latitude, r.longitude) for r in rows}
print(f" Saved {len(geocode_cache)} existing geocoded coordinates.")
except Exception as e:
print(f" Warning: could not save geocode cache: {e}")
# Only drop the core KS2 tables — leave supplementary tables (ofsted, census,
# finance, etc.) intact so a reimport doesn't wipe integrator-populated data.
# schema_version is NOT dropped: it persists so restarts don't re-trigger migration.
ks2_tables = ["school_results", "schools"]
print(f"Dropping core tables: {ks2_tables} ...")
inspector = __import__("sqlalchemy").inspect(engine)
existing = set(inspector.get_table_names())
for tname in ks2_tables:
if tname in existing:
Base.metadata.tables[tname].drop(bind=engine)
print("Creating all tables...")
Base.metadata.create_all(bind=engine)
# ALTER existing supplementary tables to add any new columns.
# create_all() only creates missing tables; it won't add columns to tables
# that already exist from an older schema version. These statements are
# idempotent (IF NOT EXISTS) so they're safe to run on every migration.
print("Applying column additions to supplementary tables...")
_apply_schema_alterations()
print("Dropping retired tables...")
_apply_schema_drops()
print("\nLoading CSV data...")
df = load_csv_data(settings.data_dir)
if df.empty:
print("Warning: No CSV data found to migrate!")
return False
migrate_data(df, geocode=geocode, geocode_cache=geocode_cache)
return True
-42
View File
@@ -296,48 +296,6 @@ def _locality_places(df, publishable: set[int],
return out return out
# Ordered authority → town/locality → outcode, widest first, because that is
# the order a breadcrumb reads. The link module re-sorts for its own purposes.
_PLACE_ORDER = {"authority": 0, "town": 1, "locality": 2, "outcode": 3}
def build_place_index(registry: dict[str, Place]) -> dict[int, tuple[Place, ...]]:
"""URN → the published places containing it, built once per registry.
The reverse of the registry, and the thing school pages link out through.
Derived from the registry rather than maintained beside it, so the two
cannot disagree about which places exist: a place below the publish
threshold is absent from the registry, so it is absent from here too, and
a link is never offered for a page that does not exist.
Built as an index rather than scanned per call because /api/schools/{urn}
is the site's highest-traffic endpoint. Scanning meant walking every place
and doing a tuple membership test against each — on the order of 10^5
comparisons per request, repeated for every school page view. One pass at
registry-build time replaces all of it with a dict lookup.
"""
grouped: dict[int, list[Place]] = {}
for place in registry.values():
for urn in place.urns:
grouped.setdefault(int(urn), []).append(place)
return {
urn: tuple(sorted(places,
key=lambda p: (_PLACE_ORDER.get(p.kind, 9), p.slug)))
for urn, places in grouped.items()
}
def places_for_urn(index: dict[int, tuple[Place, ...]], urn: int) -> tuple[Place, ...]:
"""The published places containing this school, widest first.
Empty is a real answer, not a failure: a school whose town and authority
both fall below the publish threshold has nowhere to link, and the page
renders without the module.
"""
return index.get(int(urn), ())
def build_place_registry(df) -> dict[str, Place]: def build_place_registry(df) -> dict[str, Place]:
"""Every place the site publishes, keyed by "<kind>:<slug>".""" """Every place the site publishes, keyed by "<kind>:<slug>"."""
if df.empty or "urn" not in df.columns: if df.empty or "urn" not in df.columns:
+1 -78
View File
@@ -8,8 +8,7 @@ import numpy as np
import pandas as pd import pandas as pd
import pytest import pytest
from backend.places import (MIN_SCHOOLS, build_place_index, from backend.places import MIN_SCHOOLS, build_place_registry
build_place_registry, places_for_urn)
def _df(rows: list[dict]) -> pd.DataFrame: def _df(rows: list[dict]) -> pd.DataFrame:
@@ -419,79 +418,3 @@ def test_an_authority_still_publishes_phase_variants():
and /schools/authority/[la]/[phase] is the route that serves it.""" and /schools/authority/[la]/[phase] is the route that serves it."""
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Maidstone", "Kent"))) reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Maidstone", "Kent")))
assert reg["authority:kent"].publishes_phase("primary") assert reg["authority:kent"].publishes_phase("primary")
# ── The reverse index: which published places contain a school ──────────────
#
# School pages link out to the location layer through this. It is the whole
# point of the index: before it, ~27k school pages linked to nothing on the
# site and stranded whatever authority they held.
def test_a_school_resolves_to_every_published_place_containing_it():
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
places = places_for_urn(build_place_index(reg), 100000)
kinds = {p.kind for p in places}
assert "town" in kinds
assert "authority" in kinds
def test_a_school_in_an_unpublished_town_still_resolves_to_its_authority():
# A town below the threshold has no page, so there is no link to offer —
# but the authority above it clears the threshold on the same schools and
# is where that reader should be sent.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS - 1, "Tinytown", "Essex")
+ _town(MIN_SCHOOLS, "Brentwood", "Essex", start=200000)
))
places = places_for_urn(build_place_index(reg), 100000)
# The town is below the threshold, so it has no page and must not be
# offered as a link. The authority above it does, and is the right target.
assert all(p.slug != "tinytown" for p in places)
assert "authority" in {p.kind for p in places}
def test_an_unknown_urn_resolves_to_nothing_rather_than_raising():
# A school page renders for any URN the API knows; the link module is not
# entitled to take the page down when it has nothing to say.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
assert places_for_urn(build_place_index(reg), 999999) == ()
def test_the_index_is_consistent_with_the_registry_it_was_built_from():
# The invariant that matters: a link module must never offer a place whose
# page does not exist, and never omit one that does.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS, "Bedford", "Bedford", start=300000)
))
index = build_place_index(reg)
for key, place in reg.items():
for urn in place.urns:
assert place in places_for_urn(index, urn), (
f"{urn} is in {key} but the index does not say so")
def test_the_index_holds_no_school_the_registry_does_not():
# The reverse direction of the invariant above. An index entry for a URN
# no published place contains would put a link on a page for a place that
# does not list that school.
reg = build_place_registry(_df(
_town(MIN_SCHOOLS, "Brentwood", "Essex")
+ _town(MIN_SCHOOLS - 1, "Tinytown", "Essex", start=400000)
))
index = build_place_index(reg)
for urn, places in index.items():
for place in places:
assert urn in place.urns
assert place.key in reg
def test_the_index_preserves_the_widest_first_order():
# The breadcrumb reads authority then town, and takes this order as given.
reg = build_place_registry(_df(_town(MIN_SCHOOLS, "Brentwood", "Essex")))
kinds = [p.kind for p in places_for_urn(build_place_index(reg), 100000)]
assert kinds.index("authority") < kinds.index("town")
-68
View File
@@ -1,68 +0,0 @@
"""Publication must preserve the current dataset until every replacement is ready."""
import asyncio
import pandas as pd
import pytest
from fastapi.testclient import TestClient
from backend import app as api, data_loader
from backend.tests.test_sixth_form_flag import _schools_df
@pytest.fixture
def client(monkeypatch):
old = _schools_df()
monkeypatch.setattr(data_loader, '_df_cache', old)
monkeypatch.setattr(data_loader, '_df_latest_cache', old)
monkeypatch.setattr(api, '_place_registry', {'old': 'registry'})
monkeypatch.setattr(api, '_place_index', {'old': 'index'})
monkeypatch.setattr(api, '_place_index_source', api._place_registry)
monkeypatch.setattr(api, '_sitemaps', {'old.xml': 'old sitemap'})
monkeypatch.setattr(api, '_publication_lock', asyncio.Lock())
monkeypatch.setattr(api.limiter, 'enabled', False)
api.app.dependency_overrides[api.verify_admin_api_key] = lambda: True
yield TestClient(api.app, raise_server_exceptions=False)
api.app.dependency_overrides.clear()
def state():
return (data_loader._df_cache, data_loader._df_latest_cache, api._place_registry,
api._place_index, api._place_index_source, api._sitemaps)
@pytest.mark.parametrize('failure', ['empty', 'database', 'sitemap', 'duplicate'])
def test_failed_reload_preserves_every_published_object(client, monkeypatch, failure):
before = state()
df = _schools_df()
if failure == 'empty':
df = pd.DataFrame()
if failure == 'duplicate':
df = pd.concat([df, df.iloc[:1]], ignore_index=True)
def load():
if failure == 'database':
raise RuntimeError('database unavailable')
return df
monkeypatch.setattr(api, 'load_school_data_as_dataframe', load)
if failure == 'sitemap':
monkeypatch.setattr(api, 'build_sitemaps', lambda *args: (_ for _ in ()).throw(RuntimeError('bad XML')))
response = client.post('/api/admin/reload')
assert response.status_code == 503
assert all(a is b for a, b in zip(before, state()))
def test_success_publishes_school_data_places_and_sitemaps(client, monkeypatch):
df = _schools_df()
df.loc[0, 'school_name'] = 'Replacement School'
monkeypatch.setattr(api, 'load_school_data_as_dataframe', lambda: df)
response = client.post('/api/admin/reload')
assert response.status_code == 200
assert data_loader.load_school_data() is df
assert data_loader.load_latest_school_data().iloc[0].school_name == 'Replacement School'
assert api._place_index_source is api._place_registry
assert 'old.xml' not in api._sitemaps
assert 'replacement-school' in api._sitemaps['schools-1.xml']
def test_failed_sitemap_regeneration_keeps_existing_publication(client, monkeypatch):
before = state()
monkeypatch.setattr(api, 'build_sitemaps', lambda *args: (_ for _ in ()).throw(RuntimeError('bad XML')))
assert client.post('/api/admin/regenerate-sitemap').status_code == 503
assert all(a is b for a, b in zip(before, state()))
-176
View File
@@ -56,11 +56,6 @@ def client(monkeypatch):
monkeypatch.setattr( monkeypatch.setattr(
app_module, "get_supplementary_data", lambda db, urn: {} app_module, "get_supplementary_data", lambda db, urn: {}
) )
# The place registry is a module-level cache, so without this the endpoint
# answers from whatever registry an earlier test happened to leave behind
# — and a `places == []` assertion is satisfied by a stale registry just
# as well as by this fixture's own data, which makes it prove nothing.
monkeypatch.setattr(app_module, "_place_registry", None)
return TestClient(app_module.app, raise_server_exceptions=False) return TestClient(app_module.app, raise_server_exceptions=False)
@@ -74,174 +69,3 @@ def test_nan_gias_fields_serialize_as_null(client):
assert info["capacity"] is None assert info["capacity"] is None
assert info["total_pupils"] is None assert info["total_pupils"] is None
assert info["school_name"] == "West London Performing Arts Academy" assert info["school_name"] == "West London Performing Arts Academy"
# ── Links out to the location layer ─────────────────────────────────────────
#
# School pages carried no link into the site at all: the only anchor on the
# template pointed at the school's own website, so ~27k pages received
# whatever authority the site had and sent it off-site. `places` is what the
# link module and the breadcrumb are built from.
def test_places_is_present_even_when_the_school_belongs_to_none(client):
# This fixture's single school cannot clear any publish threshold, so the
# honest answer is an empty list. The key must still be there: a missing
# key and "no places" are different things to the page rendering it.
body = client.get("/api/schools/150275").json()
assert body["places"] == []
def test_places_names_only_pages_that_exist(monkeypatch):
from backend import app as app_module
from backend.places import MIN_SCHOOLS
def _df():
return pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood",
"local_authority": "Essex",
"postcode": "CM15 8AA",
"phase": "Primary",
"year": 202425,
"rwm_expected_pct": 60.0,
"attainment_8_score": np.nan,
"ofsted_grade": 2.0,
"ofsted_date": None,
}
for i in range(MIN_SCHOOLS)
])
monkeypatch.setattr(app_module, "load_school_data", _df)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
places = client.get("/api/schools/100000").json()["places"]
assert places, "a school in a published town must offer links"
by_kind = {p["kind"]: p for p in places}
assert by_kind["town"]["url"] == "/schools/brentwood"
assert by_kind["authority"]["url"] == "/schools/authority/essex"
# Every entry carries what the link text needs, and a count, so the anchor
# can say what it leads to rather than "click here".
for place in places:
assert place["name"]
assert place["count"] >= 1
assert place["url"].startswith("/schools/")
def _brentwood_df(phase: str = "Primary", n: int = None):
from backend.places import MIN_SCHOOLS
n = n if n is not None else MIN_SCHOOLS
return lambda: pd.DataFrame([
{
"urn": 100000 + i,
"school_name": f"Brentwood School {i}",
"town": "Brentwood", "local_authority": "Essex",
"postcode": "CM15 8AA", "phase": phase, "year": 202425,
"rwm_expected_pct": 60.0, "attainment_8_score": 50.0,
"ofsted_grade": 2.0, "ofsted_date": None,
}
for i in range(n)
])
def _places_for(monkeypatch, df_factory, urn: int):
from backend import app as app_module
monkeypatch.setattr(app_module, "load_school_data", df_factory)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
monkeypatch.setattr(app_module, "_place_registry", None)
client = TestClient(app_module.app, raise_server_exceptions=False)
return client.get(f"/api/schools/{urn}").json()["places"]
def test_a_place_offers_the_phase_page_this_school_appears_on(monkeypatch):
# "primary schools in brentwood" is the query the phase pages exist for,
# and ~950 of them were once reachable by nothing at all.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert town["phases"], "a primary school in a published primary town has a link"
assert town["phases"][0]["url"] == "/schools/brentwood/primary"
assert town["phases"][0]["count"] >= 1
def test_an_all_through_school_offers_both_phase_pages(monkeypatch):
# It genuinely appears on both, so there is no tie to break.
places = _places_for(monkeypatch, _brentwood_df("All-through"), 100000)
town = next(p for p in places if p["kind"] == "town")
assert {p["phase"] for p in town["phases"]} == {"primary", "secondary"}
def test_outcodes_never_offer_a_phase_page(monkeypatch):
# The registry gives outcodes no phase route — nobody searches "primary
# schools in SW11" — and computing them anyway once put a link to a
# nonexistent route on all 1,720 outcode pages.
places = _places_for(monkeypatch, _brentwood_df("Primary"), 100000)
outcode = next((p for p in places if p["kind"] == "outcode"), None)
if outcode is not None:
assert outcode["phases"] == []
def test_a_school_absent_from_the_phase_page_is_not_linked_to_it(monkeypatch):
# The check is URN membership in the registry's own phase list, not a
# re-derivation of the phase mapping. A secondary school must not be sent
# to a primary phase page that does not list it.
from backend.places import MIN_SCHOOLS
def df():
rows = [
{"urn": 100000 + i, "school_name": f"P{i}", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Primary", "year": 202425, "rwm_expected_pct": 60.0,
"attainment_8_score": np.nan, "ofsted_grade": 2.0,
"ofsted_date": None}
for i in range(MIN_SCHOOLS)
]
rows.append({
"urn": 900000, "school_name": "Lone Secondary", "town": "Brentwood",
"local_authority": "Essex", "postcode": "CM15 8AA",
"phase": "Secondary", "year": 202425, "rwm_expected_pct": np.nan,
"attainment_8_score": 50.0, "ofsted_grade": 2.0, "ofsted_date": None,
})
return pd.DataFrame(rows)
places = _places_for(monkeypatch, df, 900000)
town = next(p for p in places if p["kind"] == "town")
# The town publishes a primary page, but this secondary school is not on
# it, and there are too few secondaries for a secondary page.
assert town["phases"] == []
def test_the_place_index_rebuilds_when_the_registry_is_replaced(monkeypatch):
"""The reverse index is cached; a stale one would put another dataset's
places on a school page. Invalidation is an identity check against the
registry rather than a second flag, so this asserts the check works."""
from backend import app as app_module
monkeypatch.setattr(app_module, "_place_registry", None)
monkeypatch.setattr(app_module, "_place_index", None)
monkeypatch.setattr(app_module, "_place_index_source", None)
monkeypatch.setattr(app_module, "load_school_data", _brentwood_df("Primary"))
first = app_module.get_place_index()
assert 100000 in first
# Same registry object, so the index is reused rather than rebuilt.
assert app_module.get_place_index() is first
# Drop the registry the way every test that touches place data does. The
# index must follow it, not survive it.
app_module._place_registry = None
monkeypatch.setattr(app_module, "load_school_data",
_brentwood_df("Primary", n=0))
rebuilt = app_module.get_place_index()
assert rebuilt is not first
assert 100000 not in rebuilt, "the index outlived the registry it came from"
-114
View File
@@ -1,114 +0,0 @@
from types import SimpleNamespace
import pytest
from fastapi.testclient import TestClient
from backend import app as api, data_loader
from backend.tests.test_sixth_form_flag import _schools_df
def client_for(monkeypatch, search):
client = SimpleNamespace(collections={'schools': SimpleNamespace(documents=SimpleNamespace(search=search))})
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: client)
def test_search_returns_matches_beyond_first_page(monkeypatch):
pages = []
def search(params):
pages.append(params['page'])
urns = range(100000, 100250) if params['page'] == 1 else [100999]
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
client_for(monkeypatch, search)
result = data_loader.search_schools_typesense('academy')
assert len(result) == 251
assert result[-1] == 100999
assert pages == [1, 2]
def test_search_caps_broad_queries_at_a_bounded_number_of_pages(monkeypatch):
requests = []
def search(params):
requests.append(params)
return {
'found': 10_000,
'hits': [
{'document': {'urn': 100000 + params['page'] * 1000 + i}}
for i in range(params['per_page'])
],
}
client_for(monkeypatch, search)
result = data_loader.search_schools_typesense('school')
assert len(result) == data_loader.SEARCH_MAX_CANDIDATES
assert len(requests) == data_loader.SEARCH_MAX_CANDIDATES // data_loader.SEARCH_PAGE_SIZE
assert all(request['per_page'] == data_loader.SEARCH_PAGE_SIZE for request in requests)
assert requests[-1]['page'] == len(requests)
def test_search_uses_a_smaller_final_page_when_the_cap_is_not_a_page_multiple(monkeypatch):
monkeypatch.setattr(data_loader, 'SEARCH_MAX_CANDIDATES', 251)
requests = []
def search(params):
requests.append(params)
return {
'found': 10_000,
'hits': [{'document': {'urn': 100000 + len(requests) * 1000 + i}}
for i in range(params['per_page'])],
}
client_for(monkeypatch, search)
result = data_loader.search_schools_typesense('school')
assert len(result) == 251
assert [request['per_page'] for request in requests] == [250, 1]
def test_later_page_failure_does_not_return_partial_results(monkeypatch):
def search(params):
if params['page'] == 2:
raise RuntimeError('timeout')
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in range(100000, 100250)]}
client_for(monkeypatch, search)
assert data_loader.search_schools_typesense('academy') is None
def test_zero_matches_are_distinct_from_unavailable(monkeypatch):
client_for(monkeypatch, lambda _: {'found': 0, 'hits': []})
assert data_loader.search_schools_typesense('academy') == []
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: None)
assert data_loader.search_schools_typesense('academy') is None
@pytest.mark.parametrize('matches, expected', [([], []), (None, [100001])])
def test_fallback_only_on_dependency_failure(monkeypatch, matches, expected):
monkeypatch.setattr(api.limiter, 'enabled', False)
monkeypatch.setattr(api, 'load_latest_school_data', _schools_df)
monkeypatch.setattr(api, 'search_schools_typesense', lambda _: matches)
response = TestClient(api.app).get('/api/schools?search=Alpha')
assert response.status_code == 200
assert [s['urn'] for s in response.json()['schools']] == expected
def test_filtered_api_keeps_match_from_second_search_page(monkeypatch):
monkeypatch.setattr(api.limiter, 'enabled', False)
df = _schools_df()
monkeypatch.setattr(api, 'load_latest_school_data', lambda: df)
def search(params):
urns = range(200000, 200250) if params['page'] == 1 else [100001]
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
client_for(monkeypatch, search)
response = TestClient(api.app).get('/api/schools?search=Alpha&local_authority=Testshire')
assert response.status_code == 200
assert response.json()['total'] == 1
assert response.json()['schools'][0]['urn'] == 100001
def test_unavailable_dataset_is_not_a_missing_school_or_empty_search(monkeypatch):
import pandas as pd
monkeypatch.setattr(api.limiter, 'enabled', False)
monkeypatch.setattr(api, 'load_school_data', lambda: pd.DataFrame())
monkeypatch.setattr(api, 'load_latest_school_data', lambda: pd.DataFrame())
client = TestClient(api.app)
assert client.get('/api/schools/100001').status_code == 503
assert client.get('/api/schools?search=school').status_code == 503
+26
View File
@@ -0,0 +1,26 @@
"""
Schema versioning for database migrations.
HOW TO USE:
- Bump SCHEMA_VERSION when making changes to database models
- This triggers an automatic full data reimport on next app startup
WHEN TO BUMP:
- Adding/removing columns in models.py
- Changing column types or constraints
- Modifying CSV column mappings in schemas.py
- Any change that requires fresh data import
"""
# Current schema version - increment when models change
SCHEMA_VERSION = 6
# Changelog for documentation
SCHEMA_CHANGELOG = {
1: "Initial schema with School and SchoolResult tables",
2: "Added pupil absence fields (reading, maths, gps, writing, science)",
3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools",
4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)",
5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables",
6: "Removed the Ofsted Parent View feature: dropped fact_parent_view table and model",
}
+172 -29
View File
@@ -1,37 +1,180 @@
# SchoolCompare project context # SchoolCompare.co.uk - Project Context
## Maintained documentation ## Overview
Read [README.md](README.md), [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) and SchoolCompare is a web application for comparing UK primary school (KS2) performance data. It allows users to:
[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the current implementation. - Search and browse schools by name, location (postcode), or local authority
[docs/LEGACY_CODE.md](docs/LEGACY_CODE.md) records obsolete paths and deliberate - Compare multiple schools side-by-side with charts and tables
compatibility code. Historical design documents are not current setup instructions. - View school rankings by various KS2 metrics
- See historical performance trends across years
## Architecture constraints ## Architecture
- Next.js serves the public UI. FastAPI serves school data from dbt-built `marts.*`. ### Backend (Python/FastAPI)
The backend does not create school tables or import CSVs at startup. - **Framework**: FastAPI with uvicorn
- School coverage spans England and multiple phases, not only primary schools in - **Database**: PostgreSQL with SQLAlchemy ORM
Wandsworth and Merton. - **Data Source**: UK Government "Compare School Performance" CSV downloads
- `/api/*` belongs to the FastAPI proxy. Payload uses `/cms-api` and `/admin`.
- Payload runs inside Next.js, with its own `payload` schema and persistent media. Key files:
Keep CMS migrations independent of school-data transformations. - `backend/app.py` - Main FastAPI application, API routes
- Public and Payload route groups have separate root layouts. Do not introduce - `backend/config.py` - Configuration via pydantic-settings (env vars, .env file)
`app/layout.tsx`. Keep site-wide metadata files at the `app/` root. - `backend/database.py` - SQLAlchemy engine, session management
- Builds must succeed with `DATABASE_URL` unset. Do not call `getCachedPayload()` - `backend/models.py` - Database models (School, SchoolResult)
at module scope or add DB-backed `generateStaticParams`. - `backend/data_loader.py` - Data queries, geocoding, legacy DataFrame compatibility
- After changing CMS fields/editors, run `npm run generate:importmap` and commit - `backend/schemas.py` - Column mappings, metric definitions, LA code mappings
the generated import map. See `nextjs-app/docs/PUBLISHING.md`.
- The backend and pipeline GIAS dictionary copies are generated together; preserve ### Content / CMS (Payload)
their parity. Tests enforce it.
Payload CMS runs **inside** the Next.js app — one image, one container, no
separate service. It powers `/blog`; `/about` is a plain coded page.
- **Admin panel:** `/admin`. The only authenticated surface on the site.
`noindex` via both `robots.txt` and `X-Robots-Tag`.
- **CMS API:** `/cms-api`, **not** `/api`. `/api/*` is a catch-all proxy to
FastAPI (`app/(frontend)/api/[...path]`) which would silently swallow every
admin call and forward it to the backend. Mount points are defined once in
`lib/payloadRoutes.ts`.
- **Database:** the existing Postgres, in its own `payload` schema, so no
pipeline operation on `public` — including
`scripts/migrate_csv_to_db.py --drop` — can reach blog content.
- **Uploads:** the `payload_media` Docker volume at `/app/media`. Not
reproducible from the pipeline; must be backed up.
- **New env vars:** `DATABASE_URL` and `PAYLOAD_SECRET` on the frontend service.
Staging must use a different `PAYLOAD_SECRET` from production.
- Publishing workflow and house style: `nextjs-app/docs/PUBLISHING.md`.
- **Admin field components resolve through a generated import map**
(`app/(payload)/admin/importMap.js`). Payload hands the client a *path* per
field and looks it up there; a missing entry renders no field and reports no
error, while `required` still blocks the save. After adding or changing any
field, editor or lexical feature, run `npm run generate:importmap` in
`nextjs-app/` and commit the result.
### Two route groups
`nextjs-app/app/` has no root `layout.tsx`. It cannot: Payload's admin panel
ships its own root layout rendering `<html>`/`<body>`, and Next permits
multiple root layouts only when no `app/layout.tsx` exists.
- `app/(frontend)/` — the site. Its `layout.tsx` is the site's root layout.
- `app/(payload)/` — the admin panel and `/cms-api`.
Route groups are invisible to routing, so every public URL is unchanged.
**The metadata file conventions stay at the `app/` root** — `robots.ts`,
`opengraph-image.tsx`, `icon.png`, `apple-icon.png`. Inside a route group Next
treats them as segment-scoped: it renames `/icon.png` to `/icon-<hash>.png` and
drops `/robots.txt` entirely. Route handlers are unaffected.
The build must succeed with `DATABASE_URL` unset, because CI builds it that
way. Never call `getCachedPayload()` at module scope, and never add
`generateStaticParams` to a DB-backed route.
### Frontend (Vanilla JS)
- Single-page application with hash-based routing
- Chart.js for data visualization
- No build step required
Key files:
- `frontend/index.html` - Main HTML structure
- `frontend/app.js` - All application logic, API calls, rendering
- `frontend/styles.css` - Styling (CSS variables, responsive design)
### Database Schema
```
schools school_results
├── id (PK) ├── id (PK)
├── urn (unique, indexed) ├── school_id (FK → schools.id)
├── school_name ├── year (indexed)
├── local_authority ├── rwm_expected_pct
├── school_type ├── reading_expected_pct
├── postcode ├── ... (all KS2 metrics)
├── latitude, longitude └── unique(school_id, year)
└── results → SchoolResult[]
```
## Configuration
Environment variables (or `.env` file):
- `DATABASE_URL` - PostgreSQL connection string (default: `postgresql://schoolcompare:schoolcompare@localhost:5432/schoolcompare`)
- `HOST`, `PORT` - Server binding (default: `0.0.0.0:80`)
- `ALLOWED_ORIGINS` - CORS origins
## Running Locally
1. Start PostgreSQL:
```bash
docker compose up -d db
```
2. Run migration to import CSV data:
```bash
python scripts/migrate_csv_to_db.py --drop
# Add --geocode to geocode postcodes (slower, adds lat/long)
```
3. Start the app:
```bash
uvicorn backend.app:app --host 0.0.0.0 --port 8000
```
## Docker Deployment
```bash
docker compose up -d
```
This starts:
- `db` - PostgreSQL 16 with persistent volume
- `app` - FastAPI application on port 80
## Data
- Source: UK Government Compare School Performance downloads
- Location: `data/` directory with year folders (e.g., `2023-2024/england_ks2final.csv`)
- The `scripts/download_data.py` can fetch data from the government website
## Key Features
- **Location Search**: Enter postcode to find nearby schools (uses postcodes.io API)
- **Multi-school Comparison**: Select multiple schools, view metrics across years
- **Rankings**: Top schools by any KS2 metric, filterable by local authority
- **Variability Analysis**: Shows standard deviation of scores across years
## API Endpoints
- `GET /api/schools` - List/search schools (supports pagination, location search)
- `GET /api/schools/{urn}` - School details with all yearly data
- `GET /api/compare?urns=123,456` - Compare multiple schools
- `GET /api/rankings` - School rankings by metric
- `GET /api/filters` - Available filter options (LAs, types, years)
- `GET /api/metrics` - Metric definitions (single source of truth)
- `GET /api/data-info` - Database stats
## SDLC ## SDLC
Follow [docs/DEPLOY.md](docs/DEPLOY.md). Full details in `docs/DEPLOY.md`. The short version:
- Never push directly to `main`. Use a feature branch and a PR with passing checks. - **Never push to `main` directly.** Work on a feature branch and open a PR;
- Merges deploy staging only. Production promotion is a separate human decision; branch protection requires the PR checks (typecheck, tests, builds, AI review)
do not trigger the promotion workflow yourself. to pass before merge.
- Update E2E journeys in the same PR when changing user-facing behaviour. - Merging to `main` deploys automatically **to staging only**: images are
- Do not attempt to start a local server to test the application; use unit checks built once, deployed to the staging Portainer stack, and verified by the
and the configured integration environment. Playwright journeys in `e2e/`. Production is a second, manual approval:
the "Promote to Production (manual)" workflow in Gitea Actions, run after
testing the feature on staging. It refuses commits whose staging E2E gate
isn't green. Never trigger it yourself — promotion is the human's call.
- If you change user-facing behaviour, update or extend the `e2e/` journey
tests in the same PR — they gate whether staging is fit for human testing
and whether a commit is promotable.
## Recent Changes
- Added staging environment + automated staging→prod pipeline (Gitea Actions)
- Migrated from CSV file storage to PostgreSQL database
- Added location-based search using postcode geocoding
- Added local authority filter to rankings
- Improved frontend with featured schools, loading states, API caching
# Important
- Do not attempt to start a local server to test the application, it does not work
-120
View File
@@ -1,120 +0,0 @@
# Architecture
This describes the implementation as reviewed on 2026-09-14. It distinguishes
current behaviour from improvements still to be implemented.
## Request flow
```text
Browser → Next.js public routes
├─ /api/* proxy → FastAPI → cached DataFrames / PostgreSQL marts
│ ├─ Typesense (search and suggestions)
│ └─ postcodes.io (postcode lookup)
└─ /admin, /cms-api, /blog → Payload → payload schema + media volume
Next.js server rendering → FastAPI directly through FASTAPI_URL
```
`nextjs-app/lib/api.ts` contains typed fetch wrappers and revalidation defaults.
The proxy is `nextjs-app/app/(frontend)/api/[...path]/route.ts`. Payload uses
`/cms-api` so its routes do not collide with the FastAPI proxy. The proxy denies
`/api/flags`; server-side rendering reads flags directly from FastAPI.
## Data ownership
| Layer | Owner and role |
|---|---|
| Source data | GIAS, DfE EES, Ofsted, finance, deprivation and council admission-distance sources |
| `raw` | Singer taps and the PostgreSQL target configured in `pipeline/meltano.yml` |
| Staging/intermediate/marts | dbt models in `pipeline/transform`; marts are materialized tables |
| `marts.dim_school`, `marts.dim_location` | School identity and location, filtered to supported England establishments |
| `marts.fact_*` | Performance and supplementary datasets; coverage and years vary |
| Typesense `schools` alias | Search documents built by `pipeline/scripts/sync_typesense.py` |
| `payload` | CMS collections and migrations in `nextjs-app/`; independent of dbt |
| Media volume | Uploaded blog media; requires backup and cannot be regenerated from school datasets |
`backend/models.py` maps existing marts for reading. It does not create the school
schema. There is no startup schema-version migration or CSV reimport. Payload's
`nextjs-app/migrations/` is active and must not be confused with the removed
legacy backend migration code.
Coordinates normally come from GIAS British National Grid coordinates transformed
by PostGIS in `dim_location.sql`. `pipeline/scripts/geocode_postcodes.py` is a
manual fallback utility, not a task wired into the current school-data DAG.
Backend postcode searches also use postcodes.io; that lookup does not populate
school coordinates in the database.
## Backend boundaries
- `app.py`: routes, middleware, search filtering, sitemap/place publication and response assembly.
- `data_loader.py`: SQL loading, process-local DataFrame caches, Typesense calls,
postcode lookups, supplementary queries and benchmark calculation.
- `database.py`: synchronous SQLAlchemy engine and sessions.
- `schemas.py`: metric definitions, column mappings and display metadata; despite
its name this is not a collection of Pydantic API response models.
- `places.py` and `localities.py`: place registry and curated locality information.
- `flags.py`: Unleash-backed feature flags, disabled when no server is configured.
- `gias_codes.py` / `ofsted_codes.py`: source-code translation and display rules.
Search starts from a cached latest-row-per-school snapshot. Detail pages read
history from the full DataFrame and supplementary data from marts. Comparisons
batch supplementary queries across selected URNs. Async routes still contain
synchronous dependency calls; a fully asynchronous database layer is not present.
## Frontend boundaries
`app/(frontend)` owns the public root layout and pages. `app/(payload)` owns the
CMS root layout. Do not add a shared `app/layout.tsx`: these groups deliberately
have separate root layouts. Root metadata files remain in `app/`.
Server pages fetch initial data and pass it to client views. Client state uses
React hooks, URL search parameters and the comparison context/localStorage.
There is no SWR dependency. Leaflet maps are loaded through dynamic wrappers;
Chart.js renders performance and comparison charts.
`components/school/` contains detail sections, with section decisions and data
preparation in `lib/schoolSections.ts`. `lib/types.ts` contains manually maintained
API types. `payload-types.ts` and the Payload import map are generated artifacts.
## Publication and caching today
1. Airflow DAGs extract and validate source data, then run selected dbt builds.
2. Relevant DAGs rebuild Typesense and swap the `schools` alias.
3. They call `POST /api/admin/reload` with `X-API-Key`. It builds and validates
replacement DataFrames, places, reverse membership and sitemaps off the request
loop, then publishes them together. Failure returns 503 and preserves live data.
4. A separate weekly sitemap DAG can regenerate the derived publication from the
current DataFrame without clearing the live registry first.
GIAS is scheduled daily, Ofsted monthly, and annual datasets are manually
triggered. The DAG definitions are authoritative for selectors and dependencies.
Caches exist in several independent layers: backend DataFrames and registries,
backend HTTP Cache-Control/ETags, Next.js fetch/page revalidation, and browser or
shared HTTP caches where configured. Place fetches request a one-week revalidation
interval. HTTP ETags are computed after route execution, not before database work.
Typesense publication validates every import response and the final document
count before switching aliases. A session-scoped PostgreSQL advisory lock
serialises index reads/publication across DAGs. The previous collection remains
available for rollback; old unaliased collections are pruned after success.
Failed drafts are retained until a later successful cleanup, because an uncertain
alias-update response must never cause deletion of a potentially live index.
The backend snapshot swap is process-local and assumes the current single-worker
deployment. It is not an atomic transaction spanning PostgreSQL marts, Typesense
and Next.js caches. Next.js caches are not explicitly purged by the pipeline.
School search retrieves a relevance-ordered candidate prefix (currently capped at
1,000 URNs) before applying API filters. This keeps scoped searches useful while
putting a hard ceiling on Typesense round trips; only a dependency failure invokes
substring fallback, not a valid empty match set.
## Deployment references
See [DEPLOY.md](DEPLOY.md). PR checks include frontend typechecking/tests, backend
unit tests, image builds and AI review. Staging journeys run after merging.
Staging runs are serialised across builds, deployment and E2E. Build-stamped
frontend/backend identities are checked before and after journeys. Only then are
the captured image digests marked verified. Promotion resolves and validates the
complete verified image set before retagging production. See the runbook for
first-rollout requirements and remaining integration checks.
+4 -53
View File
@@ -19,9 +19,8 @@ PR checks (.gitea/workflows/pr-checks.yml)
▼ ▼
Stage pipeline (.gitea/workflows/deploy.yml) — automatic Stage pipeline (.gitea/workflows/deploy.yml) — automatic
1. build & push images → tags sha-<sha>, staging 1. build & push images → tags sha-<sha>, staging
2. staging Portainer webhook → verify frontend/backend SHA + build ID 2. staging Portainer webhook → wait for staging health
3. Playwright E2E journeys against staging ← gate before human testing 3. Playwright E2E journeys against staging ← gate before human testing
4. verify identity again; tag tested digests verified-<full-sha>
▼ ▼
Manual testing on staging (stx.schoolcompare.co.uk) Manual testing on staging (stx.schoolcompare.co.uk)
│ Actions → "Promote to Production (manual)" ← approval #2 │ Actions → "Promote to Production (manual)" ← approval #2
@@ -29,15 +28,14 @@ Manual testing on staging (stx.schoolcompare.co.uk)
Promote pipeline (.gitea/workflows/promote.yml) — manual dispatch Promote pipeline (.gitea/workflows/promote.yml) — manual dispatch
1. resolve target sha (input, or latest main if empty) 1. resolve target sha (input, or latest main if empty)
2. REFUSE unless that commit's "E2E Journeys against Staging" status is green 2. REFUSE unless that commit's "E2E Journeys against Staging" status is green
3. resolve verified-<full-sha> digests, validate labels, retag digests → :prod 3. retag sha-<sha> → :prod (same bytes — build once, promote the image)
previous :prod saved as :prod-previous previous :prod saved as :prod-previous
4. prod Portainer webhook → verify expected SHA + build ID 4. prod Portainer webhook → wait for prod health
``` ```
Key principle: **build once, promote the exact image**. Production pins `:prod`, Key principle: **build once, promote the exact image**. Production pins `:prod`,
which only moves when a human runs the promote workflow — and the workflow which only moves when a human runs the promote workflow — and the workflow
only accepts commits that passed the staging E2E gate and have a complete verified only accepts commits that passed the staging E2E gate. Nothing tags `:latest`
image set. Nothing tags `:latest`
anymore. anymore.
## Branch & PR workflow ## Branch & PR workflow
@@ -258,50 +256,3 @@ how long any feature is exposed to this.
If `UNLEASH_URL` is unset, every flag is `False` and no connection is If `UNLEASH_URL` is unset, every flag is `False` and no connection is
attempted. That is the correct behaviour for local development and CI, and it attempted. That is the correct behaviour for local development and CI, and it
means the test suites need no flag server. means the test suites need no flag server.
## Release identity and the P1 reliability gate
Every staging run creates a random build ID before building its three images.
Each image carries the commit and build ID as labels. Frontend/backend images
also contain a build-time JSON file; environment overrides cannot rewrite it.
`/release.json` returns both identities with `Cache-Control: no-store`. It fails
with 503 when either identity cannot be read. FastAPI's internal endpoint is
`/api/release`.
The entire staging workflow shares one concurrency group, with cancellation
disabled. This needs Gitea 1.26 or newer, where workflow concurrency is supported
([release notes](https://blog.gitea.com/release-of-1.26.0/)); the configured server
reported 1.27.3 during this change. Do not run the workflow on an older server
that ignores the concurrency key. Manual deployments outside this workflow must
also avoid changing staging during journeys.
The gate checks both identities before and after Playwright. It then validates
labels on the captured build output digests and tags them `verified-<full-sha>`.
The manual promotion script resolves all three verified tags to immutable digests
and confirms one matching commit/build ID before moving any `:prod` tag. It polls
production for that same identity using a locally saved release manifest.
A registry error can still interrupt the three tag writes; the Portainer webhook
only runs after successful promotion, and rerunning promotion resolves the full
verified set again. There is no cross-registry atomic tag transaction.
**First rollout:** old green commits without verified tags/build identities are
not promotable through this gate. Build and test a commit containing the new
workflow first. The release route must be reachable through the configured
`STAGING_BASE_URL`/`PROD_BASE_URL`; it deliberately avoids the public staging
`/api` proxy limitation. No new deployment secret is required.
`scripts/ci/release.py` implements identity polling and digest verification.
The poller identifies itself as `SchoolCompare-Release-Check/1.0`: the public
staging proxy has returned HTTP 403 to Python's default urllib user agent even
while the release endpoint was healthy. It logs changes in HTTP/connection
failures or observed release identities, and includes the last observation in
the timeout error. If verification fails, use that observation to distinguish
proxy rejection (403), an unavailable release endpoint (503), and containers
still reporting an older SHA/build ID. Check the configured base URL from the
CI runner; a successful request from another machine does not establish runner
connectivity. Do not bypass identity verification to unblock a deployment.
Its mocked tests run in PR checks alongside backend and index-publication tests.
The new Playwright journeys also check deployed identity and stale pagination.
Local unit checks do not validate registry credentials, Portainer behaviour,
proxy routing or a deployed image; those require the staging run. Production
promotion remains a separate human action.
-94
View File
@@ -1,94 +0,0 @@
# Development and validation
## Prerequisites and environment boundaries
Use a feature branch. The deployed stack is the integration environment; do not
assume a local server can run from a fresh checkout. This cleanup did not start
local servers or provision databases. Unit tests use fixtures and mocks.
The current versions are not yet aligned:
| Component | Container | PR checks |
|---|---|---|
| Backend | Python 3.11 | Python 3.12 |
| Frontend | Node 24 | Node 22 |
| Pipeline | Python 3.13 | Pipeline image build |
Use the component's container version when reproducing deployment behaviour.
The backend dependency pins predate Python 3.14; do not assume the system Python
can install or run them. Version alignment is a separate maintenance task.
## Frontend checks
```sh
cd nextjs-app
npm ci
npm run typecheck
npm test -- --runInBand
```
`npm run build` is the production build check. There is no `lint` script.
Tests live in `__tests__/` and use Jest/React Testing Library. These checks do not
prove that live PostgreSQL queries, Typesense or a deployed proxy work.
The frontend `.env.example` documents runtime variables. Browser traffic normally
uses `/api`; `FASTAPI_URL` is an absolute server-side URL ending in `/api`.
Payload additionally needs `DATABASE_URL` and `PAYLOAD_SECRET` when used at runtime.
Never commit credentials or real `.env` files.
## Backend checks
From the repository root, using an available Python 3.11 or 3.12 interpreter:
```sh
python3.11 -m venv /tmp/schoolcompare-backend-venv
/tmp/schoolcompare-backend-venv/bin/python -m pip install -r requirements.txt pytest 'httpx<0.28' pyyaml
/tmp/schoolcompare-backend-venv/bin/python -m pytest backend/tests pipeline/tests scripts/ci/tests -q
```
Substitute `python3.12` if matching PR CI. The test dependencies above match the
current workflow; they are not yet captured in a dedicated development lockfile.
Backend configuration is defined in `backend/config.py`; `.env.example` documents
commonly used values. `ALLOWED_ORIGINS` uses a JSON array, not a comma-separated string.
## Data and pipeline work
The app needs populated `marts.*` tables. A new Postgres instance alone is not a
working school-data environment. Use the existing managed pipeline or an approved
snapshot; the removed CSV importer cannot build the current schema.
The pipeline container includes Meltano, dbt/Postgres, Airflow and the custom taps.
Airflow commands/selectors live in `pipeline/dags/`. Schema tests live in
`pipeline/transform/tests/` and model YAML files. Run the relevant `dbt build`
selector in an isolated data environment for model changes; it writes tables and
is not a read-only smoke test. Prefer `python -m dbt.cli.main` as the DAGs do.
GIAS dictionaries are generated together by
`pipeline/scripts/generate_gias_codes.py`. The backend and pipeline copies are
intentional; `backend/tests/test_gias_codes.py` checks that they stay identical.
For Payload collection/editor changes, run `npm run generate:importmap` in
`nextjs-app/` and include the generated map. Preserve CMS migrations and the
separate `payload` schema. See [publishing](../nextjs-app/docs/PUBLISHING.md).
## End-to-end checks
Against an existing, authorised test environment:
```sh
cd e2e
npm ci
npx playwright install chromium
BASE_URL=https://your-test-environment.example npx playwright test
```
The suite does not start a web server. CI installs Chromium with system dependencies
and runs against staging. Use the configured staging target: `docs/DEPLOY.md`
records the public staging proxy limitation. User-visible behaviour changes should
update the corresponding journeys.
## Before requesting review
Run checks relevant to the change, inspect `git diff --check`, and report checks
that could not run. Do not publish or promote as part of local validation.
[DEPLOY.md](DEPLOY.md) documents the PR and human promotion gates.
-89
View File
@@ -1,89 +0,0 @@
# Legacy and unused-code inventory
Reviewed 2026-09-14. This inventory records source evidence, not production usage
telemetry. A command with no repository caller may still be run manually or from
an external scheduler. Historical specs and prototypes are not runtime imports.
## Method and scope
Searched backend imports, tests, CLI scripts, Airflow DAGs, Meltano configuration,
Gitea workflows, Dockerfiles and documentation. For frontend candidates, inspected
TypeScript imports, re-exports, literal dynamic imports and `require` calls,
resolving relative and `@/` paths while excluding tests, dependencies and build
output. Checked candidates again with text searches including tests.
Next.js route files, generated Payload import-map entries and plugin discovery
are entry points even without ordinary imports. This is why a zero-import count
alone is not sufficient grounds for deletion. Computed imports and external
operators are outside this static audit.
## Removed in this cleanup
These names are recorded for Git-history lookup; they are no longer file links.
| Removed path or symbol | Evidence and replacement |
|---|---|
| `backend/migration.py` | Imported `School` and `SchoolResult`, which no longer exist in `backend/models.py`. Only the legacy CSV CLI imported it. Current tables are built by dbt. |
| `backend/version.py` | Only the legacy importer consumed `SCHEMA_VERSION`. FastAPI lifespan does not perform version-triggered imports. This is unrelated to active Payload migrations. |
| `scripts/migrate_csv_to_db.py` | Imported removed `init_db`/`set_db_schema_version` helpers and the obsolete models indirectly. No runtime, DAG or workflow calls it. Use the managed pipeline for current marts. |
| `scripts/geocode_schools.py` | Imported the removed `School` ORM model. No pipeline/workflow calls it. Coordinates now come from GIAS/PostGIS; a separate mart-aware manual utility remains under `pipeline/scripts/`. |
| `backend.data_loader.haversine_distance` | No callers. Search uses its inline vectorised NumPy calculation. |
| `nextjs-app/lib/api.ts: fetcher` | No callers; SWR is not installed. Application fetches use the named API wrappers. |
| `nextjs-app/lib/api.ts: kmToMiles` | No callers. `calculateDistance` remains because `CutoffMapPanel` uses it. |
The removed command files could not import successfully against the current
backend. This cleanup does not run replacements, migrate data or modify databases.
Their previous implementations remain recoverable from Git history.
## Unused candidates retained for a separate cleanup
| Candidate | Evidence | Recommended next step |
|---|---|---|
| `nextjs-app/components/LoadingSkeleton.tsx` and its CSS | No application or test imports found. | Remove together after confirming no planned use. |
| `nextjs-app/components/Pagination.tsx` and its CSS | No application or test imports found; HomeView implements load-more behaviour. | Remove as a pair if numbered pagination will not return. |
| `nextjs-app/components/SchoolCard.tsx` and its CSS | Imported by its own tests, not application code. HomeView uses SchoolRow/SecondarySchoolRow. | Decide whether to retire the card design; if removed, remove its dedicated tests as well. Passing tests do not establish runtime use. |
| `backend/database.py: get_db`, `get_db_session` | No remaining callers after removing the importer. Current code creates SessionLocal directly. | Either adopt these helpers during session-lifecycle cleanup or remove them; do not rewrite active sessions in a documentation change. |
| `backend/schemas.py: COLUMN_MAPPINGS`, `NULL_VALUES`, `LA_CODE_TO_NAME` | No remaining Python consumers found after importer removal. Other constants in this module are active. | Remove individual constants after checking external data utilities; retain the module. |
| `backend/config.py: data_dir`, `max_page_size`, `rate_limit_burst` | No active consumers found. `default_page_size` appears only in a branch that expects None, although the route supplies a concrete default. | Reconcile settings with route validation in a focused API change. |
## Legacy/manual paths requiring operational verification
| Path | Status and reason to retain for now |
|---|---|
| FastAPI `/`, `/compare`, `/rankings`, `/favicon.svg`, `/robots.txt`, and conditional `/static` | Old frontend-serving routes reference a `frontend/` directory absent from the checkout and backend image. Next.js owns these public surfaces. Removal changes externally callable routes, so first check proxy/operator usage and define replacement responses. |
| `scripts/fetch_real_data.py`, `scripts/download_data.py` | Historical standalone CSV utilities. The fetch script targets Wandsworth/Merton; neither is wired into the managed pipeline. Marked historical, retained pending confirmation of manual use. |
| `pipeline/scripts/geocode_postcodes.py` | Mart-aware postcode fallback, not called by the current DAGs. Do not confuse it with the removed legacy ORM geocoder. Verify the target schema before manual use. |
| `docker-compose.yml` | Uses unpublished `:latest` release tags and lacks frontend Payload DB/secret/media configuration. Retained as an old development topology, not recommended onboarding. |
| `nextjs-app/docker-compose.yml` | Standalone legacy recipe with old backend port assumptions and no CMS persistence setup. Retained until its consumers are checked. |
| `MIGRATION_SUMMARY.md`, `docs/superpowers/`, `mockups/` | Historical designs and prototypes. Retain as history; do not follow as current deployment instructions. |
| `scripts/sql/drop_fact_parent_view.sql` | One-off maintenance SQL. Not an application entry point; repository call-site searches cannot establish whether it is still needed operationally. |
## Active code that can look obsolete
- `backend/data_loader.py` older-mart query fallbacks are covered by backend tests
and support databases at different migration stages. Remove only after verifying
the deployed schemas in every supported environment.
- `backend/gias_codes.py` and `pipeline/scripts/gias_codes.py` are intentionally
generated copies for separate runtime images. Their parity is tested.
- `nextjs-app/migrations/`, `payload-types.ts` and the Payload import map are active
CMS artifacts, not remnants of the removed school importer.
- `get_available_years`, `get_available_local_authorities` and `get_schools_count`
in `data_loader.py` are called through `get_data_info`, which serves the backend
data-info endpoint. They are not dead functions.
- `get_supplementary_data` is an intentional single-school wrapper around the
batch implementation.
- `pipeline/transform` models named `legacy` can be active data sources: annual
DAG selectors explicitly include legacy KS2/KS4 lineage. Names alone do not
establish obsolescence.
## Suggested next passes
1. Decide the fate of the three unused UI components and remove paired assets/tests.
2. Consolidate backend session usage and remove abandoned settings/constants.
3. Verify external consumers, then retire static-serving API routes and old compose recipes.
4. Audit manual data utilities with pipeline operators before deleting them.
5. Revisit compatibility fallbacks only after documenting supported schema versions.
Validation for this cleanup should include frontend typechecking/tests, Python
syntax checks, reference searches and documentation link checks. Live database,
external scheduler and deployed route usage require separate integration evidence.
+1 -113
View File
@@ -1935,63 +1935,6 @@ async function firstPlaceOfKind(page: Page, kind: string) {
return hit as { kind: string; slug: string; name: string; count: number }; return hit as { kind: string; slug: string; name: string; count: number };
} }
/**
* The round trip. Place pages always linked down to school pages; school
* pages linked nowhere on the site, so the ~27k of them that carry most of
* the inbound authority stranded it — their only anchor pointed at the
* school's own website.
*
* Asserting both directions is the point. A one-way link is what already
* existed and is not what this journey is for.
*/
test('a school page links back into the location layer, and the place page links down', async ({ page }) => {
const town = await firstPlaceOfKind(page, 'town');
// Start from the place page and take its first school, so the pair is
// guaranteed to be genuinely related rather than a hardcoded guess.
await page.goto(`/schools/${town.slug}`);
const schoolHref = await page.locator('a[href^="/school/"]').first()
.getAttribute('href');
expect(schoolHref, 'the town page listed no school to follow').toBeTruthy();
await page.goto(schoolHref!);
// Down: the school page must offer a link back to the town it sits in.
const backToTown = page.locator(`a[href="/schools/${town.slug}"]`);
await expect(backToTown).toHaveCount(1);
await expect(backToTown).toBeVisible();
// The anchor says what it leads to, which is worth more than "see more".
await expect(backToTown).toContainText(town.name, { ignoreCase: true });
await expect(backToTown).toContainText(/\d+ schools?/);
// And the breadcrumb resolves the school into a real hierarchy.
const blocks = await page.locator('script[type="application/ld+json"]')
.allTextContents();
const graph = blocks.join(' ');
expect(graph).toContain('"BreadcrumbList"');
// The narrower type, not the EducationalOrganization parent it used to be.
expect(graph).toContain('"School"');
/*
* The phase variants are the pages this most needs to reach: ~950 of them
* were once reachable by nothing at all, absent from every sitemap and
* unlinked from the place page. Conditional because not every school sits
* in a town that publishes one.
*/
const phaseLink = page.locator(`a[href^="/schools/${town.slug}/"]`).first();
if (await phaseLink.count()) {
const phaseHref = await phaseLink.getAttribute('href');
expect((await page.request.get(phaseHref!)).status()).toBe(200);
await expect(phaseLink).toContainText(/primary|secondary/);
}
// Following it lands on a real page, not a 404.
await backToTown.click();
await page.waitForURL(new RegExp(`/schools/${town.slug}$`));
await expect(page.locator('h1')).toContainText(town.name, { ignoreCase: true });
});
for (const [kind, prefix, article] of [ for (const [kind, prefix, article] of [
['town', '/schools/', 'a'], ['town', '/schools/', 'a'],
['authority', '/schools/authority/', 'an'], ['authority', '/schools/authority/', 'an'],
@@ -2617,56 +2560,8 @@ test('the destinations section never claims a pupil stayed at this school', asyn
* These journeys assert the load-bearing parts of that — a name, a face, the * These journeys assert the load-bearing parts of that — a name, a face, the
* honesty claim, and a resolvable Person entity — rather than exact copy, * honesty claim, and a resolvable Person entity — rather than exact copy,
* which will be edited. * which will be edited.
*
* Both are behind flags (about_page, blog), so each has a lit journey and a
* dark one. Flag state is read from the observable effect rather than from
* /api/flags, which the public proxy denies on purpose — the same approach
* distanceFeatureIsOn() takes above.
*/ */
async function aboutPageIsOn(page: Page): Promise<boolean> {
return (await page.request.get('/about')).ok();
}
async function blogIsOn(page: Page): Promise<boolean> {
return (await page.request.get('/blog')).ok();
}
test('with the about page off, it is absent rather than empty', async ({ page }) => {
test.skip(await aboutPageIsOn(page), 'the about_page flag is on in this environment');
// Dark means the URL does not exist, not that it renders empty: a 404 is
// what stops a crawler keeping the page in its index.
expect((await page.request.get('/about')).status()).toBe(404);
// A footer link into a 404 is the failure this flag has to avoid.
await page.goto('/');
await expect(page.locator('footer a[href="/about"]')).toHaveCount(0);
// And a sitemap must never advertise a URL that 404s.
const sitemap = await page.request.get('/content-sitemap.xml');
expect(await sitemap.text()).not.toContain('/about');
});
test('with the blog off, it is absent rather than empty', async ({ page }) => {
test.skip(await blogIsOn(page), 'the blog flag is on in this environment');
expect((await page.request.get('/blog')).status()).toBe(404);
expect((await page.request.get('/blog/rss.xml')).status()).toBe(404);
await page.goto('/');
await expect(page.locator('footer a[href="/blog"]')).toHaveCount(0);
const sitemap = await page.request.get('/content-sitemap.xml');
expect(await sitemap.text()).not.toContain('/blog');
// The admin panel is deliberately NOT flagged: posts have to be writable
// before the blog is readable, or there is nothing to turn on.
expect((await page.request.get('/admin')).status()).not.toBe(404);
});
test('the about page names a human author and is reachable from the footer', async ({ page }) => { test('the about page names a human author and is reachable from the footer', async ({ page }) => {
test.skip(!(await aboutPageIsOn(page)), 'the about_page flag is off in this environment');
await page.goto('/'); await page.goto('/');
const aboutLink = page.locator('footer a[href="/about"]'); const aboutLink = page.locator('footer a[href="/about"]');
await expect(aboutLink).toBeVisible(); await expect(aboutLink).toBeVisible();
@@ -2690,8 +2585,6 @@ test('the about page names a human author and is reachable from the footer', asy
}); });
test('the blog lists posts and each one renders with a byline', async ({ page }) => { test('the blog lists posts and each one renders with a byline', async ({ page }) => {
test.skip(!(await blogIsOn(page)), 'the blog flag is off in this environment');
await page.goto('/blog'); await page.goto('/blog');
await expect(page.getByRole('heading', { level: 1 })).toBeVisible(); await expect(page.getByRole('heading', { level: 1 })).toBeVisible();
@@ -2719,13 +2612,8 @@ test('the admin panel is not indexable', async ({ page }) => {
test('the content sitemap lists the about page and is advertised in robots', async ({ page }) => { test('the content sitemap lists the about page and is advertised in robots', async ({ page }) => {
const sitemap = await page.request.get('/content-sitemap.xml'); const sitemap = await page.request.get('/content-sitemap.xml');
// Served whatever the flags say: robots.txt names it unconditionally, and
// with both dark it is a valid empty urlset rather than a 404.
expect(sitemap.ok()).toBeTruthy(); expect(sitemap.ok()).toBeTruthy();
expect(await sitemap.text()).toContain('/about');
if (await aboutPageIsOn(page)) {
expect(await sitemap.text()).toContain('/about');
}
// The school corpus sitemap is proxied from FastAPI; this one is Next's. // The school corpus sitemap is proxied from FastAPI; this one is Next's.
// robots.txt must advertise both or the blog never gets discovered. // robots.txt must advertise both or the blog never gets discovered.
-41
View File
@@ -1,41 +0,0 @@
import { test, expect, Route } from '@playwright/test';
test('the deployed frontend and backend report the tested build', async ({ request }) => {
const response = await request.get('/release.json');
expect(response.ok()).toBeTruthy();
expect(response.headers()['cache-control']).toContain('no-store');
const identity = await response.json();
expect(identity.frontend).toEqual(identity.backend);
expect(identity.frontend.sha).toMatch(/^[a-f0-9]{40}$/);
expect(identity.frontend.build_id).toMatch(/^[a-f0-9]{32}$/);
if (process.env.EXPECTED_SHA) expect(identity.frontend.sha).toBe(process.env.EXPECTED_SHA);
if (process.env.EXPECTED_BUILD_ID) expect(identity.frontend.build_id).toBe(process.env.EXPECTED_BUILD_ID);
});
test('changing search while loading another page does not append old results', async ({ page }) => {
await page.goto('/?phase=primary');
await expect(page.getByRole('button', { name: 'Load more schools' })).toBeVisible();
let received!: (route: Route) => void;
const pending = new Promise<Route>(resolve => { received = resolve; });
await page.route('**/api/schools?**', async route => {
if (new URL(route.request().url()).searchParams.get('page') === '2') {
received(route);
return;
}
await route.continue();
});
await page.getByRole('button', { name: 'Load more schools' }).click();
const oldRequest = await pending;
const search = page.getByPlaceholder('School name or postcode').first();
await search.fill('secondary');
await search.press('Enter');
await page.waitForURL(/search=secondary/);
// A cancelled fetch may prevent route fulfilment altogether; either way,
// this deliberately late response must not become part of the new results.
await oldRequest.fulfill({ json: {
schools: [{ urn: 999998, school_name: 'P1 stale result sentinel', phase: 'Primary' }],
total: 2, page: 2, page_size: 1, total_pages: 2,
} }).catch(() => {});
await expect(page.getByText('P1 stale result sentinel')).toHaveCount(0);
await expect(page.getByRole('button', { name: 'Loading...' })).toHaveCount(0);
});
+4 -12
View File
@@ -1,16 +1,8 @@
# Browser requests use the same-origin Next.js proxy. # API Configuration
NEXT_PUBLIC_API_URL=/api NEXT_PUBLIC_API_URL=http://localhost:8000/api
# Absolute URL for server-side fetching and the proxy; include /api. # Production API URL (for deployment)
# In the managed container network this is http://backend:80/api (staging differs). # NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api
FASTAPI_URL=http://localhost:8000/api
# Payload CMS runtime configuration. Use the managed environment's database;
# Payload owns the payload schema, independently of the school marts.
DATABASE_URL=postgresql://schoolcompare:CHANGE_THIS_PASSWORD@localhost:5432/schoolcompare
# Generate a secret: python -c "import secrets; print(secrets.token_urlsafe(32))"
# Use distinct secrets for staging and production.
PAYLOAD_SECRET=CHANGE_THIS_TO_A_SECURE_RANDOM_SECRET
# Node Environment # Node Environment
NODE_ENV=development NODE_ENV=development
+288 -12
View File
@@ -1,15 +1,291 @@
# Frontend deployment # Deployment Guide
Next.js and Payload run in the same frontend container. The maintained deployment This guide covers deployment options for the SchoolCompare Next.js application.
procedure is [docs/DEPLOY.md](../docs/DEPLOY.md), with the production and staging
Portainer compose files at the repository root.
The frontend Dockerfile builds a standalone Next.js image. Runtime configuration ## Deployment Options
supplies `FASTAPI_URL`, `DATABASE_URL` and `PAYLOAD_SECRET`; uploaded CMS media is
persisted in a volume. Promote the built image through the repository's Gitea
workflow after human staging approval.
Earlier Vercel and standalone deployment recipes have been retired from this file ### Option 1: Vercel (Recommended for Next.js)
because they do not describe the current CMS, persistence and promotion setup.
See [development](../docs/DEVELOPMENT.md) for checks and Vercel is the easiest and most optimized platform for Next.js applications.
[publishing](docs/PUBLISHING.md) for CMS operations.
#### Steps:
1. **Install Vercel CLI**:
```bash
npm install -g vercel
```
2. **Login to Vercel**:
```bash
vercel login
```
3. **Deploy**:
```bash
vercel --prod
```
4. **Configure Environment Variables** in Vercel dashboard:
- `NEXT_PUBLIC_API_URL`: Your FastAPI endpoint (e.g., `https://api.schoolcompare.co.uk/api`)
- `FASTAPI_URL`: Same as above for server-side requests
#### Benefits:
- Automatic HTTPS
- Global CDN
- Zero-config deployment
- Automatic preview deployments
- Built-in analytics
---
### Option 2: Docker (Self-hosted)
Deploy using Docker containers for full control.
#### Prerequisites:
- Docker 20+
- Docker Compose 2+
#### Steps:
1. **Build Docker Image**:
```bash
docker build -t schoolcompare-nextjs:latest .
```
2. **Run with Docker Compose**:
```bash
# Create .env file with production variables
echo "NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api" > .env
echo "FASTAPI_URL=http://backend:8000/api" >> .env
# Start services
docker-compose up -d
```
3. **Verify Deployment**:
```bash
curl http://localhost:3000
```
#### Environment Variables:
- `NEXT_PUBLIC_API_URL`: Public API endpoint (client-side)
- `FASTAPI_URL`: Internal API endpoint (server-side)
- `NODE_ENV`: `production`
---
### Option 3: PM2 (Node.js Process Manager)
Deploy directly on a Node.js server using PM2.
#### Prerequisites:
- Node.js 24+
- PM2 (`npm install -g pm2`)
#### Steps:
1. **Build Application**:
```bash
npm run build
```
2. **Create PM2 Ecosystem File** (`ecosystem.config.js`):
```javascript
module.exports = {
apps: [{
name: 'schoolcompare-nextjs',
script: 'npm',
args: 'start',
cwd: '/path/to/nextjs-app',
instances: 'max',
exec_mode: 'cluster',
env: {
NODE_ENV: 'production',
PORT: 3000,
NEXT_PUBLIC_API_URL: 'https://api.schoolcompare.co.uk/api',
FASTAPI_URL: 'http://localhost:8000/api',
},
}],
};
```
3. **Start with PM2**:
```bash
pm2 start ecosystem.config.js
pm2 save
pm2 startup
```
---
### Option 4: Nginx Reverse Proxy
Use Nginx as a reverse proxy in front of Next.js.
#### Nginx Configuration:
```nginx
server {
listen 80;
server_name schoolcompare.co.uk;
# Redirect to HTTPS
return 301 https://$server_name$request_uri;
}
server {
listen 443 ssl http2;
server_name schoolcompare.co.uk;
# SSL Configuration
ssl_certificate /etc/ssl/certs/schoolcompare.crt;
ssl_certificate_key /etc/ssl/private/schoolcompare.key;
# Security Headers
# frame-ancestors replaces X-Frame-Options so the analytics subdomain
# (Umami heatmap/recorder) can embed the site in an iframe.
add_header Content-Security-Policy "frame-ancestors 'self' https://analytics.schoolcompare.co.uk" always;
add_header X-Content-Type-Options "nosniff" always;
add_header X-XSS-Protection "1; mode=block" always;
# Proxy to Next.js
location / {
proxy_pass http://localhost:3000;
proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection 'upgrade';
proxy_set_header Host $host;
proxy_cache_bypass $http_upgrade;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
}
# Proxy to FastAPI
location /api/ {
proxy_pass http://localhost:8000;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
}
# Cache static files
location /_next/static/ {
proxy_pass http://localhost:3000;
add_header Cache-Control "public, max-age=31536000, immutable";
}
}
```
---
## Pre-Deployment Checklist
- [ ] Run `npm run build` successfully
- [ ] Run `npm test` - all tests pass
- [ ] Environment variables configured
- [ ] FastAPI backend accessible
- [ ] Database migrations applied
- [ ] SSL certificates configured (production)
- [ ] Domain DNS configured
- [ ] Monitoring/logging set up
- [ ] Backup strategy in place
---
## Post-Deployment Verification
1. **Health Check**:
```bash
curl https://schoolcompare.co.uk
```
2. **Test Routes**:
- Home: `https://schoolcompare.co.uk/`
- School Page: `https://schoolcompare.co.uk/school/100001`
- Compare: `https://schoolcompare.co.uk/compare`
- Rankings: `https://schoolcompare.co.uk/rankings`
3. **Check SEO**:
- Sitemap: `https://schoolcompare.co.uk/sitemap.xml`
- Robots: `https://schoolcompare.co.uk/robots.txt`
4. **Performance Audit**:
- Run Lighthouse in Chrome DevTools
- Target scores: 90+ for Performance, Accessibility, Best Practices, SEO
---
## Monitoring
### Recommended Tools:
- **Vercel Analytics** (if using Vercel)
- **Sentry** for error tracking
- **Google Analytics** for user analytics
- **Uptime Robot** for uptime monitoring
### Health Check Endpoint:
The application automatically serves health data at the root route.
---
## Rollback Procedure
### Vercel:
```bash
vercel rollback
```
### Docker:
```bash
docker-compose down
docker-compose up -d --force-recreate
```
### PM2:
```bash
pm2 stop schoolcompare-nextjs
# Restore previous build
pm2 start schoolcompare-nextjs
```
---
## Troubleshooting
### Issue: API requests failing
- **Solution**: Check `NEXT_PUBLIC_API_URL` and `FASTAPI_URL` environment variables
- **Verify**: FastAPI backend is accessible from Next.js container/server
### Issue: Build fails
- **Solution**: Check Node.js version (requires 24+)
- **Clear cache**: `rm -rf .next node_modules && npm install && npm run build`
### Issue: Slow page loads
- **Solution**: Enable caching in API calls
- **Check**: Network latency to FastAPI backend
- **Verify**: CDN is serving static assets
---
## Security Considerations
- ✅ HTTPS enabled
- ✅ Security headers configured (X-Frame-Options, CSP, etc.)
- ✅ API keys in environment variables (never in code)
- ✅ CORS properly configured
- ✅ Rate limiting on API endpoints
- ✅ Regular security updates
- ✅ Dependency vulnerability scanning
---
## Support
For deployment issues, contact the DevOps team or refer to:
- [Next.js Deployment Docs](https://nextjs.org/docs/deployment)
- [Vercel Documentation](https://vercel.com/docs)
- [Docker Documentation](https://docs.docker.com/)
-10
View File
@@ -28,10 +28,6 @@ ENV NODE_ENV=production
ARG FASTAPI_URL=http://backend:80/api ARG FASTAPI_URL=http://backend:80/api
ENV FASTAPI_URL=${FASTAPI_URL} ENV FASTAPI_URL=${FASTAPI_URL}
ARG BUILD_SHA=development
ARG BUILD_ID=development
RUN node -e 'require("fs").writeFileSync("build-info.json", JSON.stringify({sha:process.argv[1],build_id:process.argv[2]}))' "$BUILD_SHA" "$BUILD_ID"
# Build application # Build application
RUN npm run build RUN npm run build
@@ -74,12 +70,6 @@ USER nextjs
EXPOSE 3000 EXPOSE 3000
# Set environment variables # Set environment variables
ARG BUILD_SHA=development
ARG BUILD_ID=development
LABEL io.schoolcompare.build-id=$BUILD_ID
LABEL io.schoolcompare.commit=$BUILD_SHA
COPY --from=builder /app/build-info.json ./build-info.json
ENV PORT=3000 ENV PORT=3000
ENV HOSTNAME="0.0.0.0" ENV HOSTNAME="0.0.0.0"
+142 -44
View File
@@ -1,58 +1,156 @@
# SchoolCompare frontend and CMS # SchoolCompare Next.js Application
Next.js App Router with React, TypeScript, CSS Modules, Chart.js, Leaflet and Modern Next.js application for comparing primary school KS2 performance across England.
Payload CMS. It serves school search, comparisons, rankings, school/place detail
pages and editorial content across England.
Start with the [repository overview](../README.md), ## Features
[architecture](../docs/ARCHITECTURE.md) and [development checks](../docs/DEVELOPMENT.md).
## Source map - **Server-Side Rendering (SSR)**: Fast initial page loads with pre-rendered content
- **Individual School Pages**: Dedicated pages for each school with full SEO optimization
- **Side-by-Side Comparison**: Compare up to 5 schools simultaneously
- **School Rankings**: Top-performing schools by various metrics
- **Interactive Maps**: Leaflet integration for geographic visualization
- **Performance Charts**: Chart.js visualizations for historical data
- **Responsive Design**: Mobile-first approach with full responsive support
- **SEO Optimized**: Dynamic sitemaps, meta tags, and structured data
| Path | Purpose | ## Tech Stack
|---|---|
| `app/(frontend)/` | Public root layout, server pages and FastAPI proxy |
| `app/(payload)/` | Payload root layout, `/admin` and `/cms-api` |
| `app/robots.ts`, `app/opengraph-image.tsx`, root icons | Site-wide metadata endpoints |
| `components/` | Client views and reusable display components |
| `components/school/` | School detail sections |
| `lib/api.ts`, `lib/types.ts` | Fetch wrappers and manual school API types |
| `lib/schoolSections.ts`, `lib/compareLogic.ts` | Presentation decisions and data preparation |
| `context/`, `hooks/` | Comparison state, suggestion state and responsive behaviour |
| `collections/`, `blocks/`, `migrations/` | CMS schema and production migrations |
| `__tests__/` | Jest and React Testing Library tests |
Do not introduce a shared `app/layout.tsx`: public pages and Payload have separate - **Framework**: Next.js 16 (App Router)
root layouts. Keep root metadata files outside the route groups. - **Language**: TypeScript 5
- **Styling**: CSS Modules + CSS Variables
- **State Management**: React Context API + URL state
- **Data Fetching**: SWR (client-side) + Next.js fetch (server-side)
- **Charts**: Chart.js + react-chartjs-2
- **Maps**: Leaflet + react-leaflet
- **Testing**: Jest + React Testing Library
- **Validation**: Zod
## Data and state ## Getting Started
Server pages fetch initial data directly from `FASTAPI_URL`. Browser fetches use ### Prerequisites
`/api` by default, forwarded by `app/(frontend)/api/[...path]/route.ts`.
`FASTAPI_URL` must include `/api`. See `.env.example` for CMS and API settings.
State uses React hooks/context, URL search parameters and localStorage for the - Node.js 24+ (using nvm recommended)
comparison basket. SWR is not installed. Maps use dynamic Leaflet wrappers. - FastAPI backend running on port 8000
Revalidation intervals are configured in fetch wrappers and pages; they vary by
resource. Backend reloads do not automatically invalidate every Next.js cache.
## Commands ### Installation
```sh ```bash
npm ci # Install dependencies
npm run typecheck npm install
npm test -- --runInBand
npm run build # Copy environment variables
cp .env.example .env.local
# Update .env.local with your configuration
``` ```
`test:watch` and `test:coverage` are also available. There is no `lint` script. ### Development
A running application needs the backend/data environment described in the
[development guide](../docs/DEVELOPMENT.md).
After CMS field or editor changes, run `npm run generate:importmap`. Keep ```bash
`payload-types.ts` generated from the CMS schema rather than editing it by hand. # Start development server
The build must work without a database connection; avoid module-scope CMS queries npm run dev
and DB-backed `generateStaticParams` functions.
See [publishing](docs/PUBLISHING.md) for CMS operations and # Open http://localhost:3000
[deployment](../docs/DEPLOY.md) for staging and production promotion. ```
### Building
```bash
# Build for production
npm run build
# Start production server
npm start
```
### Testing
```bash
# Run tests
npm test
# Run tests in watch mode
npm run test:watch
# Run tests with coverage
npm run test:coverage
```
### Linting
```bash
# Run ESLint
npm run lint
```
## Project Structure
```
nextjs-app/
├── app/ # App Router pages
│ ├── layout.tsx # Root layout
│ ├── page.tsx # Home page
│ ├── compare/ # Compare page
│ ├── rankings/ # Rankings page
│ ├── school/[urn]/ # Individual school pages
│ ├── sitemap.ts # Dynamic sitemap
│ └── robots.ts # Robots.txt
├── components/ # React components
│ ├── SchoolCard.tsx # School card component
│ ├── FilterBar.tsx # Search/filter controls
│ ├── ComparisonView.tsx # Comparison interface
│ ├── RankingsView.tsx # Rankings table
│ └── ...
├── lib/ # Utility libraries
│ ├── api.ts # API client
│ ├── types.ts # TypeScript types
│ └── utils.ts # Helper functions
├── hooks/ # Custom React hooks
├── context/ # React Context providers
├── styles/ # Global styles
├── public/ # Static assets
└── __tests__/ # Test files
```
## Environment Variables
| Variable | Description | Default |
|----------|-------------|---------|
| `NEXT_PUBLIC_API_URL` | Public API endpoint (client-side) | `http://localhost:8000/api` |
| `FASTAPI_URL` | Server-side API endpoint | `http://localhost:8000/api` |
| `NODE_ENV` | Environment mode | `development` |
## Performance Optimizations
- **Server-Side Rendering**: Initial HTML rendered on server
- **Static Generation**: Where possible, pages are pre-generated
- **Image Optimization**: Next.js Image component with AVIF/WebP support
- **Code Splitting**: Automatic route-based code splitting
- **Dynamic Imports**: Heavy components loaded on demand
- **API Caching**: Configurable revalidation for data fetching
- **Bundle Optimization**: Tree shaking and minification
- **Compression**: Gzip compression enabled
## SEO Features
- **Dynamic Meta Tags**: Generated per page with Next.js Metadata API
- **Open Graph**: Social media optimization
- **JSON-LD**: Structured data for search engines
- **Sitemap**: Auto-generated from database
- **Robots.txt**: Search engine crawling rules
- **Canonical URLs**: Duplicate content prevention
## Browser Support
- Chrome (latest)
- Firefox (latest)
- Safari (latest)
- Edge (latest)
## License
Proprietary - SchoolCompare
## Support
For issues and questions, please contact the development team.
+2 -15
View File
@@ -27,27 +27,14 @@ describe('BlogPosting structured data', () => {
it('names the same Person entity the about page declares', () => { it('names the same Person entity the about page declares', () => {
// By @id, not by repeating the person: search engines must resolve every // By @id, not by repeating the person: search engines must resolve every
// post and the about page to one author entity, or the site has several. // post and the about page to one author entity, or the site has several.
const ld = blogPostingJsonLd(post, { namedAuthor: true }); const ld = blogPostingJsonLd(post);
expect(ld['@type']).toBe('BlogPosting'); expect(ld['@type']).toBe('BlogPosting');
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk/about#tudor'); expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk/about#tudor');
expect(ld.publisher['@id']).toBe('https://www.schoolcompare.co.uk#organization'); expect(ld.publisher['@id']).toBe('https://www.schoolcompare.co.uk#organization');
}); });
it('attributes to the organization when the about page is dark', () => {
/*
* The two flags are independent, so blog-on-about-off is a reachable
* state. The Person entity lives at /about#tudor and that URL 404s while
* the flag is dark, so claiming it would declare an author that resolves
* to nothing — worse for the blog's credibility than having no named
* author at all. Attribute to the publisher instead.
*/
const ld = blogPostingJsonLd(post, { namedAuthor: false });
expect(ld.author['@id']).toBe('https://www.schoolcompare.co.uk#organization');
expect(JSON.stringify(ld)).not.toContain('/about');
});
it('carries a self-referencing canonical url and the publish date', () => { it('carries a self-referencing canonical url and the publish date', () => {
const ld = blogPostingJsonLd(post, { namedAuthor: true }); const ld = blogPostingJsonLd(post);
expect(ld.url).toBe( expect(ld.url).toBe(
'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you', 'https://www.schoolcompare.co.uk/blog/what-the-data-cannot-tell-you',
); );
@@ -1,51 +0,0 @@
import { fireEvent, render, screen } from '@testing-library/react';
import HomePage from '@/app/(frontend)/page';
import SchoolPage from '@/app/(frontend)/school/[slug]/page';
import ErrorPage from '@/app/(frontend)/error';
import { APIFetchError, fetchSchools, fetchFilters, fetchSchoolDetails } from '@/lib/api';
import { fetchPlace, fetchPlaces } from '@/lib/places';
jest.mock('@/lib/api', () => ({
...jest.requireActual('@/lib/api'),
fetchSchools: jest.fn(),
fetchSchoolDetails: jest.fn(),
fetchFilters: jest.fn(async () => ({})),
fetchDataInfo: jest.fn(async () => null),
fetchNationalAverages: jest.fn(async () => null),
}));
jest.mock('@/lib/flags', () => ({ getFlags: jest.fn(async () => ({})) }));
jest.mock('next/navigation', () => ({
notFound: () => { throw new Error('NEXT_NOT_FOUND'); },
redirect: jest.fn(),
}));
const realFetch = global.fetch;
afterEach(() => { global.fetch = realFetch; jest.clearAllMocks(); });
test('school outages propagate; only a real 404 becomes not found', async () => {
const request = { params: Promise.resolve({ slug: '100001-school' }) };
const outage = new APIFetchError('unavailable', 503);
jest.mocked(fetchSchoolDetails).mockRejectedValueOnce(outage);
await expect(SchoolPage(request)).rejects.toBe(outage);
jest.mocked(fetchSchoolDetails).mockRejectedValueOnce(new APIFetchError('missing', 404));
await expect(SchoolPage(request)).rejects.toThrow('NEXT_NOT_FOUND');
});
test('homepage search failure is not returned as an empty successful page', async () => {
jest.mocked(fetchSchools).mockRejectedValueOnce(new APIFetchError('unavailable', 503));
await expect(HomePage({ searchParams: Promise.resolve({ search: 'school' }) })).rejects.toThrow('unavailable');
});
test('a place is absent only on 404; other failures propagate', async () => {
global.fetch = jest.fn().mockResolvedValue({ ok: false, status: 404 });
await expect(fetchPlace('town', 'example')).resolves.toBeNull();
jest.mocked(global.fetch).mockResolvedValue({ ok: false, status: 503 } as Response);
await expect(fetchPlace('town', 'example')).rejects.toMatchObject({ status: 503 });
await expect(fetchPlaces()).rejects.toMatchObject({ status: 503 });
});
test('the error boundary offers a retry without showing an empty search', () => {
const reset = jest.fn();
render(<ErrorPage error={new Error('offline')} reset={reset} />);
fireEvent.click(screen.getByRole('button', { name: 'Try again' }));
expect(reset).toHaveBeenCalledTimes(1);
});
-39
View File
@@ -2,7 +2,6 @@ import { metadata as homeMetadata } from '@/app/(frontend)/page';
import { metadata as rankingsMetadata } from '@/app/(frontend)/rankings/page'; import { metadata as rankingsMetadata } from '@/app/(frontend)/rankings/page';
import { metadata as admissionsMetadata } from '@/app/(frontend)/admissions/page'; import { metadata as admissionsMetadata } from '@/app/(frontend)/admissions/page';
import { generateMetadata as compareMetadata } from '@/app/(frontend)/compare/page'; import { generateMetadata as compareMetadata } from '@/app/(frontend)/compare/page';
import { metadata as rootMetadata } from '@/app/(frontend)/layout';
describe('canonical URLs', () => { describe('canonical URLs', () => {
it('the homepage canonicalises to the bare root', () => { it('the homepage canonicalises to the bare root', () => {
@@ -129,41 +128,3 @@ describe('C1 snippet copy', () => {
} }
}); });
}); });
/**
* The share card must be declared, not inherited.
*
* `app/opengraph-image.tsx` is a metadata file convention, and it does attach
* to routes in the app root segment — `_not-found` gets an og:image from it.
* It does NOT attach to the site's pages, which live in the `(frontend)`
* route group whose own layout is a root layout. Staging served og:title,
* og:description, og:url, og:site_name and og:type and no og:image at all,
* so every link pasted into a chat rendered bare.
*
* The file stays at the app root, because /robots.txt and /icon.png depend on
* it being there. The site's root layout points at the route it generates.
*/
describe('the share card', () => {
it('declares an opengraph image on the site root layout', () => {
// No og:image means every link pasted into a chat renders bare.
const images = rootMetadata.openGraph?.images;
expect(images).toBeTruthy();
expect(JSON.stringify(images)).toContain('/opengraph-image');
});
it('declares a twitter image too', () => {
// twitter.card is summary_large_image. Claiming a large-image card and
// supplying no image is worse than claiming a summary card.
// Metadata['twitter'] is a union and `card` is not on every member, so
// this reads the serialised shape rather than narrowing the type.
const twitter = JSON.stringify(rootMetadata.twitter);
expect(twitter).toContain('summary_large_image');
expect(twitter).toContain('/opengraph-image');
});
it('resolves the card to an absolute url via metadataBase', () => {
// The e2e journey does `new URL(ogUrl)`, which throws on a relative path.
expect(rootMetadata.metadataBase?.toString())
.toBe('https://www.schoolcompare.co.uk/');
});
});
@@ -1,30 +0,0 @@
/** @jest-environment node */
import { GET } from '@/app/(frontend)/release.json/route';
import { readFile } from 'node:fs/promises';
jest.mock('node:fs/promises', () => ({ readFile: jest.fn() }));
const realFetch = global.fetch;
const identity = { sha: 'a'.repeat(40), build_id: 'b'.repeat(32) };
beforeEach(() => {
jest.mocked(readFile).mockResolvedValue(JSON.stringify(identity));
global.fetch = jest.fn(async () => Response.json(identity));
});
afterEach(() => { global.fetch = realFetch; jest.resetAllMocks(); });
test('reports immutable file identity and backend identity without caching', async () => {
const response = await GET();
expect(response.status).toBe(200);
expect(response.headers.get('Cache-Control')).toBe('no-store');
expect(await response.json()).toEqual({ frontend: identity, backend: identity });
expect(fetch).toHaveBeenCalledWith(expect.stringMatching(/\/api\/release$/), expect.objectContaining({ cache: 'no-store', signal: expect.anything() }));
});
test('missing build metadata cannot pass the release gate', async () => {
jest.mocked(readFile).mockRejectedValueOnce(new Error('missing file'));
expect((await GET()).status).toBe(503);
});
test('backend failure cannot pass the release gate', async () => {
jest.mocked(fetch).mockResolvedValueOnce(new Response('', { status: 503 }));
expect((await GET()).status).toBe(503);
});
@@ -1,47 +0,0 @@
/**
* The footer is the only navigational route to /about and /blog, so it is
* where a dark flag would otherwise leave a link into a 404.
*
* Both props default to false. A caller that forgets to pass them hides the
* links, which is the direction that cannot break a page — the same reasoning
* as backend/flags.py's "every flag defaults to False".
*/
import { render, screen } from '@testing-library/react';
import { Footer } from '@/components/Footer';
describe('footer feature links', () => {
it('links to both when both flags are on', () => {
render(<Footer aboutEnabled blogEnabled />);
expect(screen.getByRole('link', { name: /who's behind this/i }))
.toHaveAttribute('href', '/about');
expect(screen.getByRole('link', { name: /^blog$/i }))
.toHaveAttribute('href', '/blog');
});
it('omits the about link when that flag is dark', () => {
render(<Footer blogEnabled />);
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
expect(screen.getByRole('link', { name: /^blog$/i })).toBeInTheDocument();
});
it('omits the blog link when that flag is dark', () => {
render(<Footer aboutEnabled />);
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
expect(screen.getByRole('link', { name: /who's behind this/i }))
.toBeInTheDocument();
});
it('drops the whole section when both are dark, not an empty heading', () => {
// Shipping dark means the footer renders as it did before the feature
// existed, not as a section with its contents removed.
render(<Footer />);
expect(screen.queryByRole('heading', { name: /^about$/i })).toBeNull();
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
expect(screen.queryByRole('link', { name: /^blog$/i })).toBeNull();
});
it('defaults to dark when a caller passes nothing', () => {
render(<Footer />);
expect(screen.queryByRole('link', { name: /who's behind this/i })).toBeNull();
});
});
@@ -1,81 +0,0 @@
import { act, fireEvent, render, screen } from '@testing-library/react';
import { HomeView } from '@/components/HomeView';
import { fetchSchools } from '@/lib/api';
import { primaryFixture } from '../support/schoolFixtures';
import type { SchoolsResponse, School } from '@/lib/types';
let params = new URLSearchParams('postcode=SW1A+1AA');
jest.mock('next/navigation', () => ({
useSearchParams: () => params,
usePathname: () => '/',
useRouter: () => ({ push: jest.fn(), replace: jest.fn() }),
}));
jest.mock('@/context/ComparisonContext', () => ({
useComparisonContext: () => ({ addSchool: jest.fn(), removeSchool: jest.fn(), selectedSchools: [] }),
}));
jest.mock('@/lib/api', () => ({
fetchSchools: jest.fn(),
fetchNationalAverages: jest.fn(async () => ({})),
fetchLAaverages: jest.fn(async () => ({ secondary: { attainment_8_by_la: {} } })),
}));
jest.mock('@/components/FilterBar', () => ({ FilterBar: () => null }));
jest.mock('@/components/SchoolRow', () => ({ SchoolRow: ({ school }: {school: School}) => <div>{school.school_name}</div> }));
jest.mock('@/components/SchoolMap', () => ({ SchoolMap: ({ schools }: {schools: School[]}) => <div data-testid="map">{schools.map(s => s.school_name).join(',')}</div> }));
const filters = { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
function response(name: string): SchoolsResponse {
return { schools: [{ ...primaryFixture.schoolInfo, school_name: name }],
total: 2, page: 1, page_size: 1, total_pages: 2 };
}
function deferred() {
let resolve!: (value: SchoolsResponse) => void;
let reject!: (error: Error) => void;
const promise = new Promise<SchoolsResponse>((yes, no) => { resolve = yes; reject = no; });
return { promise, resolve, reject };
}
beforeEach(() => {
params = new URLSearchParams('postcode=SW1A+1AA');
jest.mocked(fetchSchools).mockReset();
});
test('load-more results from an old search are discarded, even after returning to it', async () => {
const pending = deferred();
jest.mocked(fetchSchools).mockReturnValueOnce(pending.promise);
const view = render(<HomeView initialSchools={response('Initial A')} filters={filters} />);
fireEvent.click(screen.getByRole('button', { name: 'Load more schools' }));
const signal = jest.mocked(fetchSchools).mock.calls[0][1]?.signal;
params = new URLSearchParams('postcode=SW2+1AA');
view.rerender(<HomeView initialSchools={response('Initial B')} filters={filters} />);
expect(signal?.aborted).toBe(true);
params = new URLSearchParams('postcode=SW1A+1AA');
view.rerender(<HomeView initialSchools={response('Fresh A')} filters={filters} />);
await act(async () => pending.resolve(response('Stale append')));
expect(screen.queryByText('Stale append')).not.toBeInTheDocument();
expect(screen.getByRole('button', { name: 'Load more schools' })).toBeEnabled();
});
test('an older map response cannot overwrite the current search', async () => {
const first = deferred(), second = deferred();
jest.mocked(fetchSchools).mockReturnValueOnce(first.promise).mockReturnValueOnce(second.promise);
const initial = response('Initial A');
const view = render(<HomeView initialSchools={initial} filters={filters} />);
fireEvent.click(screen.getByRole('button', { name: 'Map' }));
params = new URLSearchParams('postcode=SW2+1AA');
view.rerender(<HomeView initialSchools={response('Initial B')} filters={filters} />);
await act(async () => second.resolve(response('Current map')));
await act(async () => first.resolve(response('Stale map')));
expect(screen.getByTestId('map')).toHaveTextContent('Current map');
expect(screen.getByTestId('map')).not.toHaveTextContent('Stale map');
});
test('failed map requests can be retried by reopening the map', async () => {
const pending = deferred();
jest.mocked(fetchSchools).mockReturnValueOnce(pending.promise).mockResolvedValue(response('Retry result'));
render(<HomeView initialSchools={response('Initial')} filters={filters} />);
fireEvent.click(screen.getByRole('button', { name: 'Map' }));
await act(async () => pending.reject(new Error('offline')));
fireEvent.click(screen.getByRole('button', { name: 'List' }));
await act(async () => fireEvent.click(screen.getByRole('button', { name: 'Map' })));
expect(fetchSchools).toHaveBeenCalledTimes(2);
expect(screen.getByTestId('map')).toHaveTextContent('Retry result');
});
@@ -1,92 +0,0 @@
/**
* The module that ends the stranding: before it, a school page's only anchor
* pointed at the school's own website, so ~27k pages sent authority off-site
* and none of it reached the location layer.
*/
import { render, screen } from '@testing-library/react';
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const cm15 = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('NearbyPlaces', () => {
it('links to every place the school belongs to', () => {
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
expect(screen.getByRole('link', { name: /Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood');
expect(screen.getByRole('link', { name: /Essex/ }))
.toHaveAttribute('href', '/schools/authority/essex');
expect(screen.getByRole('link', { name: /CM15/ }))
.toHaveAttribute('href', '/schools/near/cm15');
});
it('says how many schools each link leads to', () => {
// An anchor that states its destination's size is worth more to a reader
// and to a crawler than "see more".
render(<NearbyPlaces places={[brentwood]} />);
expect(screen.getByRole('link', { name: /37 schools in Brentwood/ }))
.toBeInTheDocument();
});
it('renders nothing at all when the school has no published places', () => {
// Not an empty heading. A school whose town and authority both fall below
// the threshold has nowhere to point, and the page should look as it did
// before the module existed.
const { container } = render(<NearbyPlaces places={[]} />);
expect(container).toBeEmptyDOMElement();
});
it('puts the narrowest place first, which is the most useful link', () => {
// The API orders widest-first for the breadcrumb; a reader on a school
// page wants its town before its county.
render(<NearbyPlaces places={[essex, brentwood, cm15]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood'))
.toBeLessThan(hrefs.indexOf('/schools/authority/essex'));
});
it('handles a singular count without saying "1 schools"', () => {
render(<NearbyPlaces places={[{ ...brentwood, count: 1 }]} />);
expect(screen.getByRole('link', { name: /1 school in Brentwood/ }))
.toBeInTheDocument();
});
it('links the phase page the school appears on', () => {
// "primary schools in brentwood" is the query these pages exist for.
render(<NearbyPlaces places={[{
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools in Brentwood/ }))
.toHaveAttribute('href', '/schools/brentwood/primary');
});
it('links both phase pages for an all-through school', () => {
render(<NearbyPlaces places={[{
...brentwood,
phases: [
{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' },
{ phase: 'secondary', count: 9, url: '/schools/brentwood/secondary' },
],
}]} />);
expect(screen.getByRole('link', { name: /22 primary schools/ })).toBeInTheDocument();
expect(screen.getByRole('link', { name: /9 secondary schools/ })).toBeInTheDocument();
});
it('keeps a phase link next to the place it belongs to', () => {
// Grouping matters: "22 primary schools in Brentwood" directly after
// "37 schools in Brentwood" reads as one place, not two unrelated links.
render(<NearbyPlaces places={[essex, {
...brentwood,
phases: [{ phase: 'primary', count: 22, url: '/schools/brentwood/primary' }],
}]} />);
const hrefs = screen.getAllByRole('link').map((a) => a.getAttribute('href'));
expect(hrefs.indexOf('/schools/brentwood/primary'))
.toBe(hrefs.indexOf('/schools/brentwood') + 1);
});
});
+1 -25
View File
@@ -1,4 +1,4 @@
import { getFlags, FLAGS_REVALIDATE } from '@/lib/flags'; import { getFlags } from '@/lib/flags';
// jsdom provides no global fetch, so there is nothing for jest.spyOn to attach // jsdom provides no global fetch, so there is nothing for jest.spyOn to attach
// to — assign it and restore the original afterwards. This is the first test // to — assign it and restore the original afterwards. This is the first test
@@ -31,28 +31,4 @@ describe('getFlags', () => {
mockFetch(async () => ({ ok: false, status: 503 })); mockFetch(async () => ({ ok: false, status: 503 }));
await expect(getFlags()).resolves.toEqual({}); await expect(getFlags()).resolves.toEqual({});
}); });
/*
* Reading a flag pins the calling route's ISR floor: Next uses the LOWEST
* revalidate among a route's fetches for the whole route. That is why the
* revalidate is an argument rather than the constant.
*
* Every SEO route here declares `revalidate = 604800`. A gate that read
* flags at the 300s default would drop the whole school and place corpus
* from a weekly cache to a 5-minute one, which is a large origin-load
* regression to pay for a feature flag.
*/
it('reads at the 300s floor by default', async () => {
mockFetch(async () => ({ ok: true, json: async () => ({}) }));
await getFlags();
expect((global.fetch as jest.Mock).mock.calls[0][1])
.toEqual({ next: { revalidate: FLAGS_REVALIDATE } });
});
it('lets a caller pass its own route floor instead', async () => {
mockFetch(async () => ({ ok: true, json: async () => ({}) }));
await getFlags(604800);
expect((global.fetch as jest.Mock).mock.calls[0][1])
.toEqual({ next: { revalidate: 604800 } });
});
}); });
@@ -1,68 +0,0 @@
/**
* School pages had no BreadcrumbList and no links into the location layer.
* Both are fixed by the same data — the `places` array the API now returns —
* so they are tested together.
*/
import { schoolBreadcrumbJsonLd } from '@/lib/jsonld';
const essex = { kind: 'authority', slug: 'essex', name: 'Essex', count: 480, url: '/schools/authority/essex', phases: [] };
const brentwood = { kind: 'town', slug: 'brentwood', name: 'Brentwood', count: 37, url: '/schools/brentwood', phases: [] };
const outcode = { kind: 'outcode', slug: 'cm15', name: 'CM15', count: 12, url: '/schools/near/cm15', phases: [] };
describe('school breadcrumbs', () => {
it('reads home to authority to town to school', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
expect(ld['@type']).toBe('BreadcrumbList');
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Brentwood', 'Brentwood School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3, 4]);
});
it('skips a level the school has no published place for', () => {
// A school whose town falls below the publish threshold has no town page.
// The trail closes over the gap rather than linking to a 404.
const ld = schoolBreadcrumbJsonLd({
name: 'Lone School', url: '/school/1-lone-school', places: [essex],
});
expect(ld.itemListElement.map((i) => i.name))
.toEqual(['schoolcompare', 'Essex', 'Lone School']);
expect(ld.itemListElement.map((i) => i.position)).toEqual([1, 2, 3]);
});
it('omits outcodes, which are not a place a breadcrumb reads through', () => {
// CM15 is a useful link in the module but nonsense in a trail: nobody
// navigates Essex → CM15 → school.
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood, outcode],
});
expect(JSON.stringify(ld)).not.toContain('cm15');
});
it('still produces a valid trail when the school has no places at all', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Orphan School', url: '/school/2-orphan-school', places: [],
});
expect(ld.itemListElement.map((i) => i.name)).toEqual(['schoolcompare', 'Orphan School']);
});
it('uses absolute urls, as every other entity on the site does', () => {
const ld = schoolBreadcrumbJsonLd({
name: 'Brentwood School', url: '/school/100000-brentwood-school',
places: [essex, brentwood],
});
for (const item of ld.itemListElement) {
expect(item.item).toMatch(/^https:\/\/www\.schoolcompare\.co\.uk\//);
}
// The root is the homepage: there is no /schools index page to link to.
expect(ld.itemListElement[0].item).toBe('https://www.schoolcompare.co.uk/');
});
});
+2 -1
View File
@@ -38,7 +38,8 @@ describe('payload mount points', () => {
}); });
it('isolates CMS tables in their own postgres schema', () => { it('isolates CMS tables in their own postgres schema', () => {
// Blog content must stay separate from school marts and Airflow metadata. // Blog content must sit outside `public`, where the app tables, Airflow's
// metadata and scripts/migrate_csv_to_db.py --drop all live.
expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/); expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/);
}); });
}); });
+1 -14
View File
@@ -1,8 +1,6 @@
import type { Metadata } from 'next'; import type { Metadata } from 'next';
import Image from 'next/image'; import Image from 'next/image';
import { notFound } from 'next/navigation';
import { absoluteUrl } from '@/lib/site'; import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import { personJsonLd, organizationJsonLd } from '@/lib/jsonld'; import { personJsonLd, organizationJsonLd } from '@/lib/jsonld';
import styles from './About.module.css'; import styles from './About.module.css';
@@ -13,18 +11,7 @@ export const metadata: Metadata = {
alternates: { canonical: absoluteUrl('/about') }, alternates: { canonical: absoluteUrl('/about') },
}; };
/* export default function AboutPage() {
* Gated on about_page. The default 300s read is the right floor here: this
* page declares no revalidate of its own, so nothing is lost by it, and a flip
* lands within five minutes.
*
* notFound(), not a redirect: while the flag is dark this URL does not exist,
* and a 404 is what tells a crawler not to keep it.
*/
export default async function AboutPage() {
const flags = await getFlags();
if (flags.about_page !== true) notFound();
const jsonLd = { const jsonLd = {
'@context': 'https://schema.org', '@context': 'https://schema.org',
'@graph': [personJsonLd(), organizationJsonLd()], '@graph': [personJsonLd(), organizationJsonLd()],
+3 -21
View File
@@ -7,7 +7,6 @@ import type { JSXConvertersFunction } from '@payloadcms/richtext-lexical/react';
import { getCachedPayload } from '@/lib/payload'; import { getCachedPayload } from '@/lib/payload';
import type { Post, Media } from '@/payload-types'; import type { Post, Media } from '@/payload-types';
import { absoluteUrl } from '@/lib/site'; import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import { import {
blogPostingJsonLd, blogPostingJsonLd,
breadcrumbJsonLd, breadcrumbJsonLd,
@@ -113,15 +112,6 @@ export default async function PostPage(
{ params }: { params: Promise<{ slug: string }> }, { params }: { params: Promise<{ slug: string }> },
) { ) {
const { slug } = await params; const { slug } = await params;
/*
* Flags read at this route's own declared floor, so gating costs it nothing.
* Checked before the post is fetched: a dark blog should not query Payload.
*/
const flags = await getFlags(3600);
if (flags.blog !== true) notFound();
const namedAuthor = flags.about_page === true;
const post = await findPost(slug); const post = await findPost(slug);
if (!post) notFound(); if (!post) notFound();
@@ -130,15 +120,10 @@ export default async function PostPage(
const jsonLd = { const jsonLd = {
'@context': 'https://schema.org', '@context': 'https://schema.org',
/*
* The Person entity is anchored at /about#tudor, so it is declared only
* when that page exists. Claiming an author whose URL 404s is a worse
* signal than attributing the post to the publisher.
*/
'@graph': [ '@graph': [
blogPostingJsonLd(summary, { namedAuthor }), blogPostingJsonLd(summary),
breadcrumbJsonLd(summary), breadcrumbJsonLd(summary),
...(namedAuthor ? [personJsonLd()] : []), personJsonLd(),
organizationJsonLd(), organizationJsonLd(),
], ],
}; };
@@ -157,10 +142,7 @@ export default async function PostPage(
<h1 className={styles.heading}>{summary.title}</h1> <h1 className={styles.heading}>{summary.title}</h1>
<p className={styles.byline}> <p className={styles.byline}>
{/* Unlinked while about_page is dark; the flags are independent. */} By <Link href="/about" className={styles.link}>Tudor</Link>
By {namedAuthor
? <Link href="/about" className={styles.link}>Tudor</Link>
: 'Tudor'}
{' · '} {' · '}
<time dateTime={summary.publishedAt}> <time dateTime={summary.publishedAt}>
{new Date(summary.publishedAt).toLocaleDateString('en-GB', { {new Date(summary.publishedAt).toLocaleDateString('en-GB', {
+1 -10
View File
@@ -1,9 +1,7 @@
import type { Metadata } from 'next'; import type { Metadata } from 'next';
import Link from 'next/link'; import Link from 'next/link';
import { notFound } from 'next/navigation';
import { getCachedPayload } from '@/lib/payload'; import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site'; import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import styles from './Blog.module.css'; import styles from './Blog.module.css';
/* /*
@@ -33,9 +31,6 @@ function formatDate(value: string) {
} }
export default async function BlogIndexPage() { export default async function BlogIndexPage() {
const flags = await getFlags();
if (flags.blog !== true) notFound();
const payload = await getCachedPayload(); const payload = await getCachedPayload();
const { docs } = await payload.find({ const { docs } = await payload.find({
collection: 'posts', collection: 'posts',
@@ -53,11 +48,7 @@ export default async function BlogIndexPage() {
<p className={styles.standfirst}> <p className={styles.standfirst}>
What school performance data shows, what it doesn&apos;t, and how to What school performance data shows, what it doesn&apos;t, and how to
read it without being misled. Written by{' '} read it without being misled. Written by{' '}
{/* Plain text when about_page is dark: the two flags are <Link href="/about" className={styles.link}>Tudor</Link>.
independent, so this link would otherwise point at a 404. */}
{flags.about_page === true
? <Link href="/about" className={styles.link}>Tudor</Link>
: 'Tudor'}.
</p> </p>
</header> </header>
@@ -1,6 +1,5 @@
import { getCachedPayload } from '@/lib/payload'; import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site'; import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
/* /*
* Dynamic, not ISR. * Dynamic, not ISR.
@@ -19,11 +18,6 @@ function escapeXml(value: string): string {
} }
export async function GET() { export async function GET() {
// A dark blog has no feed. 404 rather than an empty channel: an empty feed
// is a live feed with nothing in it, which a reader would keep polling.
const flags = await getFlags();
if (flags.blog !== true) return new Response('Not found', { status: 404 });
const payload = await getCachedPayload(); const payload = await getCachedPayload();
const { docs } = await payload.find({ const { docs } = await payload.find({
collection: 'posts', collection: 'posts',
@@ -8,7 +8,6 @@
*/ */
import { getCachedPayload } from '@/lib/payload'; import { getCachedPayload } from '@/lib/payload';
import { absoluteUrl } from '@/lib/site'; import { absoluteUrl } from '@/lib/site';
import { getFlags } from '@/lib/flags';
/* /*
* Dynamic, not ISR. * Dynamic, not ISR.
@@ -22,33 +21,18 @@ import { getFlags } from '@/lib/flags';
export const dynamic = 'force-dynamic'; export const dynamic = 'force-dynamic';
export async function GET() { export async function GET() {
/* const payload = await getCachedPayload();
* A dark page must not be advertised. Submitting a URL that 404s is the one const { docs } = await payload.find({
* thing a sitemap is not allowed to do, so each entry is gated on the same collection: 'posts',
* flag that gates the page itself. where: { _status: { equals: 'published' } },
* sort: '-publishedAt',
* With both flags dark this emits a valid, empty <urlset> rather than a 404: limit: 500,
* robots.txt names this sitemap unconditionally, and an empty sitemap is a depth: 0,
* well-formed statement that there is nothing here yet. });
*/
const flags = await getFlags();
const aboutEnabled = flags.about_page === true;
const blogEnabled = flags.blog === true;
// Only query Payload when the blog is actually being advertised.
const docs = blogEnabled
? (await (await getCachedPayload()).find({
collection: 'posts',
where: { _status: { equals: 'published' } },
sort: '-publishedAt',
limit: 500,
depth: 0,
})).docs
: [];
const urls: Array<{ loc: string; lastmod: string | null }> = [ const urls: Array<{ loc: string; lastmod: string | null }> = [
...(aboutEnabled ? [{ loc: absoluteUrl('/about'), lastmod: null }] : []), { loc: absoluteUrl('/about'), lastmod: null },
...(blogEnabled ? [{ loc: absoluteUrl('/blog'), lastmod: null }] : []), { loc: absoluteUrl('/blog'), lastmod: null },
...docs.map((post) => ({ ...docs.map((post) => ({
loc: absoluteUrl(`/blog/${post.slug}`), loc: absoluteUrl(`/blog/${post.slug}`),
lastmod: new Date(String(post.updatedAt ?? post.publishedAt)).toISOString(), lastmod: new Date(String(post.updatedAt ?? post.publishedAt)).toISOString(),
-12
View File
@@ -1,12 +0,0 @@
'use client';
export default function ErrorPage({ reset }: { error: Error & { digest?: string }; reset: () => void }) {
return (
<main style={{ maxWidth: '48rem', margin: '4rem auto', padding: '1.5rem' }}>
<h1>We couldn’t load this page</h1>
<p>School information is temporarily unavailable. Please try again.</p>
<button type="button" onClick={reset}>Try again</button>
<p><a href="/">Return to school search</a></p>
</main>
);
}
+4 -46
View File
@@ -7,7 +7,6 @@ import { ComparisonToast } from '@/components/ComparisonToast';
import { RouteTrail } from '@/components/RouteTrail'; import { RouteTrail } from '@/components/RouteTrail';
import { ComparisonProvider } from '@/context/ComparisonProvider'; import { ComparisonProvider } from '@/context/ComparisonProvider';
import { SITE_URL } from '@/lib/site'; import { SITE_URL } from '@/lib/site';
import { getFlags } from '@/lib/flags';
import './globals.css'; import './globals.css';
// Manrope carries headings and key messaging — the guideline's "friendly, // Manrope carries headings and key messaging — the guideline's "friendly,
@@ -59,32 +58,14 @@ export const metadata: Metadata = {
authors: [{ name: 'schoolcompare' }], authors: [{ name: 'schoolcompare' }],
manifest: '/manifest.json', manifest: '/manifest.json',
// No `icons` key on purpose: setting it here would override the file // No `icons` key on purpose: setting it here would override the file
// conventions. app/icon.png and app/apple-icon.png are the source. // conventions. app/icon.svg and app/apple-icon.tsx are the source, and
// // app/opengraph-image.tsx supplies og:image and twitter:image.
// og:image and twitter:image are NOT inherited from
// app/opengraph-image.tsx — see the note on openGraph.images below. The
// icon conventions do reach these pages; the opengraph-image one does not.
metadataBase: new URL(SITE_URL), metadataBase: new URL(SITE_URL),
openGraph: { openGraph: {
type: 'website', type: 'website',
title: 'Compare Schools Side by Side | schoolcompare', title: 'Compare Schools Side by Side | schoolcompare',
description: description:
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.', 'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
/*
* Declared, not inherited.
*
* app/opengraph-image.tsx is a metadata file convention, and it does
* attach to routes in the app root segment — _not-found gets an og:image
* from it. It does not reach the site's pages, which live in the
* (frontend) route group whose own layout.tsx is a root layout. Staging
* served og:title, og:description, og:url, og:site_name and og:type with
* no og:image at all, so every link pasted into a chat rendered bare.
*
* The file stays at the app root: /robots.txt and /icon.png depend on it
* being there, and moving it is what broke those before. This points at
* the route it generates instead. metadataBase makes it absolute.
*/
images: ['/opengraph-image'],
url: SITE_URL, url: SITE_URL,
siteName: 'schoolcompare', siteName: 'schoolcompare',
}, },
@@ -94,34 +75,14 @@ export const metadata: Metadata = {
title: 'Compare Schools Side by Side | schoolcompare', title: 'Compare Schools Side by Side | schoolcompare',
description: description:
'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.', 'Put five English schools on one screen — SATs, GCSE results, Ofsted grades, and how close you had to live to get a place.',
// The card is summary_large_image; claiming that and supplying no image
// is worse than claiming a summary card.
images: ['/opengraph-image'],
}, },
}; };
/* export default function RootLayout({
* The footer's About and Blog links are flagged, which makes this the one
* place on the site that reads a flag on every route.
*
* 604800 is deliberate and load-bearing: it is the revalidate every SEO route
* here already declares. Next pins a route to the LOWEST revalidate among its
* fetches, so reading flags at the 300s default would drop the whole school
* and place corpus from a weekly cache to a 5-minute one — a large origin-load
* regression to hide two footer links.
*
* The cost is latency in one direction only. The pages themselves read the
* same flags at their own floors and flip within minutes; the footer links
* follow within a week. Turning a feature on early therefore shows the page
* before its footer link, which is harmless. Turning one off leaves a link to
* a 404 until the cache turns over, so a rollback that matters wants a purge.
*/
export default async function RootLayout({
children, children,
}: Readonly<{ }: Readonly<{
children: React.ReactNode; children: React.ReactNode;
}>) { }>) {
const flags = await getFlags(604800);
return ( return (
// The font variable classes must sit on <html>, not <body>. globals.css // The font variable classes must sit on <html>, not <body>. globals.css
// declares --font-display on :root as var(--font-manrope) and --font-ui as // declares --font-display on :root as var(--font-manrope) and --font-ui as
@@ -165,10 +126,7 @@ export default async function RootLayout({
{children} {children}
</main> </main>
<ComparisonToast /> <ComparisonToast />
<Footer <Footer />
aboutEnabled={flags.about_page === true}
blogEnabled={flags.blog === true}
/>
</ComparisonProvider> </ComparisonProvider>
</body> </body>
</html> </html>
+67 -44
View File
@@ -85,50 +85,73 @@ export default async function HomePage({ searchParams }: HomePageProps) {
params.has_sixth_form params.has_sixth_form
); );
// Failures propagate to the retryable error boundary. // Fetch data on server with error handling
const [filtersData, dataInfo] = await Promise.all([fetchFilters(), fetchDataInfo().catch(() => null)]); try {
const [filtersData, dataInfo] = await Promise.all([fetchFilters(), fetchDataInfo().catch(() => null)]);
// Only fetch schools if there are search parameters // Only fetch schools if there are search parameters
let schoolsData; let schoolsData;
if (hasSearchParams) { if (hasSearchParams) {
schoolsData = await fetchSchools({ schoolsData = await fetchSchools({
search: params.search, search: params.search,
local_authority: params.local_authority, local_authority: params.local_authority,
school_type: params.school_type, school_type: params.school_type,
phase: params.phase, phase: params.phase,
postcode: params.postcode, postcode: params.postcode,
radius, radius,
page, page,
page_size: 50, page_size: 50,
gender: params.gender, gender: params.gender,
admissions_policy: params.admissions_policy, admissions_policy: params.admissions_policy,
has_sixth_form: params.has_sixth_form, has_sixth_form: params.has_sixth_form,
}); });
} else { } else {
// Empty state by default // Empty state by default
schoolsData = { schools: [], page: 1, page_size: 50, total: 0, total_pages: 0 }; schoolsData = { schools: [], page: 1, page_size: 50, total: 0, total_pages: 0 };
}
const resolvedFilters = filtersData || { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
// `unique_schools`, not `total_schools` — the latter is not a field this
// endpoint returns, and reading it silently yielded null on every request.
const total = dataInfo?.unique_schools ?? null;
const years = dataInfo?.years_available ?? [];
return (
<HomeView
autosuggest={autosuggest}
initialSchools={schoolsData}
filters={resolvedFilters}
totalSchools={total}
howItWorks={hasSearchParams ? null : <HowItWorksSection />}
editorial={hasSearchParams ? null : (
<EditorialSection
totalSchools={total}
localAuthorityCount={resolvedFilters.local_authorities.length}
earliestYearLabel={years.length ? formatAcademicYear(years[0]) : null}
latestYearLabel={years.length ? formatAcademicYear(years[years.length - 1]) : null}
/>
)}
/>
);
} catch (error) {
console.error('Error fetching data for home page:', error);
const emptyFilters = { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
return (
<HomeView
autosuggest={autosuggest}
initialSchools={{ schools: [], page: 1, page_size: 50, total: 0, total_pages: 0 }}
filters={emptyFilters}
totalSchools={null}
howItWorks={hasSearchParams ? null : <HowItWorksSection />}
editorial={hasSearchParams ? null : (
<EditorialSection
totalSchools={null}
localAuthorityCount={0}
earliestYearLabel={null}
latestYearLabel={null}
/>
)}
/>
);
} }
const resolvedFilters = filtersData || { local_authorities: [], school_types: [], years: [], phases: [], genders: [], admissions_policies: [] };
// `unique_schools`, not `total_schools` — the latter is not a field this
// endpoint returns, and reading it silently yielded null on every request.
const total = dataInfo?.unique_schools ?? null;
const years = dataInfo?.years_available ?? [];
return (
<HomeView
autosuggest={autosuggest}
initialSchools={schoolsData}
filters={resolvedFilters}
totalSchools={total}
howItWorks={hasSearchParams ? null : <HowItWorksSection />}
editorial={hasSearchParams ? null : (
<EditorialSection
totalSchools={total}
localAuthorityCount={resolvedFilters.local_authorities.length}
earliestYearLabel={years.length ? formatAcademicYear(years[0]) : null}
latestYearLabel={years.length ? formatAcademicYear(years[years.length - 1]) : null}
/>
)}
/>
);
} }
@@ -1,17 +0,0 @@
import { readFile } from 'node:fs/promises';
import path from 'node:path';
export const dynamic = 'force-dynamic';
export const runtime = 'nodejs';
export async function GET() {
try {
const frontend = JSON.parse(await readFile(path.join(process.cwd(), 'build-info.json'), 'utf8'));
const base = process.env.FASTAPI_URL || 'http://localhost:8000/api';
const res = await fetch(`${base}/release`, { cache: 'no-store', signal: AbortSignal.timeout(5000) });
if (!res.ok) throw new Error('Backend identity unavailable');
return Response.json({ frontend, backend: await res.json() }, { headers: { 'Cache-Control': 'no-store', 'X-Robots-Tag': 'noindex' } });
} catch {
return Response.json({ detail: 'Release identity unavailable' }, { status: 503, headers: { 'Cache-Control': 'no-store', 'X-Robots-Tag': 'noindex' } });
}
}
@@ -4,11 +4,9 @@
* URL format: /school/138267-school-name-here * URL format: /school/138267-school-name-here
*/ */
import { APIFetchError, fetchSchoolDetails, fetchSchools, fetchNationalAverages } from '@/lib/api'; import { fetchSchoolDetails, fetchSchools, fetchNationalAverages } from '@/lib/api';
import { notFound, redirect } from 'next/navigation'; import { notFound, redirect } from 'next/navigation';
import { SchoolDetailShell } from '@/components/school/SchoolDetailShell'; import { SchoolDetailShell } from '@/components/school/SchoolDetailShell';
import { NearbyPlaces } from '@/components/school/NearbyPlaces';
import { schoolBreadcrumbJsonLd, type SchoolPlace } from '@/lib/jsonld';
import { PrimarySchoolSections } from '@/components/school/PrimarySchoolSections'; import { PrimarySchoolSections } from '@/components/school/PrimarySchoolSections';
import { SecondarySchoolSections } from '@/components/school/SecondarySchoolSections'; import { SecondarySchoolSections } from '@/components/school/SecondarySchoolSections';
import { import {
@@ -146,15 +144,11 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
fetchNationalAverages().catch(() => null), fetchNationalAverages().catch(() => null),
]); ]);
} catch (error) { } catch (error) {
if (error instanceof APIFetchError && error.status === 404) notFound(); console.error(`Failed to fetch school ${urn}:`, error);
throw error; notFound();
} }
const { school_info, yearly_data, absence_data, ofsted, census, admissions, admissions_history, admission_distance, deprivation, finance, destinations } = data; const { school_info, yearly_data, absence_data, ofsted, census, admissions, admissions_history, admission_distance, deprivation, finance, destinations } = data;
// Absent on an older API build; the module and the trail both degrade to
// nothing rather than throwing, which is how this shipped without a
// lockstep deploy of the two images.
const places: SchoolPlace[] = data.places ?? [];
// Redirect bare URN to canonical slug URL // Redirect bare URN to canonical slug URL
const canonicalSlug = schoolUrl(urn, school_info.school_name).replace('/school/', ''); const canonicalSlug = schoolUrl(urn, school_info.school_name).replace('/school/', '');
@@ -191,19 +185,10 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
const primaryNavItems = buildNavItems(primaryFlags, navInput); const primaryNavItems = buildNavItems(primaryFlags, navInput);
const secondaryNavItems = buildSecondaryNavItems(secondaryFlags, navInput); const secondaryNavItems = buildSecondaryNavItems(secondaryFlags, navInput);
/* // Generate JSON-LD structured data for SEO
* `School`, not `EducationalOrganization`.
*
* Both are valid, but EducationalOrganization is the parent type covering
* universities, training providers and nurseries alike. School is the
* specific one, and a type that says what the page is about is the whole
* point of declaring it. Google's own guidance treats the narrower type as
* the correct choice where it applies.
*/
const structuredData = { const structuredData = {
'@context': 'https://schema.org', '@context': 'https://schema.org',
'@graph': [{ '@type': 'EducationalOrganization',
'@type': 'School',
name: school_info.school_name, name: school_info.school_name,
identifier: school_info.urn.toString(), identifier: school_info.urn.toString(),
...(school_info.address && { ...(school_info.address && {
@@ -225,15 +210,6 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
...(school_info.school_type && { ...(school_info.school_type && {
additionalType: school_info.school_type, additionalType: school_info.school_type,
}), }),
},
// The trail the page sits at the end of. School pages carried no
// breadcrumb at all, while every place page already emitted one.
schoolBreadcrumbJsonLd({
name: school_info.school_name,
url: `/school/${slug}`,
places,
}),
],
}; };
return ( return (
@@ -288,7 +264,6 @@ export default async function SchoolPage({ params }: SchoolPageProps) {
/> />
</SchoolDetailShell> </SchoolDetailShell>
)} )}
<NearbyPlaces places={places} />
</> </>
); );
} }
+12 -31
View File
@@ -10,17 +10,7 @@
import { LogoMark } from './Logo'; import { LogoMark } from './Logo';
import styles from './Footer.module.css'; import styles from './Footer.module.css';
/** export function Footer() {
* Both default to false so a caller that forgets a prop hides the link rather
* than pointing it at a page that 404s. Same reasoning as backend/flags.py:
* "Every flag defaults to False."
*/
interface FooterProps {
aboutEnabled?: boolean;
blogEnabled?: boolean;
}
export function Footer({ aboutEnabled = false, blogEnabled = false }: FooterProps = {}) {
const currentYear = new Date().getFullYear(); const currentYear = new Date().getFullYear();
return ( return (
@@ -104,26 +94,17 @@ export function Footer({ aboutEnabled = false, blogEnabled = false }: FooterProp
</ul> </ul>
</div> </div>
{/* Dropped entirely when both flags are dark, rather than left as an <div className={styles.section}>
empty heading: shipping dark means the footer renders as it did <h4 className={styles.sectionTitle}>About</h4>
before the feature existed. */} <ul className={styles.links}>
{(aboutEnabled || blogEnabled) && ( {/* The only route to a named human. Deliberately not in the nav:
<div className={styles.section}> the mobile bottom bar already carries four items, and both of
<h4 className={styles.sectionTitle}>About</h4> these are lower intent than any of them. Post bylines link
<ul className={styles.links}> here too, which is where a reader actually asks the question. */}
{/* The only route to a named human. Deliberately not in the nav: <li><a href="/about" className={styles.link}>Who&apos;s behind this</a></li>
the mobile bottom bar already carries four items, and both of <li><a href="/blog" className={styles.link}>Blog</a></li>
these are lower intent than any of them. Post bylines link </ul>
here too, which is where a reader actually asks the question. */} </div>
{aboutEnabled && (
<li><a href="/about" className={styles.link}>Who&apos;s behind this</a></li>
)}
{blogEnabled && (
<li><a href="/blog" className={styles.link}>Blog</a></li>
)}
</ul>
</div>
)}
</div> </div>
<div className={styles.bottom}> <div className={styles.bottom}>
+8 -34
View File
@@ -213,16 +213,6 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
const [isLoadingMap, setIsLoadingMap] = useState(false); const [isLoadingMap, setIsLoadingMap] = useState(false);
const prevSearchParamsRef = useRef(searchParams.toString()); const prevSearchParamsRef = useRef(searchParams.toString());
const mapParamsRef = useRef<string>(''); const mapParamsRef = useRef<string>('');
const loadMoreController = useRef<AbortController | null>(null);
// Identity changes even for A → B → A, so an old A response stays stale.
const searchScope = useRef({ key: searchParams.toString() });
if (searchScope.current.key !== searchParams.toString()) {
searchScope.current = { key: searchParams.toString() };
}
useEffect(() => {
setIsLoadingMore(false);
return () => { loadMoreController.current?.abort(); };
}, [searchParams]);
const [geoState, setGeoState] = useState<'idle' | 'requesting' | 'error'>('idle'); const [geoState, setGeoState] = useState<'idle' | 'requesting' | 'error'>('idle');
const [geoError, setGeoError] = useState<string | null>(null); const [geoError, setGeoError] = useState<string | null>(null);
/* /*
@@ -284,27 +274,17 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
if (resultsView !== 'map' || !isLocationSearch) return; if (resultsView !== 'map' || !isLocationSearch) return;
const paramsKey = searchParams.toString(); const paramsKey = searchParams.toString();
if (paramsKey === mapParamsRef.current) return; if (paramsKey === mapParamsRef.current) return;
const controller = new AbortController(); mapParamsRef.current = paramsKey;
const scope = searchScope.current;
const current = () => !controller.signal.aborted && searchScope.current === scope;
setIsLoadingMap(true); setIsLoadingMap(true);
const params: Record<string, any> = {}; const params: Record<string, any> = {};
searchParams.forEach((value, key) => { params[key] = value; }); searchParams.forEach((value, key) => { params[key] = value; });
params.page = 1; params.page = 1;
params.page_size = 500; params.page_size = 500;
fetchSchools(params, { cache: 'no-store', signal: controller.signal }) fetchSchools(params, { cache: 'no-store' })
.then(r => { .then(r => setMapSchools(r.schools))
if (!current()) return; .catch(() => setMapSchools(initialSchools.schools))
mapParamsRef.current = paramsKey; .finally(() => setIsLoadingMap(false));
setMapSchools(r.schools); }, [resultsView, searchParams]);
})
.catch(() => {
if (current()) setMapSchools(initialSchools.schools);
// No cache marker on failure: opening the map again retries.
})
.finally(() => { if (current()) setIsLoadingMap(false); });
return () => controller.abort();
}, [resultsView, searchParams, initialSchools.schools]);
// Fetch LA averages when secondary or mixed schools are visible // Fetch LA averages when secondary or mixed schools are visible
useEffect(() => { useEffect(() => {
@@ -325,25 +305,19 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
if (isLoadingMore || !hasMore) return; if (isLoadingMore || !hasMore) return;
track('results_load_more', { next_page: currentPage + 1 }); track('results_load_more', { next_page: currentPage + 1 });
setIsLoadingMore(true); setIsLoadingMore(true);
const scope = searchScope.current;
const controller = new AbortController();
loadMoreController.current?.abort();
loadMoreController.current = controller;
const current = () => !controller.signal.aborted && searchScope.current === scope;
try { try {
const params: Record<string, any> = {}; const params: Record<string, any> = {};
searchParams.forEach((value, key) => { params[key] = value; }); searchParams.forEach((value, key) => { params[key] = value; });
params.page = currentPage + 1; params.page = currentPage + 1;
params.page_size = initialSchools.page_size; params.page_size = initialSchools.page_size;
const response = await fetchSchools(params, { cache: 'no-store', signal: controller.signal }); const response = await fetchSchools(params, { cache: 'no-store' });
if (!current()) return;
setAllSchools(prev => [...prev, ...response.schools]); setAllSchools(prev => [...prev, ...response.schools]);
setCurrentPage(response.page); setCurrentPage(response.page);
setHasMore(response.page < response.total_pages); setHasMore(response.page < response.total_pages);
} catch { } catch {
// silently ignore // silently ignore
} finally { } finally {
if (current()) setIsLoadingMore(false); setIsLoadingMore(false);
} }
}; };
@@ -1,53 +0,0 @@
/* Tokens only — the same vocabulary schoolSections.module.css uses, so the
module follows both themes without a rule of its own. No hardcoded colour
appears here; darkThemeSafety asserts that across the codebase. */
.section {
margin-top: 2rem;
}
/* Matches .sectionTitle in schoolSections.module.css, including the brand
rule before the text, so this reads as one more section of the page
rather than a footer bolted underneath it. */
.heading {
font-size: 1.125rem;
font-weight: 600;
color: var(--text-primary);
margin-bottom: 0.875rem;
padding-bottom: 0.5rem;
border-bottom: 2px solid var(--border);
font-family: var(--font-display);
display: flex;
align-items: center;
gap: 0.375rem;
}
.heading::before {
content: "";
display: inline-block;
width: 3px;
height: 1em;
background: var(--brand);
border-radius: 2px;
flex-shrink: 0;
}
.list {
display: flex;
flex-wrap: wrap;
gap: 0.5rem 1.25rem;
list-style: none;
margin: 0;
padding: 0;
}
.link {
color: var(--brand-strong);
font-weight: 500;
text-decoration: underline;
text-underline-offset: 2px;
}
.link:hover {
text-decoration-thickness: 2px;
}
@@ -1,67 +0,0 @@
import Link from 'next/link';
import type { SchoolPlace, SchoolPhasePage } from '@/lib/jsonld';
import styles from './NearbyPlaces.module.css';
/**
* Links from a school page into the location layer.
*
* This exists for a structural reason rather than a decorative one. Before
* it, the only anchor on a school page pointed at the school's own website,
* so the ~27k pages that carry most of the site's inbound authority passed it
* straight off-site and none of it reached the place pages. These links are
* what circulate it instead.
*
* Every entry comes from the place registry via the API, so a link is only
* ever offered for a page that exists: a place below the publish threshold is
* absent from the registry and therefore absent here.
*/
/** Narrowest first: a reader on a school page wants its town before its
* county. The API orders widest-first because that is what the breadcrumb
* reads, so the two orders are deliberately different. */
const ORDER: Record<string, number> = {
town: 0, locality: 0, outcode: 1, authority: 2,
};
function label(place: SchoolPlace): string {
const noun = place.count === 1 ? 'school' : 'schools';
const preposition = place.kind === 'outcode' ? 'near' : 'in';
return `${place.count} ${noun} ${preposition} ${place.name}`;
}
/** "22 primary schools in Brentwood" — the phrasing the query itself uses. */
function phaseLabel(place: SchoolPlace, page: SchoolPhasePage): string {
const noun = page.count === 1 ? 'school' : 'schools';
return `${page.count} ${page.phase} ${noun} in ${place.name}`;
}
export function NearbyPlaces({ places }: { places: SchoolPlace[] }) {
if (places.length === 0) return null;
const sorted = [...places].sort(
(a, b) => (ORDER[a.kind] ?? 9) - (ORDER[b.kind] ?? 9),
);
return (
<section className={styles.section} aria-labelledby="nearby-places">
<h2 id="nearby-places" className={styles.heading}>More schools near here</h2>
<ul className={styles.list}>
{sorted.flatMap((place) => [
<li key={`${place.kind}:${place.slug}`}>
<Link href={place.url} className={styles.link}>{label(place)}</Link>
</li>,
/* Immediately after its own place, so "22 primary schools in
Brentwood" reads as part of Brentwood rather than as an
unrelated link further down the row. */
...place.phases.map((page) => (
<li key={`${place.kind}:${place.slug}:${page.phase}`}>
<Link href={page.url} className={styles.link}>
{phaseLabel(place, page)}
</Link>
</li>
)),
])}
</ul>
</section>
);
}
-20
View File
@@ -4,26 +4,6 @@ The blog is Payload CMS, running inside the Next.js app. There is no separate
service and no second deploy. Writing a post is done in the browser and takes service and no second deploy. Writing a post is done in the browser and takes
effect on the live site within seconds. effect on the live site within seconds.
## Before any of this: the flags
`/blog` and `/about` are behind feature flags (`blog` and `about_page`), and
every flag in this system starts off. While `blog` is dark, `/blog`, every post
page and the RSS feed return 404, and neither appears in the sitemap or the
footer. Posts still save normally, because `/admin` is deliberately **not**
flagged: you have to be able to write a post before there is anything worth
switching on.
So a new post published to a dark blog is invisible, and that is working as
intended, not a bug. Flip the flag in Unleash when the content is ready.
Staging and production hold their own values (`development` and `production`
environments), so you can light it on staging first. The page follows a flip
within five minutes; the footer link takes up to a week, because it renders in
the root layout and is cached at the same weekly floor as the school corpus.
Both flags are temporary scaffolding, like every flag here: a test starts
failing once one is older than 90 days, at which point either the feature is
permanent and the flag comes out, or it was never going to ship.
## Signing in ## Signing in
`https://www.schoolcompare.co.uk/admin`, one account, no registration. If you `https://www.schoolcompare.co.uk/admin`, one account, no registration. If you
-1
View File
@@ -9,7 +9,6 @@ const createJestConfig = nextJest({
const customJestConfig = { const customJestConfig = {
setupFilesAfterEnv: ['<rootDir>/jest.setup.js'], setupFilesAfterEnv: ['<rootDir>/jest.setup.js'],
testEnvironment: 'jest-environment-jsdom', testEnvironment: 'jest-environment-jsdom',
modulePathIgnorePatterns: ['<rootDir>/.next/'],
moduleNameMapper: { moduleNameMapper: {
'^@/(.*)$': '<rootDir>/$1', '^@/(.*)$': '<rootDir>/$1',
}, },
+27
View File
@@ -311,6 +311,26 @@ export async function fetchDataInfo(
return handleResponse<DataInfoResponse>(response); return handleResponse<DataInfoResponse>(response);
} }
// ============================================================================
// Client-Side Fetcher (for SWR)
// ============================================================================
/**
* Generic fetcher function for use with SWR
* @example
* ```tsx
* const { data, error } = useSWR('/api/schools', fetcher);
* ```
*/
export async function fetcher<T>(url: string): Promise<T> {
// If it's already a full URL, use it directly
// Otherwise, prepend the API_BASE_URL
const fullUrl = url.startsWith('http') ? url : `${API_BASE_URL}${url.startsWith('/') ? url : `/${url}`}`;
const response = await fetch(fullUrl);
return handleResponse<T>(response);
}
// ============================================================================ // ============================================================================
// Geocoding API // Geocoding API
// ============================================================================ // ============================================================================
@@ -376,3 +396,10 @@ export function calculateDistance(
const c = 2 * Math.atan2(Math.sqrt(a), Math.sqrt(1 - a)); const c = 2 * Math.atan2(Math.sqrt(a), Math.sqrt(1 - a));
return R * c; return R * c;
} }
/**
* Convert kilometers to miles
*/
export function kmToMiles(km: number): number {
return km * 0.621371;
}
+3 -13
View File
@@ -26,21 +26,11 @@ export const FLAGS_REVALIDATE = 300;
const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL
|| 'http://localhost:8000/api'; || 'http://localhost:8000/api';
/** /** Every flag and its value. Never throws: an unreadable flag is a dark one. */
* Every flag and its value. Never throws: an unreadable flag is a dark one. export async function getFlags(): Promise<Flags> {
*
* `revalidate` is the caller's, because reading flags pins the whole route to
* the lowest revalidate among its fetches. Every SEO route here declares
* 604800; gating one at the 300s default would drop it from a weekly cache to
* a 5-minute one. Pass the route's own floor and gating costs it nothing.
*
* The trade is flag-flip latency: a route that revalidates weekly takes up to
* a week to notice a flip. Pass a smaller number where a flip must land fast.
*/
export async function getFlags(revalidate: number = FLAGS_REVALIDATE): Promise<Flags> {
try { try {
const res = await fetch(`${API}/flags`, { const res = await fetch(`${API}/flags`, {
next: { revalidate }, next: { revalidate: FLAGS_REVALIDATE },
}); });
if (!res.ok) return {}; if (!res.ok) return {};
return await res.json(); return await res.json();
+2 -83
View File
@@ -44,100 +44,19 @@ interface PostSummary {
* References the Person and Organization by @id rather than repeating them, so * References the Person and Organization by @id rather than repeating them, so
* search engines resolve every post and the About page to the one author * search engines resolve every post and the About page to the one author
* entity. Repeating the shape would declare several people with one name. * entity. Repeating the shape would declare several people with one name.
*
* `namedAuthor` is the about_page flag. The Person entity is anchored at
* /about#tudor, and that URL 404s while the flag is dark, so a post published
* in that state must not claim it: an author @id resolving to nothing is a
* worse signal than no named author. It falls back to the publisher, which is
* always live. The parameter is required rather than defaulted because every
* call site has the flag to hand and the wrong default is silent.
*/ */
export function blogPostingJsonLd( export function blogPostingJsonLd(post: PostSummary) {
post: PostSummary,
{ namedAuthor }: { namedAuthor: boolean },
) {
return { return {
'@type': 'BlogPosting', '@type': 'BlogPosting',
headline: post.title, headline: post.title,
description: post.excerpt, description: post.excerpt,
url: absoluteUrl(`/blog/${post.slug}`), url: absoluteUrl(`/blog/${post.slug}`),
datePublished: post.publishedAt, datePublished: post.publishedAt,
author: { author: { '@id': `${SITE_URL}/about#tudor` },
'@id': namedAuthor ? `${SITE_URL}/about#tudor` : `${SITE_URL}#organization`,
},
publisher: { '@id': `${SITE_URL}#organization` }, publisher: { '@id': `${SITE_URL}#organization` },
} as const; } as const;
} }
/**
* A place the location layer publishes a page for, as the school API reports
* it. `count` is what lets a link say "All 37 schools in Brentwood" rather
* than "click here".
*/
export interface SchoolPhasePage {
phase: string;
count: number;
url: string;
}
export interface SchoolPlace {
kind: string;
slug: string;
name: string;
count: number;
url: string;
/**
* The phase variants this school is actually listed on: usually one, two
* for an all-through school, none for an outcode, which publishes no phase
* route. Decided by the place registry, never re-derived here.
*/
phases: SchoolPhasePage[];
}
/**
* The trail a school page sits at the end of: Schools → authority → town.
*
* Only authority and town/locality appear. An outcode is a useful link in the
* module beside this — a parent does search "schools near CM15" — but it is
* not a step anyone navigates through, and a breadcrumb that claims otherwise
* describes a hierarchy the site does not have.
*
* Levels are skipped rather than faked. A school whose town falls below the
* publish threshold has no town page, so the trail closes over the gap; the
* alternative is a breadcrumb linking to a 404.
*/
export function schoolBreadcrumbJsonLd(
school: { name: string; url: string; places: SchoolPlace[] },
) {
/*
* Rooted at the homepage, not at /schools. There is no /schools index page
* — the location layer is /schools/[place], /schools/authority/[la] and
* /schools/near/[outcode], with nothing at the bare path — so a trail
* starting there would open with a link to a 404.
*/
const trail: Array<{ name: string; url: string }> = [
{ name: 'schoolcompare', url: '/' },
];
const authority = school.places.find((p) => p.kind === 'authority');
if (authority) trail.push({ name: authority.name, url: authority.url });
const town = school.places.find((p) => p.kind === 'town' || p.kind === 'locality');
if (town) trail.push({ name: town.name, url: town.url });
trail.push({ name: school.name, url: school.url });
return {
'@type': 'BreadcrumbList',
itemListElement: trail.map((step, index) => ({
'@type': 'ListItem',
position: index + 1,
name: step.name,
item: absoluteUrl(step.url),
})),
} as const;
}
export function breadcrumbJsonLd(post: PostSummary) { export function breadcrumbJsonLd(post: PostSummary) {
return { return {
'@type': 'BreadcrumbList', '@type': 'BreadcrumbList',
+2 -5
View File
@@ -1,5 +1,3 @@
import { APIFetchError } from './api';
/** /**
* Client for the places API. * Client for the places API.
* *
@@ -71,7 +69,7 @@ const API = process.env.FASTAPI_URL || process.env.NEXT_PUBLIC_API_URL
export async function fetchPlaces(): Promise<PlaceSummary[]> { export async function fetchPlaces(): Promise<PlaceSummary[]> {
const res = await fetch(`${API}/places`, { next: { revalidate: 604800 } }); const res = await fetch(`${API}/places`, { next: { revalidate: 604800 } });
if (!res.ok) throw new APIFetchError("Unable to load places", res.status); if (!res.ok) return [];
return (await res.json()).places ?? []; return (await res.json()).places ?? [];
} }
@@ -81,7 +79,6 @@ export async function fetchPlace(
const q = phase ? `?phase=${encodeURIComponent(phase)}` : ''; const q = phase ? `?phase=${encodeURIComponent(phase)}` : '';
const res = await fetch(`${API}/places/${kind}/${slug}${q}`, const res = await fetch(`${API}/places/${kind}/${slug}${q}`,
{ next: { revalidate: 604800 } }); { next: { revalidate: 604800 } });
if (res.status === 404) return null; if (!res.ok) return null;
if (!res.ok) throw new APIFetchError("Unable to load place", res.status);
return res.json(); return res.json();
} }
-11
View File
@@ -1,5 +1,3 @@
import type { SchoolPlace } from '@/lib/jsonld';
/** /**
* TypeScript type definitions for SchoolCompare API * TypeScript type definitions for SchoolCompare API
* Generated from backend/models.py and backend/schemas.py * Generated from backend/models.py and backend/schemas.py
@@ -348,15 +346,6 @@ export interface SchoolsResponse {
export interface SchoolDetailsResponse { export interface SchoolDetailsResponse {
school_info: School; school_info: School;
/**
* The published location-layer pages containing this school, widest first.
*
* Optional because the frontend and backend ship as separate images: a
* frontend deployed ahead of the API that serves this must render without
* it, not throw. Empty is also a real answer — a school whose town and
* authority both fall below the publish threshold has nowhere to link.
*/
places?: SchoolPlace[];
yearly_data: SchoolResult[]; yearly_data: SchoolResult[];
absence_data: AbsenceData | null; absence_data: AbsenceData | null;
// Supplementary data (null until Kestra populates) // Supplementary data (null until Kestra populates)
+3 -2
View File
@@ -24,8 +24,9 @@ export default buildConfig({
typescript: { outputFile: path.resolve(dirname, 'payload-types.ts') }, typescript: { outputFile: path.resolve(dirname, 'payload-types.ts') },
db: postgresAdapter({ db: postgresAdapter({
pool: { connectionString: process.env.DATABASE_URL }, pool: { connectionString: process.env.DATABASE_URL },
// Its own schema separates blog content from pipeline-managed school // Its own schema, so no pipeline operation on `public` can reach blog
// tables and Airflow metadata. The schema is created by the initial // content. scripts/migrate_csv_to_db.py --drop lives in that blast radius,
// as does Airflow's metadata. The schema itself is created by the initial
// migration: schemaName says where tables go, it does not create anything. // migration: schemaName says where tables go, it does not create anything.
// //
// prodMigrations runs pending migrations during server init. Without it a // prodMigrations runs pending migrations during server init. Without it a
-5
View File
@@ -40,9 +40,4 @@ ENV AIRFLOW_HOME=/opt/airflow
ENV AIRFLOW__CORE__DAGS_FOLDER=/opt/pipeline/dags ENV AIRFLOW__CORE__DAGS_FOLDER=/opt/pipeline/dags
ENV PYTHONPATH=/opt/pipeline ENV PYTHONPATH=/opt/pipeline
ARG BUILD_SHA=development
ARG BUILD_ID=development
LABEL io.schoolcompare.build-id=$BUILD_ID
LABEL io.schoolcompare.commit=$BUILD_SHA
CMD ["airflow", "api-server"] CMD ["airflow", "api-server"]
+50 -69
View File
@@ -11,10 +11,6 @@ Usage:
from __future__ import annotations from __future__ import annotations
import argparse import argparse
import json
import logging
import re
import uuid
import os import os
import sys import sys
import time import time
@@ -116,78 +112,63 @@ def build_document(row: dict) -> dict:
return doc return doc
def publish_collection(client, rows: list[dict]) -> str:
"""Validate a new collection before moving the alias; keep rollback data.
Caller holds the database advisory lock across reading and publication so
overlapping school-data DAGs cannot publish or prune each other's work.
Failed drafts are left for the next successful publication to prune: a lost
alias-update response must never cause deletion of a potentially live index.
"""
if not rows or len({r["urn"] for r in rows}) != len(rows):
raise ValueError("Search source must contain nonempty, unique school URNs")
name = f"schools_{int(time.time())}_{uuid.uuid4().hex[:12]}"
try:
previous = client.aliases["schools"].retrieve()["collection_name"]
except typesense.exceptions.ObjectNotFound:
previous = None
client.collections.create({**COLLECTION_SCHEMA, "name": name})
for i in range(0, len(rows), 500):
batch = [build_document(r) for r in rows[i:i + 500]]
results = client.collections[name].documents.import_(batch, {"action": "upsert"})
if isinstance(results, str):
results = [json.loads(line) for line in results.splitlines() if line.strip()]
if len(results) != len(batch) or any(r.get("success") is not True for r in results):
raise ValueError("Search import failed; live alias unchanged")
if client.collections[name].retrieve()["num_documents"] != len(rows):
raise ValueError("Search document count mismatch; live alias unchanged")
client.aliases.upsert("schools", {"collection_name": name})
# Retention is best-effort and must not make successful publication fail.
try:
keep = {name, previous}
keep.update(a["collection_name"] for a in client.aliases.retrieve()["aliases"])
for collection in client.collections.retrieve():
old = collection["name"]
if old not in keep and re.fullmatch(r"schools_\d+(?:_[0-9a-f]+)?", old):
client.collections[old].delete()
except Exception:
logging.getLogger(__name__).exception("Search published, but old collection cleanup failed")
return name
def sync(typesense_url: str, api_key: str): def sync(typesense_url: str, api_key: str):
from urllib.parse import urlparse
url = urlparse(typesense_url)
client = typesense.Client({ client = typesense.Client({
"nodes": [{"host": url.hostname, "port": str(url.port or 8108), "nodes": [{"host": typesense_url.split("//")[-1].split(":")[0],
"protocol": url.scheme}], "port": typesense_url.split(":")[-1],
"protocol": "http"}],
"api_key": api_key, "api_key": api_key,
"connection_timeout_seconds": 10, "connection_timeout_seconds": 10,
}) })
# Create timestamped collection for zero-downtime swap
ts = int(time.time())
collection_name = f"schools_{ts}"
print(f"Creating collection: {collection_name}")
schema = {**COLLECTION_SCHEMA, "name": collection_name}
client.collections.create(schema)
# Fetch data from marts — join fact_performance if it exists
conn = get_db_connection() conn = get_db_connection()
with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur:
# Check whether the merged fact table exists
cur.execute("""
SELECT table_name FROM information_schema.tables
WHERE table_schema = 'marts' AND table_name = 'fact_performance'
""")
has_fact_performance = cur.fetchone() is not None
query = QUERY_BASE
if has_fact_performance:
query = query.replace(
"l.longitude as lng",
"l.longitude as lng,\n p.rwm_expected_pct,\n p.progress_8_score",
)
query += QUERY_PERFORMANCE_JOIN
cur.execute(query)
rows = cur.fetchall()
conn.close()
print(f"Indexing {len(rows)} schools...")
# Batch import
batch_size = 500
for i in range(0, len(rows), batch_size):
batch = [build_document(r) for r in rows[i : i + batch_size]]
client.collections[collection_name].documents.import_(batch, {"action": "upsert"})
print(f" Indexed {min(i + batch_size, len(rows))}/{len(rows)}")
# Swap alias
print("Swapping alias 'schools' → new collection")
try: try:
with conn.cursor(cursor_factory=psycopg2.extras.RealDictCursor) as cur: client.aliases.upsert("schools", {"collection_name": collection_name})
# Session-scoped lock is released even on errors when conn closes. except Exception:
cur.execute("SELECT pg_advisory_lock(731042019)") # If alias doesn't exist yet, create it
cur.execute(""" client.aliases.upsert("schools", {"collection_name": collection_name})
SELECT table_name FROM information_schema.tables
WHERE table_schema = 'marts' AND table_name = 'fact_performance' print("Done.")
""")
has_fact_performance = cur.fetchone() is not None
query = QUERY_BASE
if has_fact_performance:
query = query.replace(
"l.longitude as lng",
"l.longitude as lng, p.rwm_expected_pct, p.progress_8_score",
)
query += QUERY_PERFORMANCE_JOIN
cur.execute(query)
rows = cur.fetchall()
name = publish_collection(client, rows)
print(f"Published {len(rows)} schools in {name}")
finally:
conn.close()
def main(): def main():
-92
View File
@@ -1,92 +0,0 @@
import importlib.util
from pathlib import Path
from unittest.mock import MagicMock
import pytest
@pytest.fixture
def sync_module(monkeypatch):
folder = Path(__file__).resolve().parents[1] / 'scripts'
monkeypatch.syspath_prepend(str(folder))
spec = importlib.util.spec_from_file_location('sync_typesense', folder / 'sync_typesense.py')
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
@pytest.fixture
def client():
c = MagicMock()
c.aliases.__getitem__.return_value.retrieve.return_value = {'collection_name': 'schools_2'}
c.aliases.retrieve.return_value = {'aliases': [{'collection_name': 'schools_99'}]}
c.collections.__getitem__.return_value.documents.import_.return_value = [{'success': True}]
c.collections.__getitem__.return_value.retrieve.return_value = {'num_documents': 1}
c.collections.retrieve.return_value = [{'name': n} for n in ['schools_1', 'schools_2', 'schools_99', 'unrelated']]
return c
@pytest.fixture
def rows():
return [{'urn': 100001, 'school_name': 'Example', 'phase_code': 2,
'school_type_code': 1, 'local_authority': 'Testshire', 'postcode': 'TS1 1AA',
'total_pupils': 250}]
@pytest.mark.parametrize('results', [[{'success': False}], [], [{'success': True}, {'success': True}]])
def test_partial_import_never_moves_alias(sync_module, client, rows, results):
client.collections.__getitem__.return_value.documents.import_.return_value = results
with pytest.raises(ValueError):
sync_module.publish_collection(client, rows)
client.aliases.upsert.assert_not_called()
client.collections.__getitem__.return_value.delete.assert_not_called()
def test_count_mismatch_does_not_publish(sync_module, client, rows):
client.collections.__getitem__.return_value.retrieve.return_value = {'num_documents': 0}
with pytest.raises(ValueError):
sync_module.publish_collection(client, rows)
client.aliases.upsert.assert_not_called()
@pytest.mark.parametrize('empty', [True, False])
def test_invalid_source_never_creates_collection(sync_module, client, rows, empty):
with pytest.raises(ValueError):
sync_module.publish_collection(client, [] if empty else rows + rows)
client.collections.create.assert_not_called()
def test_success_retains_previous_and_other_live_aliases(sync_module, client, rows):
# Distinct mock per collection allows checking exactly which one was deleted.
collections = {}
def get(name):
if name not in collections:
c = MagicMock()
c.documents.import_.return_value = '{"success":true}\n'
c.retrieve.return_value = {'num_documents': 1}
collections[name] = c
return collections[name]
client.collections.__getitem__.side_effect = get
name = sync_module.publish_collection(client, rows)
client.aliases.upsert.assert_called_once_with('schools', {'collection_name': name})
assert set(collections) == {name, 'schools_1'}
collections['schools_1'].delete.assert_called_once()
def test_uncertain_alias_update_does_not_delete_candidate(sync_module, client, rows):
client.aliases.upsert.side_effect = RuntimeError('response lost')
with pytest.raises(RuntimeError):
sync_module.publish_collection(client, rows)
client.collections.__getitem__.return_value.delete.assert_not_called()
def test_sync_closes_database_when_publication_fails(sync_module, monkeypatch):
conn = MagicMock()
monkeypatch.setattr(sync_module, 'get_db_connection', lambda: conn)
monkeypatch.setattr(sync_module.typesense, 'Client', lambda _: MagicMock())
def fail(*args): raise ValueError('import rejected')
monkeypatch.setattr(sync_module, 'publish_collection', fail)
with pytest.raises(ValueError):
sync_module.sync('http://localhost:8108', 'dummy')
conn.close.assert_called_once()
statements = [call.args[0] for call in conn.cursor.return_value.__enter__.return_value.execute.call_args_list]
assert statements[0] == 'SELECT pg_advisory_lock(731042019)'
-174
View File
@@ -1,174 +0,0 @@
"""Release identity checks and promotion of the exact digests that passed E2E.
Uses only the standard library and Docker Buildx. No registry mutation happens
until every image in the set has been resolved and its build labels validated.
"""
import argparse
import json
import os
from pathlib import Path
import re
import subprocess
import time
from urllib.error import HTTPError, URLError
from urllib.request import Request, urlopen
COMPONENTS = ('BACKEND', 'FRONTEND', 'PIPELINE')
def docker(*args):
return subprocess.check_output(['docker', 'buildx', 'imagetools', *args], text=True).strip()
def check_sha(sha):
if not re.fullmatch(r'[0-9a-f]{40}', sha):
raise ValueError('Expected a full commit SHA')
return sha
def check_digest(digest):
if not re.fullmatch(r'sha256:[0-9a-f]{64}', digest):
raise ValueError('Expected an immutable image digest')
return digest
def image_identity(ref):
image = json.loads(docker('inspect', ref, '--format', '{{json .Image}}'))
configs = [image] if 'config' in image else list(image.values())
identities = set()
for config in configs:
labels = config['config']['Labels']
identities.add((labels['io.schoolcompare.commit'], labels['io.schoolcompare.build-id']))
if len(identities) != 1:
raise ValueError('Image platforms disagree about their release identity')
return next(iter(identities))
def resolve_images(sha, verified=False, expected_build_id=None):
check_sha(sha)
refs = []
build_ids = set()
for component in COMPONENTS:
image = f"{os.environ['REGISTRY']}/{os.environ[component + '_IMAGE_NAME']}"
if verified:
manifest = json.loads(docker('inspect', f'{image}:verified-{sha}', '--format', '{{json .Manifest}}'))
digest = check_digest(manifest['digest'])
else:
digest = check_digest(os.environ[component + '_DIGEST'])
ref = f'{image}@{digest}'
actual_sha, build_id = image_identity(ref)
if actual_sha != sha or not re.fullmatch(r'[0-9a-f]{32}', build_id):
raise ValueError(f'Unrecognised release identity for {component}')
if expected_build_id is not None and build_id != expected_build_id:
raise ValueError(f'Build identity mismatch for {component}')
build_ids.add(build_id)
refs.append((image, ref))
if len(build_ids) != 1:
raise ValueError('Refusing a mixed image set')
return {'sha': sha, 'build_id': build_ids.pop(), 'images': refs}
def verify(sha, build_id):
release = resolve_images(sha, expected_build_id=build_id)
for image, ref in release['images']:
docker('create', '-t', f'{image}:verified-{sha}', ref)
return release
def promote(sha):
release = resolve_images(sha, verified=True)
# Resolve all targets first; never discover a missing candidate halfway through.
for image, _ in release['images']:
try:
docker('create', '-t', f'{image}:prod-previous', f'{image}:prod')
except subprocess.CalledProcessError:
print(f'No rollback pointer saved for {image}', flush=True)
for image, ref in release['images']:
docker('create', '-t', f'{image}:prod', ref)
return release
def matches(payload, sha, build_id):
return isinstance(payload, dict) and all(
payload.get(component) == {'sha': sha, 'build_id': build_id}
for component in ('frontend', 'backend'))
def describe_identity(payload):
"""Log only release fields, never arbitrary response bodies or secret URLs."""
if not isinstance(payload, dict):
return 'Invalid release response: expected a JSON object'
identities = []
for component in ('frontend', 'backend'):
identity = payload.get(component)
if not isinstance(identity, dict):
identities.append(f'{component}=missing or invalid')
continue
values = []
for field, length in (('sha', 40), ('build_id', 32)):
value = identity.get(field)
valid = isinstance(value, str) and (
value == 'development' or re.fullmatch(r'[0-9a-f]{' + str(length) + '}', value))
values.append(f'{field}={value if valid else "missing or invalid"}')
identities.append(f'{component}: {", ".join(values)}')
return 'Release mismatch: ' + '; '.join(identities)
def wait(base_url, sha, build_id, timeout):
check_sha(sha)
if not re.fullmatch(r'[0-9a-f]{32}', build_id):
raise ValueError('Missing expected build identity')
deadline = time.monotonic() + timeout
last_observation = 'No response received'
print(f'Waiting for deployed release {sha} / {build_id}', flush=True)
while time.monotonic() < deadline:
try:
req = Request(f'{base_url.rstrip("/")}/release.json?check={time.time_ns()}',
headers={'Cache-Control': 'no-cache',
'User-Agent': 'SchoolCompare-Release-Check/1.0',
'Accept': 'application/json'})
with urlopen(req, timeout=min(10, max(.1, deadline - time.monotonic()))) as response:
payload = json.load(response)
if matches(payload, sha, build_id):
print(f'Verified deployed release {sha} / {build_id}')
return
observation = describe_identity(payload)
except HTTPError as exc:
observation = f'Release endpoint returned HTTP {exc.code}'
exc.close()
except URLError as exc:
observation = f'Release endpoint connection failed ({type(exc.reason).__name__})'
except OSError as exc:
observation = f'Release endpoint request failed ({type(exc).__name__})'
except ValueError:
observation = 'Release endpoint returned invalid JSON or request configuration'
if observation != last_observation:
print(observation, flush=True)
last_observation = observation
time.sleep(min(5, max(0, deadline - time.monotonic())))
raise RuntimeError('Deployment did not report the expected frontend/backend release '
f'{sha} / {build_id}. Last observation: {last_observation}')
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('action', choices=['wait', 'verify', 'promote'])
parser.add_argument('--timeout', type=float, default=300)
parser.add_argument('--release', type=Path)
parser.add_argument('--output', type=Path)
args = parser.parse_args()
identity = json.loads(args.release.read_text()) if args.release else {
'sha': os.environ.get('EXPECTED_SHA', ''),
'build_id': os.environ.get('EXPECTED_BUILD_ID', ''),
}
if args.action == 'wait':
wait(os.environ['BASE_URL'], identity['sha'], identity['build_id'], args.timeout)
return
result = (verify(identity['sha'], identity['build_id']) if args.action == 'verify'
else promote(identity['sha']))
if args.output:
args.output.write_text(json.dumps(result))
if __name__ == '__main__':
main()
-130
View File
@@ -1,130 +0,0 @@
import json
from io import BytesIO
from urllib.error import HTTPError, URLError
from unittest.mock import Mock
import pytest
from scripts.ci import release
SHA = 'a' * 40
BUILD = 'b' * 32
DIGESTS = ['sha256:' + c * 64 for c in '123']
@pytest.fixture
def docker(monkeypatch):
monkeypatch.setenv('REGISTRY', 'registry.example')
refs = {}
for component, digest in zip(release.COMPONENTS, DIGESTS):
monkeypatch.setenv(component + '_IMAGE_NAME', component.lower())
monkeypatch.setenv(component + '_DIGEST', digest)
refs[f'registry.example/{component.lower()}'] = digest
def run(*args):
if args[0] == 'create': return ''
if args[-1] == '{{json .Manifest}}':
return json.dumps({'digest': refs[args[1].split(':')[0]]})
return json.dumps({'config': {'Labels': {'io.schoolcompare.commit': SHA,
'io.schoolcompare.build-id': BUILD}}})
mock = Mock(side_effect=run)
monkeypatch.setattr(release, 'docker', mock)
return mock
def test_wrong_deployed_build_is_rejected_even_at_same_commit():
assert not release.matches({'frontend': {'sha': SHA, 'build_id': BUILD},
'backend': {'sha': SHA, 'build_id': 'c' * 32}}, SHA, BUILD)
assert release.matches({c: {'sha': SHA, 'build_id': BUILD} for c in ('frontend', 'backend')}, SHA, BUILD)
def test_verification_tags_the_captured_digests(docker):
release.verify(SHA, BUILD)
creates = [c.args for c in docker.call_args_list if c.args[0] == 'create']
assert len(creates) == 3
for call, digest in zip(creates, DIGESTS):
assert call[-1].endswith('@' + digest)
assert call[2].endswith(':verified-' + SHA)
def test_promotion_resolves_all_verified_images_before_mutation(docker):
result = release.promote(SHA)
assert result['build_id'] == BUILD
calls = [c.args for c in docker.call_args_list]
first_write = next(i for i, c in enumerate(calls) if c[0] == 'create')
assert first_write == 6 # each of three candidates needs manifest + config
assert all(c[-1].endswith('@' + d) for c, d in zip(calls[-3:], DIGESTS))
def test_mixed_builds_fail_before_any_tag_is_changed(docker, monkeypatch):
identities = iter([(SHA, BUILD), (SHA, 'c' * 32), (SHA, BUILD)])
monkeypatch.setattr(release, 'image_identity', lambda _: next(identities))
with pytest.raises(ValueError, match='mixed'):
release.promote(SHA)
assert not any(c.args[0] == 'create' for c in docker.call_args_list)
def test_missing_candidate_fails_before_any_tag_is_changed(docker):
docker.side_effect = RuntimeError('missing verified tag')
with pytest.raises(RuntimeError): release.promote(SHA)
assert not any(c.args[0] == 'create' for c in docker.call_args_list)
@pytest.fixture
def poll(monkeypatch):
now = [0.0]
monkeypatch.setattr(release.time, 'monotonic', lambda: now[0])
monkeypatch.setattr(release.time, 'sleep', lambda seconds: now.__setitem__(0, now[0] + seconds))
opener = Mock()
monkeypatch.setattr(release, 'urlopen', opener)
return opener
def response(payload):
return BytesIO(json.dumps(payload).encode())
def test_wait_identifies_its_client_and_retries_until_both_services_match(poll, capsys):
poll.side_effect = [
HTTPError('https://secret.example', 503, 'unavailable', {}, None),
response({'frontend': {'sha': SHA, 'build_id': BUILD},
'backend': {'sha': SHA, 'build_id': 'c' * 32}}),
response({component: {'sha': SHA, 'build_id': BUILD}
for component in ('frontend', 'backend')}),
]
release.wait('https://secret.example/', SHA, BUILD, 15)
assert poll.call_count == 3
request = poll.call_args.args[0]
assert request.get_header('User-agent') == 'SchoolCompare-Release-Check/1.0'
assert request.get_header('Cache-control') == 'no-cache'
assert request.get_header('Accept') == 'application/json'
assert '/release.json?check=' in request.full_url
output = capsys.readouterr().out
assert 'HTTP 503' in output
assert 'backend: sha=' + SHA + ', build_id=' + 'c' * 32 in output
assert 'Verified deployed release' in output
assert 'secret.example' not in output
@pytest.mark.parametrize('failure, expected', [
(lambda: HTTPError('https://secret.example', 403, 'secret response', {}, None), 'HTTP 403'),
(lambda: URLError(OSError('secret address')), 'connection failed (OSError)'),
(lambda: TimeoutError('secret address'), 'request failed (TimeoutError)'),
(lambda: BytesIO(b'<html>secret response</html>'), 'invalid JSON'),
(lambda: response([]), 'expected a JSON object'),
(lambda: response({'frontend': {'sha': 'secret response'}}), 'missing or invalid'),
])
def test_wait_timeout_reports_last_failure_without_leaking_response_or_url(poll, capsys, failure, expected):
poll.side_effect = lambda *args, **kwargs: result_or_raise(failure())
with pytest.raises(RuntimeError) as error:
release.wait('https://secret.example', SHA, BUILD, 10)
assert expected in str(error.value)
assert SHA in str(error.value)
assert BUILD in str(error.value)
output = capsys.readouterr().out
assert sum(expected in line for line in output.splitlines()) == 1
assert 'secret' not in output + str(error.value)
assert poll.call_count == 2
def result_or_raise(result):
if isinstance(result, Exception):
raise result
return result
-34
View File
@@ -1,34 +0,0 @@
"""Check the dependency graph that ties tested digests to deployable images."""
from pathlib import Path
import yaml
ROOT = Path(__file__).resolve().parents[3]
def test_staging_verifies_identity_before_and_after_journeys():
workflow = yaml.safe_load((ROOT / '.gitea/workflows/deploy.yml').read_text())
assert workflow['concurrency'] == {'group': 'staging-release', 'cancel-in-progress': False}
jobs = workflow['jobs']
for component in ('backend', 'frontend', 'pipeline'):
job = jobs['build-' + component]
assert 'prepare' in job['needs']
assert job['outputs']['digest'] == '${{ steps.build.outputs.digest }}'
build = next(step for step in job['steps'] if step.get('id') == 'build')
assert 'BUILD_ID=${{ needs.prepare.outputs.build_id }}' in build['with']['build-args']
steps = jobs['e2e-staging']['steps']
runs = [step.get('run', '') for step in steps]
test = runs.index('npx playwright test')
assert 'release.py wait' in runs[test - 1]
assert 'release.py wait' in runs[test + 1]
assert 'release.py verify' in runs[-1]
for component in ('backend', 'frontend', 'pipeline'):
assert 'build-' + component in jobs['e2e-staging']['needs']
assert component.upper() + '_DIGEST' in steps[-1]['env']
def test_promotion_uses_verified_digest_resolver_and_build_identity_poll():
workflow = yaml.safe_load((ROOT / '.gitea/workflows/promote.yml').read_text())
steps = workflow['jobs']['promote-prod']['steps']
runs = [step.get('run', '') for step in steps]
assert 'python3 scripts/ci/release.py promote --output release.json' in runs
assert runs[-1] == 'python3 scripts/ci/release.py wait --release release.json'
-3
View File
@@ -1,8 +1,5 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
""" """
Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline.
See docs/LEGACY_CODE.md before using it for current school data.
Data Download Helper Script Data Download Helper Script
This script provides instructions and utilities for downloading This script provides instructions and utilities for downloading
-3
View File
@@ -1,8 +1,5 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
""" """
Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline.
See docs/LEGACY_CODE.md before using it for current school data.
Fetch real school performance data from UK Government sources. Fetch real school performance data from UK Government sources.
This script downloads KS2 (Key Stage 2) primary school data from: This script downloads KS2 (Key Stage 2) primary school data from:
+184
View File
@@ -0,0 +1,184 @@
#!/usr/bin/env python3
"""
Geocode all school postcodes and update the database.
This script should be run as a weekly cron job to ensure all schools
have up-to-date latitude/longitude coordinates.
Usage:
python scripts/geocode_schools.py [--force]
Options:
--force Re-geocode all postcodes, even if already geocoded
Crontab example (run every Sunday at 2am):
0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1
"""
import argparse
import sys
from datetime import datetime
from pathlib import Path
from typing import Dict, Tuple
import requests
# Add parent directory to path for imports
sys.path.insert(0, str(Path(__file__).parent.parent))
from backend.database import SessionLocal
from backend.models import School
def geocode_postcodes_bulk(postcodes: list) -> Dict[str, Tuple[float, float]]:
"""
Geocode postcodes in bulk using postcodes.io API.
Returns dict of postcode -> (latitude, longitude).
"""
results = {}
valid_postcodes = [
p.strip().upper()
for p in postcodes
if p and isinstance(p, str) and len(p.strip()) >= 5
]
valid_postcodes = list(set(valid_postcodes))
if not valid_postcodes:
return results
batch_size = 100
total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size
for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)):
batch = valid_postcodes[batch_start : batch_start + batch_size]
print(f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)...")
try:
response = requests.post(
"https://api.postcodes.io/postcodes",
json={"postcodes": batch},
timeout=30,
)
if response.status_code == 200:
data = response.json()
for item in data.get("result", []):
if item and item.get("result"):
pc = item["query"].upper()
lat = item["result"].get("latitude")
lon = item["result"].get("longitude")
if lat and lon:
results[pc] = (lat, lon)
else:
print(f" Warning: API returned status {response.status_code}")
except Exception as e:
print(f" Warning: Geocoding batch failed: {e}")
return results
def geocode_schools(force: bool = False) -> None:
"""
Geocode all schools in the database.
Args:
force: If True, re-geocode all postcodes even if already geocoded
"""
print(f"\n{'='*60}")
print(f"School Geocoding Job - {datetime.now().isoformat()}")
print(f"{'='*60}\n")
db = SessionLocal()
try:
# Get schools that need geocoding
if force:
schools = db.query(School).filter(School.postcode.isnot(None)).all()
print(f"Force mode: Processing all {len(schools)} schools with postcodes")
else:
schools = db.query(School).filter(
School.postcode.isnot(None),
(School.latitude.is_(None)) | (School.longitude.is_(None))
).all()
print(f"Found {len(schools)} schools without coordinates")
if not schools:
print("No schools to geocode. Exiting.")
return
# Extract unique postcodes
postcodes = list(set(
s.postcode.strip().upper()
for s in schools
if s.postcode
))
print(f"Unique postcodes to geocode: {len(postcodes)}")
# Geocode in bulk
print("\nGeocoding postcodes...")
geocoded = geocode_postcodes_bulk(postcodes)
print(f"Successfully geocoded: {len(geocoded)} postcodes")
# Update database
print("\nUpdating database...")
updated_count = 0
failed_count = 0
for school in schools:
if not school.postcode:
continue
pc_upper = school.postcode.strip().upper()
coords = geocoded.get(pc_upper)
if coords:
school.latitude = coords[0]
school.longitude = coords[1]
updated_count += 1
else:
failed_count += 1
db.commit()
print(f"\nResults:")
print(f" - Updated: {updated_count} schools")
print(f" - Failed (invalid/not found): {failed_count} postcodes")
# Summary stats
total_with_coords = db.query(School).filter(
School.latitude.isnot(None),
School.longitude.isnot(None)
).count()
total_schools = db.query(School).count()
print(f"\nDatabase summary:")
print(f" - Total schools: {total_schools}")
print(f" - Schools with coordinates: {total_with_coords}")
print(f" - Coverage: {100*total_with_coords/total_schools:.1f}%")
except Exception as e:
print(f"Error during geocoding: {e}")
db.rollback()
raise
finally:
db.close()
print(f"\n{'='*60}")
print(f"Geocoding job completed - {datetime.now().isoformat()}")
print(f"{'='*60}\n")
def main():
parser = argparse.ArgumentParser(
description="Geocode school postcodes and update database"
)
parser.add_argument(
"--force",
action="store_true",
help="Re-geocode all postcodes, even if already geocoded"
)
args = parser.parse_args()
geocode_schools(force=args.force)
if __name__ == "__main__":
main()
+68
View File
@@ -0,0 +1,68 @@
#!/usr/bin/env python3
"""
CLI script for manual database migration.
Usage:
python scripts/migrate_csv_to_db.py [--drop] [--geocode]
Options:
--drop Drop existing tables before migration (full reimport)
--geocode Geocode postcodes (requires network access)
"""
import sys
from pathlib import Path
# Add parent directory to path for imports
sys.path.insert(0, str(Path(__file__).parent.parent))
import argparse
from backend.config import settings
from backend.database import Base, engine, init_db, set_db_schema_version
from backend.migration import load_csv_data, migrate_data, run_full_migration
from backend.version import SCHEMA_VERSION
def main():
parser = argparse.ArgumentParser(
description="Migrate CSV data to PostgreSQL database"
)
parser.add_argument(
"--drop", action="store_true", help="Drop existing tables before migration"
)
parser.add_argument("--geocode", action="store_true", help="Geocode postcodes")
args = parser.parse_args()
print("=" * 60)
print("School Data Migration: CSV -> PostgreSQL")
print("=" * 60)
print(f"\nDatabase: {settings.database_url.split('@')[-1]}")
print(f"Data directory: {settings.data_dir}")
print(f"Target schema version: {SCHEMA_VERSION}")
if args.drop:
print("\nRunning full migration (drop + reimport)...")
success = run_full_migration(geocode=args.geocode)
else:
print("\nCreating tables (preserving existing data)...")
init_db()
print("\nLoading CSV data...")
df = load_csv_data(settings.data_dir)
if df.empty:
print("No data found to migrate!")
return 1
migrate_data(df, geocode=args.geocode)
success = True
if success:
# Ensure schema_version table exists
init_db()
set_db_schema_version(SCHEMA_VERSION)
print(f"\nSchema version set to {SCHEMA_VERSION}")
return 0 if success else 1
if __name__ == "__main__":
sys.exit(main())