From eaf5e5d180f35b3c88dbbd85b7f9bf428aa3641b Mon Sep 17 00:00:00 2001 From: Tudor Date: Mon, 14 Sep 2026 23:01:15 +0100 Subject: [PATCH 1/2] docs: describe the system that exists, not the one we started with The README still opened on "Primary School Compass", a KS2 tool for Wandsworth and Merton served by FastAPI and vanilla JavaScript with Chart.js. Every layer of that sentence is now wrong: coverage is England-wide across KS2, KS4, all-through and post-16, Next.js owns the public UI, and school data comes from dbt-built `marts.*` rather than CSVs loaded at startup. The setup instructions walked a reader into a virtualenv and a CSV import that cannot build the current schema, so following the docs produced an empty database and a wrong mental model at the same time. Replace the narrative docs with two reference documents that were checked against the code: docs/ARCHITECTURE.md for request flow, data ownership, the backend/frontend module boundaries and the real publication sequence, and docs/DEVELOPMENT.md for the checks that actually run, including the container and CI version skew that makes "just run pytest" misleading. The env examples drifted the same way. ALLOWED_ORIGINS is a JSON array, not a comma-separated list; the frontend needs FASTAPI_URL, DATABASE_URL and PAYLOAD_SECRET, none of which were documented; and RATE_LIMIT_BURST, DEFAULT_PAGE_SIZE and MAX_PAGE_SIZE were presented as tuning controls the routes do not consult. Each is now stated as it behaves. MIGRATION_SUMMARY.md keeps its content but gains a banner, because it reads like setup instructions and is not. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_016y2J6bs8gbuSJbH18w7Tan --- .env.example | 18 ++- DOCKER_DEPLOY.md | 200 ++------------------------ MIGRATION_SUMMARY.md | 2 + README.md | 251 +++++++------------------------- claude.md | 201 ++++---------------------- docs/ARCHITECTURE.md | 107 ++++++++++++++ docs/DEVELOPMENT.md | 94 ++++++++++++ nextjs-app/.env.example | 16 ++- nextjs-app/DEPLOYMENT.md | 300 ++------------------------------------- nextjs-app/README.md | 184 ++++++------------------ 10 files changed, 377 insertions(+), 996 deletions(-) create mode 100644 docs/ARCHITECTURE.md create mode 100644 docs/DEVELOPMENT.md diff --git a/.env.example b/.env.example index 7a8c4a2..6ac529f 100644 --- a/.env.example +++ b/.env.example @@ -20,7 +20,7 @@ PORT=80 # ============================================================================= # CORS # ============================================================================= -# Comma-separated list of allowed origins +# JSON array of allowed origins (pydantic-settings format) # In production, only include your actual domain ALLOWED_ORIGINS=["https://schoolcompare.co.uk"] @@ -33,13 +33,21 @@ ADMIN_API_KEY=CHANGE_THIS_TO_A_SECURE_RANDOM_KEY # Rate limiting (requests per minute per IP) RATE_LIMIT_PER_MINUTE=60 -RATE_LIMIT_BURST=10 +GLOBAL_RATE_LIMIT_PER_MINUTE=3000 # Maximum request body size in bytes (default 1MB) MAX_REQUEST_SIZE=1048576 # ============================================================================= -# API +# SEARCH AND OPTIONAL FEATURE FLAGS # ============================================================================= -DEFAULT_PAGE_SIZE=50 -MAX_PAGE_SIZE=100 +TYPESENSE_URL=http://localhost:8108 +TYPESENSE_API_KEY=CHANGE_THIS_TO_YOUR_TYPESENSE_KEY + +# Empty URL disables Unleash-backed flags. Match the managed environment when used. +UNLEASH_URL= +UNLEASH_API_TOKEN= + +# Page-size limits are currently declared by route Query parameters. +# DEFAULT_PAGE_SIZE, MAX_PAGE_SIZE and RATE_LIMIT_BURST are not reliable tuning +# controls in the current routes; see docs/LEGACY_CODE.md. diff --git a/DOCKER_DEPLOY.md b/DOCKER_DEPLOY.md index a92a220..2318bf9 100644 --- a/DOCKER_DEPLOY.md +++ b/DOCKER_DEPLOY.md @@ -1,191 +1,17 @@ -# Docker Deployment Guide +# Docker deployment -## Quick Start +The maintained deployment runbook is [docs/DEPLOY.md](docs/DEPLOY.md). -Deploy the complete SchoolCompare stack (PostgreSQL + FastAPI + Next.js) with one command: +- Production: `docker-compose.portainer.yml`, using `:prod` images. +- Staging: `docker-compose.portainer.staging.yml`, using `:staging` images. +- Builds and deployment: `.gitea/workflows/deploy.yml`. +- Human-approved production promotion: `.gitea/workflows/promote.yml`. -```bash -docker-compose up -d -``` +The generic `docker-compose.yml` is not a supported one-command onboarding path: +it still uses `:latest` tags that the release workflow no longer publishes and +lacks the full current CMS setup. Review the [legacy inventory](docs/LEGACY_CODE.md) +before using old compose examples. Starting an empty database does not populate +school marts. -This will start: -- **PostgreSQL** on port 5432 (database) -- **FastAPI** on port 8000 (backend API) -- **Next.js** on port 3000 (frontend) - -## Service Details - -### PostgreSQL Database -- **Port**: 5432 -- **Container**: `schoolcompare_db` -- **Credentials**: - - User: `schoolcompare` - - Password: `schoolcompare` - - Database: `schoolcompare` -- **Volume**: `postgres_data` (persistent storage) - -### FastAPI Backend -- **Port**: 8000 → 80 (container) -- **Container**: `schoolcompare_backend` -- **Built from**: Root `Dockerfile` -- **API Endpoint**: http://localhost:8000/api -- **Health Check**: http://localhost:8000/api/data-info - -### Next.js Frontend -- **Port**: 3000 -- **Container**: `schoolcompare_nextjs` -- **Built from**: `nextjs-app/Dockerfile` -- **URL**: http://localhost:3000 -- **Connects to**: Backend via internal network - -## Commands - -### Start all services -```bash -docker-compose up -d -``` - -### View logs -```bash -# All services -docker-compose logs -f - -# Specific service -docker-compose logs -f nextjs -docker-compose logs -f backend -docker-compose logs -f db -``` - -### Check status -```bash -docker-compose ps -``` - -### Stop all services -```bash -docker-compose down -``` - -### Rebuild after code changes -```bash -# Rebuild and restart specific service -docker-compose up -d --build nextjs - -# Rebuild all services -docker-compose up -d --build -``` - -### Clean restart (remove volumes) -```bash -docker-compose down -v -docker-compose up -d -``` - -## Initial Database Setup - -After first start, you may need to initialize the database: - -```bash -# Enter the backend container -docker exec -it schoolcompare_backend bash - -# Run migrations or data loading -python -m backend.data_loader -``` - -## Accessing Services - -Once running: -- **Frontend**: http://localhost:3000 -- **Backend API**: http://localhost:8000/api -- **API Docs**: http://localhost:8000/docs (Swagger UI) -- **Database**: localhost:5432 (use any PostgreSQL client) - -## Environment Variables - -Create a `.env` file in the root directory to customize: - -```env -# Database -POSTGRES_USER=schoolcompare -POSTGRES_PASSWORD=your_secure_password -POSTGRES_DB=schoolcompare - -# Backend -DATABASE_URL=postgresql://schoolcompare:your_secure_password@db:5432/schoolcompare - -# Frontend (for client-side access) -NEXT_PUBLIC_API_URL=http://localhost:8000/api -``` - -Then run: -```bash -docker-compose up -d -``` - -## Troubleshooting - -### Backend not connecting to database -```bash -# Check database health -docker-compose ps - -# View backend logs -docker-compose logs backend - -# Restart backend -docker-compose restart backend -``` - -### Frontend not connecting to backend -```bash -# Check backend health -curl http://localhost:8000/api/data-info - -# Check Next.js environment variables -docker exec schoolcompare_nextjs env | grep API -``` - -### Port already in use -```bash -# Change ports in docker-compose.yml -# For example, change "3000:3000" to "3001:3000" -``` - -### Rebuild from scratch -```bash -docker-compose down -v -docker system prune -a -docker-compose up -d --build -``` - -## Production Deployment - -For production, update the following: - -1. **Use secure passwords** in `.env` file -2. **Configure reverse proxy** (Nginx) in front of Next.js -3. **Enable HTTPS** with SSL certificates -4. **Set production environment variables**: - ```env - NODE_ENV=production - POSTGRES_PASSWORD= - ``` -5. **Backup database** regularly: - ```bash - docker exec schoolcompare_db pg_dump -U schoolcompare schoolcompare > backup.sql - ``` - -## Network Architecture - -``` -Internet - ↓ -Next.js (port 3000) ← User browsers - ↓ (internal network) -FastAPI (port 8000) ← API calls - ↓ (internal network) -PostgreSQL (port 5432) ← Data queries -``` - -All services communicate via the `schoolcompare-network` Docker network. +For architecture, configuration and test commands, see +[ARCHITECTURE.md](docs/ARCHITECTURE.md) and [DEVELOPMENT.md](docs/DEVELOPMENT.md). diff --git a/MIGRATION_SUMMARY.md b/MIGRATION_SUMMARY.md index ad49683..0c2bf40 100644 --- a/MIGRATION_SUMMARY.md +++ b/MIGRATION_SUMMARY.md @@ -1,3 +1,5 @@ +> Historical migration record, retained for context. Setup and architecture claims below may be obsolete. Use [README.md](README.md), [architecture](docs/ARCHITECTURE.md) and [deployment](docs/DEPLOY.md) for current guidance. + # SchoolCompare: Vanilla JS → Next.js Migration Summary ## Overview diff --git a/README.md b/README.md index 1ab8f59..2e6eaaa 100644 --- a/README.md +++ b/README.md @@ -1,214 +1,67 @@ -# Primary School Compass 🧒📚 +# SchoolCompare -A modern web application for comparing **primary school (KS2)** performance data in **Wandsworth and Merton** over the last 5 years. Built with FastAPI and vanilla JavaScript with Chart.js visualizations. +SchoolCompare compares schools across England: primary (KS2), secondary (KS4), +all-through and post-16 provision, with coverage depending on the source dataset. +It provides school search, postcode maps, comparisons, rankings, place pages, +Ofsted information, admissions and destination measures. Editorial content lives +in a Payload CMS blog. -![Python](https://img.shields.io/badge/Python-3.9+-blue) -![FastAPI](https://img.shields.io/badge/FastAPI-0.109-green) -![License](https://img.shields.io/badge/License-MIT-yellow) +## Start here -## Features +- [Architecture and data flow](docs/ARCHITECTURE.md) +- [Development and validation](docs/DEVELOPMENT.md) +- [Deployment and promotion](docs/DEPLOY.md) +- [Legacy and unused-code inventory](docs/LEGACY_CODE.md) +- [Frontend conventions](nextjs-app/README.md) +- [CMS publishing](nextjs-app/docs/PUBLISHING.md) -- 📊 **Interactive Charts** - Visualize KS2 performance trends over time -- 🔍 **Smart Search** - Find primary schools by name in Wandsworth & Merton -- ⚖️ **Side-by-Side Comparison** - Compare up to 5 schools simultaneously -- 🏆 **Rankings** - View top-performing primary schools by various KS2 metrics -- 📱 **Responsive Design** - Works beautifully on desktop and mobile +## Repository map -## Key Metrics (KS2) +| Path | Responsibility | +|---|---| +| `backend/` | FastAPI routes, cached school data, read-only SQLAlchemy mappings, feature flags | +| `nextjs-app/` | Next.js App Router, React UI, Payload CMS, frontend tests | +| `pipeline/plugins/extractors/` | Custom Singer taps for GIAS, EES, Ofsted and other datasets | +| `pipeline/transform/` | dbt staging/intermediate models, marts, seeds and data tests | +| `pipeline/dags/` | Airflow extraction, transformation and publication workflows | +| `pipeline/scripts/` | Search indexing, code generation and operational diagnostics | +| `e2e/` | Playwright journeys against a running environment | +| `.gitea/workflows/` | PR checks, staging deployment and manual production promotion | +| `scripts/` | CI review tooling and historical data utilities; see the legacy inventory | +| `docs/superpowers/`, `mockups/` | Design history and prototypes, not application entry points | -The application tracks these Key Stage 2 performance indicators: +## Runtime -| Metric | Description | -|--------|-------------| -| **Reading Progress** | Progress in reading from KS1 to KS2 | -| **Writing Progress** | Progress in writing from KS1 to KS2 | -| **Maths Progress** | Progress in maths from KS1 to KS2 | -| **Reading Expected %** | Percentage meeting expected standard in reading | -| **Writing Expected %** | Percentage meeting expected standard in writing | -| **Maths Expected %** | Percentage meeting expected standard in maths | -| **Reading, Writing & Maths Combined %** | Percentage meeting expected standard in all three subjects | +The public site is **Next.js**, not the FastAPI root page. Browser `/api/*` +requests pass through a Next.js route handler to FastAPI. Server-rendered pages +call FastAPI directly using `FASTAPI_URL`, including its `/api` suffix. -## Quick Start +PostgreSQL/PostGIS stores school data. Meltano/Singer extracts source data; +dbt builds `marts.*`; FastAPI reads those tables. Typesense serves text search +and autocomplete. Payload runs inside Next.js and owns a separate `payload` +database schema and uploaded media. -### 1. Clone and Setup +There is **no automatic CSV import or sample dataset on startup**. A working +school-data environment needs populated marts from the pipeline or an approved +database snapshot. See [development](docs/DEVELOPMENT.md) before choosing a setup. -```bash -cd school_results +## Validation -# Create virtual environment -python -m venv venv -source venv/bin/activate # On Windows: venv\Scripts\activate - -# Install dependencies -pip install -r requirements.txt +```sh +cd nextjs-app +npm ci +npm run typecheck +npm test -- --runInBand ``` -### 2. Run the Application +Backend checks, pipeline validation, runtime versions and E2E requirements are +listed in [DEVELOPMENT.md](docs/DEVELOPMENT.md). No `npm run lint` script is +currently defined. -```bash -# Start the server -python -m uvicorn backend.app:app --reload --port 8000 -``` - -Then open http://localhost:8000 in your browser. - -The app will run with **sample data** by default, showing **110 primary schools** (66 in Wandsworth, 44 in Merton) with 5 years of KS2 performance data. - -### 3. (Optional) Use Real Data - -To use real UK school performance data: - -1. Visit [Compare School Performance - Download Data](https://www.compare-school-performance.service.gov.uk/download-data) - -2. Download **Key Stage 2** data for the years you want (2019-2024) - - Select "Key Stage 2" as the data type - -3. Place the CSV files in the `data/` folder - -4. Restart the server - it will automatically load and filter to Wandsworth & Merton schools - -**Note:** The app only displays schools in Wandsworth and Merton. Data from other areas will be filtered out. - -See the helper script for more details: -```bash -python scripts/download_data.py -``` - -## Project Structure - -``` -school_results/ -├── backend/ -│ └── app.py # FastAPI application with all API endpoints -├── frontend/ -│ ├── index.html # Main HTML page -│ ├── styles.css # Styling (warm, editorial design) -│ └── app.js # Frontend JavaScript -├── data/ -│ └── .gitkeep # Place CSV data files here -├── scripts/ -│ └── download_data.py # Helper for downloading/processing data -├── requirements.txt # Python dependencies -└── README.md -``` - -## API Endpoints - -| Endpoint | Description | -|----------|-------------| -| `GET /api/schools` | List schools with optional search/filter | -| `GET /api/schools/{urn}` | Get detailed data for a specific school | -| `GET /api/compare?urns=...` | Compare multiple schools | -| `GET /api/rankings` | Get school rankings by metric | -| `GET /api/filters` | Get available filter options | -| `GET /api/metrics` | Get available performance metrics | - -### Example API Usage - -```bash -# Search for schools -curl "http://localhost:8000/api/schools?search=academy" - -# Get school details -curl "http://localhost:8000/api/schools/100001" - -# Compare schools -curl "http://localhost:8000/api/compare?urns=100001,100002,100003" - -# Get rankings -curl "http://localhost:8000/api/rankings?metric=rwm_expected_pct&year=2024" -``` - -## Data Format - -If using your own CSV data, ensure it includes these columns (or similar): - -| Column | Type | Description | -|--------|------|-------------| -| URN | Integer | Unique Reference Number | -| SCHNAME | String | School name | -| LA | String | Local Authority (must be Wandsworth or Merton) | -| READPROG | Float | Reading progress score | -| WRITPROG | Float | Writing progress score | -| MATPROG | Float | Maths progress score | -| PTRWM_EXP | Float | % meeting expected standard in reading, writing & maths | -| PTREAD_EXP | Float | % meeting expected standard in reading | -| PTWRIT_EXP | Float | % meeting expected standard in writing | -| PTMAT_EXP | Float | % meeting expected standard in maths | - -The application normalizes column names automatically and filters to only show Wandsworth and Merton schools. - -## Technology Stack - -- **Backend**: FastAPI (Python) - High-performance async API framework -- **Frontend**: Vanilla JavaScript with Chart.js -- **Styling**: Custom CSS with CSS variables for theming -- **Data**: Pandas for CSV processing - -## Design Philosophy - -The UI features a warm, editorial design inspired by quality publications: -- **Typography**: DM Sans for body text, Playfair Display for headings -- **Color Palette**: Warm cream background with coral and teal accents -- **Interactions**: Smooth animations and hover effects -- **Charts**: Clean, readable data visualizations - -## Development - -```bash -# Run with auto-reload -python -m uvicorn backend.app:app --reload --port 8000 - -# Or run directly -python backend/app.py -``` - -## Coverage - -This application is specifically designed for: - -- **School Phase**: Primary schools only (Key Stage 2) -- **Geographic Area**: Wandsworth and Merton (London boroughs) -- **Time Period**: Last 5 years of data (2020-2024) - -Note: 2021 data shows as unavailable because SATs were cancelled due to COVID-19. - -## Data Source - -Data is sourced from the UK Government's [Compare School Performance](https://www.compare-school-performance.service.gov.uk/) service, which provides official school performance data for England. - -**Important**: When using real data, please comply with the [terms of use](https://www.compare-school-performance.service.gov.uk/download-data) and data protection regulations. - -## Scheduled Jobs - -### Geocoding Schools (Cron Job) - -School postcodes are geocoded by a scheduled job, not on-demand. This improves performance and reduces API calls. - -**Setup the cron job** (runs weekly on Sunday at 2am): - -```bash -# Edit crontab -crontab -e - -# Add this line (adjust paths as needed): -0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1 -``` - -**Manual run:** -```bash -# Geocode only schools missing coordinates -python scripts/geocode_schools.py - -# Force re-geocode all schools -python scripts/geocode_schools.py --force -``` - -## License - -MIT License - feel free to use this project for educational purposes. - ---- - -Built with ❤️ for Wandsworth & Merton families +## Deployment +Work on a feature branch and open a PR. Merging to `main` builds images and +deploys staging. Production promotion is a separate, human-triggered Gitea +workflow. Use [DEPLOY.md](docs/DEPLOY.md) and the Portainer compose files as the +operational references. The generic compose examples still reference `:latest`, +which the current release workflow does not publish. diff --git a/claude.md b/claude.md index 38b62af..29d6f7f 100644 --- a/claude.md +++ b/claude.md @@ -1,180 +1,37 @@ -# SchoolCompare.co.uk - Project Context +# SchoolCompare project context -## Overview +## Maintained documentation -SchoolCompare is a web application for comparing UK primary school (KS2) performance data. It allows users to: -- Search and browse schools by name, location (postcode), or local authority -- Compare multiple schools side-by-side with charts and tables -- View school rankings by various KS2 metrics -- See historical performance trends across years +Read [README.md](README.md), [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) and +[docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the current implementation. +[docs/LEGACY_CODE.md](docs/LEGACY_CODE.md) records obsolete paths and deliberate +compatibility code. Historical design documents are not current setup instructions. -## Architecture +## Architecture constraints -### Backend (Python/FastAPI) -- **Framework**: FastAPI with uvicorn -- **Database**: PostgreSQL with SQLAlchemy ORM -- **Data Source**: UK Government "Compare School Performance" CSV downloads - -Key files: -- `backend/app.py` - Main FastAPI application, API routes -- `backend/config.py` - Configuration via pydantic-settings (env vars, .env file) -- `backend/database.py` - SQLAlchemy engine, session management -- `backend/models.py` - Database models (School, SchoolResult) -- `backend/data_loader.py` - Data queries, geocoding, legacy DataFrame compatibility -- `backend/schemas.py` - Column mappings, metric definitions, LA code mappings - -### Content / CMS (Payload) - -Payload CMS runs **inside** the Next.js app — one image, one container, no -separate service. It powers `/blog`; `/about` is a plain coded page. - -- **Admin panel:** `/admin`. The only authenticated surface on the site. - `noindex` via both `robots.txt` and `X-Robots-Tag`. -- **CMS API:** `/cms-api`, **not** `/api`. `/api/*` is a catch-all proxy to - FastAPI (`app/(frontend)/api/[...path]`) which would silently swallow every - admin call and forward it to the backend. Mount points are defined once in - `lib/payloadRoutes.ts`. -- **Database:** the existing Postgres, in its own `payload` schema, so no - pipeline operation on `public` — including - `scripts/migrate_csv_to_db.py --drop` — can reach blog content. -- **Uploads:** the `payload_media` Docker volume at `/app/media`. Not - reproducible from the pipeline; must be backed up. -- **New env vars:** `DATABASE_URL` and `PAYLOAD_SECRET` on the frontend service. - Staging must use a different `PAYLOAD_SECRET` from production. -- Publishing workflow and house style: `nextjs-app/docs/PUBLISHING.md`. -- **Admin field components resolve through a generated import map** - (`app/(payload)/admin/importMap.js`). Payload hands the client a *path* per - field and looks it up there; a missing entry renders no field and reports no - error, while `required` still blocks the save. After adding or changing any - field, editor or lexical feature, run `npm run generate:importmap` in - `nextjs-app/` and commit the result. - -### Two route groups - -`nextjs-app/app/` has no root `layout.tsx`. It cannot: Payload's admin panel -ships its own root layout rendering ``/``, and Next permits -multiple root layouts only when no `app/layout.tsx` exists. - -- `app/(frontend)/` — the site. Its `layout.tsx` is the site's root layout. -- `app/(payload)/` — the admin panel and `/cms-api`. - -Route groups are invisible to routing, so every public URL is unchanged. - -**The metadata file conventions stay at the `app/` root** — `robots.ts`, -`opengraph-image.tsx`, `icon.png`, `apple-icon.png`. Inside a route group Next -treats them as segment-scoped: it renames `/icon.png` to `/icon-.png` and -drops `/robots.txt` entirely. Route handlers are unaffected. - -The build must succeed with `DATABASE_URL` unset, because CI builds it that -way. Never call `getCachedPayload()` at module scope, and never add -`generateStaticParams` to a DB-backed route. - -### Frontend (Vanilla JS) -- Single-page application with hash-based routing -- Chart.js for data visualization -- No build step required - -Key files: -- `frontend/index.html` - Main HTML structure -- `frontend/app.js` - All application logic, API calls, rendering -- `frontend/styles.css` - Styling (CSS variables, responsive design) - -### Database Schema - -``` -schools school_results -├── id (PK) ├── id (PK) -├── urn (unique, indexed) ├── school_id (FK → schools.id) -├── school_name ├── year (indexed) -├── local_authority ├── rwm_expected_pct -├── school_type ├── reading_expected_pct -├── postcode ├── ... (all KS2 metrics) -├── latitude, longitude └── unique(school_id, year) -└── results → SchoolResult[] -``` - -## Configuration - -Environment variables (or `.env` file): -- `DATABASE_URL` - PostgreSQL connection string (default: `postgresql://schoolcompare:schoolcompare@localhost:5432/schoolcompare`) -- `HOST`, `PORT` - Server binding (default: `0.0.0.0:80`) -- `ALLOWED_ORIGINS` - CORS origins - -## Running Locally - -1. Start PostgreSQL: - ```bash - docker compose up -d db - ``` - -2. Run migration to import CSV data: - ```bash - python scripts/migrate_csv_to_db.py --drop - # Add --geocode to geocode postcodes (slower, adds lat/long) - ``` - -3. Start the app: - ```bash - uvicorn backend.app:app --host 0.0.0.0 --port 8000 - ``` - -## Docker Deployment - -```bash -docker compose up -d -``` - -This starts: -- `db` - PostgreSQL 16 with persistent volume -- `app` - FastAPI application on port 80 - -## Data - -- Source: UK Government Compare School Performance downloads -- Location: `data/` directory with year folders (e.g., `2023-2024/england_ks2final.csv`) -- The `scripts/download_data.py` can fetch data from the government website - -## Key Features - -- **Location Search**: Enter postcode to find nearby schools (uses postcodes.io API) -- **Multi-school Comparison**: Select multiple schools, view metrics across years -- **Rankings**: Top schools by any KS2 metric, filterable by local authority -- **Variability Analysis**: Shows standard deviation of scores across years - -## API Endpoints - -- `GET /api/schools` - List/search schools (supports pagination, location search) -- `GET /api/schools/{urn}` - School details with all yearly data -- `GET /api/compare?urns=123,456` - Compare multiple schools -- `GET /api/rankings` - School rankings by metric -- `GET /api/filters` - Available filter options (LAs, types, years) -- `GET /api/metrics` - Metric definitions (single source of truth) -- `GET /api/data-info` - Database stats +- Next.js serves the public UI. FastAPI serves school data from dbt-built `marts.*`. + The backend does not create school tables or import CSVs at startup. +- School coverage spans England and multiple phases, not only primary schools in + Wandsworth and Merton. +- `/api/*` belongs to the FastAPI proxy. Payload uses `/cms-api` and `/admin`. +- Payload runs inside Next.js, with its own `payload` schema and persistent media. + Keep CMS migrations independent of school-data transformations. +- Public and Payload route groups have separate root layouts. Do not introduce + `app/layout.tsx`. Keep site-wide metadata files at the `app/` root. +- Builds must succeed with `DATABASE_URL` unset. Do not call `getCachedPayload()` + at module scope or add DB-backed `generateStaticParams`. +- After changing CMS fields/editors, run `npm run generate:importmap` and commit + the generated import map. See `nextjs-app/docs/PUBLISHING.md`. +- The backend and pipeline GIAS dictionary copies are generated together; preserve + their parity. Tests enforce it. ## SDLC -Full details in `docs/DEPLOY.md`. The short version: +Follow [docs/DEPLOY.md](docs/DEPLOY.md). -- **Never push to `main` directly.** Work on a feature branch and open a PR; - branch protection requires the PR checks (typecheck, tests, builds, AI review) - to pass before merge. -- Merging to `main` deploys automatically **to staging only**: images are - built once, deployed to the staging Portainer stack, and verified by the - Playwright journeys in `e2e/`. Production is a second, manual approval: - the "Promote to Production (manual)" workflow in Gitea Actions, run after - testing the feature on staging. It refuses commits whose staging E2E gate - isn't green. Never trigger it yourself — promotion is the human's call. -- If you change user-facing behaviour, update or extend the `e2e/` journey - tests in the same PR — they gate whether staging is fit for human testing - and whether a commit is promotable. - -## Recent Changes - -- Added staging environment + automated staging→prod pipeline (Gitea Actions) -- Migrated from CSV file storage to PostgreSQL database -- Added location-based search using postcode geocoding -- Added local authority filter to rankings -- Improved frontend with featured schools, loading states, API caching - -# Important -- Do not attempt to start a local server to test the application, it does not work +- Never push directly to `main`. Use a feature branch and a PR with passing checks. +- Merges deploy staging only. Production promotion is a separate human decision; + do not trigger the promotion workflow yourself. +- Update E2E journeys in the same PR when changing user-facing behaviour. +- Do not attempt to start a local server to test the application; use unit checks + and the configured integration environment. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..4ea42cd --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,107 @@ +# Architecture + +This describes the implementation as reviewed on 2026-09-14. It distinguishes +current behaviour from improvements still to be implemented. + +## Request flow + +```text +Browser → Next.js public routes + ├─ /api/* proxy → FastAPI → cached DataFrames / PostgreSQL marts + │ ├─ Typesense (search and suggestions) + │ └─ postcodes.io (postcode lookup) + └─ /admin, /cms-api, /blog → Payload → payload schema + media volume + +Next.js server rendering → FastAPI directly through FASTAPI_URL +``` + +`nextjs-app/lib/api.ts` contains typed fetch wrappers and revalidation defaults. +The proxy is `nextjs-app/app/(frontend)/api/[...path]/route.ts`. Payload uses +`/cms-api` so its routes do not collide with the FastAPI proxy. The proxy denies +`/api/flags`; server-side rendering reads flags directly from FastAPI. + +## Data ownership + +| Layer | Owner and role | +|---|---| +| Source data | GIAS, DfE EES, Ofsted, finance, deprivation and council admission-distance sources | +| `raw` | Singer taps and the PostgreSQL target configured in `pipeline/meltano.yml` | +| Staging/intermediate/marts | dbt models in `pipeline/transform`; marts are materialized tables | +| `marts.dim_school`, `marts.dim_location` | School identity and location, filtered to supported England establishments | +| `marts.fact_*` | Performance and supplementary datasets; coverage and years vary | +| Typesense `schools` alias | Search documents built by `pipeline/scripts/sync_typesense.py` | +| `payload` | CMS collections and migrations in `nextjs-app/`; independent of dbt | +| Media volume | Uploaded blog media; requires backup and cannot be regenerated from school datasets | + +`backend/models.py` maps existing marts for reading. It does not create the school +schema. There is no startup schema-version migration or CSV reimport. Payload's +`nextjs-app/migrations/` is active and must not be confused with the removed +legacy backend migration code. + +Coordinates normally come from GIAS British National Grid coordinates transformed +by PostGIS in `dim_location.sql`. `pipeline/scripts/geocode_postcodes.py` is a +manual fallback utility, not a task wired into the current school-data DAG. +Backend postcode searches also use postcodes.io; that lookup does not populate +school coordinates in the database. + +## Backend boundaries + +- `app.py`: routes, middleware, search filtering, sitemap/place publication and response assembly. +- `data_loader.py`: SQL loading, process-local DataFrame caches, Typesense calls, + postcode lookups, supplementary queries and benchmark calculation. +- `database.py`: synchronous SQLAlchemy engine and sessions. +- `schemas.py`: metric definitions, column mappings and display metadata; despite + its name this is not a collection of Pydantic API response models. +- `places.py` and `localities.py`: place registry and curated locality information. +- `flags.py`: Unleash-backed feature flags, disabled when no server is configured. +- `gias_codes.py` / `ofsted_codes.py`: source-code translation and display rules. + +Search starts from a cached latest-row-per-school snapshot. Detail pages read +history from the full DataFrame and supplementary data from marts. Comparisons +batch supplementary queries across selected URNs. Async routes still contain +synchronous dependency calls; a fully asynchronous database layer is not present. + +## Frontend boundaries + +`app/(frontend)` owns the public root layout and pages. `app/(payload)` owns the +CMS root layout. Do not add a shared `app/layout.tsx`: these groups deliberately +have separate root layouts. Root metadata files remain in `app/`. + +Server pages fetch initial data and pass it to client views. Client state uses +React hooks, URL search parameters and the comparison context/localStorage. +There is no SWR dependency. Leaflet maps are loaded through dynamic wrappers; +Chart.js renders performance and comparison charts. + +`components/school/` contains detail sections, with section decisions and data +preparation in `lib/schoolSections.ts`. `lib/types.ts` contains manually maintained +API types. `payload-types.ts` and the Payload import map are generated artifacts. + +## Publication and caching today + +1. Airflow DAGs extract and validate source data, then run selected dbt builds. +2. Relevant DAGs rebuild Typesense and swap the `schools` alias. +3. They call `POST /api/admin/reload` with `X-API-Key` to refresh school DataFrames. +4. A separate weekly sitemap DAG calls `POST /api/admin/regenerate-sitemap`, + rebuilding places and sitemaps. + +GIAS is scheduled daily, Ofsted monthly, and annual datasets are manually +triggered. The DAG definitions are authoritative for selectors and dependencies. + +Caches exist in several independent layers: backend DataFrames and registries, +backend HTTP Cache-Control/ETags, Next.js fetch/page revalidation, and browser or +shared HTTP caches where configured. Place fetches request a one-week revalidation +interval. HTTP ETags are computed after route execution, not before database work. + +Known limitations: reload clears the old DataFrames before verifying replacement +data; places/sitemaps refresh separately; Next.js caches are not explicitly purged +by the pipeline; Typesense import results are not validated before alias publication. +Do not describe this sequence as an atomic dataset release. These are follow-up +reliability tasks, not changes implemented by the documentation cleanup. + +## Deployment references + +See [DEPLOY.md](DEPLOY.md). PR checks include frontend typechecking/tests, backend +unit tests, image builds and AI review. Staging journeys run after merging. +Production promotion retags a selected commit's images. Current health polling +checks HTTP success, not the deployed commit identity; overlapping staging runs +remain a release-verification concern. diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md new file mode 100644 index 0000000..f1ebb0c --- /dev/null +++ b/docs/DEVELOPMENT.md @@ -0,0 +1,94 @@ +# Development and validation + +## Prerequisites and environment boundaries + +Use a feature branch. The deployed stack is the integration environment; do not +assume a local server can run from a fresh checkout. This cleanup did not start +local servers or provision databases. Unit tests use fixtures and mocks. + +The current versions are not yet aligned: + +| Component | Container | PR checks | +|---|---|---| +| Backend | Python 3.11 | Python 3.12 | +| Frontend | Node 24 | Node 22 | +| Pipeline | Python 3.13 | Pipeline image build | + +Use the component's container version when reproducing deployment behaviour. +The backend dependency pins predate Python 3.14; do not assume the system Python +can install or run them. Version alignment is a separate maintenance task. + +## Frontend checks + +```sh +cd nextjs-app +npm ci +npm run typecheck +npm test -- --runInBand +``` + +`npm run build` is the production build check. There is no `lint` script. +Tests live in `__tests__/` and use Jest/React Testing Library. These checks do not +prove that live PostgreSQL queries, Typesense or a deployed proxy work. + +The frontend `.env.example` documents runtime variables. Browser traffic normally +uses `/api`; `FASTAPI_URL` is an absolute server-side URL ending in `/api`. +Payload additionally needs `DATABASE_URL` and `PAYLOAD_SECRET` when used at runtime. +Never commit credentials or real `.env` files. + +## Backend checks + +From the repository root, using an available Python 3.11 or 3.12 interpreter: + +```sh +python3.11 -m venv /tmp/schoolcompare-backend-venv +/tmp/schoolcompare-backend-venv/bin/python -m pip install -r requirements.txt pytest 'httpx<0.28' +/tmp/schoolcompare-backend-venv/bin/python -m pytest backend/tests -q +``` + +Substitute `python3.12` if matching PR CI. The test dependencies above match the +current workflow; they are not yet captured in a dedicated development lockfile. +Backend configuration is defined in `backend/config.py`; `.env.example` documents +commonly used values. `ALLOWED_ORIGINS` uses a JSON array, not a comma-separated string. + +## Data and pipeline work + +The app needs populated `marts.*` tables. A new Postgres instance alone is not a +working school-data environment. Use the existing managed pipeline or an approved +snapshot; the removed CSV importer cannot build the current schema. + +The pipeline container includes Meltano, dbt/Postgres, Airflow and the custom taps. +Airflow commands/selectors live in `pipeline/dags/`. Schema tests live in +`pipeline/transform/tests/` and model YAML files. Run the relevant `dbt build` +selector in an isolated data environment for model changes; it writes tables and +is not a read-only smoke test. Prefer `python -m dbt.cli.main` as the DAGs do. + +GIAS dictionaries are generated together by +`pipeline/scripts/generate_gias_codes.py`. The backend and pipeline copies are +intentional; `backend/tests/test_gias_codes.py` checks that they stay identical. + +For Payload collection/editor changes, run `npm run generate:importmap` in +`nextjs-app/` and include the generated map. Preserve CMS migrations and the +separate `payload` schema. See [publishing](../nextjs-app/docs/PUBLISHING.md). + +## End-to-end checks + +Against an existing, authorised test environment: + +```sh +cd e2e +npm ci +npx playwright install chromium +BASE_URL=https://your-test-environment.example npx playwright test +``` + +The suite does not start a web server. CI installs Chromium with system dependencies +and runs against staging. Use the configured staging target: `docs/DEPLOY.md` +records the public staging proxy limitation. User-visible behaviour changes should +update the corresponding journeys. + +## Before requesting review + +Run checks relevant to the change, inspect `git diff --check`, and report checks +that could not run. Do not publish or promote as part of local validation. +[DEPLOY.md](DEPLOY.md) documents the PR and human promotion gates. diff --git a/nextjs-app/.env.example b/nextjs-app/.env.example index 83984a3..af1004b 100644 --- a/nextjs-app/.env.example +++ b/nextjs-app/.env.example @@ -1,8 +1,16 @@ -# API Configuration -NEXT_PUBLIC_API_URL=http://localhost:8000/api +# Browser requests use the same-origin Next.js proxy. +NEXT_PUBLIC_API_URL=/api -# Production API URL (for deployment) -# NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api +# Absolute URL for server-side fetching and the proxy; include /api. +# In the managed container network this is http://backend:80/api (staging differs). +FASTAPI_URL=http://localhost:8000/api + +# Payload CMS runtime configuration. Use the managed environment's database; +# Payload owns the payload schema, independently of the school marts. +DATABASE_URL=postgresql://schoolcompare:CHANGE_THIS_PASSWORD@localhost:5432/schoolcompare +# Generate a secret: python -c "import secrets; print(secrets.token_urlsafe(32))" +# Use distinct secrets for staging and production. +PAYLOAD_SECRET=CHANGE_THIS_TO_A_SECURE_RANDOM_SECRET # Node Environment NODE_ENV=development diff --git a/nextjs-app/DEPLOYMENT.md b/nextjs-app/DEPLOYMENT.md index 4878fa1..b0acb0e 100644 --- a/nextjs-app/DEPLOYMENT.md +++ b/nextjs-app/DEPLOYMENT.md @@ -1,291 +1,15 @@ -# Deployment Guide +# Frontend deployment -This guide covers deployment options for the SchoolCompare Next.js application. +Next.js and Payload run in the same frontend container. The maintained deployment +procedure is [docs/DEPLOY.md](../docs/DEPLOY.md), with the production and staging +Portainer compose files at the repository root. -## Deployment Options +The frontend Dockerfile builds a standalone Next.js image. Runtime configuration +supplies `FASTAPI_URL`, `DATABASE_URL` and `PAYLOAD_SECRET`; uploaded CMS media is +persisted in a volume. Promote the built image through the repository's Gitea +workflow after human staging approval. -### Option 1: Vercel (Recommended for Next.js) - -Vercel is the easiest and most optimized platform for Next.js applications. - -#### Steps: - -1. **Install Vercel CLI**: - ```bash - npm install -g vercel - ``` - -2. **Login to Vercel**: - ```bash - vercel login - ``` - -3. **Deploy**: - ```bash - vercel --prod - ``` - -4. **Configure Environment Variables** in Vercel dashboard: - - `NEXT_PUBLIC_API_URL`: Your FastAPI endpoint (e.g., `https://api.schoolcompare.co.uk/api`) - - `FASTAPI_URL`: Same as above for server-side requests - -#### Benefits: -- Automatic HTTPS -- Global CDN -- Zero-config deployment -- Automatic preview deployments -- Built-in analytics - ---- - -### Option 2: Docker (Self-hosted) - -Deploy using Docker containers for full control. - -#### Prerequisites: -- Docker 20+ -- Docker Compose 2+ - -#### Steps: - -1. **Build Docker Image**: - ```bash - docker build -t schoolcompare-nextjs:latest . - ``` - -2. **Run with Docker Compose**: - ```bash - # Create .env file with production variables - echo "NEXT_PUBLIC_API_URL=https://api.schoolcompare.co.uk/api" > .env - echo "FASTAPI_URL=http://backend:8000/api" >> .env - - # Start services - docker-compose up -d - ``` - -3. **Verify Deployment**: - ```bash - curl http://localhost:3000 - ``` - -#### Environment Variables: -- `NEXT_PUBLIC_API_URL`: Public API endpoint (client-side) -- `FASTAPI_URL`: Internal API endpoint (server-side) -- `NODE_ENV`: `production` - ---- - -### Option 3: PM2 (Node.js Process Manager) - -Deploy directly on a Node.js server using PM2. - -#### Prerequisites: -- Node.js 24+ -- PM2 (`npm install -g pm2`) - -#### Steps: - -1. **Build Application**: - ```bash - npm run build - ``` - -2. **Create PM2 Ecosystem File** (`ecosystem.config.js`): - ```javascript - module.exports = { - apps: [{ - name: 'schoolcompare-nextjs', - script: 'npm', - args: 'start', - cwd: '/path/to/nextjs-app', - instances: 'max', - exec_mode: 'cluster', - env: { - NODE_ENV: 'production', - PORT: 3000, - NEXT_PUBLIC_API_URL: 'https://api.schoolcompare.co.uk/api', - FASTAPI_URL: 'http://localhost:8000/api', - }, - }], - }; - ``` - -3. **Start with PM2**: - ```bash - pm2 start ecosystem.config.js - pm2 save - pm2 startup - ``` - ---- - -### Option 4: Nginx Reverse Proxy - -Use Nginx as a reverse proxy in front of Next.js. - -#### Nginx Configuration: - -```nginx -server { - listen 80; - server_name schoolcompare.co.uk; - - # Redirect to HTTPS - return 301 https://$server_name$request_uri; -} - -server { - listen 443 ssl http2; - server_name schoolcompare.co.uk; - - # SSL Configuration - ssl_certificate /etc/ssl/certs/schoolcompare.crt; - ssl_certificate_key /etc/ssl/private/schoolcompare.key; - - # Security Headers - # frame-ancestors replaces X-Frame-Options so the analytics subdomain - # (Umami heatmap/recorder) can embed the site in an iframe. - add_header Content-Security-Policy "frame-ancestors 'self' https://analytics.schoolcompare.co.uk" always; - add_header X-Content-Type-Options "nosniff" always; - add_header X-XSS-Protection "1; mode=block" always; - - # Proxy to Next.js - location / { - proxy_pass http://localhost:3000; - proxy_http_version 1.1; - proxy_set_header Upgrade $http_upgrade; - proxy_set_header Connection 'upgrade'; - proxy_set_header Host $host; - proxy_cache_bypass $http_upgrade; - proxy_set_header X-Real-IP $remote_addr; - proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - proxy_set_header X-Forwarded-Proto $scheme; - } - - # Proxy to FastAPI - location /api/ { - proxy_pass http://localhost:8000; - proxy_http_version 1.1; - proxy_set_header Host $host; - proxy_set_header X-Real-IP $remote_addr; - proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - proxy_set_header X-Forwarded-Proto $scheme; - } - - # Cache static files - location /_next/static/ { - proxy_pass http://localhost:3000; - add_header Cache-Control "public, max-age=31536000, immutable"; - } -} -``` - ---- - -## Pre-Deployment Checklist - -- [ ] Run `npm run build` successfully -- [ ] Run `npm test` - all tests pass -- [ ] Environment variables configured -- [ ] FastAPI backend accessible -- [ ] Database migrations applied -- [ ] SSL certificates configured (production) -- [ ] Domain DNS configured -- [ ] Monitoring/logging set up -- [ ] Backup strategy in place - ---- - -## Post-Deployment Verification - -1. **Health Check**: - ```bash - curl https://schoolcompare.co.uk - ``` - -2. **Test Routes**: - - Home: `https://schoolcompare.co.uk/` - - School Page: `https://schoolcompare.co.uk/school/100001` - - Compare: `https://schoolcompare.co.uk/compare` - - Rankings: `https://schoolcompare.co.uk/rankings` - -3. **Check SEO**: - - Sitemap: `https://schoolcompare.co.uk/sitemap.xml` - - Robots: `https://schoolcompare.co.uk/robots.txt` - -4. **Performance Audit**: - - Run Lighthouse in Chrome DevTools - - Target scores: 90+ for Performance, Accessibility, Best Practices, SEO - ---- - -## Monitoring - -### Recommended Tools: -- **Vercel Analytics** (if using Vercel) -- **Sentry** for error tracking -- **Google Analytics** for user analytics -- **Uptime Robot** for uptime monitoring - -### Health Check Endpoint: -The application automatically serves health data at the root route. - ---- - -## Rollback Procedure - -### Vercel: -```bash -vercel rollback -``` - -### Docker: -```bash -docker-compose down -docker-compose up -d --force-recreate -``` - -### PM2: -```bash -pm2 stop schoolcompare-nextjs -# Restore previous build -pm2 start schoolcompare-nextjs -``` - ---- - -## Troubleshooting - -### Issue: API requests failing -- **Solution**: Check `NEXT_PUBLIC_API_URL` and `FASTAPI_URL` environment variables -- **Verify**: FastAPI backend is accessible from Next.js container/server - -### Issue: Build fails -- **Solution**: Check Node.js version (requires 24+) -- **Clear cache**: `rm -rf .next node_modules && npm install && npm run build` - -### Issue: Slow page loads -- **Solution**: Enable caching in API calls -- **Check**: Network latency to FastAPI backend -- **Verify**: CDN is serving static assets - ---- - -## Security Considerations - -- ✅ HTTPS enabled -- ✅ Security headers configured (X-Frame-Options, CSP, etc.) -- ✅ API keys in environment variables (never in code) -- ✅ CORS properly configured -- ✅ Rate limiting on API endpoints -- ✅ Regular security updates -- ✅ Dependency vulnerability scanning - ---- - -## Support - -For deployment issues, contact the DevOps team or refer to: -- [Next.js Deployment Docs](https://nextjs.org/docs/deployment) -- [Vercel Documentation](https://vercel.com/docs) -- [Docker Documentation](https://docs.docker.com/) +Earlier Vercel and standalone deployment recipes have been retired from this file +because they do not describe the current CMS, persistence and promotion setup. +See [development](../docs/DEVELOPMENT.md) for checks and +[publishing](docs/PUBLISHING.md) for CMS operations. diff --git a/nextjs-app/README.md b/nextjs-app/README.md index f91a2af..1cfb8e9 100644 --- a/nextjs-app/README.md +++ b/nextjs-app/README.md @@ -1,156 +1,58 @@ -# SchoolCompare Next.js Application +# SchoolCompare frontend and CMS -Modern Next.js application for comparing primary school KS2 performance across England. +Next.js App Router with React, TypeScript, CSS Modules, Chart.js, Leaflet and +Payload CMS. It serves school search, comparisons, rankings, school/place detail +pages and editorial content across England. -## Features +Start with the [repository overview](../README.md), +[architecture](../docs/ARCHITECTURE.md) and [development checks](../docs/DEVELOPMENT.md). -- **Server-Side Rendering (SSR)**: Fast initial page loads with pre-rendered content -- **Individual School Pages**: Dedicated pages for each school with full SEO optimization -- **Side-by-Side Comparison**: Compare up to 5 schools simultaneously -- **School Rankings**: Top-performing schools by various metrics -- **Interactive Maps**: Leaflet integration for geographic visualization -- **Performance Charts**: Chart.js visualizations for historical data -- **Responsive Design**: Mobile-first approach with full responsive support -- **SEO Optimized**: Dynamic sitemaps, meta tags, and structured data +## Source map -## Tech Stack +| Path | Purpose | +|---|---| +| `app/(frontend)/` | Public root layout, server pages and FastAPI proxy | +| `app/(payload)/` | Payload root layout, `/admin` and `/cms-api` | +| `app/robots.ts`, `app/opengraph-image.tsx`, root icons | Site-wide metadata endpoints | +| `components/` | Client views and reusable display components | +| `components/school/` | School detail sections | +| `lib/api.ts`, `lib/types.ts` | Fetch wrappers and manual school API types | +| `lib/schoolSections.ts`, `lib/compareLogic.ts` | Presentation decisions and data preparation | +| `context/`, `hooks/` | Comparison state, suggestion state and responsive behaviour | +| `collections/`, `blocks/`, `migrations/` | CMS schema and production migrations | +| `__tests__/` | Jest and React Testing Library tests | -- **Framework**: Next.js 16 (App Router) -- **Language**: TypeScript 5 -- **Styling**: CSS Modules + CSS Variables -- **State Management**: React Context API + URL state -- **Data Fetching**: SWR (client-side) + Next.js fetch (server-side) -- **Charts**: Chart.js + react-chartjs-2 -- **Maps**: Leaflet + react-leaflet -- **Testing**: Jest + React Testing Library -- **Validation**: Zod +Do not introduce a shared `app/layout.tsx`: public pages and Payload have separate +root layouts. Keep root metadata files outside the route groups. -## Getting Started +## Data and state -### Prerequisites +Server pages fetch initial data directly from `FASTAPI_URL`. Browser fetches use +`/api` by default, forwarded by `app/(frontend)/api/[...path]/route.ts`. +`FASTAPI_URL` must include `/api`. See `.env.example` for CMS and API settings. -- Node.js 24+ (using nvm recommended) -- FastAPI backend running on port 8000 +State uses React hooks/context, URL search parameters and localStorage for the +comparison basket. SWR is not installed. Maps use dynamic Leaflet wrappers. +Revalidation intervals are configured in fetch wrappers and pages; they vary by +resource. Backend reloads do not automatically invalidate every Next.js cache. -### Installation +## Commands -```bash -# Install dependencies -npm install - -# Copy environment variables -cp .env.example .env.local - -# Update .env.local with your configuration -``` - -### Development - -```bash -# Start development server -npm run dev - -# Open http://localhost:3000 -``` - -### Building - -```bash -# Build for production +```sh +npm ci +npm run typecheck +npm test -- --runInBand npm run build - -# Start production server -npm start ``` -### Testing +`test:watch` and `test:coverage` are also available. There is no `lint` script. +A running application needs the backend/data environment described in the +[development guide](../docs/DEVELOPMENT.md). -```bash -# Run tests -npm test +After CMS field or editor changes, run `npm run generate:importmap`. Keep +`payload-types.ts` generated from the CMS schema rather than editing it by hand. +The build must work without a database connection; avoid module-scope CMS queries +and DB-backed `generateStaticParams` functions. -# Run tests in watch mode -npm run test:watch - -# Run tests with coverage -npm run test:coverage -``` - -### Linting - -```bash -# Run ESLint -npm run lint -``` - -## Project Structure - -``` -nextjs-app/ -├── app/ # App Router pages -│ ├── layout.tsx # Root layout -│ ├── page.tsx # Home page -│ ├── compare/ # Compare page -│ ├── rankings/ # Rankings page -│ ├── school/[urn]/ # Individual school pages -│ ├── sitemap.ts # Dynamic sitemap -│ └── robots.ts # Robots.txt -├── components/ # React components -│ ├── SchoolCard.tsx # School card component -│ ├── FilterBar.tsx # Search/filter controls -│ ├── ComparisonView.tsx # Comparison interface -│ ├── RankingsView.tsx # Rankings table -│ └── ... -├── lib/ # Utility libraries -│ ├── api.ts # API client -│ ├── types.ts # TypeScript types -│ └── utils.ts # Helper functions -├── hooks/ # Custom React hooks -├── context/ # React Context providers -├── styles/ # Global styles -├── public/ # Static assets -└── __tests__/ # Test files -``` - -## Environment Variables - -| Variable | Description | Default | -|----------|-------------|---------| -| `NEXT_PUBLIC_API_URL` | Public API endpoint (client-side) | `http://localhost:8000/api` | -| `FASTAPI_URL` | Server-side API endpoint | `http://localhost:8000/api` | -| `NODE_ENV` | Environment mode | `development` | - -## Performance Optimizations - -- **Server-Side Rendering**: Initial HTML rendered on server -- **Static Generation**: Where possible, pages are pre-generated -- **Image Optimization**: Next.js Image component with AVIF/WebP support -- **Code Splitting**: Automatic route-based code splitting -- **Dynamic Imports**: Heavy components loaded on demand -- **API Caching**: Configurable revalidation for data fetching -- **Bundle Optimization**: Tree shaking and minification -- **Compression**: Gzip compression enabled - -## SEO Features - -- **Dynamic Meta Tags**: Generated per page with Next.js Metadata API -- **Open Graph**: Social media optimization -- **JSON-LD**: Structured data for search engines -- **Sitemap**: Auto-generated from database -- **Robots.txt**: Search engine crawling rules -- **Canonical URLs**: Duplicate content prevention - -## Browser Support - -- Chrome (latest) -- Firefox (latest) -- Safari (latest) -- Edge (latest) - -## License - -Proprietary - SchoolCompare - -## Support - -For issues and questions, please contact the development team. +See [publishing](docs/PUBLISHING.md) for CMS operations and +[deployment](../docs/DEPLOY.md) for staging and production promotion. From 1d8858fbda47ed272ca537065d960502e8bd336e Mon Sep 17 00:00:00 2001 From: Tudor Date: Mon, 14 Sep 2026 23:01:22 +0100 Subject: [PATCH 2/2] chore: remove the code the legacy CSV importer left behind MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `backend/migration.py` and `scripts/migrate_csv_to_db.py` import `School`, `SchoolResult`, `init_db` and `set_db_schema_version` — names that no longer exist. `scripts/geocode_schools.py` imports the same removed ORM model. None of the three can be imported against the current backend, so they were not dormant utilities anyone could fall back on; they were files that would fail on the first line. `backend/version.py` existed only to hand `SCHEMA_VERSION` to that importer, and the FastAPI lifespan performs no version-triggered import. Three symbols go with them, each confirmed to have no caller: the unvectorised `haversine_distance`, superseded by the inline NumPy calculation in search; `fetcher`, an SWR helper for a dependency this project does not install; and `kmToMiles`. `calculateDistance` stays — CutoffMapPanel uses it. Two comments pointed at `migrate_csv_to_db.py --drop` to explain why Payload owns its own schema. The reason survives the script: blog content must stay clear of the school marts and Airflow's metadata. Reworded rather than deleted, so the constraint keeps its justification. docs/LEGACY_CODE.md records what was removed and where to find it in history. It also records what was deliberately *not* removed, which is the more useful half: unused UI components awaiting a design decision, manual data utilities whose operators a repository search cannot see, and fallbacks that look obsolete but are load-bearing — `data_loader.py`'s older-mart branches, the generated GIAS dictionary copies, and the `legacy`-named dbt models that annual DAG selectors explicitly include. A zero-import count is evidence, not a verdict. The scripts that fetch DfE CSVs are marked historical and kept, pending confirmation that nobody runs them by hand. Checked: 190 backend tests, 429 frontend tests, `tsc --noEmit` clean. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_016y2J6bs8gbuSJbH18w7Tan --- backend/data_loader.py | 10 - backend/migration.py | 512 -------------------- backend/version.py | 26 - docs/LEGACY_CODE.md | 89 ++++ nextjs-app/__tests__/payload/routes.test.ts | 3 +- nextjs-app/lib/api.ts | 27 -- nextjs-app/payload.config.ts | 5 +- scripts/download_data.py | 3 + scripts/fetch_real_data.py | 3 + scripts/geocode_schools.py | 184 ------- scripts/migrate_csv_to_db.py | 68 --- 11 files changed, 98 insertions(+), 832 deletions(-) delete mode 100644 backend/migration.py delete mode 100644 backend/version.py create mode 100644 docs/LEGACY_CODE.md delete mode 100755 scripts/geocode_schools.py delete mode 100644 scripts/migrate_csv_to_db.py diff --git a/backend/data_loader.py b/backend/data_loader.py index cad154e..ebd8d05 100644 --- a/backend/data_loader.py +++ b/backend/data_loader.py @@ -188,16 +188,6 @@ def geocode_single_postcode(postcode: str) -> Optional[Tuple[float, float]]: return None -def haversine_distance(lat1: float, lon1: float, lat2: float, lon2: float) -> float: - """Calculate great-circle distance between two points (miles).""" - from math import radians, cos, sin, asin, sqrt - lat1, lon1, lat2, lon2 = map(radians, [lat1, lon1, lat2, lon2]) - dlat = lat2 - lat1 - dlon = lon2 - lon1 - a = sin(dlat / 2) ** 2 + cos(lat1) * cos(lat2) * sin(dlon / 2) ** 2 - return 2 * asin(sqrt(a)) * 3956 - - # ============================================================================= # MAIN DATA LOAD — joins dim_school + dim_location + fact_performance # fact_performance is a merged KS2+KS4 table (one row per URN per year). diff --git a/backend/migration.py b/backend/migration.py deleted file mode 100644 index 73dffea..0000000 --- a/backend/migration.py +++ /dev/null @@ -1,512 +0,0 @@ -""" -Database migration logic for importing CSV data. -Used by both CLI script and automatic startup migration. -""" - -import re -from pathlib import Path -from typing import Dict, Optional - -import numpy as np -import pandas as pd -import requests - -from .config import settings -from .database import Base, engine, get_db_session -from .models import School, SchoolResult -from .schemas import ( - COLUMN_MAPPINGS, - LA_CODE_TO_NAME, - NULL_VALUES, - SCHOOL_TYPE_MAP, -) - - -def parse_numeric(value) -> Optional[float]: - """Parse a numeric value, handling special cases.""" - if pd.isna(value): - return None - if isinstance(value, (int, float)): - return float(value) if not np.isnan(value) else None - str_val = str(value).strip().upper() - if str_val in NULL_VALUES or str_val == "": - return None - # Remove percentage signs if present - str_val = str_val.replace("%", "") - try: - return float(str_val) - except ValueError: - return None - - -def extract_year_from_folder(folder_name: str) -> Optional[int]: - """Extract year from folder name like '2023-2024'.""" - match = re.search(r"(\d{4})-(\d{4})", folder_name) - if match: - return int(match.group(2)) - match = re.search(r"(\d{4})", folder_name) - if match: - return int(match.group(1)) - return None - - -def geocode_postcodes_bulk(postcodes: list) -> Dict[str, tuple]: - """ - Geocode postcodes in bulk using postcodes.io API. - Returns dict of postcode -> (latitude, longitude). - """ - results = {} - valid_postcodes = [ - p.strip().upper() - for p in postcodes - if p and isinstance(p, str) and len(p.strip()) >= 5 - ] - valid_postcodes = list(set(valid_postcodes)) - - if not valid_postcodes: - return results - - batch_size = 100 - total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size - - for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)): - batch = valid_postcodes[batch_start : batch_start + batch_size] - print( - f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)..." - ) - - try: - response = requests.post( - "https://api.postcodes.io/postcodes", - json={"postcodes": batch}, - timeout=30, - ) - if response.status_code == 200: - data = response.json() - for item in data.get("result", []): - if item and item.get("result"): - pc = item["query"].upper() - lat = item["result"].get("latitude") - lon = item["result"].get("longitude") - if lat and lon: - results[pc] = (lat, lon) - except Exception as e: - print(f" Warning: Geocoding batch failed: {e}") - - return results - - -def load_csv_data(data_dir: Path) -> pd.DataFrame: - """Load all CSV data from data directory.""" - all_data = [] - - for folder in sorted(data_dir.iterdir()): - if not folder.is_dir(): - continue - - year = extract_year_from_folder(folder.name) - if not year: - continue - - # Specifically look for the KS2 results file - ks2_file = folder / "england_ks2final.csv" - if not ks2_file.exists(): - continue - - csv_file = ks2_file - print(f" Loading {csv_file.name} (year {year})...") - - try: - df = pd.read_csv(csv_file, encoding="latin-1", low_memory=False) - except Exception as e: - print(f" Error loading {csv_file}: {e}") - continue - - # Rename columns - df.rename(columns=COLUMN_MAPPINGS, inplace=True) - df["year"] = year - - # Handle local authority name - la_name_cols = ["LANAME", "LA (name)", "LA_NAME", "LA NAME"] - la_name_col = next((c for c in la_name_cols if c in df.columns), None) - - if la_name_col and la_name_col != "local_authority": - df["local_authority"] = df[la_name_col] - elif "LEA" in df.columns: - df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce") - df["local_authority"] = ( - df["local_authority_code"] - .map(LA_CODE_TO_NAME) - .fillna(df["LEA"].astype(str)) - ) - - # Store LEA code - if "LEA" in df.columns: - df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce") - - # Map school type - if "school_type_code" in df.columns: - df["school_type"] = ( - df["school_type_code"] - .map(SCHOOL_TYPE_MAP) - .fillna(df["school_type_code"]) - ) - - # Create combined address - addr_parts = ["address1", "address2", "town", "postcode"] - for col in addr_parts: - if col not in df.columns: - df[col] = None - - df["address"] = df.apply( - lambda r: ", ".join( - str(v) - for v in [ - r.get("address1"), - r.get("address2"), - r.get("town"), - r.get("postcode"), - ] - if pd.notna(v) and str(v).strip() - ), - axis=1, - ) - - all_data.append(df) - print(f" Loaded {len(df)} records") - - if all_data: - result = pd.concat(all_data, ignore_index=True) - print(f"\nTotal records loaded: {len(result)}") - print(f"Unique schools: {result['urn'].nunique()}") - print(f"Years: {sorted(result['year'].unique())}") - return result - - return pd.DataFrame() - - -def migrate_data(df: pd.DataFrame, geocode: bool = False, geocode_cache: dict = None): - """Migrate DataFrame data to database.""" - - if geocode_cache is None: - geocode_cache = {} - - # Clean URN column - convert to integer, drop invalid values - df = df.copy() - df["urn"] = pd.to_numeric(df["urn"], errors="coerce") - df = df.dropna(subset=["urn"]) - df["urn"] = df["urn"].astype(int) - - # Group by URN to get unique schools (use latest year's data) - school_data = ( - df.sort_values("year", ascending=False).groupby("urn").first().reset_index() - ) - print(f"\nMigrating {len(school_data)} unique schools...") - - # Geocode postcodes that aren't already in the cache - geocoded = dict(geocode_cache) # start with preserved coordinates - if geocode and "postcode" in df.columns: - cached_postcodes = { - str(row.get("postcode", "")).strip().upper() - for _, row in school_data.iterrows() - if int(float(str(row.get("urn", 0) or 0))) in geocode_cache - } - postcodes_needed = [ - p for p in df["postcode"].dropna().unique() - if str(p).strip().upper() not in cached_postcodes - ] - if postcodes_needed: - print(f"\nGeocoding {len(postcodes_needed)} postcodes ({len(geocode_cache)} restored from cache)...") - fresh = geocode_postcodes_bulk(postcodes_needed) - geocoded.update(fresh) - print(f" Successfully geocoded {len(fresh)} new postcodes") - else: - print(f"\nAll {len(geocode_cache)} postcodes restored from cache, skipping geocoding.") - - with get_db_session() as db: - # Create schools - urn_to_school_id = {} - schools_created = 0 - - for _, row in school_data.iterrows(): - # Safely parse URN - handle None, NaN, whitespace, and invalid values - urn_val = row.get("urn") - urn = None - if pd.notna(urn_val): - try: - urn_str = str(urn_val).strip() - if urn_str: - urn = int(float(urn_str)) # Handle "12345.0" format - except (ValueError, TypeError): - pass - if not urn: - continue - - # Skip if we've already added this URN (handles duplicates in source data) - if urn in urn_to_school_id: - continue - - # Get geocoding data - postcode = row.get("postcode") - lat, lon = None, None - if postcode and pd.notna(postcode): - coords = geocoded.get(str(postcode).strip().upper()) - if coords: - lat, lon = coords - - # Safely parse local_authority_code - la_code = None - la_code_val = row.get("local_authority_code") - if pd.notna(la_code_val): - try: - la_code_str = str(la_code_val).strip() - if la_code_str: - la_code = int(float(la_code_str)) - except (ValueError, TypeError): - pass - - school = School( - urn=urn, - school_name=row.get("school_name") - if pd.notna(row.get("school_name")) - else "Unknown", - local_authority=row.get("local_authority") - if pd.notna(row.get("local_authority")) - else None, - local_authority_code=la_code, - school_type=row.get("school_type") - if pd.notna(row.get("school_type")) - else None, - school_type_code=row.get("school_type_code") - if pd.notna(row.get("school_type_code")) - else None, - religious_denomination=row.get("religious_denomination") - if pd.notna(row.get("religious_denomination")) - else None, - age_range=row.get("age_range") - if pd.notna(row.get("age_range")) - else None, - address1=row.get("address1") if pd.notna(row.get("address1")) else None, - address2=row.get("address2") if pd.notna(row.get("address2")) else None, - town=row.get("town") if pd.notna(row.get("town")) else None, - postcode=row.get("postcode") if pd.notna(row.get("postcode")) else None, - latitude=lat, - longitude=lon, - ) - db.add(school) - db.flush() # Get the ID - urn_to_school_id[urn] = school.id - schools_created += 1 - - if schools_created % 1000 == 0: - print(f" Created {schools_created} schools...") - - print(f" Created {schools_created} schools") - - # Create results - print(f"\nMigrating {len(df)} yearly results...") - results_created = 0 - - for _, row in df.iterrows(): - # Safely parse URN - urn_val = row.get("urn") - urn = None - if pd.notna(urn_val): - try: - urn_str = str(urn_val).strip() - if urn_str: - urn = int(float(urn_str)) - except (ValueError, TypeError): - pass - if not urn or urn not in urn_to_school_id: - continue - - school_id = urn_to_school_id[urn] - - # Safely parse year - year_val = row.get("year") - year = None - if pd.notna(year_val): - try: - year = int(float(str(year_val).strip())) - except (ValueError, TypeError): - pass - if not year: - continue - - result = SchoolResult( - school_id=school_id, - year=year, - total_pupils=parse_numeric(row.get("total_pupils")), - eligible_pupils=parse_numeric(row.get("eligible_pupils")), - # Expected Standard - rwm_expected_pct=parse_numeric(row.get("rwm_expected_pct")), - reading_expected_pct=parse_numeric(row.get("reading_expected_pct")), - writing_expected_pct=parse_numeric(row.get("writing_expected_pct")), - maths_expected_pct=parse_numeric(row.get("maths_expected_pct")), - gps_expected_pct=parse_numeric(row.get("gps_expected_pct")), - science_expected_pct=parse_numeric(row.get("science_expected_pct")), - # Higher Standard - rwm_high_pct=parse_numeric(row.get("rwm_high_pct")), - reading_high_pct=parse_numeric(row.get("reading_high_pct")), - writing_high_pct=parse_numeric(row.get("writing_high_pct")), - maths_high_pct=parse_numeric(row.get("maths_high_pct")), - gps_high_pct=parse_numeric(row.get("gps_high_pct")), - # Progress - reading_progress=parse_numeric(row.get("reading_progress")), - writing_progress=parse_numeric(row.get("writing_progress")), - maths_progress=parse_numeric(row.get("maths_progress")), - # Averages - reading_avg_score=parse_numeric(row.get("reading_avg_score")), - maths_avg_score=parse_numeric(row.get("maths_avg_score")), - gps_avg_score=parse_numeric(row.get("gps_avg_score")), - # Context - disadvantaged_pct=parse_numeric(row.get("disadvantaged_pct")), - eal_pct=parse_numeric(row.get("eal_pct")), - sen_support_pct=parse_numeric(row.get("sen_support_pct")), - sen_ehcp_pct=parse_numeric(row.get("sen_ehcp_pct")), - stability_pct=parse_numeric(row.get("stability_pct")), - # Absence - reading_absence_pct=parse_numeric(row.get("reading_absence_pct")), - gps_absence_pct=parse_numeric(row.get("gps_absence_pct")), - maths_absence_pct=parse_numeric(row.get("maths_absence_pct")), - writing_absence_pct=parse_numeric(row.get("writing_absence_pct")), - science_absence_pct=parse_numeric(row.get("science_absence_pct")), - # Gender - rwm_expected_boys_pct=parse_numeric(row.get("rwm_expected_boys_pct")), - rwm_expected_girls_pct=parse_numeric(row.get("rwm_expected_girls_pct")), - rwm_high_boys_pct=parse_numeric(row.get("rwm_high_boys_pct")), - rwm_high_girls_pct=parse_numeric(row.get("rwm_high_girls_pct")), - # Disadvantaged - rwm_expected_disadvantaged_pct=parse_numeric( - row.get("rwm_expected_disadvantaged_pct") - ), - rwm_expected_non_disadvantaged_pct=parse_numeric( - row.get("rwm_expected_non_disadvantaged_pct") - ), - disadvantaged_gap=parse_numeric(row.get("disadvantaged_gap")), - # 3-Year - rwm_expected_3yr_pct=parse_numeric(row.get("rwm_expected_3yr_pct")), - reading_avg_3yr=parse_numeric(row.get("reading_avg_3yr")), - maths_avg_3yr=parse_numeric(row.get("maths_avg_3yr")), - ) - db.add(result) - results_created += 1 - - if results_created % 10000 == 0: - print(f" Created {results_created} results...") - db.flush() - - print(f" Created {results_created} results") - - # Commit all changes - db.commit() - print("\nMigration complete!") - - -def _apply_schema_alterations(): - """ - Add new columns to existing tables using ALTER TABLE … ADD COLUMN IF NOT EXISTS. - Safe to run on every migration — no-ops if the column already exists. - Add entries here whenever models.py gains new columns on an existing table. - """ - alterations = [ - # v4: Ofsted Report Card columns - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS framework VARCHAR(20)", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_safeguarding_met BOOLEAN", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_inclusion INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_curriculum_teaching INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_achievement INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_attendance_behaviour INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_personal_development INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_leadership_governance INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_early_years INTEGER", - "ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_sixth_form INTEGER", - ] - from sqlalchemy import text as sa_text - with engine.connect() as conn: - for stmt in alterations: - try: - conn.execute(sa_text(stmt)) - except Exception as e: - print(f" Warning: alteration skipped ({e})") - conn.commit() - - -def _apply_schema_drops(): - """ - Drop tables retired from the schema. Idempotent (DROP … IF EXISTS), so it's - safe to run on every migration. Add entries here when a model is removed. - """ - drops = [ - # v6: Ofsted Parent View feature removed - "DROP TABLE IF EXISTS marts.fact_parent_view CASCADE", - ] - from sqlalchemy import text as sa_text - with engine.connect() as conn: - for stmt in drops: - try: - conn.execute(sa_text(stmt)) - except Exception as e: - print(f" Warning: drop skipped ({e})") - conn.commit() - - -def run_full_migration(geocode: bool = False) -> bool: - """ - Run a complete migration: drop all tables and reimport from CSV. - - Returns True if successful, False if no data found. - Raises exception on error. - """ - # Preserve existing geocoding so a reimport doesn't throw away coordinates - # that took a long time to compute. - geocode_cache: dict[int, tuple[float, float]] = {} - inspector = __import__("sqlalchemy").inspect(engine) - if "schools" in inspector.get_table_names(): - try: - with get_db_session() as db: - rows = db.execute( - __import__("sqlalchemy").text( - "SELECT urn, latitude, longitude FROM schools " - "WHERE latitude IS NOT NULL AND longitude IS NOT NULL" - ) - ).fetchall() - geocode_cache = {r.urn: (r.latitude, r.longitude) for r in rows} - print(f" Saved {len(geocode_cache)} existing geocoded coordinates.") - except Exception as e: - print(f" Warning: could not save geocode cache: {e}") - - # Only drop the core KS2 tables — leave supplementary tables (ofsted, census, - # finance, etc.) intact so a reimport doesn't wipe integrator-populated data. - # schema_version is NOT dropped: it persists so restarts don't re-trigger migration. - ks2_tables = ["school_results", "schools"] - print(f"Dropping core tables: {ks2_tables} ...") - inspector = __import__("sqlalchemy").inspect(engine) - existing = set(inspector.get_table_names()) - for tname in ks2_tables: - if tname in existing: - Base.metadata.tables[tname].drop(bind=engine) - - print("Creating all tables...") - Base.metadata.create_all(bind=engine) - - # ALTER existing supplementary tables to add any new columns. - # create_all() only creates missing tables; it won't add columns to tables - # that already exist from an older schema version. These statements are - # idempotent (IF NOT EXISTS) so they're safe to run on every migration. - print("Applying column additions to supplementary tables...") - _apply_schema_alterations() - - print("Dropping retired tables...") - _apply_schema_drops() - - print("\nLoading CSV data...") - df = load_csv_data(settings.data_dir) - - if df.empty: - print("Warning: No CSV data found to migrate!") - return False - - migrate_data(df, geocode=geocode, geocode_cache=geocode_cache) - return True diff --git a/backend/version.py b/backend/version.py deleted file mode 100644 index 56aaffe..0000000 --- a/backend/version.py +++ /dev/null @@ -1,26 +0,0 @@ -""" -Schema versioning for database migrations. - -HOW TO USE: -- Bump SCHEMA_VERSION when making changes to database models -- This triggers an automatic full data reimport on next app startup - -WHEN TO BUMP: -- Adding/removing columns in models.py -- Changing column types or constraints -- Modifying CSV column mappings in schemas.py -- Any change that requires fresh data import -""" - -# Current schema version - increment when models change -SCHEMA_VERSION = 6 - -# Changelog for documentation -SCHEMA_CHANGELOG = { - 1: "Initial schema with School and SchoolResult tables", - 2: "Added pupil absence fields (reading, maths, gps, writing, science)", - 3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools", - 4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)", - 5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables", - 6: "Removed the Ofsted Parent View feature: dropped fact_parent_view table and model", -} diff --git a/docs/LEGACY_CODE.md b/docs/LEGACY_CODE.md new file mode 100644 index 0000000..165eb09 --- /dev/null +++ b/docs/LEGACY_CODE.md @@ -0,0 +1,89 @@ +# Legacy and unused-code inventory + +Reviewed 2026-09-14. This inventory records source evidence, not production usage +telemetry. A command with no repository caller may still be run manually or from +an external scheduler. Historical specs and prototypes are not runtime imports. + +## Method and scope + +Searched backend imports, tests, CLI scripts, Airflow DAGs, Meltano configuration, +Gitea workflows, Dockerfiles and documentation. For frontend candidates, inspected +TypeScript imports, re-exports, literal dynamic imports and `require` calls, +resolving relative and `@/` paths while excluding tests, dependencies and build +output. Checked candidates again with text searches including tests. + +Next.js route files, generated Payload import-map entries and plugin discovery +are entry points even without ordinary imports. This is why a zero-import count +alone is not sufficient grounds for deletion. Computed imports and external +operators are outside this static audit. + +## Removed in this cleanup + +These names are recorded for Git-history lookup; they are no longer file links. + +| Removed path or symbol | Evidence and replacement | +|---|---| +| `backend/migration.py` | Imported `School` and `SchoolResult`, which no longer exist in `backend/models.py`. Only the legacy CSV CLI imported it. Current tables are built by dbt. | +| `backend/version.py` | Only the legacy importer consumed `SCHEMA_VERSION`. FastAPI lifespan does not perform version-triggered imports. This is unrelated to active Payload migrations. | +| `scripts/migrate_csv_to_db.py` | Imported removed `init_db`/`set_db_schema_version` helpers and the obsolete models indirectly. No runtime, DAG or workflow calls it. Use the managed pipeline for current marts. | +| `scripts/geocode_schools.py` | Imported the removed `School` ORM model. No pipeline/workflow calls it. Coordinates now come from GIAS/PostGIS; a separate mart-aware manual utility remains under `pipeline/scripts/`. | +| `backend.data_loader.haversine_distance` | No callers. Search uses its inline vectorised NumPy calculation. | +| `nextjs-app/lib/api.ts: fetcher` | No callers; SWR is not installed. Application fetches use the named API wrappers. | +| `nextjs-app/lib/api.ts: kmToMiles` | No callers. `calculateDistance` remains because `CutoffMapPanel` uses it. | + +The removed command files could not import successfully against the current +backend. This cleanup does not run replacements, migrate data or modify databases. +Their previous implementations remain recoverable from Git history. + +## Unused candidates retained for a separate cleanup + +| Candidate | Evidence | Recommended next step | +|---|---|---| +| `nextjs-app/components/LoadingSkeleton.tsx` and its CSS | No application or test imports found. | Remove together after confirming no planned use. | +| `nextjs-app/components/Pagination.tsx` and its CSS | No application or test imports found; HomeView implements load-more behaviour. | Remove as a pair if numbered pagination will not return. | +| `nextjs-app/components/SchoolCard.tsx` and its CSS | Imported by its own tests, not application code. HomeView uses SchoolRow/SecondarySchoolRow. | Decide whether to retire the card design; if removed, remove its dedicated tests as well. Passing tests do not establish runtime use. | +| `backend/database.py: get_db`, `get_db_session` | No remaining callers after removing the importer. Current code creates SessionLocal directly. | Either adopt these helpers during session-lifecycle cleanup or remove them; do not rewrite active sessions in a documentation change. | +| `backend/schemas.py: COLUMN_MAPPINGS`, `NULL_VALUES`, `LA_CODE_TO_NAME` | No remaining Python consumers found after importer removal. Other constants in this module are active. | Remove individual constants after checking external data utilities; retain the module. | +| `backend/config.py: data_dir`, `max_page_size`, `rate_limit_burst` | No active consumers found. `default_page_size` appears only in a branch that expects None, although the route supplies a concrete default. | Reconcile settings with route validation in a focused API change. | + +## Legacy/manual paths requiring operational verification + +| Path | Status and reason to retain for now | +|---|---| +| FastAPI `/`, `/compare`, `/rankings`, `/favicon.svg`, `/robots.txt`, and conditional `/static` | Old frontend-serving routes reference a `frontend/` directory absent from the checkout and backend image. Next.js owns these public surfaces. Removal changes externally callable routes, so first check proxy/operator usage and define replacement responses. | +| `scripts/fetch_real_data.py`, `scripts/download_data.py` | Historical standalone CSV utilities. The fetch script targets Wandsworth/Merton; neither is wired into the managed pipeline. Marked historical, retained pending confirmation of manual use. | +| `pipeline/scripts/geocode_postcodes.py` | Mart-aware postcode fallback, not called by the current DAGs. Do not confuse it with the removed legacy ORM geocoder. Verify the target schema before manual use. | +| `docker-compose.yml` | Uses unpublished `:latest` release tags and lacks frontend Payload DB/secret/media configuration. Retained as an old development topology, not recommended onboarding. | +| `nextjs-app/docker-compose.yml` | Standalone legacy recipe with old backend port assumptions and no CMS persistence setup. Retained until its consumers are checked. | +| `MIGRATION_SUMMARY.md`, `docs/superpowers/`, `mockups/` | Historical designs and prototypes. Retain as history; do not follow as current deployment instructions. | +| `scripts/sql/drop_fact_parent_view.sql` | One-off maintenance SQL. Not an application entry point; repository call-site searches cannot establish whether it is still needed operationally. | + +## Active code that can look obsolete + +- `backend/data_loader.py` older-mart query fallbacks are covered by backend tests + and support databases at different migration stages. Remove only after verifying + the deployed schemas in every supported environment. +- `backend/gias_codes.py` and `pipeline/scripts/gias_codes.py` are intentionally + generated copies for separate runtime images. Their parity is tested. +- `nextjs-app/migrations/`, `payload-types.ts` and the Payload import map are active + CMS artifacts, not remnants of the removed school importer. +- `get_available_years`, `get_available_local_authorities` and `get_schools_count` + in `data_loader.py` are called through `get_data_info`, which serves the backend + data-info endpoint. They are not dead functions. +- `get_supplementary_data` is an intentional single-school wrapper around the + batch implementation. +- `pipeline/transform` models named `legacy` can be active data sources: annual + DAG selectors explicitly include legacy KS2/KS4 lineage. Names alone do not + establish obsolescence. + +## Suggested next passes + +1. Decide the fate of the three unused UI components and remove paired assets/tests. +2. Consolidate backend session usage and remove abandoned settings/constants. +3. Verify external consumers, then retire static-serving API routes and old compose recipes. +4. Audit manual data utilities with pipeline operators before deleting them. +5. Revisit compatibility fallbacks only after documenting supported schema versions. + +Validation for this cleanup should include frontend typechecking/tests, Python +syntax checks, reference searches and documentation link checks. Live database, +external scheduler and deployed route usage require separate integration evidence. diff --git a/nextjs-app/__tests__/payload/routes.test.ts b/nextjs-app/__tests__/payload/routes.test.ts index 8356450..6085796 100644 --- a/nextjs-app/__tests__/payload/routes.test.ts +++ b/nextjs-app/__tests__/payload/routes.test.ts @@ -38,8 +38,7 @@ describe('payload mount points', () => { }); it('isolates CMS tables in their own postgres schema', () => { - // Blog content must sit outside `public`, where the app tables, Airflow's - // metadata and scripts/migrate_csv_to_db.py --drop all live. + // Blog content must stay separate from school marts and Airflow metadata. expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/); }); }); diff --git a/nextjs-app/lib/api.ts b/nextjs-app/lib/api.ts index 016750b..33df0e4 100644 --- a/nextjs-app/lib/api.ts +++ b/nextjs-app/lib/api.ts @@ -311,26 +311,6 @@ export async function fetchDataInfo( return handleResponse(response); } -// ============================================================================ -// Client-Side Fetcher (for SWR) -// ============================================================================ - -/** - * Generic fetcher function for use with SWR - * @example - * ```tsx - * const { data, error } = useSWR('/api/schools', fetcher); - * ``` - */ -export async function fetcher(url: string): Promise { - // If it's already a full URL, use it directly - // Otherwise, prepend the API_BASE_URL - const fullUrl = url.startsWith('http') ? url : `${API_BASE_URL}${url.startsWith('/') ? url : `/${url}`}`; - - const response = await fetch(fullUrl); - return handleResponse(response); -} - // ============================================================================ // Geocoding API // ============================================================================ @@ -396,10 +376,3 @@ export function calculateDistance( const c = 2 * Math.atan2(Math.sqrt(a), Math.sqrt(1 - a)); return R * c; } - -/** - * Convert kilometers to miles - */ -export function kmToMiles(km: number): number { - return km * 0.621371; -} diff --git a/nextjs-app/payload.config.ts b/nextjs-app/payload.config.ts index 5d53597..2887ef3 100644 --- a/nextjs-app/payload.config.ts +++ b/nextjs-app/payload.config.ts @@ -24,9 +24,8 @@ export default buildConfig({ typescript: { outputFile: path.resolve(dirname, 'payload-types.ts') }, db: postgresAdapter({ pool: { connectionString: process.env.DATABASE_URL }, - // Its own schema, so no pipeline operation on `public` can reach blog - // content. scripts/migrate_csv_to_db.py --drop lives in that blast radius, - // as does Airflow's metadata. The schema itself is created by the initial + // Its own schema separates blog content from pipeline-managed school + // tables and Airflow metadata. The schema is created by the initial // migration: schemaName says where tables go, it does not create anything. // // prodMigrations runs pending migrations during server init. Without it a diff --git a/scripts/download_data.py b/scripts/download_data.py index 4f2af46..a7350a3 100644 --- a/scripts/download_data.py +++ b/scripts/download_data.py @@ -1,5 +1,8 @@ #!/usr/bin/env python3 """ +Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline. +See docs/LEGACY_CODE.md before using it for current school data. + Data Download Helper Script This script provides instructions and utilities for downloading diff --git a/scripts/fetch_real_data.py b/scripts/fetch_real_data.py index 0507cc9..38a9876 100644 --- a/scripts/fetch_real_data.py +++ b/scripts/fetch_real_data.py @@ -1,5 +1,8 @@ #!/usr/bin/env python3 """ +Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline. +See docs/LEGACY_CODE.md before using it for current school data. + Fetch real school performance data from UK Government sources. This script downloads KS2 (Key Stage 2) primary school data from: diff --git a/scripts/geocode_schools.py b/scripts/geocode_schools.py deleted file mode 100755 index 9468bab..0000000 --- a/scripts/geocode_schools.py +++ /dev/null @@ -1,184 +0,0 @@ -#!/usr/bin/env python3 -""" -Geocode all school postcodes and update the database. - -This script should be run as a weekly cron job to ensure all schools -have up-to-date latitude/longitude coordinates. - -Usage: - python scripts/geocode_schools.py [--force] - -Options: - --force Re-geocode all postcodes, even if already geocoded - -Crontab example (run every Sunday at 2am): - 0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1 -""" - -import argparse -import sys -from datetime import datetime -from pathlib import Path -from typing import Dict, Tuple - -import requests - -# Add parent directory to path for imports -sys.path.insert(0, str(Path(__file__).parent.parent)) - -from backend.database import SessionLocal -from backend.models import School - - -def geocode_postcodes_bulk(postcodes: list) -> Dict[str, Tuple[float, float]]: - """ - Geocode postcodes in bulk using postcodes.io API. - Returns dict of postcode -> (latitude, longitude). - """ - results = {} - valid_postcodes = [ - p.strip().upper() - for p in postcodes - if p and isinstance(p, str) and len(p.strip()) >= 5 - ] - valid_postcodes = list(set(valid_postcodes)) - - if not valid_postcodes: - return results - - batch_size = 100 - total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size - - for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)): - batch = valid_postcodes[batch_start : batch_start + batch_size] - print(f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)...") - - try: - response = requests.post( - "https://api.postcodes.io/postcodes", - json={"postcodes": batch}, - timeout=30, - ) - if response.status_code == 200: - data = response.json() - for item in data.get("result", []): - if item and item.get("result"): - pc = item["query"].upper() - lat = item["result"].get("latitude") - lon = item["result"].get("longitude") - if lat and lon: - results[pc] = (lat, lon) - else: - print(f" Warning: API returned status {response.status_code}") - except Exception as e: - print(f" Warning: Geocoding batch failed: {e}") - - return results - - -def geocode_schools(force: bool = False) -> None: - """ - Geocode all schools in the database. - - Args: - force: If True, re-geocode all postcodes even if already geocoded - """ - print(f"\n{'='*60}") - print(f"School Geocoding Job - {datetime.now().isoformat()}") - print(f"{'='*60}\n") - - db = SessionLocal() - - try: - # Get schools that need geocoding - if force: - schools = db.query(School).filter(School.postcode.isnot(None)).all() - print(f"Force mode: Processing all {len(schools)} schools with postcodes") - else: - schools = db.query(School).filter( - School.postcode.isnot(None), - (School.latitude.is_(None)) | (School.longitude.is_(None)) - ).all() - print(f"Found {len(schools)} schools without coordinates") - - if not schools: - print("No schools to geocode. Exiting.") - return - - # Extract unique postcodes - postcodes = list(set( - s.postcode.strip().upper() - for s in schools - if s.postcode - )) - print(f"Unique postcodes to geocode: {len(postcodes)}") - - # Geocode in bulk - print("\nGeocoding postcodes...") - geocoded = geocode_postcodes_bulk(postcodes) - print(f"Successfully geocoded: {len(geocoded)} postcodes") - - # Update database - print("\nUpdating database...") - updated_count = 0 - failed_count = 0 - - for school in schools: - if not school.postcode: - continue - - pc_upper = school.postcode.strip().upper() - coords = geocoded.get(pc_upper) - - if coords: - school.latitude = coords[0] - school.longitude = coords[1] - updated_count += 1 - else: - failed_count += 1 - - db.commit() - - print(f"\nResults:") - print(f" - Updated: {updated_count} schools") - print(f" - Failed (invalid/not found): {failed_count} postcodes") - - # Summary stats - total_with_coords = db.query(School).filter( - School.latitude.isnot(None), - School.longitude.isnot(None) - ).count() - total_schools = db.query(School).count() - - print(f"\nDatabase summary:") - print(f" - Total schools: {total_schools}") - print(f" - Schools with coordinates: {total_with_coords}") - print(f" - Coverage: {100*total_with_coords/total_schools:.1f}%") - - except Exception as e: - print(f"Error during geocoding: {e}") - db.rollback() - raise - finally: - db.close() - print(f"\n{'='*60}") - print(f"Geocoding job completed - {datetime.now().isoformat()}") - print(f"{'='*60}\n") - - -def main(): - parser = argparse.ArgumentParser( - description="Geocode school postcodes and update database" - ) - parser.add_argument( - "--force", - action="store_true", - help="Re-geocode all postcodes, even if already geocoded" - ) - args = parser.parse_args() - - geocode_schools(force=args.force) - - -if __name__ == "__main__": - main() diff --git a/scripts/migrate_csv_to_db.py b/scripts/migrate_csv_to_db.py deleted file mode 100644 index c4879d2..0000000 --- a/scripts/migrate_csv_to_db.py +++ /dev/null @@ -1,68 +0,0 @@ -#!/usr/bin/env python3 -""" -CLI script for manual database migration. - -Usage: - python scripts/migrate_csv_to_db.py [--drop] [--geocode] - -Options: - --drop Drop existing tables before migration (full reimport) - --geocode Geocode postcodes (requires network access) -""" - -import sys -from pathlib import Path - -# Add parent directory to path for imports -sys.path.insert(0, str(Path(__file__).parent.parent)) - -import argparse - -from backend.config import settings -from backend.database import Base, engine, init_db, set_db_schema_version -from backend.migration import load_csv_data, migrate_data, run_full_migration -from backend.version import SCHEMA_VERSION - - -def main(): - parser = argparse.ArgumentParser( - description="Migrate CSV data to PostgreSQL database" - ) - parser.add_argument( - "--drop", action="store_true", help="Drop existing tables before migration" - ) - parser.add_argument("--geocode", action="store_true", help="Geocode postcodes") - args = parser.parse_args() - - print("=" * 60) - print("School Data Migration: CSV -> PostgreSQL") - print("=" * 60) - print(f"\nDatabase: {settings.database_url.split('@')[-1]}") - print(f"Data directory: {settings.data_dir}") - print(f"Target schema version: {SCHEMA_VERSION}") - - if args.drop: - print("\nRunning full migration (drop + reimport)...") - success = run_full_migration(geocode=args.geocode) - else: - print("\nCreating tables (preserving existing data)...") - init_db() - print("\nLoading CSV data...") - df = load_csv_data(settings.data_dir) - if df.empty: - print("No data found to migrate!") - return 1 - migrate_data(df, geocode=args.geocode) - success = True - - if success: - # Ensure schema_version table exists - init_db() - set_db_schema_version(SCHEMA_VERSION) - print(f"\nSchema version set to {SCHEMA_VERSION}") - - return 0 if success else 1 - - -if __name__ == "__main__": - sys.exit(main())