chore: remove the code the legacy CSV importer left behind
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m11s
PR Checks / Backend Smoke (pull_request) Successful in 9s
PR Checks / Build Backend (no push) (pull_request) Successful in 18s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m18s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 11s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 1m2s
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m11s
PR Checks / Backend Smoke (pull_request) Successful in 9s
PR Checks / Build Backend (no push) (pull_request) Successful in 18s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m18s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 11s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 1m2s
`backend/migration.py` and `scripts/migrate_csv_to_db.py` import `School`, `SchoolResult`, `init_db` and `set_db_schema_version` — names that no longer exist. `scripts/geocode_schools.py` imports the same removed ORM model. None of the three can be imported against the current backend, so they were not dormant utilities anyone could fall back on; they were files that would fail on the first line. `backend/version.py` existed only to hand `SCHEMA_VERSION` to that importer, and the FastAPI lifespan performs no version-triggered import. Three symbols go with them, each confirmed to have no caller: the unvectorised `haversine_distance`, superseded by the inline NumPy calculation in search; `fetcher`, an SWR helper for a dependency this project does not install; and `kmToMiles`. `calculateDistance` stays — CutoffMapPanel uses it. Two comments pointed at `migrate_csv_to_db.py --drop` to explain why Payload owns its own schema. The reason survives the script: blog content must stay clear of the school marts and Airflow's metadata. Reworded rather than deleted, so the constraint keeps its justification. docs/LEGACY_CODE.md records what was removed and where to find it in history. It also records what was deliberately *not* removed, which is the more useful half: unused UI components awaiting a design decision, manual data utilities whose operators a repository search cannot see, and fallbacks that look obsolete but are load-bearing — `data_loader.py`'s older-mart branches, the generated GIAS dictionary copies, and the `legacy`-named dbt models that annual DAG selectors explicitly include. A zero-import count is evidence, not a verdict. The scripts that fetch DfE CSVs are marked historical and kept, pending confirmation that nobody runs them by hand. Checked: 190 backend tests, 429 frontend tests, `tsc --noEmit` clean. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016y2J6bs8gbuSJbH18w7Tan
This commit is contained in:
1 parent
eaf5e5d180
commit
1d8858fbda
11 files changed
+98
-832
No files matched your search
@@ -188,16 +188,6 @@ def geocode_single_postcode(postcode: str) -> Optional[Tuple[float, float]]:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def haversine_distance(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
|
|
||||||
"""Calculate great-circle distance between two points (miles)."""
|
|
||||||
from math import radians, cos, sin, asin, sqrt
|
|
||||||
lat1, lon1, lat2, lon2 = map(radians, [lat1, lon1, lat2, lon2])
|
|
||||||
dlat = lat2 - lat1
|
|
||||||
dlon = lon2 - lon1
|
|
||||||
a = sin(dlat / 2) ** 2 + cos(lat1) * cos(lat2) * sin(dlon / 2) ** 2
|
|
||||||
return 2 * asin(sqrt(a)) * 3956
|
|
||||||
|
|
||||||
|
|
||||||
# =============================================================================
|
# =============================================================================
|
||||||
# MAIN DATA LOAD — joins dim_school + dim_location + fact_performance
|
# MAIN DATA LOAD — joins dim_school + dim_location + fact_performance
|
||||||
# fact_performance is a merged KS2+KS4 table (one row per URN per year).
|
# fact_performance is a merged KS2+KS4 table (one row per URN per year).
|
||||||
|
|||||||
@@ -1,512 +0,0 @@
|
|||||||
"""
|
|
||||||
Database migration logic for importing CSV data.
|
|
||||||
Used by both CLI script and automatic startup migration.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import re
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Dict, Optional
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
|
|
||||||
from .config import settings
|
|
||||||
from .database import Base, engine, get_db_session
|
|
||||||
from .models import School, SchoolResult
|
|
||||||
from .schemas import (
|
|
||||||
COLUMN_MAPPINGS,
|
|
||||||
LA_CODE_TO_NAME,
|
|
||||||
NULL_VALUES,
|
|
||||||
SCHOOL_TYPE_MAP,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def parse_numeric(value) -> Optional[float]:
|
|
||||||
"""Parse a numeric value, handling special cases."""
|
|
||||||
if pd.isna(value):
|
|
||||||
return None
|
|
||||||
if isinstance(value, (int, float)):
|
|
||||||
return float(value) if not np.isnan(value) else None
|
|
||||||
str_val = str(value).strip().upper()
|
|
||||||
if str_val in NULL_VALUES or str_val == "":
|
|
||||||
return None
|
|
||||||
# Remove percentage signs if present
|
|
||||||
str_val = str_val.replace("%", "")
|
|
||||||
try:
|
|
||||||
return float(str_val)
|
|
||||||
except ValueError:
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def extract_year_from_folder(folder_name: str) -> Optional[int]:
|
|
||||||
"""Extract year from folder name like '2023-2024'."""
|
|
||||||
match = re.search(r"(\d{4})-(\d{4})", folder_name)
|
|
||||||
if match:
|
|
||||||
return int(match.group(2))
|
|
||||||
match = re.search(r"(\d{4})", folder_name)
|
|
||||||
if match:
|
|
||||||
return int(match.group(1))
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def geocode_postcodes_bulk(postcodes: list) -> Dict[str, tuple]:
|
|
||||||
"""
|
|
||||||
Geocode postcodes in bulk using postcodes.io API.
|
|
||||||
Returns dict of postcode -> (latitude, longitude).
|
|
||||||
"""
|
|
||||||
results = {}
|
|
||||||
valid_postcodes = [
|
|
||||||
p.strip().upper()
|
|
||||||
for p in postcodes
|
|
||||||
if p and isinstance(p, str) and len(p.strip()) >= 5
|
|
||||||
]
|
|
||||||
valid_postcodes = list(set(valid_postcodes))
|
|
||||||
|
|
||||||
if not valid_postcodes:
|
|
||||||
return results
|
|
||||||
|
|
||||||
batch_size = 100
|
|
||||||
total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size
|
|
||||||
|
|
||||||
for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)):
|
|
||||||
batch = valid_postcodes[batch_start : batch_start + batch_size]
|
|
||||||
print(
|
|
||||||
f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)..."
|
|
||||||
)
|
|
||||||
|
|
||||||
try:
|
|
||||||
response = requests.post(
|
|
||||||
"https://api.postcodes.io/postcodes",
|
|
||||||
json={"postcodes": batch},
|
|
||||||
timeout=30,
|
|
||||||
)
|
|
||||||
if response.status_code == 200:
|
|
||||||
data = response.json()
|
|
||||||
for item in data.get("result", []):
|
|
||||||
if item and item.get("result"):
|
|
||||||
pc = item["query"].upper()
|
|
||||||
lat = item["result"].get("latitude")
|
|
||||||
lon = item["result"].get("longitude")
|
|
||||||
if lat and lon:
|
|
||||||
results[pc] = (lat, lon)
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: Geocoding batch failed: {e}")
|
|
||||||
|
|
||||||
return results
|
|
||||||
|
|
||||||
|
|
||||||
def load_csv_data(data_dir: Path) -> pd.DataFrame:
|
|
||||||
"""Load all CSV data from data directory."""
|
|
||||||
all_data = []
|
|
||||||
|
|
||||||
for folder in sorted(data_dir.iterdir()):
|
|
||||||
if not folder.is_dir():
|
|
||||||
continue
|
|
||||||
|
|
||||||
year = extract_year_from_folder(folder.name)
|
|
||||||
if not year:
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Specifically look for the KS2 results file
|
|
||||||
ks2_file = folder / "england_ks2final.csv"
|
|
||||||
if not ks2_file.exists():
|
|
||||||
continue
|
|
||||||
|
|
||||||
csv_file = ks2_file
|
|
||||||
print(f" Loading {csv_file.name} (year {year})...")
|
|
||||||
|
|
||||||
try:
|
|
||||||
df = pd.read_csv(csv_file, encoding="latin-1", low_memory=False)
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Error loading {csv_file}: {e}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Rename columns
|
|
||||||
df.rename(columns=COLUMN_MAPPINGS, inplace=True)
|
|
||||||
df["year"] = year
|
|
||||||
|
|
||||||
# Handle local authority name
|
|
||||||
la_name_cols = ["LANAME", "LA (name)", "LA_NAME", "LA NAME"]
|
|
||||||
la_name_col = next((c for c in la_name_cols if c in df.columns), None)
|
|
||||||
|
|
||||||
if la_name_col and la_name_col != "local_authority":
|
|
||||||
df["local_authority"] = df[la_name_col]
|
|
||||||
elif "LEA" in df.columns:
|
|
||||||
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
|
|
||||||
df["local_authority"] = (
|
|
||||||
df["local_authority_code"]
|
|
||||||
.map(LA_CODE_TO_NAME)
|
|
||||||
.fillna(df["LEA"].astype(str))
|
|
||||||
)
|
|
||||||
|
|
||||||
# Store LEA code
|
|
||||||
if "LEA" in df.columns:
|
|
||||||
df["local_authority_code"] = pd.to_numeric(df["LEA"], errors="coerce")
|
|
||||||
|
|
||||||
# Map school type
|
|
||||||
if "school_type_code" in df.columns:
|
|
||||||
df["school_type"] = (
|
|
||||||
df["school_type_code"]
|
|
||||||
.map(SCHOOL_TYPE_MAP)
|
|
||||||
.fillna(df["school_type_code"])
|
|
||||||
)
|
|
||||||
|
|
||||||
# Create combined address
|
|
||||||
addr_parts = ["address1", "address2", "town", "postcode"]
|
|
||||||
for col in addr_parts:
|
|
||||||
if col not in df.columns:
|
|
||||||
df[col] = None
|
|
||||||
|
|
||||||
df["address"] = df.apply(
|
|
||||||
lambda r: ", ".join(
|
|
||||||
str(v)
|
|
||||||
for v in [
|
|
||||||
r.get("address1"),
|
|
||||||
r.get("address2"),
|
|
||||||
r.get("town"),
|
|
||||||
r.get("postcode"),
|
|
||||||
]
|
|
||||||
if pd.notna(v) and str(v).strip()
|
|
||||||
),
|
|
||||||
axis=1,
|
|
||||||
)
|
|
||||||
|
|
||||||
all_data.append(df)
|
|
||||||
print(f" Loaded {len(df)} records")
|
|
||||||
|
|
||||||
if all_data:
|
|
||||||
result = pd.concat(all_data, ignore_index=True)
|
|
||||||
print(f"\nTotal records loaded: {len(result)}")
|
|
||||||
print(f"Unique schools: {result['urn'].nunique()}")
|
|
||||||
print(f"Years: {sorted(result['year'].unique())}")
|
|
||||||
return result
|
|
||||||
|
|
||||||
return pd.DataFrame()
|
|
||||||
|
|
||||||
|
|
||||||
def migrate_data(df: pd.DataFrame, geocode: bool = False, geocode_cache: dict = None):
|
|
||||||
"""Migrate DataFrame data to database."""
|
|
||||||
|
|
||||||
if geocode_cache is None:
|
|
||||||
geocode_cache = {}
|
|
||||||
|
|
||||||
# Clean URN column - convert to integer, drop invalid values
|
|
||||||
df = df.copy()
|
|
||||||
df["urn"] = pd.to_numeric(df["urn"], errors="coerce")
|
|
||||||
df = df.dropna(subset=["urn"])
|
|
||||||
df["urn"] = df["urn"].astype(int)
|
|
||||||
|
|
||||||
# Group by URN to get unique schools (use latest year's data)
|
|
||||||
school_data = (
|
|
||||||
df.sort_values("year", ascending=False).groupby("urn").first().reset_index()
|
|
||||||
)
|
|
||||||
print(f"\nMigrating {len(school_data)} unique schools...")
|
|
||||||
|
|
||||||
# Geocode postcodes that aren't already in the cache
|
|
||||||
geocoded = dict(geocode_cache) # start with preserved coordinates
|
|
||||||
if geocode and "postcode" in df.columns:
|
|
||||||
cached_postcodes = {
|
|
||||||
str(row.get("postcode", "")).strip().upper()
|
|
||||||
for _, row in school_data.iterrows()
|
|
||||||
if int(float(str(row.get("urn", 0) or 0))) in geocode_cache
|
|
||||||
}
|
|
||||||
postcodes_needed = [
|
|
||||||
p for p in df["postcode"].dropna().unique()
|
|
||||||
if str(p).strip().upper() not in cached_postcodes
|
|
||||||
]
|
|
||||||
if postcodes_needed:
|
|
||||||
print(f"\nGeocoding {len(postcodes_needed)} postcodes ({len(geocode_cache)} restored from cache)...")
|
|
||||||
fresh = geocode_postcodes_bulk(postcodes_needed)
|
|
||||||
geocoded.update(fresh)
|
|
||||||
print(f" Successfully geocoded {len(fresh)} new postcodes")
|
|
||||||
else:
|
|
||||||
print(f"\nAll {len(geocode_cache)} postcodes restored from cache, skipping geocoding.")
|
|
||||||
|
|
||||||
with get_db_session() as db:
|
|
||||||
# Create schools
|
|
||||||
urn_to_school_id = {}
|
|
||||||
schools_created = 0
|
|
||||||
|
|
||||||
for _, row in school_data.iterrows():
|
|
||||||
# Safely parse URN - handle None, NaN, whitespace, and invalid values
|
|
||||||
urn_val = row.get("urn")
|
|
||||||
urn = None
|
|
||||||
if pd.notna(urn_val):
|
|
||||||
try:
|
|
||||||
urn_str = str(urn_val).strip()
|
|
||||||
if urn_str:
|
|
||||||
urn = int(float(urn_str)) # Handle "12345.0" format
|
|
||||||
except (ValueError, TypeError):
|
|
||||||
pass
|
|
||||||
if not urn:
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Skip if we've already added this URN (handles duplicates in source data)
|
|
||||||
if urn in urn_to_school_id:
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Get geocoding data
|
|
||||||
postcode = row.get("postcode")
|
|
||||||
lat, lon = None, None
|
|
||||||
if postcode and pd.notna(postcode):
|
|
||||||
coords = geocoded.get(str(postcode).strip().upper())
|
|
||||||
if coords:
|
|
||||||
lat, lon = coords
|
|
||||||
|
|
||||||
# Safely parse local_authority_code
|
|
||||||
la_code = None
|
|
||||||
la_code_val = row.get("local_authority_code")
|
|
||||||
if pd.notna(la_code_val):
|
|
||||||
try:
|
|
||||||
la_code_str = str(la_code_val).strip()
|
|
||||||
if la_code_str:
|
|
||||||
la_code = int(float(la_code_str))
|
|
||||||
except (ValueError, TypeError):
|
|
||||||
pass
|
|
||||||
|
|
||||||
school = School(
|
|
||||||
urn=urn,
|
|
||||||
school_name=row.get("school_name")
|
|
||||||
if pd.notna(row.get("school_name"))
|
|
||||||
else "Unknown",
|
|
||||||
local_authority=row.get("local_authority")
|
|
||||||
if pd.notna(row.get("local_authority"))
|
|
||||||
else None,
|
|
||||||
local_authority_code=la_code,
|
|
||||||
school_type=row.get("school_type")
|
|
||||||
if pd.notna(row.get("school_type"))
|
|
||||||
else None,
|
|
||||||
school_type_code=row.get("school_type_code")
|
|
||||||
if pd.notna(row.get("school_type_code"))
|
|
||||||
else None,
|
|
||||||
religious_denomination=row.get("religious_denomination")
|
|
||||||
if pd.notna(row.get("religious_denomination"))
|
|
||||||
else None,
|
|
||||||
age_range=row.get("age_range")
|
|
||||||
if pd.notna(row.get("age_range"))
|
|
||||||
else None,
|
|
||||||
address1=row.get("address1") if pd.notna(row.get("address1")) else None,
|
|
||||||
address2=row.get("address2") if pd.notna(row.get("address2")) else None,
|
|
||||||
town=row.get("town") if pd.notna(row.get("town")) else None,
|
|
||||||
postcode=row.get("postcode") if pd.notna(row.get("postcode")) else None,
|
|
||||||
latitude=lat,
|
|
||||||
longitude=lon,
|
|
||||||
)
|
|
||||||
db.add(school)
|
|
||||||
db.flush() # Get the ID
|
|
||||||
urn_to_school_id[urn] = school.id
|
|
||||||
schools_created += 1
|
|
||||||
|
|
||||||
if schools_created % 1000 == 0:
|
|
||||||
print(f" Created {schools_created} schools...")
|
|
||||||
|
|
||||||
print(f" Created {schools_created} schools")
|
|
||||||
|
|
||||||
# Create results
|
|
||||||
print(f"\nMigrating {len(df)} yearly results...")
|
|
||||||
results_created = 0
|
|
||||||
|
|
||||||
for _, row in df.iterrows():
|
|
||||||
# Safely parse URN
|
|
||||||
urn_val = row.get("urn")
|
|
||||||
urn = None
|
|
||||||
if pd.notna(urn_val):
|
|
||||||
try:
|
|
||||||
urn_str = str(urn_val).strip()
|
|
||||||
if urn_str:
|
|
||||||
urn = int(float(urn_str))
|
|
||||||
except (ValueError, TypeError):
|
|
||||||
pass
|
|
||||||
if not urn or urn not in urn_to_school_id:
|
|
||||||
continue
|
|
||||||
|
|
||||||
school_id = urn_to_school_id[urn]
|
|
||||||
|
|
||||||
# Safely parse year
|
|
||||||
year_val = row.get("year")
|
|
||||||
year = None
|
|
||||||
if pd.notna(year_val):
|
|
||||||
try:
|
|
||||||
year = int(float(str(year_val).strip()))
|
|
||||||
except (ValueError, TypeError):
|
|
||||||
pass
|
|
||||||
if not year:
|
|
||||||
continue
|
|
||||||
|
|
||||||
result = SchoolResult(
|
|
||||||
school_id=school_id,
|
|
||||||
year=year,
|
|
||||||
total_pupils=parse_numeric(row.get("total_pupils")),
|
|
||||||
eligible_pupils=parse_numeric(row.get("eligible_pupils")),
|
|
||||||
# Expected Standard
|
|
||||||
rwm_expected_pct=parse_numeric(row.get("rwm_expected_pct")),
|
|
||||||
reading_expected_pct=parse_numeric(row.get("reading_expected_pct")),
|
|
||||||
writing_expected_pct=parse_numeric(row.get("writing_expected_pct")),
|
|
||||||
maths_expected_pct=parse_numeric(row.get("maths_expected_pct")),
|
|
||||||
gps_expected_pct=parse_numeric(row.get("gps_expected_pct")),
|
|
||||||
science_expected_pct=parse_numeric(row.get("science_expected_pct")),
|
|
||||||
# Higher Standard
|
|
||||||
rwm_high_pct=parse_numeric(row.get("rwm_high_pct")),
|
|
||||||
reading_high_pct=parse_numeric(row.get("reading_high_pct")),
|
|
||||||
writing_high_pct=parse_numeric(row.get("writing_high_pct")),
|
|
||||||
maths_high_pct=parse_numeric(row.get("maths_high_pct")),
|
|
||||||
gps_high_pct=parse_numeric(row.get("gps_high_pct")),
|
|
||||||
# Progress
|
|
||||||
reading_progress=parse_numeric(row.get("reading_progress")),
|
|
||||||
writing_progress=parse_numeric(row.get("writing_progress")),
|
|
||||||
maths_progress=parse_numeric(row.get("maths_progress")),
|
|
||||||
# Averages
|
|
||||||
reading_avg_score=parse_numeric(row.get("reading_avg_score")),
|
|
||||||
maths_avg_score=parse_numeric(row.get("maths_avg_score")),
|
|
||||||
gps_avg_score=parse_numeric(row.get("gps_avg_score")),
|
|
||||||
# Context
|
|
||||||
disadvantaged_pct=parse_numeric(row.get("disadvantaged_pct")),
|
|
||||||
eal_pct=parse_numeric(row.get("eal_pct")),
|
|
||||||
sen_support_pct=parse_numeric(row.get("sen_support_pct")),
|
|
||||||
sen_ehcp_pct=parse_numeric(row.get("sen_ehcp_pct")),
|
|
||||||
stability_pct=parse_numeric(row.get("stability_pct")),
|
|
||||||
# Absence
|
|
||||||
reading_absence_pct=parse_numeric(row.get("reading_absence_pct")),
|
|
||||||
gps_absence_pct=parse_numeric(row.get("gps_absence_pct")),
|
|
||||||
maths_absence_pct=parse_numeric(row.get("maths_absence_pct")),
|
|
||||||
writing_absence_pct=parse_numeric(row.get("writing_absence_pct")),
|
|
||||||
science_absence_pct=parse_numeric(row.get("science_absence_pct")),
|
|
||||||
# Gender
|
|
||||||
rwm_expected_boys_pct=parse_numeric(row.get("rwm_expected_boys_pct")),
|
|
||||||
rwm_expected_girls_pct=parse_numeric(row.get("rwm_expected_girls_pct")),
|
|
||||||
rwm_high_boys_pct=parse_numeric(row.get("rwm_high_boys_pct")),
|
|
||||||
rwm_high_girls_pct=parse_numeric(row.get("rwm_high_girls_pct")),
|
|
||||||
# Disadvantaged
|
|
||||||
rwm_expected_disadvantaged_pct=parse_numeric(
|
|
||||||
row.get("rwm_expected_disadvantaged_pct")
|
|
||||||
),
|
|
||||||
rwm_expected_non_disadvantaged_pct=parse_numeric(
|
|
||||||
row.get("rwm_expected_non_disadvantaged_pct")
|
|
||||||
),
|
|
||||||
disadvantaged_gap=parse_numeric(row.get("disadvantaged_gap")),
|
|
||||||
# 3-Year
|
|
||||||
rwm_expected_3yr_pct=parse_numeric(row.get("rwm_expected_3yr_pct")),
|
|
||||||
reading_avg_3yr=parse_numeric(row.get("reading_avg_3yr")),
|
|
||||||
maths_avg_3yr=parse_numeric(row.get("maths_avg_3yr")),
|
|
||||||
)
|
|
||||||
db.add(result)
|
|
||||||
results_created += 1
|
|
||||||
|
|
||||||
if results_created % 10000 == 0:
|
|
||||||
print(f" Created {results_created} results...")
|
|
||||||
db.flush()
|
|
||||||
|
|
||||||
print(f" Created {results_created} results")
|
|
||||||
|
|
||||||
# Commit all changes
|
|
||||||
db.commit()
|
|
||||||
print("\nMigration complete!")
|
|
||||||
|
|
||||||
|
|
||||||
def _apply_schema_alterations():
|
|
||||||
"""
|
|
||||||
Add new columns to existing tables using ALTER TABLE … ADD COLUMN IF NOT EXISTS.
|
|
||||||
Safe to run on every migration — no-ops if the column already exists.
|
|
||||||
Add entries here whenever models.py gains new columns on an existing table.
|
|
||||||
"""
|
|
||||||
alterations = [
|
|
||||||
# v4: Ofsted Report Card columns
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS framework VARCHAR(20)",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_safeguarding_met BOOLEAN",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_inclusion INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_curriculum_teaching INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_achievement INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_attendance_behaviour INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_personal_development INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_leadership_governance INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_early_years INTEGER",
|
|
||||||
"ALTER TABLE ofsted_inspections ADD COLUMN IF NOT EXISTS rc_sixth_form INTEGER",
|
|
||||||
]
|
|
||||||
from sqlalchemy import text as sa_text
|
|
||||||
with engine.connect() as conn:
|
|
||||||
for stmt in alterations:
|
|
||||||
try:
|
|
||||||
conn.execute(sa_text(stmt))
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: alteration skipped ({e})")
|
|
||||||
conn.commit()
|
|
||||||
|
|
||||||
|
|
||||||
def _apply_schema_drops():
|
|
||||||
"""
|
|
||||||
Drop tables retired from the schema. Idempotent (DROP … IF EXISTS), so it's
|
|
||||||
safe to run on every migration. Add entries here when a model is removed.
|
|
||||||
"""
|
|
||||||
drops = [
|
|
||||||
# v6: Ofsted Parent View feature removed
|
|
||||||
"DROP TABLE IF EXISTS marts.fact_parent_view CASCADE",
|
|
||||||
]
|
|
||||||
from sqlalchemy import text as sa_text
|
|
||||||
with engine.connect() as conn:
|
|
||||||
for stmt in drops:
|
|
||||||
try:
|
|
||||||
conn.execute(sa_text(stmt))
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: drop skipped ({e})")
|
|
||||||
conn.commit()
|
|
||||||
|
|
||||||
|
|
||||||
def run_full_migration(geocode: bool = False) -> bool:
|
|
||||||
"""
|
|
||||||
Run a complete migration: drop all tables and reimport from CSV.
|
|
||||||
|
|
||||||
Returns True if successful, False if no data found.
|
|
||||||
Raises exception on error.
|
|
||||||
"""
|
|
||||||
# Preserve existing geocoding so a reimport doesn't throw away coordinates
|
|
||||||
# that took a long time to compute.
|
|
||||||
geocode_cache: dict[int, tuple[float, float]] = {}
|
|
||||||
inspector = __import__("sqlalchemy").inspect(engine)
|
|
||||||
if "schools" in inspector.get_table_names():
|
|
||||||
try:
|
|
||||||
with get_db_session() as db:
|
|
||||||
rows = db.execute(
|
|
||||||
__import__("sqlalchemy").text(
|
|
||||||
"SELECT urn, latitude, longitude FROM schools "
|
|
||||||
"WHERE latitude IS NOT NULL AND longitude IS NOT NULL"
|
|
||||||
)
|
|
||||||
).fetchall()
|
|
||||||
geocode_cache = {r.urn: (r.latitude, r.longitude) for r in rows}
|
|
||||||
print(f" Saved {len(geocode_cache)} existing geocoded coordinates.")
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: could not save geocode cache: {e}")
|
|
||||||
|
|
||||||
# Only drop the core KS2 tables — leave supplementary tables (ofsted, census,
|
|
||||||
# finance, etc.) intact so a reimport doesn't wipe integrator-populated data.
|
|
||||||
# schema_version is NOT dropped: it persists so restarts don't re-trigger migration.
|
|
||||||
ks2_tables = ["school_results", "schools"]
|
|
||||||
print(f"Dropping core tables: {ks2_tables} ...")
|
|
||||||
inspector = __import__("sqlalchemy").inspect(engine)
|
|
||||||
existing = set(inspector.get_table_names())
|
|
||||||
for tname in ks2_tables:
|
|
||||||
if tname in existing:
|
|
||||||
Base.metadata.tables[tname].drop(bind=engine)
|
|
||||||
|
|
||||||
print("Creating all tables...")
|
|
||||||
Base.metadata.create_all(bind=engine)
|
|
||||||
|
|
||||||
# ALTER existing supplementary tables to add any new columns.
|
|
||||||
# create_all() only creates missing tables; it won't add columns to tables
|
|
||||||
# that already exist from an older schema version. These statements are
|
|
||||||
# idempotent (IF NOT EXISTS) so they're safe to run on every migration.
|
|
||||||
print("Applying column additions to supplementary tables...")
|
|
||||||
_apply_schema_alterations()
|
|
||||||
|
|
||||||
print("Dropping retired tables...")
|
|
||||||
_apply_schema_drops()
|
|
||||||
|
|
||||||
print("\nLoading CSV data...")
|
|
||||||
df = load_csv_data(settings.data_dir)
|
|
||||||
|
|
||||||
if df.empty:
|
|
||||||
print("Warning: No CSV data found to migrate!")
|
|
||||||
return False
|
|
||||||
|
|
||||||
migrate_data(df, geocode=geocode, geocode_cache=geocode_cache)
|
|
||||||
return True
|
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
"""
|
|
||||||
Schema versioning for database migrations.
|
|
||||||
|
|
||||||
HOW TO USE:
|
|
||||||
- Bump SCHEMA_VERSION when making changes to database models
|
|
||||||
- This triggers an automatic full data reimport on next app startup
|
|
||||||
|
|
||||||
WHEN TO BUMP:
|
|
||||||
- Adding/removing columns in models.py
|
|
||||||
- Changing column types or constraints
|
|
||||||
- Modifying CSV column mappings in schemas.py
|
|
||||||
- Any change that requires fresh data import
|
|
||||||
"""
|
|
||||||
|
|
||||||
# Current schema version - increment when models change
|
|
||||||
SCHEMA_VERSION = 6
|
|
||||||
|
|
||||||
# Changelog for documentation
|
|
||||||
SCHEMA_CHANGELOG = {
|
|
||||||
1: "Initial schema with School and SchoolResult tables",
|
|
||||||
2: "Added pupil absence fields (reading, maths, gps, writing, science)",
|
|
||||||
3: "Added supplementary data tables: ofsted, parent_view, census, admissions, sen_detail, phonics, deprivation, finance; GIAS columns on schools",
|
|
||||||
4: "Added Ofsted Report Card columns to ofsted_inspections (new framework from Nov 2025)",
|
|
||||||
5: "Apply ALTER TABLE additions for RC columns missed by create_all on existing tables",
|
|
||||||
6: "Removed the Ofsted Parent View feature: dropped fact_parent_view table and model",
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,89 @@
|
|||||||
|
# Legacy and unused-code inventory
|
||||||
|
|
||||||
|
Reviewed 2026-09-14. This inventory records source evidence, not production usage
|
||||||
|
telemetry. A command with no repository caller may still be run manually or from
|
||||||
|
an external scheduler. Historical specs and prototypes are not runtime imports.
|
||||||
|
|
||||||
|
## Method and scope
|
||||||
|
|
||||||
|
Searched backend imports, tests, CLI scripts, Airflow DAGs, Meltano configuration,
|
||||||
|
Gitea workflows, Dockerfiles and documentation. For frontend candidates, inspected
|
||||||
|
TypeScript imports, re-exports, literal dynamic imports and `require` calls,
|
||||||
|
resolving relative and `@/` paths while excluding tests, dependencies and build
|
||||||
|
output. Checked candidates again with text searches including tests.
|
||||||
|
|
||||||
|
Next.js route files, generated Payload import-map entries and plugin discovery
|
||||||
|
are entry points even without ordinary imports. This is why a zero-import count
|
||||||
|
alone is not sufficient grounds for deletion. Computed imports and external
|
||||||
|
operators are outside this static audit.
|
||||||
|
|
||||||
|
## Removed in this cleanup
|
||||||
|
|
||||||
|
These names are recorded for Git-history lookup; they are no longer file links.
|
||||||
|
|
||||||
|
| Removed path or symbol | Evidence and replacement |
|
||||||
|
|---|---|
|
||||||
|
| `backend/migration.py` | Imported `School` and `SchoolResult`, which no longer exist in `backend/models.py`. Only the legacy CSV CLI imported it. Current tables are built by dbt. |
|
||||||
|
| `backend/version.py` | Only the legacy importer consumed `SCHEMA_VERSION`. FastAPI lifespan does not perform version-triggered imports. This is unrelated to active Payload migrations. |
|
||||||
|
| `scripts/migrate_csv_to_db.py` | Imported removed `init_db`/`set_db_schema_version` helpers and the obsolete models indirectly. No runtime, DAG or workflow calls it. Use the managed pipeline for current marts. |
|
||||||
|
| `scripts/geocode_schools.py` | Imported the removed `School` ORM model. No pipeline/workflow calls it. Coordinates now come from GIAS/PostGIS; a separate mart-aware manual utility remains under `pipeline/scripts/`. |
|
||||||
|
| `backend.data_loader.haversine_distance` | No callers. Search uses its inline vectorised NumPy calculation. |
|
||||||
|
| `nextjs-app/lib/api.ts: fetcher` | No callers; SWR is not installed. Application fetches use the named API wrappers. |
|
||||||
|
| `nextjs-app/lib/api.ts: kmToMiles` | No callers. `calculateDistance` remains because `CutoffMapPanel` uses it. |
|
||||||
|
|
||||||
|
The removed command files could not import successfully against the current
|
||||||
|
backend. This cleanup does not run replacements, migrate data or modify databases.
|
||||||
|
Their previous implementations remain recoverable from Git history.
|
||||||
|
|
||||||
|
## Unused candidates retained for a separate cleanup
|
||||||
|
|
||||||
|
| Candidate | Evidence | Recommended next step |
|
||||||
|
|---|---|---|
|
||||||
|
| `nextjs-app/components/LoadingSkeleton.tsx` and its CSS | No application or test imports found. | Remove together after confirming no planned use. |
|
||||||
|
| `nextjs-app/components/Pagination.tsx` and its CSS | No application or test imports found; HomeView implements load-more behaviour. | Remove as a pair if numbered pagination will not return. |
|
||||||
|
| `nextjs-app/components/SchoolCard.tsx` and its CSS | Imported by its own tests, not application code. HomeView uses SchoolRow/SecondarySchoolRow. | Decide whether to retire the card design; if removed, remove its dedicated tests as well. Passing tests do not establish runtime use. |
|
||||||
|
| `backend/database.py: get_db`, `get_db_session` | No remaining callers after removing the importer. Current code creates SessionLocal directly. | Either adopt these helpers during session-lifecycle cleanup or remove them; do not rewrite active sessions in a documentation change. |
|
||||||
|
| `backend/schemas.py: COLUMN_MAPPINGS`, `NULL_VALUES`, `LA_CODE_TO_NAME` | No remaining Python consumers found after importer removal. Other constants in this module are active. | Remove individual constants after checking external data utilities; retain the module. |
|
||||||
|
| `backend/config.py: data_dir`, `max_page_size`, `rate_limit_burst` | No active consumers found. `default_page_size` appears only in a branch that expects None, although the route supplies a concrete default. | Reconcile settings with route validation in a focused API change. |
|
||||||
|
|
||||||
|
## Legacy/manual paths requiring operational verification
|
||||||
|
|
||||||
|
| Path | Status and reason to retain for now |
|
||||||
|
|---|---|
|
||||||
|
| FastAPI `/`, `/compare`, `/rankings`, `/favicon.svg`, `/robots.txt`, and conditional `/static` | Old frontend-serving routes reference a `frontend/` directory absent from the checkout and backend image. Next.js owns these public surfaces. Removal changes externally callable routes, so first check proxy/operator usage and define replacement responses. |
|
||||||
|
| `scripts/fetch_real_data.py`, `scripts/download_data.py` | Historical standalone CSV utilities. The fetch script targets Wandsworth/Merton; neither is wired into the managed pipeline. Marked historical, retained pending confirmation of manual use. |
|
||||||
|
| `pipeline/scripts/geocode_postcodes.py` | Mart-aware postcode fallback, not called by the current DAGs. Do not confuse it with the removed legacy ORM geocoder. Verify the target schema before manual use. |
|
||||||
|
| `docker-compose.yml` | Uses unpublished `:latest` release tags and lacks frontend Payload DB/secret/media configuration. Retained as an old development topology, not recommended onboarding. |
|
||||||
|
| `nextjs-app/docker-compose.yml` | Standalone legacy recipe with old backend port assumptions and no CMS persistence setup. Retained until its consumers are checked. |
|
||||||
|
| `MIGRATION_SUMMARY.md`, `docs/superpowers/`, `mockups/` | Historical designs and prototypes. Retain as history; do not follow as current deployment instructions. |
|
||||||
|
| `scripts/sql/drop_fact_parent_view.sql` | One-off maintenance SQL. Not an application entry point; repository call-site searches cannot establish whether it is still needed operationally. |
|
||||||
|
|
||||||
|
## Active code that can look obsolete
|
||||||
|
|
||||||
|
- `backend/data_loader.py` older-mart query fallbacks are covered by backend tests
|
||||||
|
and support databases at different migration stages. Remove only after verifying
|
||||||
|
the deployed schemas in every supported environment.
|
||||||
|
- `backend/gias_codes.py` and `pipeline/scripts/gias_codes.py` are intentionally
|
||||||
|
generated copies for separate runtime images. Their parity is tested.
|
||||||
|
- `nextjs-app/migrations/`, `payload-types.ts` and the Payload import map are active
|
||||||
|
CMS artifacts, not remnants of the removed school importer.
|
||||||
|
- `get_available_years`, `get_available_local_authorities` and `get_schools_count`
|
||||||
|
in `data_loader.py` are called through `get_data_info`, which serves the backend
|
||||||
|
data-info endpoint. They are not dead functions.
|
||||||
|
- `get_supplementary_data` is an intentional single-school wrapper around the
|
||||||
|
batch implementation.
|
||||||
|
- `pipeline/transform` models named `legacy` can be active data sources: annual
|
||||||
|
DAG selectors explicitly include legacy KS2/KS4 lineage. Names alone do not
|
||||||
|
establish obsolescence.
|
||||||
|
|
||||||
|
## Suggested next passes
|
||||||
|
|
||||||
|
1. Decide the fate of the three unused UI components and remove paired assets/tests.
|
||||||
|
2. Consolidate backend session usage and remove abandoned settings/constants.
|
||||||
|
3. Verify external consumers, then retire static-serving API routes and old compose recipes.
|
||||||
|
4. Audit manual data utilities with pipeline operators before deleting them.
|
||||||
|
5. Revisit compatibility fallbacks only after documenting supported schema versions.
|
||||||
|
|
||||||
|
Validation for this cleanup should include frontend typechecking/tests, Python
|
||||||
|
syntax checks, reference searches and documentation link checks. Live database,
|
||||||
|
external scheduler and deployed route usage require separate integration evidence.
|
||||||
@@ -38,8 +38,7 @@ describe('payload mount points', () => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
it('isolates CMS tables in their own postgres schema', () => {
|
it('isolates CMS tables in their own postgres schema', () => {
|
||||||
// Blog content must sit outside `public`, where the app tables, Airflow's
|
// Blog content must stay separate from school marts and Airflow metadata.
|
||||||
// metadata and scripts/migrate_csv_to_db.py --drop all live.
|
|
||||||
expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/);
|
expect(CONFIG).toMatch(/schemaName:\s*['"]payload['"]/);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
@@ -311,26 +311,6 @@ export async function fetchDataInfo(
|
|||||||
return handleResponse<DataInfoResponse>(response);
|
return handleResponse<DataInfoResponse>(response);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
|
||||||
// Client-Side Fetcher (for SWR)
|
|
||||||
// ============================================================================
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Generic fetcher function for use with SWR
|
|
||||||
* @example
|
|
||||||
* ```tsx
|
|
||||||
* const { data, error } = useSWR('/api/schools', fetcher);
|
|
||||||
* ```
|
|
||||||
*/
|
|
||||||
export async function fetcher<T>(url: string): Promise<T> {
|
|
||||||
// If it's already a full URL, use it directly
|
|
||||||
// Otherwise, prepend the API_BASE_URL
|
|
||||||
const fullUrl = url.startsWith('http') ? url : `${API_BASE_URL}${url.startsWith('/') ? url : `/${url}`}`;
|
|
||||||
|
|
||||||
const response = await fetch(fullUrl);
|
|
||||||
return handleResponse<T>(response);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// Geocoding API
|
// Geocoding API
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -396,10 +376,3 @@ export function calculateDistance(
|
|||||||
const c = 2 * Math.atan2(Math.sqrt(a), Math.sqrt(1 - a));
|
const c = 2 * Math.atan2(Math.sqrt(a), Math.sqrt(1 - a));
|
||||||
return R * c;
|
return R * c;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Convert kilometers to miles
|
|
||||||
*/
|
|
||||||
export function kmToMiles(km: number): number {
|
|
||||||
return km * 0.621371;
|
|
||||||
}
|
|
||||||
@@ -24,9 +24,8 @@ export default buildConfig({
|
|||||||
typescript: { outputFile: path.resolve(dirname, 'payload-types.ts') },
|
typescript: { outputFile: path.resolve(dirname, 'payload-types.ts') },
|
||||||
db: postgresAdapter({
|
db: postgresAdapter({
|
||||||
pool: { connectionString: process.env.DATABASE_URL },
|
pool: { connectionString: process.env.DATABASE_URL },
|
||||||
// Its own schema, so no pipeline operation on `public` can reach blog
|
// Its own schema separates blog content from pipeline-managed school
|
||||||
// content. scripts/migrate_csv_to_db.py --drop lives in that blast radius,
|
// tables and Airflow metadata. The schema is created by the initial
|
||||||
// as does Airflow's metadata. The schema itself is created by the initial
|
|
||||||
// migration: schemaName says where tables go, it does not create anything.
|
// migration: schemaName says where tables go, it does not create anything.
|
||||||
//
|
//
|
||||||
// prodMigrations runs pending migrations during server init. Without it a
|
// prodMigrations runs pending migrations during server init. Without it a
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
|
Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline.
|
||||||
|
See docs/LEGACY_CODE.md before using it for current school data.
|
||||||
|
|
||||||
Data Download Helper Script
|
Data Download Helper Script
|
||||||
|
|
||||||
This script provides instructions and utilities for downloading
|
This script provides instructions and utilities for downloading
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
|
Historical standalone CSV utility; not part of the managed Meltano/dbt pipeline.
|
||||||
|
See docs/LEGACY_CODE.md before using it for current school data.
|
||||||
|
|
||||||
Fetch real school performance data from UK Government sources.
|
Fetch real school performance data from UK Government sources.
|
||||||
|
|
||||||
This script downloads KS2 (Key Stage 2) primary school data from:
|
This script downloads KS2 (Key Stage 2) primary school data from:
|
||||||
|
|||||||
@@ -1,184 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
Geocode all school postcodes and update the database.
|
|
||||||
|
|
||||||
This script should be run as a weekly cron job to ensure all schools
|
|
||||||
have up-to-date latitude/longitude coordinates.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python scripts/geocode_schools.py [--force]
|
|
||||||
|
|
||||||
Options:
|
|
||||||
--force Re-geocode all postcodes, even if already geocoded
|
|
||||||
|
|
||||||
Crontab example (run every Sunday at 2am):
|
|
||||||
0 2 * * 0 cd /path/to/school_compare && /path/to/venv/bin/python scripts/geocode_schools.py >> /var/log/geocode_schools.log 2>&1
|
|
||||||
"""
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import sys
|
|
||||||
from datetime import datetime
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Dict, Tuple
|
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
# Add parent directory to path for imports
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
|
||||||
|
|
||||||
from backend.database import SessionLocal
|
|
||||||
from backend.models import School
|
|
||||||
|
|
||||||
|
|
||||||
def geocode_postcodes_bulk(postcodes: list) -> Dict[str, Tuple[float, float]]:
|
|
||||||
"""
|
|
||||||
Geocode postcodes in bulk using postcodes.io API.
|
|
||||||
Returns dict of postcode -> (latitude, longitude).
|
|
||||||
"""
|
|
||||||
results = {}
|
|
||||||
valid_postcodes = [
|
|
||||||
p.strip().upper()
|
|
||||||
for p in postcodes
|
|
||||||
if p and isinstance(p, str) and len(p.strip()) >= 5
|
|
||||||
]
|
|
||||||
valid_postcodes = list(set(valid_postcodes))
|
|
||||||
|
|
||||||
if not valid_postcodes:
|
|
||||||
return results
|
|
||||||
|
|
||||||
batch_size = 100
|
|
||||||
total_batches = (len(valid_postcodes) + batch_size - 1) // batch_size
|
|
||||||
|
|
||||||
for i, batch_start in enumerate(range(0, len(valid_postcodes), batch_size)):
|
|
||||||
batch = valid_postcodes[batch_start : batch_start + batch_size]
|
|
||||||
print(f" Geocoding batch {i + 1}/{total_batches} ({len(batch)} postcodes)...")
|
|
||||||
|
|
||||||
try:
|
|
||||||
response = requests.post(
|
|
||||||
"https://api.postcodes.io/postcodes",
|
|
||||||
json={"postcodes": batch},
|
|
||||||
timeout=30,
|
|
||||||
)
|
|
||||||
if response.status_code == 200:
|
|
||||||
data = response.json()
|
|
||||||
for item in data.get("result", []):
|
|
||||||
if item and item.get("result"):
|
|
||||||
pc = item["query"].upper()
|
|
||||||
lat = item["result"].get("latitude")
|
|
||||||
lon = item["result"].get("longitude")
|
|
||||||
if lat and lon:
|
|
||||||
results[pc] = (lat, lon)
|
|
||||||
else:
|
|
||||||
print(f" Warning: API returned status {response.status_code}")
|
|
||||||
except Exception as e:
|
|
||||||
print(f" Warning: Geocoding batch failed: {e}")
|
|
||||||
|
|
||||||
return results
|
|
||||||
|
|
||||||
|
|
||||||
def geocode_schools(force: bool = False) -> None:
|
|
||||||
"""
|
|
||||||
Geocode all schools in the database.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
force: If True, re-geocode all postcodes even if already geocoded
|
|
||||||
"""
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"School Geocoding Job - {datetime.now().isoformat()}")
|
|
||||||
print(f"{'='*60}\n")
|
|
||||||
|
|
||||||
db = SessionLocal()
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Get schools that need geocoding
|
|
||||||
if force:
|
|
||||||
schools = db.query(School).filter(School.postcode.isnot(None)).all()
|
|
||||||
print(f"Force mode: Processing all {len(schools)} schools with postcodes")
|
|
||||||
else:
|
|
||||||
schools = db.query(School).filter(
|
|
||||||
School.postcode.isnot(None),
|
|
||||||
(School.latitude.is_(None)) | (School.longitude.is_(None))
|
|
||||||
).all()
|
|
||||||
print(f"Found {len(schools)} schools without coordinates")
|
|
||||||
|
|
||||||
if not schools:
|
|
||||||
print("No schools to geocode. Exiting.")
|
|
||||||
return
|
|
||||||
|
|
||||||
# Extract unique postcodes
|
|
||||||
postcodes = list(set(
|
|
||||||
s.postcode.strip().upper()
|
|
||||||
for s in schools
|
|
||||||
if s.postcode
|
|
||||||
))
|
|
||||||
print(f"Unique postcodes to geocode: {len(postcodes)}")
|
|
||||||
|
|
||||||
# Geocode in bulk
|
|
||||||
print("\nGeocoding postcodes...")
|
|
||||||
geocoded = geocode_postcodes_bulk(postcodes)
|
|
||||||
print(f"Successfully geocoded: {len(geocoded)} postcodes")
|
|
||||||
|
|
||||||
# Update database
|
|
||||||
print("\nUpdating database...")
|
|
||||||
updated_count = 0
|
|
||||||
failed_count = 0
|
|
||||||
|
|
||||||
for school in schools:
|
|
||||||
if not school.postcode:
|
|
||||||
continue
|
|
||||||
|
|
||||||
pc_upper = school.postcode.strip().upper()
|
|
||||||
coords = geocoded.get(pc_upper)
|
|
||||||
|
|
||||||
if coords:
|
|
||||||
school.latitude = coords[0]
|
|
||||||
school.longitude = coords[1]
|
|
||||||
updated_count += 1
|
|
||||||
else:
|
|
||||||
failed_count += 1
|
|
||||||
|
|
||||||
db.commit()
|
|
||||||
|
|
||||||
print(f"\nResults:")
|
|
||||||
print(f" - Updated: {updated_count} schools")
|
|
||||||
print(f" - Failed (invalid/not found): {failed_count} postcodes")
|
|
||||||
|
|
||||||
# Summary stats
|
|
||||||
total_with_coords = db.query(School).filter(
|
|
||||||
School.latitude.isnot(None),
|
|
||||||
School.longitude.isnot(None)
|
|
||||||
).count()
|
|
||||||
total_schools = db.query(School).count()
|
|
||||||
|
|
||||||
print(f"\nDatabase summary:")
|
|
||||||
print(f" - Total schools: {total_schools}")
|
|
||||||
print(f" - Schools with coordinates: {total_with_coords}")
|
|
||||||
print(f" - Coverage: {100*total_with_coords/total_schools:.1f}%")
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error during geocoding: {e}")
|
|
||||||
db.rollback()
|
|
||||||
raise
|
|
||||||
finally:
|
|
||||||
db.close()
|
|
||||||
print(f"\n{'='*60}")
|
|
||||||
print(f"Geocoding job completed - {datetime.now().isoformat()}")
|
|
||||||
print(f"{'='*60}\n")
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(
|
|
||||||
description="Geocode school postcodes and update database"
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--force",
|
|
||||||
action="store_true",
|
|
||||||
help="Re-geocode all postcodes, even if already geocoded"
|
|
||||||
)
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
geocode_schools(force=args.force)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
@@ -1,68 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
"""
|
|
||||||
CLI script for manual database migration.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
python scripts/migrate_csv_to_db.py [--drop] [--geocode]
|
|
||||||
|
|
||||||
Options:
|
|
||||||
--drop Drop existing tables before migration (full reimport)
|
|
||||||
--geocode Geocode postcodes (requires network access)
|
|
||||||
"""
|
|
||||||
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
# Add parent directory to path for imports
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
|
|
||||||
from backend.config import settings
|
|
||||||
from backend.database import Base, engine, init_db, set_db_schema_version
|
|
||||||
from backend.migration import load_csv_data, migrate_data, run_full_migration
|
|
||||||
from backend.version import SCHEMA_VERSION
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(
|
|
||||||
description="Migrate CSV data to PostgreSQL database"
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--drop", action="store_true", help="Drop existing tables before migration"
|
|
||||||
)
|
|
||||||
parser.add_argument("--geocode", action="store_true", help="Geocode postcodes")
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
print("=" * 60)
|
|
||||||
print("School Data Migration: CSV -> PostgreSQL")
|
|
||||||
print("=" * 60)
|
|
||||||
print(f"\nDatabase: {settings.database_url.split('@')[-1]}")
|
|
||||||
print(f"Data directory: {settings.data_dir}")
|
|
||||||
print(f"Target schema version: {SCHEMA_VERSION}")
|
|
||||||
|
|
||||||
if args.drop:
|
|
||||||
print("\nRunning full migration (drop + reimport)...")
|
|
||||||
success = run_full_migration(geocode=args.geocode)
|
|
||||||
else:
|
|
||||||
print("\nCreating tables (preserving existing data)...")
|
|
||||||
init_db()
|
|
||||||
print("\nLoading CSV data...")
|
|
||||||
df = load_csv_data(settings.data_dir)
|
|
||||||
if df.empty:
|
|
||||||
print("No data found to migrate!")
|
|
||||||
return 1
|
|
||||||
migrate_data(df, geocode=args.geocode)
|
|
||||||
success = True
|
|
||||||
|
|
||||||
if success:
|
|
||||||
# Ensure schema_version table exists
|
|
||||||
init_db()
|
|
||||||
set_db_schema_version(SCHEMA_VERSION)
|
|
||||||
print(f"\nSchema version set to {SCHEMA_VERSION}")
|
|
||||||
|
|
||||||
return 0 if success else 1
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
sys.exit(main())
|
|
||||||
Reference in new issue
Block a user