Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
74ca76d150 | ||
|
|
4b75152ee0 |
+59
-2
@@ -4,6 +4,7 @@ Provides efficient queries with caching.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import re
|
||||
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
@@ -262,15 +263,68 @@ assert "NULL AS has_sixth_form" in str(_MAIN_QUERY_NO_SIXTH_FORM), (
|
||||
"expected replacement of 's.has_sixth_form,' to have taken effect"
|
||||
)
|
||||
|
||||
# Fallback used when marts.dim_school predates the GIAS code-dictionary
|
||||
# migration (i.e. the nightly dbt pipeline hasn't rebuilt the mart yet on
|
||||
# this DB, so it still has the old name columns instead of *_code columns).
|
||||
_MAIN_QUERY_LEGACY_NAMES = str(_MAIN_QUERY)
|
||||
_LEGACY_NAME_REPLACEMENTS = [
|
||||
("s.phase_code,", "s.phase,"),
|
||||
("s.school_type_code,", "s.school_type,"),
|
||||
(
|
||||
"s.religious_character_code,",
|
||||
"s.religious_character AS religious_denomination,",
|
||||
),
|
||||
("s.status_code,", "s.status,"),
|
||||
("s.admissions_policy_code,", "s.admissions_policy,"),
|
||||
]
|
||||
for _old, _new in _LEGACY_NAME_REPLACEMENTS:
|
||||
assert _old in _MAIN_QUERY_LEGACY_NAMES, (
|
||||
f"expected {_old!r} to be present in _MAIN_QUERY before replacement"
|
||||
)
|
||||
_MAIN_QUERY_LEGACY_NAMES = _MAIN_QUERY_LEGACY_NAMES.replace(_old, _new)
|
||||
_MAIN_QUERY_LEGACY_NAMES = text(_MAIN_QUERY_LEGACY_NAMES)
|
||||
|
||||
_GIAS_CODE_COLUMN_NAMES = (
|
||||
"phase_code",
|
||||
"school_type_code",
|
||||
"religious_character_code",
|
||||
"status_code",
|
||||
"admissions_policy_code",
|
||||
)
|
||||
|
||||
_MISSING_COLUMN_RE = re.compile(r'column "?(?:s\.)?(\w+)"? does not exist')
|
||||
|
||||
|
||||
def _missing_column_name(exc: Exception) -> Optional[str]:
|
||||
"""Name of the missing column from a psycopg2 UndefinedColumn error.
|
||||
|
||||
Inspects exc.orig (the DBAPI error), whose message names only the
|
||||
offending column — str(exc) also embeds the full SQL statement, which
|
||||
contains every column name and therefore must not be matched against.
|
||||
"""
|
||||
orig = getattr(exc, "orig", None)
|
||||
match = _MISSING_COLUMN_RE.search(str(orig) if orig is not None else str(exc))
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
def load_school_data_as_dataframe() -> pd.DataFrame:
|
||||
"""Load all school + KS2 data as a pandas DataFrame."""
|
||||
try:
|
||||
df = pd.read_sql(_MAIN_QUERY, engine)
|
||||
except sqlalchemy.exc.ProgrammingError as exc:
|
||||
if "has_sixth_form" not in str(exc):
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
missing = _missing_column_name(exc)
|
||||
if missing in _GIAS_CODE_COLUMN_NAMES:
|
||||
logging.getLogger(__name__).warning(
|
||||
"marts predate the GIAS code migration — falling back to "
|
||||
"legacy name-column query: %s",
|
||||
exc,
|
||||
)
|
||||
try:
|
||||
df = pd.read_sql(_MAIN_QUERY_LEGACY_NAMES, engine)
|
||||
except Exception as exc2:
|
||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
||||
return pd.DataFrame()
|
||||
elif missing == "has_sixth_form":
|
||||
logging.getLogger(__name__).warning(
|
||||
"marts.dim_school is missing has_sixth_form (pipeline hasn't "
|
||||
"rebuilt the mart yet on this DB) — retrying without it: %s",
|
||||
@@ -281,6 +335,9 @@ def load_school_data_as_dataframe() -> pd.DataFrame:
|
||||
except Exception as exc2:
|
||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
||||
return pd.DataFrame()
|
||||
else:
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
return pd.DataFrame()
|
||||
except Exception as exc:
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
return pd.DataFrame()
|
||||
|
||||
@@ -78,7 +78,6 @@ OFFICIAL_SIXTH_FORM: dict[int, str] = {
|
||||
0: "Not applicable",
|
||||
1: "Has a sixth form",
|
||||
2: "Does not have a sixth form",
|
||||
9: "",
|
||||
}
|
||||
|
||||
RELIGIOUS_CHARACTER: dict[int, str] = {
|
||||
@@ -129,14 +128,12 @@ RELIGIOUS_CHARACTER: dict[int, str] = {
|
||||
47: "Reformed Baptist",
|
||||
48: "Roman Catholic/Anglican",
|
||||
49: "Sunni Deobandi",
|
||||
99: "",
|
||||
}
|
||||
|
||||
ADMISSIONS_POLICY: dict[int, str] = {
|
||||
0: "Not applicable",
|
||||
2: "Selective",
|
||||
4: "Non-selective",
|
||||
9: "",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -81,15 +81,3 @@ def test_seed_matches_dictionaries():
|
||||
for row in csv.DictReader(fh):
|
||||
seed[row["field"]][int(row["code"])] = row["name"]
|
||||
assert seed == fields
|
||||
|
||||
|
||||
def test_blank_name_sentinel_codes_map_to_empty_string():
|
||||
"""GIAS carries codes whose (name) column is blank — e.g. ReligiousCharacter
|
||||
99 (~4k schools) and AdmissionsPolicy 9 (~5.6k schools). The old name
|
||||
pipeline served these as empty strings; the dictionaries must reproduce
|
||||
that ("" is falsy, so UI tag heuristics stay silent) rather than letting
|
||||
them hit the "Unknown (<code>)" path meant for genuinely new codes."""
|
||||
assert RELIGIOUS_CHARACTER[99] == ""
|
||||
assert ADMISSIONS_POLICY[9] == ""
|
||||
assert translate(99, RELIGIOUS_CHARACTER) == ""
|
||||
assert translate(9, ADMISSIONS_POLICY) == ""
|
||||
|
||||
@@ -4,7 +4,7 @@ rest of the backend sees must carry today's name strings."""
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from backend.data_loader import translate_gias_code_columns
|
||||
from backend.data_loader import _missing_column_name, translate_gias_code_columns
|
||||
from backend.gias_codes import ESTABLISHMENT_STATUS, PHASE_OF_EDUCATION
|
||||
|
||||
|
||||
@@ -42,3 +42,91 @@ def test_missing_code_columns_are_a_noop():
|
||||
out = translate_gias_code_columns(df)
|
||||
assert out.iloc[0]["phase"] == "Primary"
|
||||
assert out.iloc[0]["status"] == "Open"
|
||||
|
||||
|
||||
def _fake_exc(orig_message):
|
||||
"""A stand-in for sqlalchemy.exc.ProgrammingError: str(exc) embeds the
|
||||
full SQL statement (deliberately containing every column name below, to
|
||||
prove the matcher doesn't fall back to it), while .orig carries the real
|
||||
DBAPI error message naming only the offending column."""
|
||||
exc = Exception(
|
||||
"SELECT s.phase_code, s.school_type_code, s.religious_character_code, "
|
||||
"s.status_code, s.admissions_policy_code, s.has_sixth_form FROM ... "
|
||||
f"[SQL: ...] (Background on this error at: https://...)"
|
||||
)
|
||||
exc.orig = Exception(orig_message) if orig_message is not None else None
|
||||
return exc
|
||||
|
||||
|
||||
def test_missing_column_name_quoted():
|
||||
assert _missing_column_name(_fake_exc('column "phase_code" does not exist')) == "phase_code"
|
||||
|
||||
|
||||
def test_missing_column_name_unquoted():
|
||||
assert _missing_column_name(_fake_exc("column phase_code does not exist")) == "phase_code"
|
||||
|
||||
|
||||
def test_missing_column_name_table_prefixed():
|
||||
assert (
|
||||
_missing_column_name(_fake_exc("column s.has_sixth_form does not exist"))
|
||||
== "has_sixth_form"
|
||||
)
|
||||
|
||||
|
||||
def test_missing_column_name_no_match_returns_none():
|
||||
assert _missing_column_name(_fake_exc("relation \"marts.dim_school\" does not exist")) is None
|
||||
|
||||
|
||||
def test_load_school_data_survives_premigration_marts(monkeypatch):
|
||||
"""Real prod state until the nightly pipeline first rebuilds the mart with
|
||||
the GIAS code columns: marts.dim_school still has the old name columns
|
||||
(phase, school_type, religious_character, status, admissions_policy)
|
||||
instead of the new *_code columns. The first query raises UndefinedColumn
|
||||
on s.phase_code; load_school_data_as_dataframe must retry with the
|
||||
legacy name-column query rather than swallow the error and return (and
|
||||
then have load_school_data cache) an empty DataFrame."""
|
||||
import sqlalchemy.exc
|
||||
from backend import data_loader
|
||||
|
||||
data_loader._df_cache = None
|
||||
data_loader._df_latest_cache = None
|
||||
|
||||
good_df = pd.DataFrame(
|
||||
[
|
||||
{
|
||||
"urn": 1,
|
||||
"school_name": "Legacy School",
|
||||
"phase": "Primary",
|
||||
"school_type": "Academy",
|
||||
"status": "Open",
|
||||
}
|
||||
]
|
||||
)
|
||||
calls = []
|
||||
|
||||
def fake_read_sql(query, con):
|
||||
calls.append(query)
|
||||
if len(calls) == 1:
|
||||
raise sqlalchemy.exc.ProgrammingError(
|
||||
statement=str(data_loader._MAIN_QUERY),
|
||||
params=None,
|
||||
orig=Exception(
|
||||
"(psycopg2.errors.UndefinedColumn) column s.phase_code "
|
||||
"does not exist\nLINE 5: s.phase_code,"
|
||||
),
|
||||
)
|
||||
return good_df.copy()
|
||||
|
||||
monkeypatch.setattr(data_loader.pd, "read_sql", fake_read_sql)
|
||||
|
||||
try:
|
||||
df = data_loader.load_school_data_as_dataframe()
|
||||
finally:
|
||||
data_loader._df_cache = None
|
||||
data_loader._df_latest_cache = None
|
||||
|
||||
assert len(calls) == 2, "must retry with the legacy name-column query variant"
|
||||
assert calls[1] is data_loader._MAIN_QUERY_LEGACY_NAMES
|
||||
assert not df.empty
|
||||
assert df["phase"].iloc[0] == "Primary"
|
||||
assert df["status"].iloc[0] == "Open"
|
||||
|
||||
@@ -148,10 +148,14 @@ def test_load_school_data_survives_missing_has_sixth_form_column(monkeypatch):
|
||||
def fake_read_sql(query, con):
|
||||
calls.append(query)
|
||||
if len(calls) == 1:
|
||||
# The statement text still contains phase_code, school_type_code,
|
||||
# etc. (it's the full _MAIN_QUERY SELECT list) — that's exactly
|
||||
# the collision this test guards against: matching must be done
|
||||
# against exc.orig (the DBAPI error), not str(exc)/the statement.
|
||||
raise sqlalchemy.exc.ProgrammingError(
|
||||
"SELECT ...",
|
||||
None,
|
||||
Exception(
|
||||
statement=str(data_loader._MAIN_QUERY),
|
||||
params=None,
|
||||
orig=Exception(
|
||||
"(psycopg2.errors.UndefinedColumn) column s.has_sixth_form "
|
||||
"does not exist"
|
||||
),
|
||||
|
||||
@@ -89,7 +89,7 @@ These are strengths the fixes below must not regress:
|
||||
- Uplift: **home 46% exit rate — decrease, moderate** and **share of sessions reaching a school page — increase, moderate** (assists the majority entry path at its first interaction).
|
||||
|
||||
- **P1.7 — The mobile hero omits the value proposition entirely** *(J1-F6)*
|
||||
- Evidence: desktop shows the "UPDATED WITH 2026/2027 ADMISSIONS RESULTS" trust badge and the "27,000+ schools… side by side, in one place" subheading; mobile renders only the poetic H1 ("Every school in England, *compared.*") and a bare search box (`j1-home-desktop-fold.png` vs `j1-home-mobile-fold.png`).
|
||||
- Evidence: desktop shows the "UPDATED WITH 2026/2027 ADMISSIONS RESULTS" trust badge and the "24,000+ schools… side by side, in one place" subheading; mobile renders only the poetic H1 ("Every school in England, *compared.*") and a bare search box (`j1-home-desktop-fold.png` vs `j1-home-mobile-fold.png`).
|
||||
- Criterion: mobile content parity; Nielsen #1 — first-visit orientation ("what is this, why trust it") absent on the primary viewport.
|
||||
- Argument: 63% of entries land here and 56% of traffic is mobile; a first-time visitor gets no statement of coverage, data source, or freshness above the fold. Weak value proposition at first glance is a classic bounce driver and plausibly a material slice of the 46% exit rate.
|
||||
- Recommendation: restore a compact version of the badge + one-line value prop under the mobile H1 (one text block; the fold has room above the deadline rail).
|
||||
|
||||
@@ -271,10 +271,10 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
|
||||
freshness, standing in for the hidden eyebrow too) on phones,
|
||||
where every line above the fold costs. */}
|
||||
<span className={styles.heroDescriptionFull}>
|
||||
<strong>27,000+ primary and secondary schools</strong> with Key Stage 2 SATs, GCSE results, Ofsted grades, progress scores and admissions data — side by side, in one place.
|
||||
<strong>24,000+ primary and secondary schools</strong> with Key Stage 2 SATs, GCSE results, Ofsted grades, progress scores and admissions data — side by side, in one place.
|
||||
</span>
|
||||
<span className={styles.heroDescriptionCompact}>
|
||||
<strong>27,000+ English schools</strong> — SATs, GCSEs, Ofsted & admissions, side by side. Updated for 2026/27.
|
||||
<strong>24,000+ English schools</strong> — SATs, GCSEs, Ofsted & admissions, side by side. Updated for 2026/27.
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
@@ -94,23 +94,13 @@ def main() -> None:
|
||||
for code_col, name_col, dict_name, field_key in FIELDS:
|
||||
pairs = (
|
||||
df[[code_col, name_col]]
|
||||
.loc[lambda d: d[code_col] != ""]
|
||||
.loc[lambda d: (d[code_col] != "") & (d[name_col] != "")]
|
||||
.drop_duplicates()
|
||||
)
|
||||
by_code: dict[int, set] = {}
|
||||
for c, n in pairs.itertuples(index=False):
|
||||
by_code.setdefault(int(c), set()).add(n)
|
||||
mapping = []
|
||||
for code, names in sorted(by_code.items()):
|
||||
named = sorted(n for n in names if n != "")
|
||||
if len(named) > 1:
|
||||
sys.exit(f"{code_col}: code {code} maps to multiple names {named} — investigate before generating")
|
||||
# Codes that only ever appear with a blank (name) are GIAS
|
||||
# "not recorded" sentinels (e.g. ReligiousCharacter 99,
|
||||
# AdmissionsPolicy 9). Map them to "" so the API serves the same
|
||||
# empty string the old name pipeline did — the "Unknown (<code>)"
|
||||
# path is reserved for genuinely new codes.
|
||||
mapping.append((code, named[0] if named else ""))
|
||||
mapping = sorted((int(c), n) for c, n in pairs.itertuples(index=False))
|
||||
dupes = len(mapping) - len({c for c, _ in mapping})
|
||||
if dupes:
|
||||
sys.exit(f"{code_col}: {dupes} codes map to multiple names — investigate before generating")
|
||||
lines = [f"{dict_name}: dict[int, str] = {{"]
|
||||
for code, name in mapping:
|
||||
escaped = name.replace('"', '\\"')
|
||||
|
||||
@@ -78,7 +78,6 @@ OFFICIAL_SIXTH_FORM: dict[int, str] = {
|
||||
0: "Not applicable",
|
||||
1: "Has a sixth form",
|
||||
2: "Does not have a sixth form",
|
||||
9: "",
|
||||
}
|
||||
|
||||
RELIGIOUS_CHARACTER: dict[int, str] = {
|
||||
@@ -129,14 +128,12 @@ RELIGIOUS_CHARACTER: dict[int, str] = {
|
||||
47: "Reformed Baptist",
|
||||
48: "Roman Catholic/Anglican",
|
||||
49: "Sunni Deobandi",
|
||||
99: "",
|
||||
}
|
||||
|
||||
ADMISSIONS_POLICY: dict[int, str] = {
|
||||
0: "Not applicable",
|
||||
2: "Selective",
|
||||
4: "Non-selective",
|
||||
9: "",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -42,12 +42,12 @@ models:
|
||||
tests:
|
||||
- accepted_values:
|
||||
severity: warn
|
||||
values: [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 99]
|
||||
values: [0, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 24, 25, 26, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49]
|
||||
- name: admissions_policy_code
|
||||
tests:
|
||||
- accepted_values:
|
||||
severity: warn
|
||||
values: [0, 2, 4, 9]
|
||||
values: [0, 2, 4]
|
||||
|
||||
- name: dim_location
|
||||
description: School location dimension with PostGIS geometry
|
||||
|
||||
@@ -53,7 +53,6 @@ phase_of_education,7,All-through
|
||||
official_sixth_form,0,Not applicable
|
||||
official_sixth_form,1,Has a sixth form
|
||||
official_sixth_form,2,Does not have a sixth form
|
||||
official_sixth_form,9,
|
||||
religious_character,0,Does not apply
|
||||
religious_character,2,Church of England
|
||||
religious_character,3,Roman Catholic
|
||||
@@ -101,8 +100,6 @@ religious_character,46,Protestant/Evangelical
|
||||
religious_character,47,Reformed Baptist
|
||||
religious_character,48,Roman Catholic/Anglican
|
||||
religious_character,49,Sunni Deobandi
|
||||
religious_character,99,
|
||||
admissions_policy,0,Not applicable
|
||||
admissions_policy,2,Selective
|
||||
admissions_policy,4,Non-selective
|
||||
admissions_policy,9,
|
||||
|
||||
|
Reference in New Issue
Block a user