Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b44fca902f |
+43
-2
@@ -262,15 +262,53 @@ assert "NULL AS has_sixth_form" in str(_MAIN_QUERY_NO_SIXTH_FORM), (
|
||||
"expected replacement of 's.has_sixth_form,' to have taken effect"
|
||||
)
|
||||
|
||||
# Fallback used when marts.dim_school predates the GIAS code-dictionary
|
||||
# migration (i.e. the nightly dbt pipeline hasn't rebuilt the mart yet on
|
||||
# this DB, so it still has the old name columns instead of *_code columns).
|
||||
_MAIN_QUERY_LEGACY_NAMES = str(_MAIN_QUERY)
|
||||
_LEGACY_NAME_REPLACEMENTS = [
|
||||
("s.phase_code,", "s.phase,"),
|
||||
("s.school_type_code,", "s.school_type,"),
|
||||
(
|
||||
"s.religious_character_code,",
|
||||
"s.religious_character AS religious_denomination,",
|
||||
),
|
||||
("s.status_code,", "s.status,"),
|
||||
("s.admissions_policy_code,", "s.admissions_policy,"),
|
||||
]
|
||||
for _old, _new in _LEGACY_NAME_REPLACEMENTS:
|
||||
assert _old in _MAIN_QUERY_LEGACY_NAMES, (
|
||||
f"expected {_old!r} to be present in _MAIN_QUERY before replacement"
|
||||
)
|
||||
_MAIN_QUERY_LEGACY_NAMES = _MAIN_QUERY_LEGACY_NAMES.replace(_old, _new)
|
||||
_MAIN_QUERY_LEGACY_NAMES = text(_MAIN_QUERY_LEGACY_NAMES)
|
||||
|
||||
_GIAS_CODE_COLUMN_NAMES = (
|
||||
"phase_code",
|
||||
"school_type_code",
|
||||
"religious_character_code",
|
||||
"status_code",
|
||||
"admissions_policy_code",
|
||||
)
|
||||
|
||||
|
||||
def load_school_data_as_dataframe() -> pd.DataFrame:
|
||||
"""Load all school + KS2 data as a pandas DataFrame."""
|
||||
try:
|
||||
df = pd.read_sql(_MAIN_QUERY, engine)
|
||||
except sqlalchemy.exc.ProgrammingError as exc:
|
||||
if "has_sixth_form" not in str(exc):
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
if any(col in str(exc) for col in _GIAS_CODE_COLUMN_NAMES):
|
||||
logging.getLogger(__name__).warning(
|
||||
"marts predate the GIAS code migration — falling back to "
|
||||
"legacy name-column query: %s",
|
||||
exc,
|
||||
)
|
||||
try:
|
||||
df = pd.read_sql(_MAIN_QUERY_LEGACY_NAMES, engine)
|
||||
except Exception as exc2:
|
||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
||||
return pd.DataFrame()
|
||||
elif "has_sixth_form" in str(exc):
|
||||
logging.getLogger(__name__).warning(
|
||||
"marts.dim_school is missing has_sixth_form (pipeline hasn't "
|
||||
"rebuilt the mart yet on this DB) — retrying without it: %s",
|
||||
@@ -281,6 +319,9 @@ def load_school_data_as_dataframe() -> pd.DataFrame:
|
||||
except Exception as exc2:
|
||||
print(f"Warning: Could not load school data from marts: {exc2}")
|
||||
return pd.DataFrame()
|
||||
else:
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
return pd.DataFrame()
|
||||
except Exception as exc:
|
||||
print(f"Warning: Could not load school data from marts: {exc}")
|
||||
return pd.DataFrame()
|
||||
|
||||
@@ -42,3 +42,55 @@ def test_missing_code_columns_are_a_noop():
|
||||
out = translate_gias_code_columns(df)
|
||||
assert out.iloc[0]["phase"] == "Primary"
|
||||
assert out.iloc[0]["status"] == "Open"
|
||||
|
||||
|
||||
def test_load_school_data_survives_premigration_marts(monkeypatch):
|
||||
"""Real prod state until the nightly pipeline first rebuilds the mart with
|
||||
the GIAS code columns: marts.dim_school still has the old name columns
|
||||
(phase, school_type, religious_character, status, admissions_policy)
|
||||
instead of the new *_code columns. The first query raises UndefinedColumn
|
||||
on s.phase_code; load_school_data_as_dataframe must retry with the
|
||||
legacy name-column query rather than swallow the error and return (and
|
||||
then have load_school_data cache) an empty DataFrame."""
|
||||
import sqlalchemy.exc
|
||||
from backend import data_loader
|
||||
|
||||
data_loader._df_cache = None
|
||||
data_loader._df_latest_cache = None
|
||||
|
||||
good_df = pd.DataFrame(
|
||||
[
|
||||
{
|
||||
"urn": 1,
|
||||
"school_name": "Legacy School",
|
||||
"phase": "Primary",
|
||||
"school_type": "Academy",
|
||||
"status": "Open",
|
||||
}
|
||||
]
|
||||
)
|
||||
calls = []
|
||||
|
||||
def fake_read_sql(query, con):
|
||||
calls.append(query)
|
||||
if len(calls) == 1:
|
||||
raise sqlalchemy.exc.ProgrammingError(
|
||||
"(psycopg2.errors.UndefinedColumn) column s.phase_code does not exist",
|
||||
None,
|
||||
None,
|
||||
)
|
||||
return good_df.copy()
|
||||
|
||||
monkeypatch.setattr(data_loader.pd, "read_sql", fake_read_sql)
|
||||
|
||||
try:
|
||||
df = data_loader.load_school_data_as_dataframe()
|
||||
finally:
|
||||
data_loader._df_cache = None
|
||||
data_loader._df_latest_cache = None
|
||||
|
||||
assert len(calls) == 2, "must retry with the legacy name-column query variant"
|
||||
assert calls[1] is data_loader._MAIN_QUERY_LEGACY_NAMES
|
||||
assert not df.empty
|
||||
assert df["phase"].iloc[0] == "Primary"
|
||||
assert df["status"].iloc[0] == "Open"
|
||||
|
||||
Reference in New Issue
Block a user