fix(api): survive missing has_sixth_form column and numpy bool serialization

- data_loader.load_school_data_as_dataframe now catches a ProgrammingError
  whose message mentions has_sixth_form (psycopg2 UndefinedColumn) and
  retries with a NULL-AS-has_sixth_form query variant, so the API keeps
  serving data (and the app.py column-fallback branch stays reachable)
  even before the nightly pipeline has rebuilt marts.dim_school.
- utils.convert_to_native now handles numpy.bool_ so GET /api/schools/{urn}
  doesn't 500 once has_sixth_form is a populated bool-dtype column.
- Update the now-stale comment on the app.py age-range fallback branch.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Tudor
2026-07-07 13:33:25 +01:00
co-authored by Claude Fable 5
parent 3fcb1340d4
commit a524cdc591
4 changed files with 115 additions and 1 deletions
+81
View File
@@ -95,3 +95,84 @@ def test_detail_payload_includes_flag(client):
resp = client.get("/api/schools/100002")
assert resp.status_code == 200, resp.text
assert resp.json()["school_info"]["has_sixth_form"] is True
def test_detail_payload_serializes_numpy_bool(monkeypatch):
"""Once the pipeline has run, has_sixth_form is a real bool dtype column
(dbt not_null test guarantees no NULLs), so row access yields
numpy.bool_ rather than a Python bool. convert_to_native must handle it —
otherwise FastAPI's jsonable_encoder raises ValueError and the detail
endpoint 500s (C2)."""
from backend import app as app_module
df = _schools_df()
# Drop the row with a None flag — this fixture models the post-pipeline
# state where the column is a genuine, fully-populated bool dtype.
df = df[df["has_sixth_form"].notna()].reset_index(drop=True)
df["has_sixth_form"] = df["has_sixth_form"].astype(bool)
assert df["has_sixth_form"].dtype == bool
monkeypatch.setattr(app_module, "load_school_data", lambda: df)
monkeypatch.setattr(app_module, "get_supplementary_data", lambda db, urn: {})
client = TestClient(app_module.app, raise_server_exceptions=False)
resp = client.get("/api/schools/100002")
assert resp.status_code == 200, resp.text
assert resp.json()["school_info"]["has_sixth_form"] is True
class _FakeProgrammingError(Exception):
"""Stand-in for sqlalchemy.exc.ProgrammingError wrapping psycopg2's
UndefinedColumn, without needing a real DB connection to construct one."""
def test_load_school_data_survives_missing_has_sixth_form_column(monkeypatch):
"""Real prod state until the nightly pipeline first rebuilds the mart:
marts.dim_school lacks has_sixth_form entirely. The first query raises
UndefinedColumn; load_school_data_as_dataframe must retry without the
column (synthesizing it as None) rather than swallow the error and
return (and then have load_school_data cache) an empty DataFrame (C1)."""
import sqlalchemy.exc
from backend import data_loader
data_loader._df_cache = None
data_loader._df_latest_cache = None
good_df = pd.DataFrame(
[
{
"urn": 1,
"school_name": "Fallback School",
"school_type": "Academy",
"has_sixth_form": None,
}
]
)
calls = []
def fake_read_sql(query, con):
calls.append(query)
if len(calls) == 1:
raise sqlalchemy.exc.ProgrammingError(
"SELECT ...",
None,
Exception(
"(psycopg2.errors.UndefinedColumn) column s.has_sixth_form "
"does not exist"
),
)
return good_df.copy()
monkeypatch.setattr(data_loader.pd, "read_sql", fake_read_sql)
try:
df = data_loader.load_school_data_as_dataframe()
finally:
data_loader._df_cache = None
data_loader._df_latest_cache = None
assert len(calls) == 2, "must retry with the no-sixth-form query variant"
assert calls[1] is data_loader._MAIN_QUERY_NO_SIXTH_FORM
assert not df.empty
assert "has_sixth_form" in df.columns
assert df["has_sixth_form"].iloc[0] is None