feat(suggest): GET /api/suggest, cacheable and DataFrame-free

A dedicated endpoint rather than a mode of /api/schools, because that
path filters and sorts 25,000 pandas rows per query while holding the
GIL — affordable once per search, not once per keystroke. A test asserts
the distinction directly by making load_school_data raise and requiring
the endpoint to answer anyway.

Nothing errors on ordinary input: a short query, no matches, or
Typesense being down are all 200 with an empty list.

Cached deliberately. Prefix queries repeat enormously across users and
school names change once a year, so s-maxage plus the existing ETag
middleware turns most keystrokes into 304s.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-26 20:34:34 +01:00
1 parent 75d3534d82
commit 1a6d349dad
2 files changed
+91

No files matched your search

+57
View File
@@ -69,3 +69,60 @@ def test_the_limit_is_passed_through_and_clamped(monkeypatch):
_use(monkeypatch, client)
data_loader.suggest_schools_typesense("x", limit=500)
assert client.docs.last_params["per_page"] == 20
def _client(monkeypatch, rows, *, blow_up_dataframe=False):
from fastapi.testclient import TestClient
from backend import app as app_module
monkeypatch.setattr(app_module, "suggest_schools_typesense",
lambda q, limit=8: rows)
if blow_up_dataframe:
def _boom():
raise AssertionError("the suggest path must not load the DataFrame")
monkeypatch.setattr(app_module, "load_school_data", _boom)
monkeypatch.setattr(app_module, "load_latest_school_data", _boom)
return TestClient(app_module.app, raise_server_exceptions=False)
def test_the_endpoint_returns_suggestions(monkeypatch):
body = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck").json()
assert body["suggestions"][0]["school_name"] == "Brecknock Primary School"
def test_the_endpoint_never_touches_the_dataframe(monkeypatch):
"""The whole reason this is not a mode of /api/schools.
That endpoint filters and sorts 25,000 rows of pandas per query, holding
the GIL. Per keystroke, that is the cost this endpoint exists to avoid.
"""
res = _client(monkeypatch, [_HIT], blow_up_dataframe=True).get("/api/suggest?q=breck")
assert res.status_code == 200
assert res.json()["suggestions"]
def test_a_one_character_query_returns_nothing_and_does_not_error(monkeypatch):
# The keystroke path never errors on ordinary input.
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=b")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_a_blank_query_returns_nothing_and_does_not_error(monkeypatch):
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_typesense_down_is_an_empty_list_not_a_500(monkeypatch):
res = _client(monkeypatch, []).get("/api/suggest?q=breck")
assert res.status_code == 200
assert res.json() == {"suggestions": []}
def test_the_response_is_cacheable(monkeypatch):
# Prefix queries repeat enormously across users, and school names change
# once a year. Without this the endpoint pays full price every keystroke.
res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck")
assert "s-maxage" in res.headers.get("cache-control", "")
assert res.headers.get("etag")