From 1a6d349dad219191cab7f074b8aecd383f07c89a Mon Sep 17 00:00:00 2001 From: Tudor Date: Wed, 26 Aug 2026 20:34:34 +0100 Subject: [PATCH] feat(suggest): GET /api/suggest, cacheable and DataFrame-free MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A dedicated endpoint rather than a mode of /api/schools, because that path filters and sorts 25,000 pandas rows per query while holding the GIL — affordable once per search, not once per keystroke. A test asserts the distinction directly by making load_school_data raise and requiring the endpoint to answer anyway. Nothing errors on ordinary input: a short query, no matches, or Typesense being down are all 200 with an empty list. Cached deliberately. Prefix queries repeat enormously across users and school names change once a year, so s-maxage plus the existing ETag middleware turns most keystrokes into 304s. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj --- backend/app.py | 34 +++++++++++++++++++++ backend/tests/test_suggest.py | 57 +++++++++++++++++++++++++++++++++++ 2 files changed, 91 insertions(+) diff --git a/backend/app.py b/backend/app.py index ab5a3e7..832a039 100644 --- a/backend/app.py +++ b/backend/app.py @@ -33,6 +33,7 @@ from .data_loader import ( get_supplementary_data, get_supplementary_data_batch, search_schools_typesense, + suggest_schools_typesense, ) from .data_loader import get_data_info as get_db_info from . import flags @@ -383,6 +384,7 @@ CACHE_RULES: list[tuple[str, tuple[int, int, int]]] = [ ("/api/schools/", (300, 3600, 86400)), # /api/schools/{urn} ("/api/rankings", (60, 600, 3600)), ("/api/compare", (60, 600, 3600)), + ("/api/suggest", (60, 3600, 86400)), # autosuggest ("/api/schools", (30, 300, 1800)), # search list ] @@ -1299,6 +1301,38 @@ async def get_place(request: Request, kind: str, slug: str, } +# Two characters. One is not a query — it matches thousands of schools and the +# response is useless, so it is not worth a round trip. +SUGGEST_MIN_QUERY = 2 + + +@app.get("/api/suggest") +@limiter.limit("120/minute") +async def suggest_schools( + request: Request, + q: str = Query("", max_length=100), + limit: int = Query(8, ge=1, le=20), +): + """School name suggestions, from Typesense alone. + + Deliberately not a mode of /api/schools: that path filters and sorts the + full in-memory DataFrame, which is far too expensive to run per keystroke. + + Nothing here returns an error for ordinary input. A short query, no + matches, or Typesense being unreachable are all 200 with an empty list — + a dropdown that quietly does not appear is the right failure for a + keystroke path, and there is no DataFrame fallback because the 25,000-row + substring scan is precisely what this endpoint exists to avoid. + + 120/minute rather than the default 60: a 200 ms debounce makes typing + legitimately bursty. + """ + query = q.strip() + if len(query) < SUGGEST_MIN_QUERY: + return {"suggestions": []} + return {"suggestions": suggest_schools_typesense(query, limit)} + + @app.get("/api/flags") @limiter.limit(f"{settings.rate_limit_per_minute}/minute") async def get_feature_flags(request: Request): diff --git a/backend/tests/test_suggest.py b/backend/tests/test_suggest.py index b380f8d..b2bcc53 100644 --- a/backend/tests/test_suggest.py +++ b/backend/tests/test_suggest.py @@ -69,3 +69,60 @@ def test_the_limit_is_passed_through_and_clamped(monkeypatch): _use(monkeypatch, client) data_loader.suggest_schools_typesense("x", limit=500) assert client.docs.last_params["per_page"] == 20 + + +def _client(monkeypatch, rows, *, blow_up_dataframe=False): + from fastapi.testclient import TestClient + from backend import app as app_module + + monkeypatch.setattr(app_module, "suggest_schools_typesense", + lambda q, limit=8: rows) + if blow_up_dataframe: + def _boom(): + raise AssertionError("the suggest path must not load the DataFrame") + monkeypatch.setattr(app_module, "load_school_data", _boom) + monkeypatch.setattr(app_module, "load_latest_school_data", _boom) + return TestClient(app_module.app, raise_server_exceptions=False) + + +def test_the_endpoint_returns_suggestions(monkeypatch): + body = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck").json() + assert body["suggestions"][0]["school_name"] == "Brecknock Primary School" + + +def test_the_endpoint_never_touches_the_dataframe(monkeypatch): + """The whole reason this is not a mode of /api/schools. + + That endpoint filters and sorts 25,000 rows of pandas per query, holding + the GIL. Per keystroke, that is the cost this endpoint exists to avoid. + """ + res = _client(monkeypatch, [_HIT], blow_up_dataframe=True).get("/api/suggest?q=breck") + assert res.status_code == 200 + assert res.json()["suggestions"] + + +def test_a_one_character_query_returns_nothing_and_does_not_error(monkeypatch): + # The keystroke path never errors on ordinary input. + res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=b") + assert res.status_code == 200 + assert res.json() == {"suggestions": []} + + +def test_a_blank_query_returns_nothing_and_does_not_error(monkeypatch): + res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=") + assert res.status_code == 200 + assert res.json() == {"suggestions": []} + + +def test_typesense_down_is_an_empty_list_not_a_500(monkeypatch): + res = _client(monkeypatch, []).get("/api/suggest?q=breck") + assert res.status_code == 200 + assert res.json() == {"suggestions": []} + + +def test_the_response_is_cacheable(monkeypatch): + # Prefix queries repeat enormously across users, and school names change + # once a year. Without this the endpoint pays full price every keystroke. + res = _client(monkeypatch, [_HIT]).get("/api/suggest?q=breck") + assert "s-maxage" in res.headers.get("cache-control", "") + assert res.headers.get("etag")