feat(suggest): Typesense rows for autosuggest, no DataFrame

search_schools_typesense returns URNs, which forces the caller to
hydrate from the 25,000-row in-memory frame. Every field a suggestion
needs is already in the Typesense document, so this returns documents
and the caller needs no pandas at all — the difference between a query
that can run per keystroke and one that cannot.

Never raises. Typesense unreachable or erroring gives an empty list,
because a dropdown that quietly stops appearing is the right failure for
a keystroke path.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015mWQnpye9F299NVRCCSRvj
This commit is contained in:
TudorandClaude Opus 5 committed 2026-08-26 20:33:42 +01:00
1 parent ff041544f2
commit 75d3534d82
2 files changed
+111

No files matched your search

+40
View File
@@ -100,6 +100,46 @@ def search_schools_typesense(query: str, limit: int = 250) -> List[int]:
return []
# The most a public endpoint will return in one response.
SUGGEST_MAX_LIMIT = 20
# Fields a suggestion row carries, and the default when the document omits an
# optional one. phase and school_type are optional in the Typesense schema.
_SUGGEST_FIELDS = ("school_name", "local_authority", "postcode",
"phase", "school_type")
def suggest_schools_typesense(query: str, limit: int = 8) -> List[dict]:
"""Autosuggest rows straight from Typesense. Never raises.
Returns documents rather than URNs, unlike search_schools_typesense, so the
caller needs no DataFrame. Every field below is already in the index — see
pipeline/scripts/sync_typesense.py — which is what makes this cheap enough
to run per keystroke.
"""
client = _get_typesense_client()
if client is None:
return []
try:
result = client.collections["schools"].documents.search({
"q": query,
"query_by": "school_name,local_authority",
"per_page": max(1, min(limit, SUGGEST_MAX_LIMIT)),
"typo_tokens_threshold": 1,
})
except Exception:
# A dropdown that quietly stops appearing is the right failure here.
return []
rows = []
for hit in result.get("hits", []):
doc = hit.get("document", {})
row = {"urn": int(doc.get("urn", 0))}
row.update({f: str(doc.get(f, "") or "") for f in _SUGGEST_FIELDS})
rows.append(row)
return rows
def normalize_school_type(school_type: Optional[str]) -> Optional[str]:
"""Convert cryptic school type codes to user-friendly names."""
if not school_type: