fix(api): publish a validated dataset, and stop truncating search
Reload cleared the caches first and rebuilt afterwards, so any failure left the API serving nothing, and requests arriving mid-reload saw a half-swapped state. It now builds and validates the replacement frames, place registry, reverse index and sitemaps off the request loop, then publishes them in one synchronous step under a lock. A failed reload returns 503 and keeps the previous data. Sitemap regeneration takes the same path rather than clearing the live registry up front. Typesense search returned at most one page of hits and used an empty list for both "no matches" and "search is down", so a genuine empty result silently fell back to substring matching. It now pages through every candidate and returns None only on failure; the fallback matches literally, since a query containing regex metacharacters used to throw. Empty datasets answer 503 rather than 200-with-nothing or a misleading 404, so callers can tell an outage from an absent school. Adds /api/release, which reports the build identity baked into the image. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
38bc17cab3
commit
9b75f54206
4 files changed
+253
-47
No files matched your search
+34
-14
@@ -84,21 +84,37 @@ def _get_typesense_client():
|
||||
return None
|
||||
|
||||
|
||||
def search_schools_typesense(query: str, limit: int = 250) -> List[int]:
|
||||
"""Search Typesense. Returns URNs in relevance order, or [] if unavailable."""
|
||||
def search_schools_typesense(query: str) -> Optional[List[int]]:
|
||||
"""Return all matching URNs in relevance order; None means unavailable.
|
||||
|
||||
Filtering and user pagination happen in the API after this search. Returning
|
||||
only the first search page would silently discard valid local matches.
|
||||
Never return a partial candidate set if a later page fails.
|
||||
"""
|
||||
client = _get_typesense_client()
|
||||
if client is None:
|
||||
return []
|
||||
return None
|
||||
urns = []
|
||||
try:
|
||||
result = client.collections["schools"].documents.search({
|
||||
"q": query,
|
||||
"query_by": "school_name,local_authority,postcode",
|
||||
"per_page": min(limit, 250),
|
||||
"typo_tokens_threshold": 1,
|
||||
})
|
||||
return [int(h["document"]["urn"]) for h in result.get("hits", [])]
|
||||
page = 1
|
||||
while True:
|
||||
result = client.collections["schools"].documents.search({
|
||||
"q": query,
|
||||
"query_by": "school_name,local_authority,postcode",
|
||||
"per_page": 250,
|
||||
"page": page,
|
||||
"typo_tokens_threshold": 1,
|
||||
})
|
||||
hits = result.get("hits", [])
|
||||
urns.extend(int(h["document"]["urn"]) for h in hits)
|
||||
if len(urns) >= result.get("found", len(urns)):
|
||||
return list(dict.fromkeys(urns))
|
||||
if not hits:
|
||||
raise ValueError("Search pagination ended before all matches arrived")
|
||||
page += 1
|
||||
except Exception:
|
||||
return []
|
||||
logging.getLogger(__name__).exception("School search unavailable")
|
||||
return None
|
||||
|
||||
|
||||
# The most a public endpoint will return in one response.
|
||||
@@ -502,7 +518,12 @@ def load_latest_school_data() -> pd.DataFrame:
|
||||
if _df_latest_cache is not None:
|
||||
return _df_latest_cache
|
||||
|
||||
df = load_school_data()
|
||||
_df_latest_cache = build_latest_school_data(load_school_data())
|
||||
return _df_latest_cache
|
||||
|
||||
|
||||
def build_latest_school_data(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Build a replacement snapshot without mutating the published caches."""
|
||||
if df.empty:
|
||||
return df
|
||||
|
||||
@@ -535,8 +556,7 @@ def load_latest_school_data() -> pd.DataFrame:
|
||||
df_latest = pd.concat([df_latest, df_no_perf], ignore_index=True)
|
||||
|
||||
print(f"Latest-snapshot cache built: {len(df_latest)} schools")
|
||||
_df_latest_cache = df_latest
|
||||
return _df_latest_cache
|
||||
return df_latest
|
||||
|
||||
|
||||
def clear_cache():
|
||||
|
||||
Reference in new issue
Block a user