PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m12s
PR Checks / Backend Smoke (pull_request) Successful in 10s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m18s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 39s
PR Checks / AI Code Review (Claude) (pull_request) Failing after 6m44s
Fetching every match kept scoped searches correct but left the number of round trips in the caller's hands: a one-letter query, or a deliberately broad one, walked the whole collection a page at a time. Cap the candidate set at 1,000 URNs — four pages — and return the relevance-ordered prefix when the ceiling is hit. That is still far more than one page, so the API's own authority, phase and postcode filters keep the matches they need, while latency and upstream load stay bounded. A capped query is logged so a genuinely truncated search is visible. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
115 lines
4.6 KiB
Python
115 lines
4.6 KiB
Python
from types import SimpleNamespace
|
|
import pytest
|
|
from fastapi.testclient import TestClient
|
|
from backend import app as api, data_loader
|
|
from backend.tests.test_sixth_form_flag import _schools_df
|
|
|
|
|
|
def client_for(monkeypatch, search):
|
|
client = SimpleNamespace(collections={'schools': SimpleNamespace(documents=SimpleNamespace(search=search))})
|
|
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: client)
|
|
|
|
|
|
def test_search_returns_matches_beyond_first_page(monkeypatch):
|
|
pages = []
|
|
def search(params):
|
|
pages.append(params['page'])
|
|
urns = range(100000, 100250) if params['page'] == 1 else [100999]
|
|
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
|
|
client_for(monkeypatch, search)
|
|
result = data_loader.search_schools_typesense('academy')
|
|
assert len(result) == 251
|
|
assert result[-1] == 100999
|
|
assert pages == [1, 2]
|
|
|
|
|
|
def test_search_caps_broad_queries_at_a_bounded_number_of_pages(monkeypatch):
|
|
requests = []
|
|
|
|
def search(params):
|
|
requests.append(params)
|
|
return {
|
|
'found': 10_000,
|
|
'hits': [
|
|
{'document': {'urn': 100000 + params['page'] * 1000 + i}}
|
|
for i in range(params['per_page'])
|
|
],
|
|
}
|
|
|
|
client_for(monkeypatch, search)
|
|
result = data_loader.search_schools_typesense('school')
|
|
|
|
assert len(result) == data_loader.SEARCH_MAX_CANDIDATES
|
|
assert len(requests) == data_loader.SEARCH_MAX_CANDIDATES // data_loader.SEARCH_PAGE_SIZE
|
|
assert all(request['per_page'] == data_loader.SEARCH_PAGE_SIZE for request in requests)
|
|
assert requests[-1]['page'] == len(requests)
|
|
|
|
|
|
def test_search_uses_a_smaller_final_page_when_the_cap_is_not_a_page_multiple(monkeypatch):
|
|
monkeypatch.setattr(data_loader, 'SEARCH_MAX_CANDIDATES', 251)
|
|
requests = []
|
|
|
|
def search(params):
|
|
requests.append(params)
|
|
return {
|
|
'found': 10_000,
|
|
'hits': [{'document': {'urn': 100000 + len(requests) * 1000 + i}}
|
|
for i in range(params['per_page'])],
|
|
}
|
|
|
|
client_for(monkeypatch, search)
|
|
result = data_loader.search_schools_typesense('school')
|
|
|
|
assert len(result) == 251
|
|
assert [request['per_page'] for request in requests] == [250, 1]
|
|
|
|
|
|
def test_later_page_failure_does_not_return_partial_results(monkeypatch):
|
|
def search(params):
|
|
if params['page'] == 2:
|
|
raise RuntimeError('timeout')
|
|
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in range(100000, 100250)]}
|
|
client_for(monkeypatch, search)
|
|
assert data_loader.search_schools_typesense('academy') is None
|
|
|
|
|
|
def test_zero_matches_are_distinct_from_unavailable(monkeypatch):
|
|
client_for(monkeypatch, lambda _: {'found': 0, 'hits': []})
|
|
assert data_loader.search_schools_typesense('academy') == []
|
|
monkeypatch.setattr(data_loader, '_get_typesense_client', lambda: None)
|
|
assert data_loader.search_schools_typesense('academy') is None
|
|
|
|
|
|
@pytest.mark.parametrize('matches, expected', [([], []), (None, [100001])])
|
|
def test_fallback_only_on_dependency_failure(monkeypatch, matches, expected):
|
|
monkeypatch.setattr(api.limiter, 'enabled', False)
|
|
monkeypatch.setattr(api, 'load_latest_school_data', _schools_df)
|
|
monkeypatch.setattr(api, 'search_schools_typesense', lambda _: matches)
|
|
response = TestClient(api.app).get('/api/schools?search=Alpha')
|
|
assert response.status_code == 200
|
|
assert [s['urn'] for s in response.json()['schools']] == expected
|
|
|
|
|
|
def test_filtered_api_keeps_match_from_second_search_page(monkeypatch):
|
|
monkeypatch.setattr(api.limiter, 'enabled', False)
|
|
df = _schools_df()
|
|
monkeypatch.setattr(api, 'load_latest_school_data', lambda: df)
|
|
def search(params):
|
|
urns = range(200000, 200250) if params['page'] == 1 else [100001]
|
|
return {'found': 251, 'hits': [{'document': {'urn': u}} for u in urns]}
|
|
client_for(monkeypatch, search)
|
|
response = TestClient(api.app).get('/api/schools?search=Alpha&local_authority=Testshire')
|
|
assert response.status_code == 200
|
|
assert response.json()['total'] == 1
|
|
assert response.json()['schools'][0]['urn'] == 100001
|
|
|
|
|
|
def test_unavailable_dataset_is_not_a_missing_school_or_empty_search(monkeypatch):
|
|
import pandas as pd
|
|
monkeypatch.setattr(api.limiter, 'enabled', False)
|
|
monkeypatch.setattr(api, 'load_school_data', lambda: pd.DataFrame())
|
|
monkeypatch.setattr(api, 'load_latest_school_data', lambda: pd.DataFrame())
|
|
client = TestClient(api.app)
|
|
assert client.get('/api/schools/100001').status_code == 503
|
|
assert client.get('/api/schools?search=school').status_code == 503
|