Compare commits

...
Author SHA1 Message Date
TudorandClaude Opus 5 792766d308 fix(ci): make a failed release check say what it actually saw
The staging poller swallowed every failure identically, so a run that
timed out told us only that the expected release never appeared — not
whether the proxy refused us, the endpoint was down, or the containers
were still serving an older build. The public staging proxy also answers
403 to urllib's default user agent while the release endpoint is healthy,
which looked exactly like a deployment that never arrived.

Identify the poller, and report each distinct observation once: HTTP
status, connection failure type, invalid JSON, or the release identities
actually reported. The timeout error carries the last observation and the
identity it wanted. Responses and the base URL stay out of the logs —
only validated sha/build_id fields are echoed back.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 16:01:59 +01:00
TudorandClaude Opus 5 5ad1cbfb53 fix(api): bound the search candidate set instead of draining Typesense
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m12s
PR Checks / Backend Smoke (pull_request) Successful in 10s
PR Checks / Build Backend (no push) (pull_request) Successful in 17s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m18s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 39s
PR Checks / AI Code Review (Claude) (pull_request) Failing after 6m44s
Fetching every match kept scoped searches correct but left the number of
round trips in the caller's hands: a one-letter query, or a deliberately
broad one, walked the whole collection a page at a time.

Cap the candidate set at 1,000 URNs — four pages — and return the
relevance-ordered prefix when the ceiling is hit. That is still far more
than one page, so the API's own authority, phase and postcode filters
keep the matches they need, while latency and upstream load stay bounded.
A capped query is logged so a genuinely truncated search is visible.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 11:21:45 +01:00
6 changed files with 193 additions and 17 deletions

No files matched your search

+30 -9
View File
@@ -84,34 +84,55 @@ def _get_typesense_client():
return None
def search_schools_typesense(query: str) -> Optional[List[int]]:
"""Return all matching URNs in relevance order; None means unavailable.
SEARCH_PAGE_SIZE = 250
# Search results are filtered again by the API (authority, phase, postcode,
# etc.), so one page is too small for scoped searches. Keep the candidate set
# bounded, though: a broad query must not turn into an unbounded sequence of
# Typesense requests. Four pages is enough to preserve useful scoped matches
# while putting a hard ceiling on latency and upstream load.
SEARCH_MAX_CANDIDATES = 1_000
Filtering and user pagination happen in the API after this search. Returning
only the first search page would silently discard valid local matches.
Never return a partial candidate set if a later page fails.
def search_schools_typesense(query: str) -> Optional[List[int]]:
"""Return a bounded set of matching URNs in relevance order.
``None`` means Typesense is unavailable; ``[]`` is a valid zero-match
result. The API applies its remaining filters after this search, so the
first few pages are fetched rather than only the first page. Once the
candidate ceiling is reached, the relevance-ordered prefix is returned on
purpose; fetching every match would make common or adversarial queries
unbounded.
"""
client = _get_typesense_client()
if client is None:
return None
urns = []
urns: list[int] = []
fetched = 0
try:
page = 1
while True:
while fetched < SEARCH_MAX_CANDIDATES:
page_size = min(SEARCH_PAGE_SIZE, SEARCH_MAX_CANDIDATES - fetched)
result = client.collections["schools"].documents.search({
"q": query,
"query_by": "school_name,local_authority,postcode",
"per_page": 250,
"per_page": page_size,
"page": page,
"typo_tokens_threshold": 1,
})
hits = result.get("hits", [])
urns.extend(int(h["document"]["urn"]) for h in hits)
if len(urns) >= result.get("found", len(urns)):
fetched += len(hits)
if fetched >= result.get("found", fetched):
return list(dict.fromkeys(urns))
if not hits:
raise ValueError("Search pagination ended before all matches arrived")
page += 1
logging.getLogger(__name__).info(
"Typesense search capped at %d candidates for query %r",
SEARCH_MAX_CANDIDATES,
query,
)
return list(dict.fromkeys(urns))
except Exception:
logging.getLogger(__name__).exception("School search unavailable")
return None
+41
View File
@@ -23,6 +23,47 @@ def test_search_returns_matches_beyond_first_page(monkeypatch):
assert pages == [1, 2]
def test_search_caps_broad_queries_at_a_bounded_number_of_pages(monkeypatch):
requests = []
def search(params):
requests.append(params)
return {
'found': 10_000,
'hits': [
{'document': {'urn': 100000 + params['page'] * 1000 + i}}
for i in range(params['per_page'])
],
}
client_for(monkeypatch, search)
result = data_loader.search_schools_typesense('school')
assert len(result) == data_loader.SEARCH_MAX_CANDIDATES
assert len(requests) == data_loader.SEARCH_MAX_CANDIDATES // data_loader.SEARCH_PAGE_SIZE
assert all(request['per_page'] == data_loader.SEARCH_PAGE_SIZE for request in requests)
assert requests[-1]['page'] == len(requests)
def test_search_uses_a_smaller_final_page_when_the_cap_is_not_a_page_multiple(monkeypatch):
monkeypatch.setattr(data_loader, 'SEARCH_MAX_CANDIDATES', 251)
requests = []
def search(params):
requests.append(params)
return {
'found': 10_000,
'hits': [{'document': {'urn': 100000 + len(requests) * 1000 + i}}
for i in range(params['per_page'])],
}
client_for(monkeypatch, search)
result = data_loader.search_schools_typesense('school')
assert len(result) == 251
assert [request['per_page'] for request in requests] == [250, 1]
def test_later_page_failure_does_not_return_partial_results(monkeypatch):
def search(params):
if params['page'] == 2:
+4 -2
View File
@@ -104,8 +104,10 @@ alias-update response must never cause deletion of a potentially live index.
The backend snapshot swap is process-local and assumes the current single-worker
deployment. It is not an atomic transaction spanning PostgreSQL marts, Typesense
and Next.js caches. Next.js caches are not explicitly purged by the pipeline.
School search retrieves every Typesense candidate before applying API filters;
only a dependency failure invokes substring fallback, not a valid empty match set.
School search retrieves a relevance-ordered candidate prefix (currently capped at
1,000 URNs) before applying API filters. This keeps scoped searches useful while
putting a hard ceiling on Typesense round trips; only a dependency failure invokes
substring fallback, not a valid empty match set.
## Deployment references
+9
View File
@@ -291,6 +291,15 @@ workflow first. The release route must be reachable through the configured
`/api` proxy limitation. No new deployment secret is required.
`scripts/ci/release.py` implements identity polling and digest verification.
The poller identifies itself as `SchoolCompare-Release-Check/1.0`: the public
staging proxy has returned HTTP 403 to Python's default urllib user agent even
while the release endpoint was healthy. It logs changes in HTTP/connection
failures or observed release identities, and includes the last observation in
the timeout error. If verification fails, use that observation to distinguish
proxy rejection (403), an unavailable release endpoint (503), and containers
still reporting an older SHA/build ID. Check the configured base URL from the
CI runner; a successful request from another machine does not establish runner
connectivity. Do not bypass identity verification to unblock a deployment.
Its mocked tests run in PR checks alongside backend and index-publication tests.
The new Playwright journeys also check deployed identity and stale pagination.
Local unit checks do not validate registry credentials, Portainer behaviour,
+44 -6
View File
@@ -10,6 +10,7 @@ from pathlib import Path
import re
import subprocess
import time
from urllib.error import HTTPError, URLError
from urllib.request import Request, urlopen
COMPONENTS = ('BACKEND', 'FRONTEND', 'PIPELINE')
@@ -88,8 +89,29 @@ def promote(sha):
def matches(payload, sha, build_id):
return all(payload.get(component) == {'sha': sha, 'build_id': build_id}
for component in ('frontend', 'backend'))
return isinstance(payload, dict) and all(
payload.get(component) == {'sha': sha, 'build_id': build_id}
for component in ('frontend', 'backend'))
def describe_identity(payload):
"""Log only release fields, never arbitrary response bodies or secret URLs."""
if not isinstance(payload, dict):
return 'Invalid release response: expected a JSON object'
identities = []
for component in ('frontend', 'backend'):
identity = payload.get(component)
if not isinstance(identity, dict):
identities.append(f'{component}=missing or invalid')
continue
values = []
for field, length in (('sha', 40), ('build_id', 32)):
value = identity.get(field)
valid = isinstance(value, str) and (
value == 'development' or re.fullmatch(r'[0-9a-f]{' + str(length) + '}', value))
values.append(f'{field}={value if valid else "missing or invalid"}')
identities.append(f'{component}: {", ".join(values)}')
return 'Release mismatch: ' + '; '.join(identities)
def wait(base_url, sha, build_id, timeout):
@@ -97,19 +119,35 @@ def wait(base_url, sha, build_id, timeout):
if not re.fullmatch(r'[0-9a-f]{32}', build_id):
raise ValueError('Missing expected build identity')
deadline = time.monotonic() + timeout
last_observation = 'No response received'
print(f'Waiting for deployed release {sha} / {build_id}', flush=True)
while time.monotonic() < deadline:
try:
req = Request(f'{base_url.rstrip("/")}/release.json?check={time.time_ns()}',
headers={'Cache-Control': 'no-cache'})
headers={'Cache-Control': 'no-cache',
'User-Agent': 'SchoolCompare-Release-Check/1.0',
'Accept': 'application/json'})
with urlopen(req, timeout=min(10, max(.1, deadline - time.monotonic()))) as response:
payload = json.load(response)
if matches(payload, sha, build_id):
print(f'Verified deployed release {sha} / {build_id}')
return
except (OSError, ValueError):
pass
observation = describe_identity(payload)
except HTTPError as exc:
observation = f'Release endpoint returned HTTP {exc.code}'
exc.close()
except URLError as exc:
observation = f'Release endpoint connection failed ({type(exc.reason).__name__})'
except OSError as exc:
observation = f'Release endpoint request failed ({type(exc).__name__})'
except ValueError:
observation = 'Release endpoint returned invalid JSON or request configuration'
if observation != last_observation:
print(observation, flush=True)
last_observation = observation
time.sleep(min(5, max(0, deadline - time.monotonic())))
raise RuntimeError('Deployment did not report the expected frontend/backend release')
raise RuntimeError('Deployment did not report the expected frontend/backend release '
f'{sha} / {build_id}. Last observation: {last_observation}')
def main():
+65
View File
@@ -1,4 +1,6 @@
import json
from io import BytesIO
from urllib.error import HTTPError, URLError
from unittest.mock import Mock
import pytest
from scripts.ci import release
@@ -63,3 +65,66 @@ def test_missing_candidate_fails_before_any_tag_is_changed(docker):
docker.side_effect = RuntimeError('missing verified tag')
with pytest.raises(RuntimeError): release.promote(SHA)
assert not any(c.args[0] == 'create' for c in docker.call_args_list)
@pytest.fixture
def poll(monkeypatch):
now = [0.0]
monkeypatch.setattr(release.time, 'monotonic', lambda: now[0])
monkeypatch.setattr(release.time, 'sleep', lambda seconds: now.__setitem__(0, now[0] + seconds))
opener = Mock()
monkeypatch.setattr(release, 'urlopen', opener)
return opener
def response(payload):
return BytesIO(json.dumps(payload).encode())
def test_wait_identifies_its_client_and_retries_until_both_services_match(poll, capsys):
poll.side_effect = [
HTTPError('https://secret.example', 503, 'unavailable', {}, None),
response({'frontend': {'sha': SHA, 'build_id': BUILD},
'backend': {'sha': SHA, 'build_id': 'c' * 32}}),
response({component: {'sha': SHA, 'build_id': BUILD}
for component in ('frontend', 'backend')}),
]
release.wait('https://secret.example/', SHA, BUILD, 15)
assert poll.call_count == 3
request = poll.call_args.args[0]
assert request.get_header('User-agent') == 'SchoolCompare-Release-Check/1.0'
assert request.get_header('Cache-control') == 'no-cache'
assert request.get_header('Accept') == 'application/json'
assert '/release.json?check=' in request.full_url
output = capsys.readouterr().out
assert 'HTTP 503' in output
assert 'backend: sha=' + SHA + ', build_id=' + 'c' * 32 in output
assert 'Verified deployed release' in output
assert 'secret.example' not in output
@pytest.mark.parametrize('failure, expected', [
(lambda: HTTPError('https://secret.example', 403, 'secret response', {}, None), 'HTTP 403'),
(lambda: URLError(OSError('secret address')), 'connection failed (OSError)'),
(lambda: TimeoutError('secret address'), 'request failed (TimeoutError)'),
(lambda: BytesIO(b'<html>secret response</html>'), 'invalid JSON'),
(lambda: response([]), 'expected a JSON object'),
(lambda: response({'frontend': {'sha': 'secret response'}}), 'missing or invalid'),
])
def test_wait_timeout_reports_last_failure_without_leaking_response_or_url(poll, capsys, failure, expected):
poll.side_effect = lambda *args, **kwargs: result_or_raise(failure())
with pytest.raises(RuntimeError) as error:
release.wait('https://secret.example', SHA, BUILD, 10)
assert expected in str(error.value)
assert SHA in str(error.value)
assert BUILD in str(error.value)
output = capsys.readouterr().out
assert sum(expected in line for line in output.splitlines()) == 1
assert 'secret' not in output + str(error.value)
assert poll.call_count == 2
def result_or_raise(result):
if isinstance(result, Exception):
raise result
return result