Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
94bfac9caf | ||
|
|
423b27140c | ||
|
|
1c62e8247d | ||
|
|
59ea8a4bdd |
No files matched your search
@@ -410,6 +410,29 @@ test('search and the school page agree on how many pupils a secondary has', asyn
|
||||
expect(school.total_pupils).toBe(detail.school_info.total_pupils);
|
||||
});
|
||||
|
||||
test('a secondary search row compares its Attainment 8 with the LA average', async ({ page }) => {
|
||||
// The comparison vanished unnoticed: the averages were fetched with
|
||||
// force-cache, so one stored failure hid it in that browser for good.
|
||||
// Playwright disables the HTTP cache when it intercepts requests, so this
|
||||
// guards the comparison itself; the unit test pins the cache mode.
|
||||
const la = await (await page.request.get('/api/la-averages')).json();
|
||||
const averages: Record<string, number> = la.secondary?.attainment_8_by_la ?? {};
|
||||
const res = await page.request.get('/api/schools?search=school&phase=secondary&page_size=50');
|
||||
expect(res.ok()).toBeTruthy();
|
||||
const school = ((await res.json()).schools ?? []).find(
|
||||
(s: { attainment_8_score?: number | null; local_authority?: string; school_type?: string }) =>
|
||||
s.attainment_8_score != null && s.local_authority != null && averages[s.local_authority] != null
|
||||
&& !/special|pupil referral|alternative provision/i.test(s.school_type ?? ''));
|
||||
test.skip(!school, 'no mainstream secondary with an LA average here');
|
||||
|
||||
await searchByName(page, school.school_name);
|
||||
const link = page.locator(`a[href^="/school/${school.urn}-"]`).first();
|
||||
await expect(link).toBeVisible({ timeout: 15_000 });
|
||||
const stats = link.locator('xpath=ancestor::div[contains(@class, "__rowContent")][1]')
|
||||
.locator('[class*="__line3"]');
|
||||
await expect(stats.getByText(/vs LA avg/)).toBeVisible();
|
||||
});
|
||||
|
||||
test('a phase outside primary/secondary filters to that phase, not to everything', async ({ page }) => {
|
||||
// The search page offers every GIAS phase, but the API only knew the grouped
|
||||
// ones and silently dropped the rest — so "Nursery" returned primaries.
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { act, fireEvent, render, screen } from '@testing-library/react';
|
||||
import { HomeView } from '@/components/HomeView';
|
||||
import { fetchSchools } from '@/lib/api';
|
||||
import { fetchLAaverages, fetchSchools } from '@/lib/api';
|
||||
import { primaryFixture } from '../support/schoolFixtures';
|
||||
import type { SchoolsResponse, School } from '@/lib/types';
|
||||
|
||||
@@ -84,3 +84,23 @@ test('failed map requests can be retried by reopening the map', async () => {
|
||||
expect(fetchSchools).toHaveBeenCalledTimes(2);
|
||||
expect(screen.getByTestId('map')).toHaveTextContent('Retry result');
|
||||
});
|
||||
|
||||
test('LA averages are not fetched with force-cache, so one failure is not replayed for good', async () => {
|
||||
// force-cache serves any stored response, however old, without asking the
|
||||
// server. A request that failed once (a staging deploy restart, the July
|
||||
// proxy outage) was stored and replayed on every later visit, and the
|
||||
// "vs LA avg" delta vanished from every secondary row in that browser.
|
||||
// The default mode honours the API's Cache-Control and never reuses an
|
||||
// error.
|
||||
params = new URLSearchParams('search=high');
|
||||
const secondary: SchoolsResponse = {
|
||||
...response('Alpha High'),
|
||||
schools: [{ ...primaryFixture.schoolInfo, school_name: 'Alpha High', phase: 'Secondary', attainment_8_score: 50 }],
|
||||
};
|
||||
render(<HomeView initialSchools={secondary} filters={filters} />);
|
||||
await act(async () => {});
|
||||
expect(fetchLAaverages).toHaveBeenCalled();
|
||||
for (const [options] of jest.mocked(fetchLAaverages).mock.calls) {
|
||||
expect(options?.cache).not.toBe('force-cache');
|
||||
}
|
||||
});
|
||||
@@ -358,10 +358,14 @@ export function HomeView({ initialSchools, filters, totalSchools, howItWorks, ed
|
||||
return () => controller.abort();
|
||||
}, [resultsView, searchParams, initialSchools.schools]);
|
||||
|
||||
// Fetch LA averages when secondary or mixed schools are visible
|
||||
// Fetch LA averages when secondary or mixed schools are visible. Default
|
||||
// cache mode, never force-cache: force-cache replays any stored response
|
||||
// without asking the server, so one failed request (a deploy restart) hid
|
||||
// every "vs LA avg" delta in that browser for good. The API's Cache-Control
|
||||
// already lets the browser reuse a good answer for five minutes.
|
||||
useEffect(() => {
|
||||
if (!isSecondaryView && !isMixedView) return;
|
||||
fetchLAaverages({ cache: 'force-cache' })
|
||||
fetchLAaverages()
|
||||
.then(data => setLaAverages(data.secondary.attainment_8_by_la))
|
||||
.catch(() => {});
|
||||
}, [isSecondaryView, isMixedView]);
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Read a GIAS extract from the raw bytes of the download.
|
||||
|
||||
GIAS writes its CSVs in Windows-1252 and sends no charset, so `resp.text`
|
||||
leaves requests to guess the codec. On 3 Oct 2026 it guessed windows-1250 and
|
||||
"à" became "ŕ". Decode the bytes ourselves instead.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pandas as pd
|
||||
|
||||
GIAS_ENCODING = "cp1252"
|
||||
|
||||
|
||||
def read_gias_csv(content: bytes, logger=None) -> pd.DataFrame:
|
||||
"""Every column as a string; a blank cell stays ''."""
|
||||
# Windows-1252 leaves five bytes undefined. One stray byte must not stop
|
||||
# the daily refresh of every school, so it becomes U+FFFD and is logged.
|
||||
text = content.decode(GIAS_ENCODING, errors="replace")
|
||||
undecodable = text.count("�")
|
||||
if undecodable and logger is not None:
|
||||
logger.warning("%d byte(s) in the GIAS extract could not be decoded as %s",
|
||||
undecodable, GIAS_ENCODING)
|
||||
return pd.read_csv(io.StringIO(text), dtype=str, keep_default_na=False)
|
||||
@@ -7,6 +7,8 @@ from datetime import date, timedelta
|
||||
from singer_sdk import Stream, Tap
|
||||
from singer_sdk import typing as th
|
||||
|
||||
from tap_uk_gias.gias_csv import read_gias_csv
|
||||
|
||||
GIAS_URL_TEMPLATE = (
|
||||
"https://ea-edubase-api-prod.azurewebsites.net"
|
||||
"/edubase/downloads/public/edubasealldata{date}.csv"
|
||||
@@ -74,9 +76,6 @@ class GIASEstablishmentsStream(Stream):
|
||||
|
||||
def get_records(self, context):
|
||||
"""Download GIAS CSV and yield rows."""
|
||||
import io
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
|
||||
today = date.today()
|
||||
@@ -94,12 +93,7 @@ class GIASEstablishmentsStream(Stream):
|
||||
|
||||
resp.raise_for_status()
|
||||
|
||||
df = pd.read_csv(
|
||||
io.StringIO(resp.text),
|
||||
encoding="latin-1",
|
||||
dtype=str,
|
||||
keep_default_na=False,
|
||||
)
|
||||
df = read_gias_csv(resp.content, self.logger)
|
||||
|
||||
for _, row in df.iterrows():
|
||||
record = row.to_dict()
|
||||
@@ -126,9 +120,6 @@ class GIASLinksStream(Stream):
|
||||
|
||||
def get_records(self, context):
|
||||
"""Download GIAS links CSV and yield rows."""
|
||||
import io
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
|
||||
today = date.today()
|
||||
@@ -146,12 +137,7 @@ class GIASLinksStream(Stream):
|
||||
|
||||
resp.raise_for_status()
|
||||
|
||||
df = pd.read_csv(
|
||||
io.StringIO(resp.text),
|
||||
encoding="latin-1",
|
||||
dtype=str,
|
||||
keep_default_na=False,
|
||||
)
|
||||
df = read_gias_csv(resp.content, self.logger)
|
||||
|
||||
for _, row in df.iterrows():
|
||||
record = row.to_dict()
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
"""GIAS publishes its extracts in Windows-1252 and declares no charset.
|
||||
|
||||
The tap used to hand pandas `resp.text`, so requests guessed the codec.
|
||||
On 3 Oct 2026 it guessed windows-1250, and "St Thomas à Becket" was stored
|
||||
as "St Thomas ŕ Becket". The `encoding=` passed to read_csv did nothing,
|
||||
because the text was already decoded.
|
||||
"""
|
||||
import importlib.util
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-gias'
|
||||
/ 'tap_uk_gias' / 'gias_csv.py')
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def gias_csv():
|
||||
spec = importlib.util.spec_from_file_location('gias_csv', MODULE)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
# Byte for byte as GIAS writes it: 0xE0 à, 0x92 ’, 0xE9 é, 0xB0 °, 0xE7 ç.
|
||||
EXTRACT = (
|
||||
b'"URN","EstablishmentName","HeadLastName"\r\n'
|
||||
b'"138950","St Thomas \xe0 Becket Catholic Secondary School","Smith"\r\n'
|
||||
b'"100000","The Dean and Chapter of St Paul\x92s Cathedral","Pr\xe9vert"\r\n'
|
||||
b'"140677","North Star 180\xb0","Fran\xe7ois"\r\n'
|
||||
b'"100001","No head recorded",""\r\n'
|
||||
)
|
||||
|
||||
|
||||
def test_names_decode_as_windows_1252(gias_csv):
|
||||
df = gias_csv.read_gias_csv(EXTRACT)
|
||||
assert list(df['EstablishmentName']) == [
|
||||
'St Thomas à Becket Catholic Secondary School',
|
||||
'The Dean and Chapter of St Paul’s Cathedral',
|
||||
'North Star 180°',
|
||||
'No head recorded',
|
||||
]
|
||||
assert list(df['HeadLastName']) == ['Smith', 'Prévert', 'François', '']
|
||||
|
||||
|
||||
def test_the_codec_requests_guessed_is_not_used(gias_csv):
|
||||
# What the tap stored on 3 Oct: the same bytes read as windows-1250.
|
||||
assert 'ŕ' in EXTRACT.decode('cp1250')
|
||||
names = ' '.join(gias_csv.read_gias_csv(EXTRACT)['EstablishmentName'])
|
||||
assert 'ŕ' not in names
|
||||
|
||||
|
||||
def test_values_stay_strings(gias_csv):
|
||||
df = gias_csv.read_gias_csv(EXTRACT)
|
||||
assert df.loc[0, 'URN'] == '138950'
|
||||
|
||||
|
||||
def test_a_byte_windows_1252_leaves_undefined_does_not_stop_the_load(gias_csv, caplog):
|
||||
# 0x81 has no Windows-1252 character. One odd name must not block the daily
|
||||
# refresh of every school, but it must be visible in the log.
|
||||
extract = b'"URN","EstablishmentName"\r\n"100002","Odd \x81 Name"\r\n'
|
||||
with caplog.at_level(logging.WARNING):
|
||||
df = gias_csv.read_gias_csv(extract, logger=logging.getLogger('gias'))
|
||||
assert df.loc[0, 'EstablishmentName'] == 'Odd � Name'
|
||||
assert 'could not be decoded' in caplog.text
|
||||
Reference in new issue
Block a user