diff --git a/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/gias_csv.py b/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/gias_csv.py new file mode 100644 index 0000000..d76fa93 --- /dev/null +++ b/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/gias_csv.py @@ -0,0 +1,26 @@ +"""Read a GIAS extract from the raw bytes of the download. + +GIAS writes its CSVs in Windows-1252 and sends no charset, so `resp.text` +leaves requests to guess the codec. On 3 Oct 2026 it guessed windows-1250 and +"à" became "ŕ". Decode the bytes ourselves instead. +""" + +from __future__ import annotations + +import io + +import pandas as pd + +GIAS_ENCODING = "cp1252" + + +def read_gias_csv(content: bytes, logger=None) -> pd.DataFrame: + """Every column as a string; a blank cell stays ''.""" + # Windows-1252 leaves five bytes undefined. One stray byte must not stop + # the daily refresh of every school, so it becomes U+FFFD and is logged. + text = content.decode(GIAS_ENCODING, errors="replace") + undecodable = text.count("�") + if undecodable and logger is not None: + logger.warning("%d byte(s) in the GIAS extract could not be decoded as %s", + undecodable, GIAS_ENCODING) + return pd.read_csv(io.StringIO(text), dtype=str, keep_default_na=False) diff --git a/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py b/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py index 619de3b..4a217a0 100644 --- a/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py +++ b/pipeline/plugins/extractors/tap-uk-gias/tap_uk_gias/tap.py @@ -7,6 +7,8 @@ from datetime import date, timedelta from singer_sdk import Stream, Tap from singer_sdk import typing as th +from tap_uk_gias.gias_csv import read_gias_csv + GIAS_URL_TEMPLATE = ( "https://ea-edubase-api-prod.azurewebsites.net" "/edubase/downloads/public/edubasealldata{date}.csv" @@ -74,9 +76,6 @@ class GIASEstablishmentsStream(Stream): def get_records(self, context): """Download GIAS CSV and yield rows.""" - import io - - import pandas as pd import requests today = date.today() @@ -94,12 +93,7 @@ class GIASEstablishmentsStream(Stream): resp.raise_for_status() - df = pd.read_csv( - io.StringIO(resp.text), - encoding="latin-1", - dtype=str, - keep_default_na=False, - ) + df = read_gias_csv(resp.content, self.logger) for _, row in df.iterrows(): record = row.to_dict() @@ -126,9 +120,6 @@ class GIASLinksStream(Stream): def get_records(self, context): """Download GIAS links CSV and yield rows.""" - import io - - import pandas as pd import requests today = date.today() @@ -146,12 +137,7 @@ class GIASLinksStream(Stream): resp.raise_for_status() - df = pd.read_csv( - io.StringIO(resp.text), - encoding="latin-1", - dtype=str, - keep_default_na=False, - ) + df = read_gias_csv(resp.content, self.logger) for _, row in df.iterrows(): record = row.to_dict() diff --git a/pipeline/tests/test_gias_encoding.py b/pipeline/tests/test_gias_encoding.py new file mode 100644 index 0000000..038fa4b --- /dev/null +++ b/pipeline/tests/test_gias_encoding.py @@ -0,0 +1,66 @@ +"""GIAS publishes its extracts in Windows-1252 and declares no charset. + +The tap used to hand pandas `resp.text`, so requests guessed the codec. +On 3 Oct 2026 it guessed windows-1250, and "St Thomas à Becket" was stored +as "St Thomas ŕ Becket". The `encoding=` passed to read_csv did nothing, +because the text was already decoded. +""" +import importlib.util +import logging +from pathlib import Path + +import pytest + +MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-gias' + / 'tap_uk_gias' / 'gias_csv.py') + + +@pytest.fixture +def gias_csv(): + spec = importlib.util.spec_from_file_location('gias_csv', MODULE) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +# Byte for byte as GIAS writes it: 0xE0 à, 0x92 ’, 0xE9 é, 0xB0 °, 0xE7 ç. +EXTRACT = ( + b'"URN","EstablishmentName","HeadLastName"\r\n' + b'"138950","St Thomas \xe0 Becket Catholic Secondary School","Smith"\r\n' + b'"100000","The Dean and Chapter of St Paul\x92s Cathedral","Pr\xe9vert"\r\n' + b'"140677","North Star 180\xb0","Fran\xe7ois"\r\n' + b'"100001","No head recorded",""\r\n' +) + + +def test_names_decode_as_windows_1252(gias_csv): + df = gias_csv.read_gias_csv(EXTRACT) + assert list(df['EstablishmentName']) == [ + 'St Thomas à Becket Catholic Secondary School', + 'The Dean and Chapter of St Paul’s Cathedral', + 'North Star 180°', + 'No head recorded', + ] + assert list(df['HeadLastName']) == ['Smith', 'Prévert', 'François', ''] + + +def test_the_codec_requests_guessed_is_not_used(gias_csv): + # What the tap stored on 3 Oct: the same bytes read as windows-1250. + assert 'ŕ' in EXTRACT.decode('cp1250') + names = ' '.join(gias_csv.read_gias_csv(EXTRACT)['EstablishmentName']) + assert 'ŕ' not in names + + +def test_values_stay_strings(gias_csv): + df = gias_csv.read_gias_csv(EXTRACT) + assert df.loc[0, 'URN'] == '138950' + + +def test_a_byte_windows_1252_leaves_undefined_does_not_stop_the_load(gias_csv, caplog): + # 0x81 has no Windows-1252 character. One odd name must not block the daily + # refresh of every school, but it must be visible in the log. + extract = b'"URN","EstablishmentName"\r\n"100002","Odd \x81 Name"\r\n' + with caplog.at_level(logging.WARNING): + df = gias_csv.read_gias_csv(extract, logger=logging.getLogger('gias')) + assert df.loc[0, 'EstablishmentName'] == 'Odd � Name' + assert 'could not be decoded' in caplog.text