Compare commits

...
1 Commits
Author SHA1 Message Date
TudorandClaude Opus 5.5 94bfac9caf fix(pipeline): decode GIAS extracts as Windows-1252
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m16s
PR Checks / Backend Smoke (pull_request) Successful in 10s
PR Checks / Build Backend (no push) (pull_request) Successful in 18s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m27s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 1m17s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 19s
GIAS publishes its CSVs in Windows-1252 and sends no charset. The tap read
resp.text, so requests guessed the codec, and the encoding="latin-1" passed
to read_csv did nothing on already-decoded text. On 3 Oct 2026 the guess was
windows-1250, and "St Thomas à Becket" (138950, 149557) was stored as
"St Thomas ŕ Becket". A different guess on another day would garble other
accented names.

Both streams now decode the downloaded bytes themselves (gias_csv.py). A byte
Windows-1252 leaves undefined becomes U+FFFD with a logged warning instead of
failing the load, so one odd name cannot stop the daily refresh. None of the
nine extracts checked (1 Jul to 3 Oct 2026) contains such a byte.

Checked by running the tap on the real 3 Oct extract with .text forced to
windows-1250: all 52,586 rows decode, with no "ŕ" and no replacement
characters.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-04 10:09:39 +01:00
3 changed files with 96 additions and 18 deletions

No files matched your search

@@ -0,0 +1,26 @@
"""Read a GIAS extract from the raw bytes of the download.
GIAS writes its CSVs in Windows-1252 and sends no charset, so `resp.text`
leaves requests to guess the codec. On 3 Oct 2026 it guessed windows-1250 and
"à" became "ŕ". Decode the bytes ourselves instead.
"""
from __future__ import annotations
import io
import pandas as pd
GIAS_ENCODING = "cp1252"
def read_gias_csv(content: bytes, logger=None) -> pd.DataFrame:
"""Every column as a string; a blank cell stays ''."""
# Windows-1252 leaves five bytes undefined. One stray byte must not stop
# the daily refresh of every school, so it becomes U+FFFD and is logged.
text = content.decode(GIAS_ENCODING, errors="replace")
undecodable = text.count("�")
if undecodable and logger is not None:
logger.warning("%d byte(s) in the GIAS extract could not be decoded as %s",
undecodable, GIAS_ENCODING)
return pd.read_csv(io.StringIO(text), dtype=str, keep_default_na=False)
@@ -7,6 +7,8 @@ from datetime import date, timedelta
from singer_sdk import Stream, Tap from singer_sdk import Stream, Tap
from singer_sdk import typing as th from singer_sdk import typing as th
from tap_uk_gias.gias_csv import read_gias_csv
GIAS_URL_TEMPLATE = ( GIAS_URL_TEMPLATE = (
"https://ea-edubase-api-prod.azurewebsites.net" "https://ea-edubase-api-prod.azurewebsites.net"
"/edubase/downloads/public/edubasealldata{date}.csv" "/edubase/downloads/public/edubasealldata{date}.csv"
@@ -74,9 +76,6 @@ class GIASEstablishmentsStream(Stream):
def get_records(self, context): def get_records(self, context):
"""Download GIAS CSV and yield rows.""" """Download GIAS CSV and yield rows."""
import io
import pandas as pd
import requests import requests
today = date.today() today = date.today()
@@ -94,12 +93,7 @@ class GIASEstablishmentsStream(Stream):
resp.raise_for_status() resp.raise_for_status()
df = pd.read_csv( df = read_gias_csv(resp.content, self.logger)
io.StringIO(resp.text),
encoding="latin-1",
dtype=str,
keep_default_na=False,
)
for _, row in df.iterrows(): for _, row in df.iterrows():
record = row.to_dict() record = row.to_dict()
@@ -126,9 +120,6 @@ class GIASLinksStream(Stream):
def get_records(self, context): def get_records(self, context):
"""Download GIAS links CSV and yield rows.""" """Download GIAS links CSV and yield rows."""
import io
import pandas as pd
import requests import requests
today = date.today() today = date.today()
@@ -146,12 +137,7 @@ class GIASLinksStream(Stream):
resp.raise_for_status() resp.raise_for_status()
df = pd.read_csv( df = read_gias_csv(resp.content, self.logger)
io.StringIO(resp.text),
encoding="latin-1",
dtype=str,
keep_default_na=False,
)
for _, row in df.iterrows(): for _, row in df.iterrows():
record = row.to_dict() record = row.to_dict()
+66
View File
@@ -0,0 +1,66 @@
"""GIAS publishes its extracts in Windows-1252 and declares no charset.
The tap used to hand pandas `resp.text`, so requests guessed the codec.
On 3 Oct 2026 it guessed windows-1250, and "St Thomas à Becket" was stored
as "St Thomas ŕ Becket". The `encoding=` passed to read_csv did nothing,
because the text was already decoded.
"""
import importlib.util
import logging
from pathlib import Path
import pytest
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-gias'
/ 'tap_uk_gias' / 'gias_csv.py')
@pytest.fixture
def gias_csv():
spec = importlib.util.spec_from_file_location('gias_csv', MODULE)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
# Byte for byte as GIAS writes it: 0xE0 à, 0x92 ’, 0xE9 é, 0xB0 °, 0xE7 ç.
EXTRACT = (
b'"URN","EstablishmentName","HeadLastName"\r\n'
b'"138950","St Thomas \xe0 Becket Catholic Secondary School","Smith"\r\n'
b'"100000","The Dean and Chapter of St Paul\x92s Cathedral","Pr\xe9vert"\r\n'
b'"140677","North Star 180\xb0","Fran\xe7ois"\r\n'
b'"100001","No head recorded",""\r\n'
)
def test_names_decode_as_windows_1252(gias_csv):
df = gias_csv.read_gias_csv(EXTRACT)
assert list(df['EstablishmentName']) == [
'St Thomas à Becket Catholic Secondary School',
'The Dean and Chapter of St Paul’s Cathedral',
'North Star 180°',
'No head recorded',
]
assert list(df['HeadLastName']) == ['Smith', 'Prévert', 'François', '']
def test_the_codec_requests_guessed_is_not_used(gias_csv):
# What the tap stored on 3 Oct: the same bytes read as windows-1250.
assert 'ŕ' in EXTRACT.decode('cp1250')
names = ' '.join(gias_csv.read_gias_csv(EXTRACT)['EstablishmentName'])
assert 'ŕ' not in names
def test_values_stay_strings(gias_csv):
df = gias_csv.read_gias_csv(EXTRACT)
assert df.loc[0, 'URN'] == '138950'
def test_a_byte_windows_1252_leaves_undefined_does_not_stop_the_load(gias_csv, caplog):
# 0x81 has no Windows-1252 character. One odd name must not block the daily
# refresh of every school, but it must be visible in the log.
extract = b'"URN","EstablishmentName"\r\n"100002","Odd \x81 Name"\r\n'
with caplog.at_level(logging.WARNING):
df = gias_csv.read_gias_csv(extract, logger=logging.getLogger('gias'))
assert df.loc[0, 'EstablishmentName'] == 'Odd � Name'
assert 'could not be decoded' in caplog.text