Compare commits

..
1 Commits
Author SHA1 Message Date
TudorandClaude Opus 5.5 94bfac9caf fix(pipeline): decode GIAS extracts as Windows-1252
PR Checks / Frontend Typecheck + Tests (pull_request) Successful in 1m16s
PR Checks / Backend Smoke (pull_request) Successful in 10s
PR Checks / Build Backend (no push) (pull_request) Successful in 18s
PR Checks / Build Frontend (no push) (pull_request) Successful in 1m27s
PR Checks / Build Pipeline (no push) (pull_request) Successful in 1m17s
PR Checks / AI Code Review (Claude) (pull_request) Successful in 19s
GIAS publishes its CSVs in Windows-1252 and sends no charset. The tap read
resp.text, so requests guessed the codec, and the encoding="latin-1" passed
to read_csv did nothing on already-decoded text. On 3 Oct 2026 the guess was
windows-1250, and "St Thomas à Becket" (138950, 149557) was stored as
"St Thomas ŕ Becket". A different guess on another day would garble other
accented names.

Both streams now decode the downloaded bytes themselves (gias_csv.py). A byte
Windows-1252 leaves undefined becomes U+FFFD with a logged warning instead of
failing the load, so one odd name cannot stop the daily refresh. None of the
nine extracts checked (1 Jul to 3 Oct 2026) contains such a byte.

Checked by running the tap on the real 3 Oct extract with .text forced to
windows-1250: all 52,586 rows decode, with no "ŕ" and no replacement
characters.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-04 10:09:39 +01:00
9 changed files with 194 additions and 36 deletions

No files matched your search

@@ -8,6 +8,7 @@
import { render, screen } from '@testing-library/react'; import { render, screen } from '@testing-library/react';
import { import {
nearbyNoun,
NearbySchoolsSection, NearbySchoolsSection,
shouldRenderNearby, shouldRenderNearby,
} from '@/components/school/NearbySchoolsSection'; } from '@/components/school/NearbySchoolsSection';
@@ -42,7 +43,13 @@ function school(overrides: Partial<NearbySchool> = {}): NearbySchool {
function renderSection(nearby: NearbySchool[]) { function renderSection(nearby: NearbySchool[]) {
return render( return render(
<NearbySchoolsSection urn={100001} thisMetricValue={72} nearby={nearby} />, <NearbySchoolsSection
urn={100001}
schoolName="Meadowbrook Primary School"
phase="Primary"
thisMetricValue={72}
nearby={nearby}
/>,
); );
} }
@@ -83,10 +90,15 @@ describe('what the section claims', () => {
}); });
it('shows no chips at all when nothing is shared, rather than inventing one', () => { it('shows no chips at all when nothing is shared, rather than inventing one', () => {
const { container } = renderSection([ const { container } = render(
school({ shared: [] }), <NearbySchoolsSection
school({ urn: 100003, shared: [] }), urn={100001}
]); schoolName="Meadowbrook Primary School"
phase="Primary"
thisMetricValue={72}
nearby={[school({ shared: [] }), school({ urn: 100003, shared: [] })]}
/>,
);
// The card still carries its distance, name, type and figure — just no // The card still carries its distance, name, type and figure — just no
// claim of likeness. // claim of likeness.
expect(container.querySelectorAll('li ul').length).toBe(0); expect(container.querySelectorAll('li ul').length).toBe(0);
@@ -94,6 +106,37 @@ describe('what the section claims', () => {
}); });
}); });
describe('what the lede calls the set', () => {
it.each([
['Primary', 'primary schools'],
['Middle deemed primary', 'primary schools'],
['Secondary', 'secondary schools'],
['Middle deemed secondary', 'secondary schools'],
['All-through', 'all-through schools'],
// GIAS phase 6. Its candidates span the whole secondary group, so no
// single noun fits and it takes the honest general one.
['16 plus', 'schools and colleges'],
['', 'schools'],
[null, 'schools'],
])('calls a %s school\'s neighbours "%s"', (phase, expected) => {
expect(nearbyNoun(phase)).toBe(expected);
});
it('never calls a sixth form college\'s neighbours primary schools', () => {
render(
<NearbySchoolsSection
urn={100001}
schoolName="Barnet Sixth Form College"
phase="16 plus"
thisMetricValue={null}
nearby={[school(), school({ urn: 100003 })]}
/>,
);
expect(screen.getByText(/Other schools and colleges near Barnet Sixth Form College/)).toBeInTheDocument();
expect(screen.queryByText(/primary schools/)).not.toBeInTheDocument();
});
});
describe('cards', () => { describe('cards', () => {
it('links each school to its canonical slug', () => { it('links each school to its canonical slug', () => {
renderSection([school(), school({ urn: 100003, school_name: 'Oakfield Primary School' })]); renderSection([school(), school({ urn: 100003, school_name: 'Oakfield Primary School' })]);
@@ -1,7 +1,8 @@
.heading { font-family: var(--font-display); font-size: 1.4rem; letter-spacing: -0.4px; margin: 0; } .heading { font-family: var(--font-display); font-size: 1.4rem; letter-spacing: -0.4px; margin: 0; }
.lede { margin: 0.5rem 0 1.25rem; color: var(--text-secondary); max-width: 64ch; }
.caption { margin: 1rem 0 0; font-size: 0.72rem; color: var(--text-muted); } .caption { margin: 1rem 0 0; font-size: 0.72rem; color: var(--text-muted); }
.top { display: flex; align-items: center; justify-content: space-between; gap: 1rem; margin-bottom: 1.25rem; } .top { display: flex; align-items: flex-start; justify-content: space-between; gap: 1rem; }
.arrows { display: flex; gap: 0.5rem; flex: none; } .arrows { display: flex; gap: 0.5rem; flex: none; }
.arrow { width: 44px; height: 44px; display: grid; place-items: center; cursor: pointer; border: 1px solid var(--border-strong); border-radius: 999px; background: var(--bg-card); color: var(--brand); } .arrow { width: 44px; height: 44px; display: grid; place-items: center; cursor: pointer; border: 1px solid var(--border-strong); border-radius: 999px; background: var(--bg-card); color: var(--brand); }
.arrow:hover:not(:disabled) { border-color: var(--brand); background: var(--brand-bg); } .arrow:hover:not(:disabled) { border-color: var(--brand); background: var(--brand-bg); }
@@ -14,10 +15,10 @@
.scroller { display: grid; grid-auto-flow: column; grid-auto-columns: calc((100% - 1.8rem) / 3); gap: 0.9rem; overflow-x: auto; scroll-snap-type: x mandatory; padding: 2px; margin: -2px; list-style: none; scrollbar-width: none; -ms-overflow-style: none; } .scroller { display: grid; grid-auto-flow: column; grid-auto-columns: calc((100% - 1.8rem) / 3); gap: 0.9rem; overflow-x: auto; scroll-snap-type: x mandatory; padding: 2px; margin: -2px; list-style: none; scrollbar-width: none; -ms-overflow-style: none; }
.scroller::-webkit-scrollbar { display: none; } .scroller::-webkit-scrollbar { display: none; }
@media (max-width: 820px) { .scroller { grid-auto-columns: calc((100% - 0.9rem) / 2); } } @media (max-width: 820px) { .scroller { grid-auto-columns: calc((100% - 0.9rem) / 2); } }
/* Touch widths (MOBILE.md): the arrows would take 96px from a 328px card, for /* Touch widths (MOBILE.md): the arrows would take 96px from a 328px card and
a control swiping already provides. They go, and the documented right-edge crush the lede into four lines, for a control swiping already provides. They
fade carries the affordance — lifting at the end of the travel, where there go, and the documented right-edge fade carries the affordance — lifting at
is nothing more to hint at. */ the end of the travel, where there is nothing more to hint at. */
@media (max-width: 640px) { @media (max-width: 640px) {
.top { display: block; } .top { display: block; }
.arrows { display: none; } .arrows { display: none; }
@@ -4,7 +4,7 @@
* The scroller and its arrows. * The scroller and its arrows.
* *
* `children` are the server-rendered cards and `header` the server-rendered * `children` are the server-rendered cards and `header` the server-rendered
* heading: both stay server components, passed through, so this file * heading and lede: both stay server components, passed through, so this file
* owns a DOM ref and nothing else. That is what keeps all six links in the * owns a DOM ref and nothing else. That is what keeps all six links in the
* initial HTML — a carousel that mounted cards on click would put four of the * initial HTML — a carousel that mounted cards on click would put four of the
* six beyond a crawler and beyond a reader with no JavaScript. * six beyond a crawler and beyond a reader with no JavaScript.
@@ -13,10 +13,8 @@
* reached on their behalf. * reached on their behalf.
* *
* There is deliberately no "how these are chosen" panel: the method is already * There is deliberately no "how these are chosen" panel: the method is already
* visible in the chips and the distances. The single caption line is not a * visible in the lede, the chips and the distances. The single caption line is
* method note — it is the one thing a card cannot self-correct. * not a method note — it is the one thing a card cannot self-correct.
*
* Nor is there a lede: "Other primary schools near X" only restated the heading.
*/ */
import Link from 'next/link'; import Link from 'next/link';
@@ -34,6 +32,28 @@ export function shouldRenderNearby(nearby?: NearbySchool[] | null): boolean {
return (nearby?.length ?? 0) >= MINIMUM; return (nearby?.length ?? 0) >= MINIMUM;
} }
/**
* What the lede calls the set of schools it is showing.
*
* Derived from the school's own GIAS phase rather than the template it renders
* with, because those disagree for "16 plus" (GIAS phase 6): a sixth-form
* college renders the primary template — computeSchoolFlags tests for the
* substring "secondary" — while the backend correctly matches it against the
* secondary group. Taking the noun from the template would print "Other primary
* schools near <sixth form college>" above a row of secondaries.
*
* A 16-plus school's candidates span the whole secondary group, so no single
* noun fits and it gets the honest general one.
*/
export function nearbyNoun(phase: string | null | undefined): string {
const text = (phase ?? '').trim().toLowerCase();
if (text === 'all-through') return 'all-through schools';
if (text === '16 plus') return 'schools and colleges';
if (text.includes('secondary')) return 'secondary schools';
if (text.includes('primary')) return 'primary schools';
return 'schools';
}
function metricLabel(key: string): string { function metricLabel(key: string): string {
return key === 'attainment_8_score' ? 'Attainment 8' : 'Reading, writing & maths'; return key === 'attainment_8_score' ? 'Attainment 8' : 'Reading, writing & maths';
} }
@@ -45,16 +65,25 @@ function formatMetric(value: number | null, key: string): string {
export function NearbySchoolsSection({ export function NearbySchoolsSection({
urn, urn,
schoolName,
phase,
thisMetricValue, thisMetricValue,
nearby, nearby,
}: { }: {
urn: number; urn: number;
schoolName: string;
/** The school's own GIAS phase, not the template it renders with. */
phase: string | null | undefined;
thisMetricValue: number | null; thisMetricValue: number | null;
nearby?: NearbySchool[] | null; nearby?: NearbySchool[] | null;
}) { }) {
if (!shouldRenderNearby(nearby)) return null; if (!shouldRenderNearby(nearby)) return null;
const schools = nearby as NearbySchool[]; const schools = nearby as NearbySchool[];
// One card matched on phase alone, so the section may not claim the set
// shares an intake with this school.
const metricKey = schools[0].metric_key; const metricKey = schools[0].metric_key;
const noun = nearbyNoun(phase);
return ( return (
<Section id="nearby"> <Section id="nearby">
@@ -62,9 +91,12 @@ export function NearbySchoolsSection({
count={schools.length} count={schools.length}
labelledBy="nearby-schools-heading" labelledBy="nearby-schools-heading"
header={ header={
<h2 id="nearby-schools-heading" className={styles.heading}> <div>
Other schools nearby <h2 id="nearby-schools-heading" className={styles.heading}>
</h2> Other schools nearby
</h2>
<p className={styles.lede}>{`Other ${noun} near ${schoolName}.`}</p>
</div>
} }
> >
{schools.map((school) => ( {schools.map((school) => (
@@ -154,6 +154,8 @@ export function PrimarySchoolSections({
{/* Last: it is where the reader goes next, not part of this school. */} {/* Last: it is where the reader goes next, not part of this school. */}
<NearbySchoolsSection <NearbySchoolsSection
urn={schoolInfo.urn} urn={schoolInfo.urn}
schoolName={schoolInfo.school_name}
phase={schoolInfo.phase}
thisMetricValue={flags.latestResults?.rwm_expected_pct ?? null} thisMetricValue={flags.latestResults?.rwm_expected_pct ?? null}
nearby={nearbySchools} nearby={nearbySchools}
/> />
@@ -148,6 +148,8 @@ export function SecondarySchoolSections({
{/* Last: it is where the reader goes next, not part of this school. */} {/* Last: it is where the reader goes next, not part of this school. */}
<NearbySchoolsSection <NearbySchoolsSection
urn={schoolInfo.urn} urn={schoolInfo.urn}
schoolName={schoolInfo.school_name}
phase={schoolInfo.phase}
thisMetricValue={flags.latestResults?.attainment_8_score ?? null} thisMetricValue={flags.latestResults?.attainment_8_score ?? null}
nearby={nearbySchools} nearby={nearbySchools}
/> />
@@ -0,0 +1,26 @@
"""Read a GIAS extract from the raw bytes of the download.
GIAS writes its CSVs in Windows-1252 and sends no charset, so `resp.text`
leaves requests to guess the codec. On 3 Oct 2026 it guessed windows-1250 and
"à" became "ŕ". Decode the bytes ourselves instead.
"""
from __future__ import annotations
import io
import pandas as pd
GIAS_ENCODING = "cp1252"
def read_gias_csv(content: bytes, logger=None) -> pd.DataFrame:
"""Every column as a string; a blank cell stays ''."""
# Windows-1252 leaves five bytes undefined. One stray byte must not stop
# the daily refresh of every school, so it becomes U+FFFD and is logged.
text = content.decode(GIAS_ENCODING, errors="replace")
undecodable = text.count("�")
if undecodable and logger is not None:
logger.warning("%d byte(s) in the GIAS extract could not be decoded as %s",
undecodable, GIAS_ENCODING)
return pd.read_csv(io.StringIO(text), dtype=str, keep_default_na=False)
@@ -7,6 +7,8 @@ from datetime import date, timedelta
from singer_sdk import Stream, Tap from singer_sdk import Stream, Tap
from singer_sdk import typing as th from singer_sdk import typing as th
from tap_uk_gias.gias_csv import read_gias_csv
GIAS_URL_TEMPLATE = ( GIAS_URL_TEMPLATE = (
"https://ea-edubase-api-prod.azurewebsites.net" "https://ea-edubase-api-prod.azurewebsites.net"
"/edubase/downloads/public/edubasealldata{date}.csv" "/edubase/downloads/public/edubasealldata{date}.csv"
@@ -74,9 +76,6 @@ class GIASEstablishmentsStream(Stream):
def get_records(self, context): def get_records(self, context):
"""Download GIAS CSV and yield rows.""" """Download GIAS CSV and yield rows."""
import io
import pandas as pd
import requests import requests
today = date.today() today = date.today()
@@ -94,12 +93,7 @@ class GIASEstablishmentsStream(Stream):
resp.raise_for_status() resp.raise_for_status()
df = pd.read_csv( df = read_gias_csv(resp.content, self.logger)
io.StringIO(resp.text),
encoding="latin-1",
dtype=str,
keep_default_na=False,
)
for _, row in df.iterrows(): for _, row in df.iterrows():
record = row.to_dict() record = row.to_dict()
@@ -126,9 +120,6 @@ class GIASLinksStream(Stream):
def get_records(self, context): def get_records(self, context):
"""Download GIAS links CSV and yield rows.""" """Download GIAS links CSV and yield rows."""
import io
import pandas as pd
import requests import requests
today = date.today() today = date.today()
@@ -146,12 +137,7 @@ class GIASLinksStream(Stream):
resp.raise_for_status() resp.raise_for_status()
df = pd.read_csv( df = read_gias_csv(resp.content, self.logger)
io.StringIO(resp.text),
encoding="latin-1",
dtype=str,
keep_default_na=False,
)
for _, row in df.iterrows(): for _, row in df.iterrows():
record = row.to_dict() record = row.to_dict()
+66
View File
@@ -0,0 +1,66 @@
"""GIAS publishes its extracts in Windows-1252 and declares no charset.
The tap used to hand pandas `resp.text`, so requests guessed the codec.
On 3 Oct 2026 it guessed windows-1250, and "St Thomas à Becket" was stored
as "St Thomas ŕ Becket". The `encoding=` passed to read_csv did nothing,
because the text was already decoded.
"""
import importlib.util
import logging
from pathlib import Path
import pytest
MODULE = (Path(__file__).resolve().parents[1] / 'plugins' / 'extractors' / 'tap-uk-gias'
/ 'tap_uk_gias' / 'gias_csv.py')
@pytest.fixture
def gias_csv():
spec = importlib.util.spec_from_file_location('gias_csv', MODULE)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
# Byte for byte as GIAS writes it: 0xE0 à, 0x92 ’, 0xE9 é, 0xB0 °, 0xE7 ç.
EXTRACT = (
b'"URN","EstablishmentName","HeadLastName"\r\n'
b'"138950","St Thomas \xe0 Becket Catholic Secondary School","Smith"\r\n'
b'"100000","The Dean and Chapter of St Paul\x92s Cathedral","Pr\xe9vert"\r\n'
b'"140677","North Star 180\xb0","Fran\xe7ois"\r\n'
b'"100001","No head recorded",""\r\n'
)
def test_names_decode_as_windows_1252(gias_csv):
df = gias_csv.read_gias_csv(EXTRACT)
assert list(df['EstablishmentName']) == [
'St Thomas à Becket Catholic Secondary School',
'The Dean and Chapter of St Paul’s Cathedral',
'North Star 180°',
'No head recorded',
]
assert list(df['HeadLastName']) == ['Smith', 'Prévert', 'François', '']
def test_the_codec_requests_guessed_is_not_used(gias_csv):
# What the tap stored on 3 Oct: the same bytes read as windows-1250.
assert 'ŕ' in EXTRACT.decode('cp1250')
names = ' '.join(gias_csv.read_gias_csv(EXTRACT)['EstablishmentName'])
assert 'ŕ' not in names
def test_values_stay_strings(gias_csv):
df = gias_csv.read_gias_csv(EXTRACT)
assert df.loc[0, 'URN'] == '138950'
def test_a_byte_windows_1252_leaves_undefined_does_not_stop_the_load(gias_csv, caplog):
# 0x81 has no Windows-1252 character. One odd name must not block the daily
# refresh of every school, but it must be visible in the log.
extract = b'"URN","EstablishmentName"\r\n"100002","Odd \x81 Name"\r\n'
with caplog.at_level(logging.WARNING):
df = gias_csv.read_gias_csv(extract, logger=logging.getLogger('gias'))
assert df.loc[0, 'EstablishmentName'] == 'Odd � Name'
assert 'could not be decoded' in caplog.text