fix(ees): read the year from suffixed release slugs (review)

A '2025-26-revised' release read as an unknown year went last, behind the
'2025-26' release, which then owned the year and dropped every revised row.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 12:50:10 +01:00
1 parent 4ae3853277
commit e09d7a202b
3 files changed
+44 -10

No files matched your search

@@ -11,8 +11,23 @@ requirements, can load it.
"""
from __future__ import annotations
import re
import pandas as pd
_SLUG_YEAR = re.compile(r"^(\d{4})-(\d{2})(?:-|$)")
def slug_to_time_period(slug: str) -> str | None:
"""A release slug's academic year as a time_period: '2022-23' → '202223'.
Suffixed slugs ('2024-25-revised', '2025-26-provisional') give the same
year, so newest_first places them by year; read as unknown, a revised
release went last and lost its year to the first release.
"""
match = _SLUG_YEAR.match(slug or "")
return match.group(1) + match.group(2) if match else None
def newest_first(releases: list[dict]) -> list[dict]:
"""Releases by time_period, newest first. A release whose time_period is
@@ -24,7 +24,12 @@ from tap_uk_ees.ks4_summary import (
headline_record,
headline_rows,
)
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
from tap_uk_ees.release_precedence import (
drop_owned_periods,
newest_first,
periods_in,
slug_to_time_period,
)
CONTENT_API_BASE = (
"https://content.explore-education-statistics.service.gov.uk/api"
@@ -40,14 +45,6 @@ def get_content_release_id(publication_slug: str) -> str:
return resp.json()["id"]
def _slug_to_time_period(slug: str) -> str | None:
"""Convert a release slug like '2022-23' to a time_period like '202223'."""
parts = slug.split("-")
if len(parts) == 2 and len(parts[0]) == 4 and len(parts[1]) == 2:
return parts[0] + parts[1]
return None
def get_all_releases(publication_slug: str) -> list[dict]:
"""Return all releases for a publication as dicts with 'id' and 'time_period'.
@@ -72,7 +69,7 @@ def get_all_releases(publication_slug: str) -> list[dict]:
total_pages = paging.get("totalPages", 1)
for r in releases:
time_period = _slug_to_time_period(r.get("slug", ""))
time_period = slug_to_time_period(r.get("slug", ""))
result.append({"id": r["id"], "time_period": time_period})
if page >= total_pages:
@@ -76,3 +76,25 @@ def test_only_the_ks4_results_stream_opts_in():
opted_in = [chunk.split('(')[0] for chunk in source.split('\nclass ')[1:]
if re.search(r'_newest_release_owns_period\s*=\s*True', chunk)]
assert opted_in == ['EESKS4PerformanceStream']
@pytest.mark.parametrize('slug, period', [
('2024-25', '202425'),
('2024-25-revised', '202425'),
('2025-26-provisional', '202526'),
('latest', None),
('', None),
])
def test_a_release_slug_gives_its_year_whatever_its_suffix(precedence, slug, period):
# KS2 already publishes "2024-25-revised" and "2025-26-provisional". An
# unread suffix sent the release last, behind the provisional one for the
# same year, which then owned the year and dropped every revised row.
assert precedence.slug_to_time_period(slug) == period
def test_within_a_year_the_api_order_is_kept(precedence):
# The API lists a year's revised release before its first release.
releases = [{'id': 'revised', 'time_period': '202526'},
{'id': 'first', 'time_period': '202526'},
{'id': 'older', 'time_period': '202425'}]
assert [r['id'] for r in precedence.newest_first(releases)] == ['revised', 'first', 'older']