fix(ees): read the year from suffixed release slugs (review)

A '2025-26-revised' release read as an unknown year went last, behind the
'2025-26' release, which then owned the year and dropped every revised row.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-06 12:50:10 +01:00
1 parent 4ae3853277
commit e09d7a202b
3 files changed
+44 -10

No files matched your search

@@ -11,8 +11,23 @@ requirements, can load it.
"""
from __future__ import annotations
import re
import pandas as pd
_SLUG_YEAR = re.compile(r"^(\d{4})-(\d{2})(?:-|$)")
def slug_to_time_period(slug: str) -> str | None:
"""A release slug's academic year as a time_period: '2022-23' → '202223'.
Suffixed slugs ('2024-25-revised', '2025-26-provisional') give the same
year, so newest_first places them by year; read as unknown, a revised
release went last and lost its year to the first release.
"""
match = _SLUG_YEAR.match(slug or "")
return match.group(1) + match.group(2) if match else None
def newest_first(releases: list[dict]) -> list[dict]:
"""Releases by time_period, newest first. A release whose time_period is
@@ -24,7 +24,12 @@ from tap_uk_ees.ks4_summary import (
headline_record,
headline_rows,
)
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
from tap_uk_ees.release_precedence import (
drop_owned_periods,
newest_first,
periods_in,
slug_to_time_period,
)
CONTENT_API_BASE = (
"https://content.explore-education-statistics.service.gov.uk/api"
@@ -40,14 +45,6 @@ def get_content_release_id(publication_slug: str) -> str:
return resp.json()["id"]
def _slug_to_time_period(slug: str) -> str | None:
"""Convert a release slug like '2022-23' to a time_period like '202223'."""
parts = slug.split("-")
if len(parts) == 2 and len(parts[0]) == 4 and len(parts[1]) == 2:
return parts[0] + parts[1]
return None
def get_all_releases(publication_slug: str) -> list[dict]:
"""Return all releases for a publication as dicts with 'id' and 'time_period'.
@@ -72,7 +69,7 @@ def get_all_releases(publication_slug: str) -> list[dict]:
total_pages = paging.get("totalPages", 1)
for r in releases:
time_period = _slug_to_time_period(r.get("slug", ""))
time_period = slug_to_time_period(r.get("slug", ""))
result.append({"id": r["id"], "time_period": time_period})
if page >= total_pages: