fix(ees): read the year from suffixed release slugs (review)
A '2025-26-revised' release read as an unknown year went last, behind the '2025-26' release, which then owned the year and dropped every revised row. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
4ae3853277
commit
e09d7a202b
3 files changed
+44
-10
No files matched your search
@@ -11,8 +11,23 @@ requirements, can load it.
|
|||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
|
|
||||||
|
_SLUG_YEAR = re.compile(r"^(\d{4})-(\d{2})(?:-|$)")
|
||||||
|
|
||||||
|
|
||||||
|
def slug_to_time_period(slug: str) -> str | None:
|
||||||
|
"""A release slug's academic year as a time_period: '2022-23' → '202223'.
|
||||||
|
|
||||||
|
Suffixed slugs ('2024-25-revised', '2025-26-provisional') give the same
|
||||||
|
year, so newest_first places them by year; read as unknown, a revised
|
||||||
|
release went last and lost its year to the first release.
|
||||||
|
"""
|
||||||
|
match = _SLUG_YEAR.match(slug or "")
|
||||||
|
return match.group(1) + match.group(2) if match else None
|
||||||
|
|
||||||
|
|
||||||
def newest_first(releases: list[dict]) -> list[dict]:
|
def newest_first(releases: list[dict]) -> list[dict]:
|
||||||
"""Releases by time_period, newest first. A release whose time_period is
|
"""Releases by time_period, newest first. A release whose time_period is
|
||||||
|
|||||||
@@ -24,7 +24,12 @@ from tap_uk_ees.ks4_summary import (
|
|||||||
headline_record,
|
headline_record,
|
||||||
headline_rows,
|
headline_rows,
|
||||||
)
|
)
|
||||||
from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in
|
from tap_uk_ees.release_precedence import (
|
||||||
|
drop_owned_periods,
|
||||||
|
newest_first,
|
||||||
|
periods_in,
|
||||||
|
slug_to_time_period,
|
||||||
|
)
|
||||||
|
|
||||||
CONTENT_API_BASE = (
|
CONTENT_API_BASE = (
|
||||||
"https://content.explore-education-statistics.service.gov.uk/api"
|
"https://content.explore-education-statistics.service.gov.uk/api"
|
||||||
@@ -40,14 +45,6 @@ def get_content_release_id(publication_slug: str) -> str:
|
|||||||
return resp.json()["id"]
|
return resp.json()["id"]
|
||||||
|
|
||||||
|
|
||||||
def _slug_to_time_period(slug: str) -> str | None:
|
|
||||||
"""Convert a release slug like '2022-23' to a time_period like '202223'."""
|
|
||||||
parts = slug.split("-")
|
|
||||||
if len(parts) == 2 and len(parts[0]) == 4 and len(parts[1]) == 2:
|
|
||||||
return parts[0] + parts[1]
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
def get_all_releases(publication_slug: str) -> list[dict]:
|
def get_all_releases(publication_slug: str) -> list[dict]:
|
||||||
"""Return all releases for a publication as dicts with 'id' and 'time_period'.
|
"""Return all releases for a publication as dicts with 'id' and 'time_period'.
|
||||||
|
|
||||||
@@ -72,7 +69,7 @@ def get_all_releases(publication_slug: str) -> list[dict]:
|
|||||||
total_pages = paging.get("totalPages", 1)
|
total_pages = paging.get("totalPages", 1)
|
||||||
|
|
||||||
for r in releases:
|
for r in releases:
|
||||||
time_period = _slug_to_time_period(r.get("slug", ""))
|
time_period = slug_to_time_period(r.get("slug", ""))
|
||||||
result.append({"id": r["id"], "time_period": time_period})
|
result.append({"id": r["id"], "time_period": time_period})
|
||||||
|
|
||||||
if page >= total_pages:
|
if page >= total_pages:
|
||||||
|
|||||||
@@ -76,3 +76,25 @@ def test_only_the_ks4_results_stream_opts_in():
|
|||||||
opted_in = [chunk.split('(')[0] for chunk in source.split('\nclass ')[1:]
|
opted_in = [chunk.split('(')[0] for chunk in source.split('\nclass ')[1:]
|
||||||
if re.search(r'_newest_release_owns_period\s*=\s*True', chunk)]
|
if re.search(r'_newest_release_owns_period\s*=\s*True', chunk)]
|
||||||
assert opted_in == ['EESKS4PerformanceStream']
|
assert opted_in == ['EESKS4PerformanceStream']
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize('slug, period', [
|
||||||
|
('2024-25', '202425'),
|
||||||
|
('2024-25-revised', '202425'),
|
||||||
|
('2025-26-provisional', '202526'),
|
||||||
|
('latest', None),
|
||||||
|
('', None),
|
||||||
|
])
|
||||||
|
def test_a_release_slug_gives_its_year_whatever_its_suffix(precedence, slug, period):
|
||||||
|
# KS2 already publishes "2024-25-revised" and "2025-26-provisional". An
|
||||||
|
# unread suffix sent the release last, behind the provisional one for the
|
||||||
|
# same year, which then owned the year and dropped every revised row.
|
||||||
|
assert precedence.slug_to_time_period(slug) == period
|
||||||
|
|
||||||
|
|
||||||
|
def test_within_a_year_the_api_order_is_kept(precedence):
|
||||||
|
# The API lists a year's revised release before its first release.
|
||||||
|
releases = [{'id': 'revised', 'time_period': '202526'},
|
||||||
|
{'id': 'first', 'time_period': '202526'},
|
||||||
|
{'id': 'older', 'time_period': '202425'}]
|
||||||
|
assert [r['id'] for r in precedence.newest_first(releases)] == ['revised', 'first', 'older']
|
||||||
Reference in new issue
Block a user