diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/release_precedence.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/release_precedence.py index 52080c6..1a63779 100644 --- a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/release_precedence.py +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/release_precedence.py @@ -11,8 +11,23 @@ requirements, can load it. """ from __future__ import annotations +import re + import pandas as pd +_SLUG_YEAR = re.compile(r"^(\d{4})-(\d{2})(?:-|$)") + + +def slug_to_time_period(slug: str) -> str | None: + """A release slug's academic year as a time_period: '2022-23' → '202223'. + + Suffixed slugs ('2024-25-revised', '2025-26-provisional') give the same + year, so newest_first places them by year; read as unknown, a revised + release went last and lost its year to the first release. + """ + match = _SLUG_YEAR.match(slug or "") + return match.group(1) + match.group(2) if match else None + def newest_first(releases: list[dict]) -> list[dict]: """Releases by time_period, newest first. A release whose time_period is diff --git a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py index d550b80..541ef8b 100644 --- a/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py +++ b/pipeline/plugins/extractors/tap-uk-ees/tap_uk_ees/tap.py @@ -24,7 +24,12 @@ from tap_uk_ees.ks4_summary import ( headline_record, headline_rows, ) -from tap_uk_ees.release_precedence import drop_owned_periods, newest_first, periods_in +from tap_uk_ees.release_precedence import ( + drop_owned_periods, + newest_first, + periods_in, + slug_to_time_period, +) CONTENT_API_BASE = ( "https://content.explore-education-statistics.service.gov.uk/api" @@ -40,14 +45,6 @@ def get_content_release_id(publication_slug: str) -> str: return resp.json()["id"] -def _slug_to_time_period(slug: str) -> str | None: - """Convert a release slug like '2022-23' to a time_period like '202223'.""" - parts = slug.split("-") - if len(parts) == 2 and len(parts[0]) == 4 and len(parts[1]) == 2: - return parts[0] + parts[1] - return None - - def get_all_releases(publication_slug: str) -> list[dict]: """Return all releases for a publication as dicts with 'id' and 'time_period'. @@ -72,7 +69,7 @@ def get_all_releases(publication_slug: str) -> list[dict]: total_pages = paging.get("totalPages", 1) for r in releases: - time_period = _slug_to_time_period(r.get("slug", "")) + time_period = slug_to_time_period(r.get("slug", "")) result.append({"id": r["id"], "time_period": time_period}) if page >= total_pages: diff --git a/pipeline/tests/test_ees_release_precedence.py b/pipeline/tests/test_ees_release_precedence.py index 09ce3ae..104fcae 100644 --- a/pipeline/tests/test_ees_release_precedence.py +++ b/pipeline/tests/test_ees_release_precedence.py @@ -76,3 +76,25 @@ def test_only_the_ks4_results_stream_opts_in(): opted_in = [chunk.split('(')[0] for chunk in source.split('\nclass ')[1:] if re.search(r'_newest_release_owns_period\s*=\s*True', chunk)] assert opted_in == ['EESKS4PerformanceStream'] + + +@pytest.mark.parametrize('slug, period', [ + ('2024-25', '202425'), + ('2024-25-revised', '202425'), + ('2025-26-provisional', '202526'), + ('latest', None), + ('', None), +]) +def test_a_release_slug_gives_its_year_whatever_its_suffix(precedence, slug, period): + # KS2 already publishes "2024-25-revised" and "2025-26-provisional". An + # unread suffix sent the release last, behind the provisional one for the + # same year, which then owned the year and dropped every revised row. + assert precedence.slug_to_time_period(slug) == period + + +def test_within_a_year_the_api_order_is_kept(precedence): + # The API lists a year's revised release before its first release. + releases = [{'id': 'revised', 'time_period': '202526'}, + {'id': 'first', 'time_period': '202526'}, + {'id': 'older', 'time_period': '202425'}] + assert [r['id'] for r in precedence.newest_first(releases)] == ['revised', 'first', 'older']