feat(api): group GIAS school types and religions for parents
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
3e2fa4a419
commit
354244f755
3 files changed
+214
-15
No files matched your search
@@ -0,0 +1,104 @@
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
The 34 GIAS establishment types describe governance and funding, which for a
|
||||
mainstream state school barely changes what a parent experiences. The search
|
||||
filter offers six groups a parent recognises instead, and a faith filter in
|
||||
place of the faith signal that "Voluntary aided" and "Voluntary controlled"
|
||||
only half carry. See
|
||||
docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md.
|
||||
|
||||
Groups are defined over GIAS codes and looked up by the translated name,
|
||||
because the DataFrame the API filters carries names only: the codes are
|
||||
replaced at load (data_loader.translate_gias_code_columns), the legacy-name
|
||||
mart fallback never had them, and test fixtures are written in names.
|
||||
"""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
|
||||
# (key, label, GIAS TypeOfEstablishment codes), in the order shown.
|
||||
TYPE_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Academy sponsor led, Academy converter, Free schools, University technical
|
||||
# college, Studio schools, City technology college. UTCs and studio schools
|
||||
# are legally academies, and a family considering one searches by name.
|
||||
("academy", "State school: academy or free school", frozenset({28, 34, 35, 40, 41, 6})),
|
||||
# Community, Voluntary aided, Voluntary controlled, Foundation, LA nursery.
|
||||
("council", "State school: council-run", frozenset({1, 2, 3, 5, 15})),
|
||||
("independent", "Independent (fee-paying)", frozenset({11})),
|
||||
# Every special type, independent ones included (usually funded by the
|
||||
# council through an EHCP, so SEND provision to a parent, not private
|
||||
# school), and Special post 16 institutions.
|
||||
("special", "Special school (SEND)", frozenset({7, 8, 10, 12, 32, 33, 36, 44})),
|
||||
# Further education, Sixth form centres, the 16-19 academies and free schools.
|
||||
("post16", "Sixth form or college", frozenset({18, 31, 39, 45, 46})),
|
||||
# Pupil referral units and AP academies and free schools. Last: parents do
|
||||
# not apply to these; the local authority places children there.
|
||||
("alternative", "Alternative provision", frozenset({14, 38, 42, 43})),
|
||||
)
|
||||
|
||||
# In no group, so reachable only under "Any school type": Secure units,
|
||||
# Miscellaneous, Higher education institutions, Online provider, Institution
|
||||
# funded by other government department, Academy secure 16 to 19.
|
||||
UNOFFERED_TYPE_CODES: frozenset[int] = frozenset({24, 27, 29, 49, 56, 57})
|
||||
|
||||
# (key, label, GIAS ReligiousCharacter codes), in the order shown. A joint
|
||||
# school is in every faith its label names; a generic "Christian" beside a
|
||||
# named church adds nothing.
|
||||
FAITH_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
|
||||
# Does not apply, None, and 99 (a blank label).
|
||||
("none", "No religious character", frozenset({0, 6, 99})),
|
||||
("church_of_england", "Church of England",
|
||||
frozenset({2, 9, 10, 11, 12, 13, 19, 20, 30, 31, 32, 33, 34, 41, 48})),
|
||||
("roman_catholic", "Roman Catholic", frozenset({3, 11, 13, 35, 48})),
|
||||
# 28 "Inter- / non- denominational" is how GIAS files Christian schools
|
||||
# tied to no one church.
|
||||
("other_christian", "Other Christian",
|
||||
frozenset({4, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 22, 26, 28, 30, 33,
|
||||
37, 38, 39, 40, 41, 44, 45, 46, 47})),
|
||||
("jewish", "Jewish", frozenset({5, 36, 43})),
|
||||
("muslim", "Muslim", frozenset({7, 42, 49})),
|
||||
("other_faith", "Other faith", frozenset({21, 24, 25, 29})),
|
||||
)
|
||||
|
||||
TYPE_GROUP_KEYS: frozenset[str] = frozenset(k for k, _, _ in TYPE_GROUPS)
|
||||
FAITH_KEYS: frozenset[str] = frozenset(k for k, _, _ in FAITH_GROUPS)
|
||||
|
||||
|
||||
def _key(name: str) -> str:
|
||||
return name.strip().lower()
|
||||
|
||||
|
||||
_TYPE_GROUP_BY_NAME: dict[str, str] = {
|
||||
_key(SCHOOL_TYPE[code]): key
|
||||
for key, _, codes in TYPE_GROUPS
|
||||
for code in codes
|
||||
if code in SCHOOL_TYPE
|
||||
}
|
||||
|
||||
_FAITHS_BY_NAME: dict[str, tuple[str, ...]] = {}
|
||||
for _faith, _, _codes in FAITH_GROUPS:
|
||||
for _code in sorted(_codes):
|
||||
if _code in RELIGIOUS_CHARACTER:
|
||||
_name = _key(RELIGIOUS_CHARACTER[_code])
|
||||
_FAITHS_BY_NAME[_name] = _FAITHS_BY_NAME.get(_name, ()) + (_faith,)
|
||||
|
||||
|
||||
def type_group_for(name: object) -> Optional[str]:
|
||||
"""The type group of a GIAS establishment type name, or None."""
|
||||
if not isinstance(name, str):
|
||||
return None
|
||||
return _TYPE_GROUP_BY_NAME.get(_key(name))
|
||||
|
||||
|
||||
def faith_groups_for(name: object) -> tuple[str, ...]:
|
||||
"""The faith groups of a GIAS religious character name.
|
||||
|
||||
A missing or blank name is "No religious character". A name the
|
||||
dictionary does not know has no faith, so it matches no faith option.
|
||||
"""
|
||||
if not isinstance(name, str):
|
||||
return ("none",) if name is None or pd.isna(name) else ()
|
||||
return _FAITHS_BY_NAME.get(_key(name), ())
|
||||
@@ -0,0 +1,98 @@
|
||||
"""Parent-facing groups over GIAS establishment types and religious characters.
|
||||
|
||||
Every GIAS code must be accounted for, so a new DfE code fails here instead of
|
||||
silently vanishing from the filter.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
from backend.gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
|
||||
from backend.school_groups import (
|
||||
FAITH_GROUPS,
|
||||
TYPE_GROUPS,
|
||||
UNOFFERED_TYPE_CODES,
|
||||
faith_groups_for,
|
||||
type_group_for,
|
||||
)
|
||||
|
||||
DBT_PROJECT = Path(__file__).resolve().parents[2] / "pipeline" / "transform" / "dbt_project.yml"
|
||||
|
||||
|
||||
def _non_england_codes() -> set[int]:
|
||||
return set(yaml.safe_load(DBT_PROJECT.read_text())["vars"]["non_england_school_type_codes"])
|
||||
|
||||
|
||||
def test_every_type_code_is_in_exactly_one_place():
|
||||
places = [codes for _, _, codes in TYPE_GROUPS] + [UNOFFERED_TYPE_CODES, _non_england_codes()]
|
||||
for code in SCHOOL_TYPE:
|
||||
homes = sum(code in p for p in places)
|
||||
assert homes == 1, f"type code {code} ({SCHOOL_TYPE[code]}) is in {homes} places"
|
||||
|
||||
|
||||
def test_every_religion_code_has_a_faith():
|
||||
for code, name in RELIGIOUS_CHARACTER.items():
|
||||
assert any(code in codes for _, _, codes in FAITH_GROUPS), f"{code} {name!r}"
|
||||
|
||||
|
||||
def test_type_groups_in_display_order():
|
||||
assert [k for k, _, _ in TYPE_GROUPS] == [
|
||||
"academy", "council", "independent", "special", "post16", "alternative"]
|
||||
|
||||
|
||||
def test_faiths_in_display_order():
|
||||
assert [k for k, _, _ in FAITH_GROUPS] == [
|
||||
"none", "church_of_england", "roman_catholic", "other_christian",
|
||||
"jewish", "muslim", "other_faith"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, group", [
|
||||
("Academy converter", "academy"),
|
||||
("University technical college", "academy"),
|
||||
("Voluntary aided school", "council"),
|
||||
("Local authority nursery school", "council"),
|
||||
("Other independent school", "independent"),
|
||||
("Other independent special school", "special"),
|
||||
("Special post 16 institution", "special"),
|
||||
("Further education", "post16"),
|
||||
("Pupil referral unit", "alternative"),
|
||||
("academy CONVERTER", "academy"),
|
||||
])
|
||||
def test_type_group_by_name(name, group):
|
||||
assert type_group_for(name) == group
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", [
|
||||
"Higher education institutions", "Miscellaneous", "Unknown (9999)", "Academy", "", None, np.nan,
|
||||
])
|
||||
def test_unoffered_or_unknown_types_have_no_group(name):
|
||||
assert type_group_for(name) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, faiths", [
|
||||
("Roman Catholic/Church of England", ("church_of_england", "roman_catholic")),
|
||||
("Roman Catholic/Anglican", ("church_of_england", "roman_catholic")),
|
||||
("Church of England/Methodist", ("church_of_england", "other_christian")),
|
||||
("Church of England/Christian", ("church_of_england",)),
|
||||
("Catholic", ("roman_catholic",)),
|
||||
("Inter- / non- denominational", ("other_christian",)),
|
||||
("Orthodox Jewish", ("jewish",)),
|
||||
("Sunni Deobandi", ("muslim",)),
|
||||
("Hindu", ("other_faith",)),
|
||||
("Does not apply", ("none",)),
|
||||
("None", ("none",)),
|
||||
])
|
||||
def test_faiths_by_name(name, faiths):
|
||||
assert faith_groups_for(name) == faiths
|
||||
|
||||
|
||||
@pytest.mark.parametrize("missing", [None, np.nan, "", " "])
|
||||
def test_a_missing_religion_is_no_religious_character(missing):
|
||||
assert faith_groups_for(missing) == ("none",)
|
||||
|
||||
|
||||
def test_an_unknown_religion_has_no_faith():
|
||||
assert faith_groups_for("Unknown (77)") == ()
|
||||
@@ -103,8 +103,8 @@ Christian schools that are not tied to one church.
|
||||
Grouping lives in the backend, not dbt. The API already translates GIAS codes
|
||||
to names when it loads the marts (`backend/data_loader.py:
|
||||
translate_gias_code_columns`), and every filter is applied to that DataFrame.
|
||||
Grouping at the same point needs no mart change, so there is no Airflow run
|
||||
between merge and staging showing it.
|
||||
Grouping there needs no mart change, so there is no Airflow run between merge
|
||||
and staging showing it.
|
||||
|
||||
### `backend/school_groups.py` (new)
|
||||
|
||||
@@ -112,25 +112,22 @@ between merge and staging showing it.
|
||||
- `UNOFFERED_TYPE_CODES`: the not-offered codes, so that "every code is
|
||||
accounted for" is testable.
|
||||
- `FAITH_GROUPS`: ordered `(key, label, frozenset[int])` per faith group.
|
||||
- `type_group_for(code) -> str | None` and
|
||||
`faith_groups_for(code) -> tuple[str, ...]` (a missing code gives
|
||||
- `type_group_for(name) -> str | None` and
|
||||
`faith_groups_for(name) -> tuple[str, ...]` (a missing or blank name gives
|
||||
`("none",)`).
|
||||
|
||||
No dependence on the generated GIAS dictionaries beyond their codes, so the
|
||||
backend/pipeline dictionary parity test is untouched.
|
||||
|
||||
### `backend/data_loader.py`
|
||||
### Grouping by name, at filter time
|
||||
|
||||
In `translate_gias_code_columns`, before the code columns are replaced by
|
||||
names, add two columns from the codes:
|
||||
|
||||
- `school_type_group`: `type_group_for(school_type_code)`, or None.
|
||||
- `faith_groups`: `faith_groups_for(religious_character_code)`, a tuple.
|
||||
|
||||
The fallback query that reads name columns from older marts
|
||||
(`_MAIN_QUERY_NO_EXTRA_COLS` and its replacements) has no codes. There, both
|
||||
columns are derived from the names by reverse lookup through `SCHOOL_TYPE` and
|
||||
`RELIGIOUS_CHARACTER`.
|
||||
The groups are defined over codes but looked up by the translated name
|
||||
(`type_group_for(name)`, `faith_groups_for(name)`), when `/api/schools` and
|
||||
`/api/filters` filter. The DataFrame those endpoints read carries names only:
|
||||
`translate_gias_code_columns` replaces the codes at load, the legacy-name
|
||||
mart fallback never had codes, and the API test fixtures are written in
|
||||
names. The names come from the same dictionaries, so the lookup is exact.
|
||||
`data_loader.py` is unchanged.
|
||||
|
||||
### `/api/schools`
|
||||
|
||||
|
||||
Reference in new issue
Block a user