feat(api): group GIAS school types and religions for parents

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
TudorandClaude Opus 5.5 committed 2026-10-02 11:52:43 +01:00
1 parent 3e2fa4a419
commit 354244f755
3 files changed
+214 -15

No files matched your search

+104
View File
@@ -0,0 +1,104 @@
"""Parent-facing groups over GIAS establishment types and religious characters.
The 34 GIAS establishment types describe governance and funding, which for a
mainstream state school barely changes what a parent experiences. The search
filter offers six groups a parent recognises instead, and a faith filter in
place of the faith signal that "Voluntary aided" and "Voluntary controlled"
only half carry. See
docs/superpowers/specs/2026-10-02-school-type-groups-and-faith-filter-design.md.
Groups are defined over GIAS codes and looked up by the translated name,
because the DataFrame the API filters carries names only: the codes are
replaced at load (data_loader.translate_gias_code_columns), the legacy-name
mart fallback never had them, and test fixtures are written in names.
"""
from typing import Optional
import pandas as pd
from .gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
# (key, label, GIAS TypeOfEstablishment codes), in the order shown.
TYPE_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
# Academy sponsor led, Academy converter, Free schools, University technical
# college, Studio schools, City technology college. UTCs and studio schools
# are legally academies, and a family considering one searches by name.
("academy", "State school: academy or free school", frozenset({28, 34, 35, 40, 41, 6})),
# Community, Voluntary aided, Voluntary controlled, Foundation, LA nursery.
("council", "State school: council-run", frozenset({1, 2, 3, 5, 15})),
("independent", "Independent (fee-paying)", frozenset({11})),
# Every special type, independent ones included (usually funded by the
# council through an EHCP, so SEND provision to a parent, not private
# school), and Special post 16 institutions.
("special", "Special school (SEND)", frozenset({7, 8, 10, 12, 32, 33, 36, 44})),
# Further education, Sixth form centres, the 16-19 academies and free schools.
("post16", "Sixth form or college", frozenset({18, 31, 39, 45, 46})),
# Pupil referral units and AP academies and free schools. Last: parents do
# not apply to these; the local authority places children there.
("alternative", "Alternative provision", frozenset({14, 38, 42, 43})),
)
# In no group, so reachable only under "Any school type": Secure units,
# Miscellaneous, Higher education institutions, Online provider, Institution
# funded by other government department, Academy secure 16 to 19.
UNOFFERED_TYPE_CODES: frozenset[int] = frozenset({24, 27, 29, 49, 56, 57})
# (key, label, GIAS ReligiousCharacter codes), in the order shown. A joint
# school is in every faith its label names; a generic "Christian" beside a
# named church adds nothing.
FAITH_GROUPS: tuple[tuple[str, str, frozenset[int]], ...] = (
# Does not apply, None, and 99 (a blank label).
("none", "No religious character", frozenset({0, 6, 99})),
("church_of_england", "Church of England",
frozenset({2, 9, 10, 11, 12, 13, 19, 20, 30, 31, 32, 33, 34, 41, 48})),
("roman_catholic", "Roman Catholic", frozenset({3, 11, 13, 35, 48})),
# 28 "Inter- / non- denominational" is how GIAS files Christian schools
# tied to no one church.
("other_christian", "Other Christian",
frozenset({4, 8, 9, 10, 12, 14, 15, 16, 17, 18, 19, 22, 26, 28, 30, 33,
37, 38, 39, 40, 41, 44, 45, 46, 47})),
("jewish", "Jewish", frozenset({5, 36, 43})),
("muslim", "Muslim", frozenset({7, 42, 49})),
("other_faith", "Other faith", frozenset({21, 24, 25, 29})),
)
TYPE_GROUP_KEYS: frozenset[str] = frozenset(k for k, _, _ in TYPE_GROUPS)
FAITH_KEYS: frozenset[str] = frozenset(k for k, _, _ in FAITH_GROUPS)
def _key(name: str) -> str:
return name.strip().lower()
_TYPE_GROUP_BY_NAME: dict[str, str] = {
_key(SCHOOL_TYPE[code]): key
for key, _, codes in TYPE_GROUPS
for code in codes
if code in SCHOOL_TYPE
}
_FAITHS_BY_NAME: dict[str, tuple[str, ...]] = {}
for _faith, _, _codes in FAITH_GROUPS:
for _code in sorted(_codes):
if _code in RELIGIOUS_CHARACTER:
_name = _key(RELIGIOUS_CHARACTER[_code])
_FAITHS_BY_NAME[_name] = _FAITHS_BY_NAME.get(_name, ()) + (_faith,)
def type_group_for(name: object) -> Optional[str]:
"""The type group of a GIAS establishment type name, or None."""
if not isinstance(name, str):
return None
return _TYPE_GROUP_BY_NAME.get(_key(name))
def faith_groups_for(name: object) -> tuple[str, ...]:
"""The faith groups of a GIAS religious character name.
A missing or blank name is "No religious character". A name the
dictionary does not know has no faith, so it matches no faith option.
"""
if not isinstance(name, str):
return ("none",) if name is None or pd.isna(name) else ()
return _FAITHS_BY_NAME.get(_key(name), ())
+98
View File
@@ -0,0 +1,98 @@
"""Parent-facing groups over GIAS establishment types and religious characters.
Every GIAS code must be accounted for, so a new DfE code fails here instead of
silently vanishing from the filter.
"""
from pathlib import Path
import numpy as np
import pytest
import yaml
from backend.gias_codes import RELIGIOUS_CHARACTER, SCHOOL_TYPE
from backend.school_groups import (
FAITH_GROUPS,
TYPE_GROUPS,
UNOFFERED_TYPE_CODES,
faith_groups_for,
type_group_for,
)
DBT_PROJECT = Path(__file__).resolve().parents[2] / "pipeline" / "transform" / "dbt_project.yml"
def _non_england_codes() -> set[int]:
return set(yaml.safe_load(DBT_PROJECT.read_text())["vars"]["non_england_school_type_codes"])
def test_every_type_code_is_in_exactly_one_place():
places = [codes for _, _, codes in TYPE_GROUPS] + [UNOFFERED_TYPE_CODES, _non_england_codes()]
for code in SCHOOL_TYPE:
homes = sum(code in p for p in places)
assert homes == 1, f"type code {code} ({SCHOOL_TYPE[code]}) is in {homes} places"
def test_every_religion_code_has_a_faith():
for code, name in RELIGIOUS_CHARACTER.items():
assert any(code in codes for _, _, codes in FAITH_GROUPS), f"{code} {name!r}"
def test_type_groups_in_display_order():
assert [k for k, _, _ in TYPE_GROUPS] == [
"academy", "council", "independent", "special", "post16", "alternative"]
def test_faiths_in_display_order():
assert [k for k, _, _ in FAITH_GROUPS] == [
"none", "church_of_england", "roman_catholic", "other_christian",
"jewish", "muslim", "other_faith"]
@pytest.mark.parametrize("name, group", [
("Academy converter", "academy"),
("University technical college", "academy"),
("Voluntary aided school", "council"),
("Local authority nursery school", "council"),
("Other independent school", "independent"),
("Other independent special school", "special"),
("Special post 16 institution", "special"),
("Further education", "post16"),
("Pupil referral unit", "alternative"),
("academy CONVERTER", "academy"),
])
def test_type_group_by_name(name, group):
assert type_group_for(name) == group
@pytest.mark.parametrize("name", [
"Higher education institutions", "Miscellaneous", "Unknown (9999)", "Academy", "", None, np.nan,
])
def test_unoffered_or_unknown_types_have_no_group(name):
assert type_group_for(name) is None
@pytest.mark.parametrize("name, faiths", [
("Roman Catholic/Church of England", ("church_of_england", "roman_catholic")),
("Roman Catholic/Anglican", ("church_of_england", "roman_catholic")),
("Church of England/Methodist", ("church_of_england", "other_christian")),
("Church of England/Christian", ("church_of_england",)),
("Catholic", ("roman_catholic",)),
("Inter- / non- denominational", ("other_christian",)),
("Orthodox Jewish", ("jewish",)),
("Sunni Deobandi", ("muslim",)),
("Hindu", ("other_faith",)),
("Does not apply", ("none",)),
("None", ("none",)),
])
def test_faiths_by_name(name, faiths):
assert faith_groups_for(name) == faiths
@pytest.mark.parametrize("missing", [None, np.nan, "", " "])
def test_a_missing_religion_is_no_religious_character(missing):
assert faith_groups_for(missing) == ("none",)
def test_an_unknown_religion_has_no_faith():
assert faith_groups_for("Unknown (77)") == ()
@@ -103,8 +103,8 @@ Christian schools that are not tied to one church.
Grouping lives in the backend, not dbt. The API already translates GIAS codes
to names when it loads the marts (`backend/data_loader.py:
translate_gias_code_columns`), and every filter is applied to that DataFrame.
Grouping at the same point needs no mart change, so there is no Airflow run
between merge and staging showing it.
Grouping there needs no mart change, so there is no Airflow run between merge
and staging showing it.
### `backend/school_groups.py` (new)
@@ -112,25 +112,22 @@ between merge and staging showing it.
- `UNOFFERED_TYPE_CODES`: the not-offered codes, so that "every code is
accounted for" is testable.
- `FAITH_GROUPS`: ordered `(key, label, frozenset[int])` per faith group.
- `type_group_for(code) -> str | None` and
`faith_groups_for(code) -> tuple[str, ...]` (a missing code gives
- `type_group_for(name) -> str | None` and
`faith_groups_for(name) -> tuple[str, ...]` (a missing or blank name gives
`("none",)`).
No dependence on the generated GIAS dictionaries beyond their codes, so the
backend/pipeline dictionary parity test is untouched.
### `backend/data_loader.py`
### Grouping by name, at filter time
In `translate_gias_code_columns`, before the code columns are replaced by
names, add two columns from the codes:
- `school_type_group`: `type_group_for(school_type_code)`, or None.
- `faith_groups`: `faith_groups_for(religious_character_code)`, a tuple.
The fallback query that reads name columns from older marts
(`_MAIN_QUERY_NO_EXTRA_COLS` and its replacements) has no codes. There, both
columns are derived from the names by reverse lookup through `SCHOOL_TYPE` and
`RELIGIOUS_CHARACTER`.
The groups are defined over codes but looked up by the translated name
(`type_group_for(name)`, `faith_groups_for(name)`), when `/api/schools` and
`/api/filters` filter. The DataFrame those endpoints read carries names only:
`translate_gias_code_columns` replaces the codes at load, the legacy-name
mart fallback never had codes, and the API test fixtures are written in
names. The names come from the same dictionaries, so the lookup is exact.
`data_loader.py` is unchanged.
### `/api/schools`