Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e236669fde | ||
|
|
fb5a0928bd | ||
|
|
264edd2e3a | ||
|
|
cd2cbe7be6 | ||
|
|
73182d0c0c | ||
|
|
cbe3a9a772 | ||
|
|
2e9b5c83c5 | ||
|
|
102397fe69 |
No files matched your search
+158
-13
@@ -839,17 +839,151 @@ def _format_cohort_year(year) -> str | None:
|
||||
return text
|
||||
|
||||
|
||||
_PUPIL_GROUPS = ("disadvantaged", "other", "all")
|
||||
|
||||
|
||||
def _lone_hidden_groups(groups: dict) -> list:
|
||||
"""Pupil groups hiding exactly one category — solvable by subtraction."""
|
||||
return [
|
||||
key for key, group in groups.items()
|
||||
if sum(1 for c in group["categories"] if c["status"] == "suppressed") == 1
|
||||
]
|
||||
|
||||
|
||||
def _lone_hidden_categories(groups: dict) -> list:
|
||||
"""Categories hidden in exactly one of several pupil groups."""
|
||||
lone = []
|
||||
categories = {c["category"] for g in groups.values() for c in g["categories"]}
|
||||
for category in categories:
|
||||
found = [
|
||||
c for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == category
|
||||
]
|
||||
hidden = [c for c in found if c["status"] == "suppressed"]
|
||||
if len(hidden) == 1 and len(found) > 1:
|
||||
lone.append(category)
|
||||
return lone
|
||||
|
||||
|
||||
def disclosure_invariant_holds(groups: dict) -> bool:
|
||||
"""Every row and every column hides none, or at least two.
|
||||
|
||||
Public so the tests can assert it directly rather than re-deriving it.
|
||||
"""
|
||||
return not _lone_hidden_groups(groups) and not _lone_hidden_categories(groups)
|
||||
|
||||
|
||||
def _mask_for_disclosure(groups: dict) -> None:
|
||||
"""Withhold further cells until nothing suppressed can be solved for.
|
||||
|
||||
Not rendering a figure is not the same as not publishing it. This endpoint
|
||||
is public and unauthenticated, so anything left in the payload is
|
||||
published, whatever the UI chooses to draw — the same reasoning the
|
||||
admission_distance field carries in app.py.
|
||||
|
||||
Two identities let a caller solve for a withheld cell:
|
||||
|
||||
* within a pupil group, the categories sum to the cohort, so a group with
|
||||
exactly ONE suppressed category gives it away as cohort - sum(rest);
|
||||
* across groups, disadvantaged + other = all for every category, so a
|
||||
category suppressed in exactly ONE of the three gives itself away.
|
||||
|
||||
DfE's own answer is secondary suppression: withhold a second cell so the
|
||||
residual spans two unknowns and identifies neither.
|
||||
|
||||
Where no companion can do that — a sparse cohort whose every other category
|
||||
is `not_applicable`, which is common in special schools and alternative
|
||||
provision — there is nothing left to withhold, so the pupil group is
|
||||
DROPPED entirely. An earlier version simply gave up here and returned with
|
||||
the violation intact and no signal, which is the one outcome this function
|
||||
must never produce: a disclosure-control pass that fails silently is worse
|
||||
than none, because everything downstream trusts it.
|
||||
|
||||
Mutates `groups` in place. Guaranteed to return with
|
||||
disclosure_invariant_holds(groups) true.
|
||||
"""
|
||||
|
||||
def suppress(cell):
|
||||
if cell["status"] == "published":
|
||||
cell["status"] = "suppressed"
|
||||
cell["pupils"] = None
|
||||
cell["percentage"] = None
|
||||
return True
|
||||
return False
|
||||
|
||||
def add_companion(candidates) -> bool:
|
||||
"""Withhold a second cell so the residual spans two unknowns.
|
||||
|
||||
The companion must carry pupils. Suppressing a zero looks like
|
||||
secondary suppression and protects nothing: the residual still equals
|
||||
the original withheld figure exactly. Returns False when no cell can
|
||||
do the job, which escalates to dropping the group.
|
||||
"""
|
||||
published = [c for c in candidates if c["status"] == "published"]
|
||||
useful = sorted(
|
||||
(c for c in published if (c["pupils"] or 0) > 0),
|
||||
key=lambda c: c["pupils"],
|
||||
)
|
||||
if useful:
|
||||
return suppress(useful[0])
|
||||
# Every remaining cell is zero or not applicable: withholding any of
|
||||
# them leaves the residual equal to the original figure.
|
||||
return False
|
||||
|
||||
# Fixpoint: each new suppression can break the other identity. Terminates
|
||||
# because every pass either adds a suppression, drops a group, or stops.
|
||||
while not disclosure_invariant_holds(groups):
|
||||
changed = False
|
||||
|
||||
for category in _lone_hidden_categories(groups):
|
||||
siblings = [
|
||||
c for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == category
|
||||
]
|
||||
if add_companion(siblings):
|
||||
changed = True
|
||||
|
||||
for key in _lone_hidden_groups(groups):
|
||||
if add_companion(groups[key]["categories"]):
|
||||
changed = True
|
||||
|
||||
if changed:
|
||||
continue
|
||||
|
||||
# Nothing left to withhold. Drop the groups that are still solvable,
|
||||
# and any category still solvable across the groups that remain.
|
||||
for key in _lone_hidden_groups(groups):
|
||||
del groups[key]
|
||||
changed = True
|
||||
|
||||
for category in _lone_hidden_categories(groups):
|
||||
for group in groups.values():
|
||||
for cell in group["categories"]:
|
||||
if cell["category"] == category and suppress(cell):
|
||||
changed = True
|
||||
|
||||
if not changed:
|
||||
# Unreachable given the two escalations above, but a masking pass
|
||||
# must never spin or exit unsafely. Withhold everything.
|
||||
groups.clear()
|
||||
return
|
||||
|
||||
|
||||
def _destinations_block(rows: list) -> dict | None:
|
||||
"""Shape destination rows for one phase into the API's block.
|
||||
|
||||
Carries `status` through untouched and emits no computed totals. The only
|
||||
aggregates present are ones DfE published itself; whether showing one is
|
||||
safe depends on how many of its components are suppressed, which the
|
||||
frontend decides (lib/destinations.ts, rule R2).
|
||||
Applies secondary suppression before returning, so no caller of this public
|
||||
endpoint can solve for a figure DfE withheld. See _mask_for_disclosure.
|
||||
|
||||
Deliberately does NOT compute a residual, a "remaining pupils" figure, or
|
||||
any total that would close a gap left by a suppressed category — the
|
||||
categories sum to the cohort, so such a figure names the withheld cell.
|
||||
Aggregate measures are dropped entirely. DfE publishes them, and they would
|
||||
be useful for a "what is published for this group" fallback, but nothing
|
||||
renders them today and an aggregate spanning exactly one suppressed
|
||||
component names that component. An unused field that leaks is not a
|
||||
trade-off worth carrying — re-add them with their own guard if the fallback
|
||||
is ever built.
|
||||
|
||||
Deliberately computes no residual, no "remaining pupils" figure, and no
|
||||
total that would close a gap left by a suppressed category.
|
||||
"""
|
||||
if not rows:
|
||||
return None
|
||||
@@ -864,23 +998,34 @@ def _destinations_block(rows: list) -> dict | None:
|
||||
for row in rows:
|
||||
group = groups.setdefault(
|
||||
row["pupil_group"],
|
||||
{"cohort": row.get("cohort_pupils"), "categories": [], "aggregates": {}},
|
||||
{"cohort": row.get("cohort_pupils"), "categories": []},
|
||||
)
|
||||
measure = row["destination_measure"]
|
||||
published = row.get("status") == "published"
|
||||
# Belt and braces: percentage is derived from the same source cell as
|
||||
# pupils, but publishing one without the other would hand back the
|
||||
# cohort (pupils / percentage) and with it the residual.
|
||||
cell = {
|
||||
"category": measure,
|
||||
"pupils": row.get("pupils"),
|
||||
"percentage": row.get("percentage"),
|
||||
"pupils": row.get("pupils") if published else None,
|
||||
"percentage": row.get("percentage") if published else None,
|
||||
"status": row.get("status"),
|
||||
}
|
||||
if measure in _AGGREGATE_MEASURES:
|
||||
group["aggregates"][measure[len("agg_"):]] = cell
|
||||
else:
|
||||
group["categories"].append(cell)
|
||||
continue
|
||||
group["categories"].append(cell)
|
||||
|
||||
if not groups:
|
||||
return None
|
||||
|
||||
_mask_for_disclosure(groups)
|
||||
|
||||
# Masking can empty the block entirely — a sparse cohort where no group
|
||||
# could be made safe. Return None so the section is absent rather than
|
||||
# rendering an empty shell.
|
||||
if not groups:
|
||||
return None
|
||||
|
||||
return {"cohort_year": _format_cohort_year(latest_year), "groups": groups}
|
||||
|
||||
|
||||
|
||||
@@ -1,12 +1,17 @@
|
||||
"""The destinations serialiser's contract: it carries suppression through, and
|
||||
never emits a total that closes a gap left by a suppressed category.
|
||||
"""The destinations serialiser's contract.
|
||||
|
||||
The destination categories sum to the cohort, so an aggregate that happens to
|
||||
equal the residual names the withheld figure exactly. See
|
||||
docs/superpowers/specs/2026-08-28-destination-measures-design.md.
|
||||
Not rendering a figure is not the same as not publishing it. This endpoint is
|
||||
public and unauthenticated, so whatever the payload carries is published,
|
||||
whatever the UI draws. The categories sum to the cohort and the pupil groups
|
||||
sum to each other, so a lone suppressed cell is solvable by subtraction — the
|
||||
serialiser adds secondary suppression to prevent it.
|
||||
|
||||
See docs/superpowers/specs/2026-08-28-destination-measures-design.md.
|
||||
"""
|
||||
|
||||
from backend.data_loader import _destinations_block, _format_cohort_year
|
||||
from backend.data_loader import (
|
||||
_destinations_block, _format_cohort_year, disclosure_invariant_holds,
|
||||
)
|
||||
|
||||
|
||||
def _row(group, measure, pupils, status, cohort=180, percentage=None, year=202223):
|
||||
@@ -43,39 +48,6 @@ def test_published_category_keeps_its_figures():
|
||||
assert cat["status"] == "published"
|
||||
|
||||
|
||||
def test_no_closing_total_is_emitted_for_a_partially_suppressed_group():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
_row("all", "sixth_form_college", None, "suppressed"),
|
||||
_row("all", "further_education", 61, "published", percentage=33.9),
|
||||
_row("all", "apprenticeship", 8, "published", percentage=4.4),
|
||||
_row("all", "employment", 6, "published", percentage=3.3),
|
||||
_row("all", "not_sustained", 5, "published", percentage=2.8),
|
||||
_row("all", "not_captured", 4, "published", percentage=2.2),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
group = block["groups"]["all"]
|
||||
published = sum(c["pupils"] for c in group["categories"] if c["pupils"] is not None)
|
||||
residual = group["cohort"] - published
|
||||
for value in group["aggregates"].values():
|
||||
if value is None or value.get("pupils") is None:
|
||||
continue
|
||||
assert value["pupils"] != residual, (
|
||||
"an aggregate equal to the residual identifies the suppressed cell"
|
||||
)
|
||||
|
||||
|
||||
def test_aggregates_are_separated_from_categories():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
_row("all", "agg_sustained_all", 171, "published", percentage=95.0),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
group = block["groups"]["all"]
|
||||
assert [c["category"] for c in group["categories"]] == ["school_sixth_form"]
|
||||
assert group["aggregates"]["sustained_all"]["pupils"] == 171
|
||||
|
||||
|
||||
def test_only_the_latest_year_is_served():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 60, "published", year=202122),
|
||||
@@ -110,3 +82,188 @@ def test_format_cohort_year_handles_the_six_digit_form():
|
||||
|
||||
def test_empty_rows_yield_none_not_an_empty_shell():
|
||||
assert _destinations_block([]) is None
|
||||
|
||||
|
||||
# ── Disclosure control ──────────────────────────────────────────────────────
|
||||
#
|
||||
# The rendering guards in lib/destinations.ts stop a withheld figure being
|
||||
# DRAWN. They do nothing about it being COMPUTED: this endpoint is public and
|
||||
# unauthenticated, so whatever the payload carries is published. These tests
|
||||
# are the ones that matter.
|
||||
|
||||
def _solve_residual(group):
|
||||
"""What any caller can work out: cohort minus everything published."""
|
||||
published = [c["pupils"] for c in group["categories"] if c["pupils"] is not None]
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
return group["cohort"] - sum(published), len(hidden)
|
||||
|
||||
|
||||
def test_a_lone_suppressed_category_cannot_be_solved_for():
|
||||
"""Whitley Bay High School's real 2022/23 disadvantaged group: further
|
||||
education withheld, everything else published, cohort 41. Before secondary
|
||||
suppression the payload gave the answer away as 41 - 23 = 18."""
|
||||
rows = [
|
||||
_row("disadvantaged", "school_sixth_form", 15, "published", cohort=41),
|
||||
_row("disadvantaged", "sixth_form_college", 0, "published", cohort=41),
|
||||
_row("disadvantaged", "further_education", None, "suppressed", cohort=41),
|
||||
_row("disadvantaged", "apprenticeship", 1, "published", cohort=41),
|
||||
_row("disadvantaged", "employment", 2, "published", cohort=41),
|
||||
_row("disadvantaged", "not_sustained", 3, "published", cohort=41),
|
||||
_row("disadvantaged", "not_captured", 2, "published", cohort=41),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["disadvantaged"]
|
||||
residual, hidden = _solve_residual(group)
|
||||
assert hidden >= 2, "a lone suppressed cell must gain a companion"
|
||||
assert residual != 18, "the withheld figure is recoverable from the payload"
|
||||
|
||||
|
||||
def test_every_group_hides_none_or_at_least_two_categories():
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published"),
|
||||
_row("all", "sixth_form_college", None, "suppressed"),
|
||||
_row("all", "further_education", 61, "published"),
|
||||
_row("all", "apprenticeship", 8, "published"),
|
||||
_row("all", "employment", 6, "published"),
|
||||
_row("all", "not_sustained", 5, "published"),
|
||||
_row("all", "not_captured", 4, "published"),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
assert len(hidden) >= 2
|
||||
|
||||
|
||||
def test_a_category_hidden_in_one_group_is_hidden_in_a_second():
|
||||
"""disadvantaged + other = all for every category, so a category withheld
|
||||
in exactly one of the three is recoverable from the other two."""
|
||||
rows = []
|
||||
for measure, a, d, o in [
|
||||
("school_sixth_form", 75, None, 58),
|
||||
("further_education", 61, 27, 34),
|
||||
("apprenticeship", 8, 4, 4),
|
||||
("employment", 6, 1, 5),
|
||||
("not_sustained", 5, 3, 2),
|
||||
("not_captured", 4, 2, 2),
|
||||
]:
|
||||
rows.append(_row("all", measure, a, "published", cohort=159))
|
||||
rows.append(_row("disadvantaged", measure, d,
|
||||
"published" if d is not None else "suppressed", cohort=37))
|
||||
rows.append(_row("other", measure, o, "published", cohort=122))
|
||||
|
||||
groups = _destinations_block(rows)["groups"]
|
||||
measures = {c["category"] for g in groups.values() for c in g["categories"]}
|
||||
assert len(measures) == 6, "the fixture's six measures must all be checked"
|
||||
|
||||
for measure in sorted(measures):
|
||||
hidden = sum(
|
||||
1 for g in groups.values() for c in g["categories"]
|
||||
if c["category"] == measure and c["status"] == "suppressed"
|
||||
)
|
||||
# The invariant is "none, or at least two" — not "at least two".
|
||||
assert hidden != 1, f"{measure} is solvable across the pupil groups"
|
||||
|
||||
|
||||
def test_a_suppressed_cell_never_keeps_its_percentage():
|
||||
"""percentage / pupils would hand back the cohort, and with it the residual."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published", percentage=41.7),
|
||||
_row("all", "sixth_form_college", None, "suppressed", percentage=11.7),
|
||||
_row("all", "further_education", 61, "published", percentage=33.9),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
for cell in group["categories"]:
|
||||
if cell["status"] != "published":
|
||||
assert cell["pupils"] is None
|
||||
assert cell["percentage"] is None
|
||||
|
||||
|
||||
def test_aggregates_are_not_served():
|
||||
"""An aggregate spanning exactly one suppressed component names it, and
|
||||
nothing renders them today."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", 75, "published"),
|
||||
_row("all", "agg_sustained_all", 171, "published"),
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
assert [c["category"] for c in group["categories"]] == ["school_sixth_form"]
|
||||
assert "aggregates" not in group
|
||||
|
||||
|
||||
def test_a_fully_published_group_is_left_alone():
|
||||
"""Secondary suppression must not cost anything where nothing is withheld —
|
||||
this is the all-pupils view on every mainstream secondary."""
|
||||
rows = [
|
||||
_row("all", m, p, "published")
|
||||
for m, p in [("school_sixth_form", 75), ("sixth_form_college", 21),
|
||||
("further_education", 61), ("apprenticeship", 8),
|
||||
("employment", 6), ("not_sustained", 5), ("not_captured", 4)]
|
||||
]
|
||||
group = _destinations_block(rows)["groups"]["all"]
|
||||
assert all(c["status"] == "published" for c in group["categories"])
|
||||
assert len(group["categories"]) == 7
|
||||
|
||||
|
||||
def test_the_invariant_is_asserted_directly_not_re_derived():
|
||||
"""A group with one suppressed category and nothing else to withhold."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
|
||||
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
|
||||
_row("all", "further_education", None, "not_applicable", cohort=9),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert block is None or disclosure_invariant_holds(block["groups"])
|
||||
|
||||
|
||||
def test_a_sparse_cohort_with_no_companion_drops_the_group():
|
||||
"""Special schools and AP routinely have one suppressed category and every
|
||||
other one not applicable. There is nothing left to withhold, so the group
|
||||
goes — an earlier version returned here with the violation intact."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=9),
|
||||
_row("all", "sixth_form_college", None, "not_applicable", cohort=9),
|
||||
_row("all", "further_education", None, "not_applicable", cohort=9),
|
||||
_row("all", "apprenticeship", None, "not_applicable", cohort=9),
|
||||
_row("all", "employment", None, "not_applicable", cohort=9),
|
||||
_row("all", "not_sustained", None, "not_applicable", cohort=9),
|
||||
_row("all", "not_captured", None, "not_applicable", cohort=9),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
assert block is None or "all" not in block["groups"], (
|
||||
"a group that cannot be made safe must not be served"
|
||||
)
|
||||
|
||||
|
||||
def test_zeros_are_not_treated_as_a_usable_companion():
|
||||
"""Suppressing a zero protects nothing — the residual is unchanged. With
|
||||
only zeros available the group must be dropped, not falsely 'fixed'."""
|
||||
rows = [
|
||||
_row("all", "school_sixth_form", None, "suppressed", cohort=5),
|
||||
_row("all", "sixth_form_college", 0, "published", cohort=5),
|
||||
_row("all", "further_education", 0, "published", cohort=5),
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
if block and "all" in block["groups"]:
|
||||
group = block["groups"]["all"]
|
||||
published = sum(c["pupils"] for c in group["categories"]
|
||||
if c["pupils"] is not None)
|
||||
hidden = [c for c in group["categories"] if c["status"] == "suppressed"]
|
||||
assert len(hidden) != 1, "a zero companion leaves the figure solvable"
|
||||
assert group["cohort"] - published != 5
|
||||
|
||||
|
||||
def test_masking_always_terminates_in_a_safe_state():
|
||||
"""Exhaustive over every suppression pattern of a four-category group."""
|
||||
from itertools import product
|
||||
MEASURES = ["school_sixth_form", "sixth_form_college",
|
||||
"further_education", "apprenticeship"]
|
||||
for statuses in product(["published", "suppressed", "not_applicable"],
|
||||
repeat=len(MEASURES)):
|
||||
rows = [
|
||||
_row("all", m, 3 if st == "published" else None, st, cohort=12)
|
||||
for m, st in zip(MEASURES, statuses)
|
||||
]
|
||||
block = _destinations_block(rows)
|
||||
if block is None:
|
||||
continue
|
||||
assert disclosure_invariant_holds(block["groups"]), (
|
||||
f"invariant broken for {statuses}"
|
||||
)
|
||||
@@ -18,7 +18,10 @@
|
||||
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
|
||||
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
|
||||
# UNLEASH_API_TOKEN — Unleash *client* token, environment: development
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
|
||||
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
|
||||
# refuses to start without it, rather than falling
|
||||
# back to a generated one that changes on restart.
|
||||
# STAGING_DB_IP — macvlan IP for staging Postgres (default 10.0.1.190)
|
||||
# STAGING_FRONTEND_IP — macvlan IP for staging frontend (default 10.0.1.151)
|
||||
|
||||
@@ -124,7 +127,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:staging
|
||||
container_name: sc_staging_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8081:8080"
|
||||
environment:
|
||||
@@ -136,6 +155,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-staging-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
|
||||
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
|
||||
PG_HOST: sc_database
|
||||
PG_PORT: "5432"
|
||||
|
||||
@@ -9,7 +9,10 @@
|
||||
# TYPESENSE_SEARCH_KEY — Typesense search-only key (exposed to frontend)
|
||||
# UNLEASH_URL — http://<unleash-ip>:4242/api (empty = all flags off)
|
||||
# UNLEASH_API_TOKEN — Unleash *client* token, environment: production
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (password auto-generated, see api-server logs)
|
||||
# AIRFLOW_ADMIN_USER — Airflow admin username (default: admin)
|
||||
# AIRFLOW_ADMIN_PASSWORD — Airflow admin password. REQUIRED: the api-server
|
||||
# refuses to start without it, rather than falling
|
||||
# back to a generated one that changes on restart.
|
||||
|
||||
services:
|
||||
|
||||
@@ -113,7 +116,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:prod
|
||||
container_name: schoolcompare_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8080:8080"
|
||||
environment:
|
||||
@@ -125,6 +144,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "${AIRFLOW_ADMIN_USER:-admin}:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:?set AIRFLOW_ADMIN_PASSWORD in the Portainer stack environment}
|
||||
AIRFLOW__LOGGING__BASE_LOG_FOLDER: /opt/airflow/logs
|
||||
PG_HOST: sc_database
|
||||
PG_PORT: "5432"
|
||||
|
||||
+19
-1
@@ -105,7 +105,23 @@ services:
|
||||
airflow-api-server:
|
||||
image: privaterepo.sitaru.org/tudor/school_compare-pipeline:latest
|
||||
container_name: schoolcompare_airflow_api
|
||||
command: airflow api-server --port 8080
|
||||
# The simple auth manager generates a random password on first start and
|
||||
# writes it to a file, so every container restart invalidates the last one.
|
||||
# Writing the file ourselves from an environment variable makes the login
|
||||
# deterministic. Airflow does not generate anything when the file exists.
|
||||
#
|
||||
# Built with python rather than echo/printf so a password containing quotes,
|
||||
# backslashes or spaces is escaped correctly by json.dumps. An unset
|
||||
# AIRFLOW_ADMIN_PASSWORD raises KeyError and the container exits: falling
|
||||
# back to a generated password would silently undo the point of this.
|
||||
command:
|
||||
- bash
|
||||
- -c
|
||||
- |
|
||||
set -euo pipefail
|
||||
mkdir -p /opt/airflow
|
||||
python -c "import json, os, pathlib; pathlib.Path('/opt/airflow/simple_auth_manager_passwords.json').write_text(json.dumps({os.environ.get('AIRFLOW_ADMIN_USER', 'admin'): os.environ['AIRFLOW_ADMIN_PASSWORD']}))"
|
||||
exec airflow api-server --port 8080
|
||||
ports:
|
||||
- "8080:8080"
|
||||
environment: &airflow-env
|
||||
@@ -117,6 +133,8 @@ services:
|
||||
AIRFLOW__API_AUTH__JWT_SECRET: "school-compare-airflow-jwt-secret-key-long-enough-for-sha512"
|
||||
AIRFLOW__API_AUTH__JWT_ISSUER: airflow
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_USERS: "admin:admin"
|
||||
AIRFLOW__CORE__SIMPLE_AUTH_MANAGER_PASSWORDS_FILE: /opt/airflow/simple_auth_manager_passwords.json
|
||||
AIRFLOW_ADMIN_PASSWORD: ${AIRFLOW_ADMIN_PASSWORD:-admin}
|
||||
PG_HOST: db
|
||||
PG_PORT: "5432"
|
||||
PG_USER: schoolcompare
|
||||
|
||||
@@ -98,6 +98,12 @@ fail the E2E gate. That's the point: staging absorbs the risk.
|
||||
pr-checks status checks (frontend, backend, builds, ai-review) to pass.
|
||||
5. **Bootstrap staging data via Airflow** (no prod dump — staging populates
|
||||
itself from source, exercising the pipeline image end-to-end):
|
||||
- Set `AIRFLOW_ADMIN_PASSWORD` in the stack environment first. The
|
||||
api-server refuses to start without it. Airflow's simple auth manager
|
||||
otherwise generates a password on first start and writes it to a file, so
|
||||
the login changes every time the container restarts; the stack writes that
|
||||
file itself from this variable instead. `AIRFLOW_ADMIN_USER` defaults to
|
||||
`admin`.
|
||||
- Open the staging Airflow UI (`http://<host>:8081`) and trigger, in order:
|
||||
`school_data_daily`, `school_data_monthly_ofsted`, then the manual-schedule
|
||||
`school_data_annual_ees` and `school_data_annual_idaci`.
|
||||
|
||||
@@ -42,17 +42,57 @@ Those are the precise numbers the `c` exists to hide, and in a random 400-school
|
||||
sample **22% of mainstream secondaries** have exactly one suppressed category in
|
||||
their disadvantaged group. This is the normal case, not an edge case.
|
||||
|
||||
Two rules follow, and everything else in this document is downstream of them.
|
||||
Three rules follow, and everything else in this document is downstream of them.
|
||||
|
||||
**R1 — Never render a derived remainder.** Not as a number, and not as a bar
|
||||
segment: a segment sized by the residual can be read straight off the axis. Where
|
||||
any category in a pupil group is suppressed, the page shows the published
|
||||
categories, says the rest are withheld, and stops.
|
||||
**R1 — Never *publish* enough to derive a remainder.**
|
||||
|
||||
**R2 — Never aggregate across a suppression boundary.** A group total is
|
||||
publishable only when DfE published that total itself, or when the aggregate
|
||||
spans **two or more** suppressed cells. Summing published components to fill a
|
||||
gap is R1 with extra steps.
|
||||
An earlier draft of this rule said "never *render* a derived remainder", and
|
||||
that was the defect code review caught in PR #137. Not drawing a number does
|
||||
nothing to stop it being computed: `GET /api/schools/{urn}` is public and
|
||||
unauthenticated, so anything in the payload is published whatever the UI
|
||||
chooses to draw. The rendering guards shipped; the payload still carried the
|
||||
cohort and every published category, and `cohort - sum(published)` returned
|
||||
Whitley Bay's withheld figure exactly.
|
||||
|
||||
The rule is therefore about the serialiser, and the UI guards are a second line
|
||||
of defence behind it. Two identities have to be closed:
|
||||
|
||||
- within a pupil group the categories sum to the cohort, so a group with
|
||||
exactly **one** suppressed category gives it away;
|
||||
- across groups, disadvantaged + other = all for every category, so a category
|
||||
suppressed in exactly **one** of the three gives itself away.
|
||||
|
||||
`_mask_for_disclosure` applies DfE's own answer — secondary suppression —
|
||||
withholding a companion cell until every row and every column hides either none
|
||||
or at least two. It iterates, because each new suppression can break the other
|
||||
identity, and terminates because cells are only ever added.
|
||||
|
||||
The companion must carry pupils. Suppressing a zero looks like secondary
|
||||
suppression and protects nothing: the residual still equals the original
|
||||
withheld figure.
|
||||
|
||||
Where no companion can do the job — a sparse cohort whose every other category
|
||||
is `not_applicable`, routine in special schools and alternative provision — the
|
||||
pupil group is **dropped from the payload entirely**. A first version simply
|
||||
returned at that point with the violation intact and no signal, which review
|
||||
caught: a disclosure-control pass that fails silently is worse than none,
|
||||
because everything downstream trusts it. The function now cannot terminate
|
||||
except in a state where `disclosure_invariant_holds()` is true, and an
|
||||
exhaustive test sweeps all 81 suppression patterns of a four-category group to
|
||||
prove it.
|
||||
|
||||
Measured cost on the 400-school sample: the all-pupils bar survives on **94%**
|
||||
of mainstream secondaries rather than 100%. That is the price of not
|
||||
republishing what DfE withheld.
|
||||
|
||||
**R2 — Never aggregate across a suppression boundary.** Summing published
|
||||
components to fill a gap is R1 with extra steps.
|
||||
|
||||
DfE's own aggregates (`Sustained education destination`, `Sustained education,
|
||||
employment & apprenticeships`) are ingested but **not served**. An aggregate
|
||||
spanning exactly one suppressed component names it, and nothing renders them
|
||||
today — an unused field that leaks is not a trade-off worth carrying. They can
|
||||
be re-added with their own guard if the fallback ladder is ever built.
|
||||
|
||||
**R3 — The three pupil groups are one disclosure surface, not three.**
|
||||
Disadvantaged and Not-known-to-be-disadvantaged partition All pupils, so
|
||||
@@ -116,12 +156,17 @@ the counts — the published percentages do not sum to 100.
|
||||
|
||||
Random 400-school sample, 2022/23, mainstream secondaries (n=262):
|
||||
|
||||
| View | Published | Consequence |
|
||||
|---|---|---|
|
||||
| All pupils, all categories | **100%** | Full bar works everywhere |
|
||||
| Disadvantaged, headline rate | 95% | Gap panel works |
|
||||
| Disadvantaged, three grouped cards | 68% | Degrades card by card |
|
||||
| Disadvantaged, all six categories | **20%** | Bar unusable for this group |
|
||||
| View | As published by DfE | After R1–R3 masking | Consequence |
|
||||
|---|---|---|---|
|
||||
| All pupils, all categories | 100% | **94%** | Bar works nearly everywhere |
|
||||
| Disadvantaged, headline rate | 95% | 95% | Gap panel works |
|
||||
| Disadvantaged, three grouped cards | 68% | 68% | Degrades card by card |
|
||||
| Disadvantaged, all six categories | 20% | **20%** | Bar unusable for this group |
|
||||
|
||||
The middle column is what the site actually serves. Masking costs the
|
||||
all-pupils bar on 6% of mainstream secondaries — those are schools where a
|
||||
category was suppressed in exactly one pupil group and no non-zero companion
|
||||
existed below the all-pupils row.
|
||||
|
||||
Special schools and alternative provision are far worse: 13% and 41% respectively
|
||||
have the whole cohort suppressed even for all pupils. The empty state is
|
||||
@@ -325,10 +370,12 @@ and the staging E2E gate runs post-merge.
|
||||
|
||||
## Risks
|
||||
|
||||
**A later change reintroduces the disclosure.** The likeliest route is someone
|
||||
applying `safe_numeric` to a destination column for consistency, or adding a
|
||||
`coalesce` in a mart. Mitigation is the dbt test plus the unit tests on
|
||||
`canAggregate()` — the rule has to be executable, not documentary.
|
||||
**A later change reintroduces the disclosure.** The likeliest routes are
|
||||
applying `safe_numeric` to a destination column for consistency, adding a
|
||||
`coalesce` in a mart, or — as happened in review — enforcing a disclosure rule
|
||||
at the rendering layer instead of the publishing layer. Mitigation is the dbt
|
||||
tests plus `backend/tests/test_destinations_api.py`, which reconstructs the
|
||||
residual the way an attacker would and asserts it no longer resolves.
|
||||
|
||||
**The two-year lag reads as staleness.** Mitigated by dating the cohort in the
|
||||
section header rather than only in a tooltip.
|
||||
|
||||
@@ -2415,11 +2415,20 @@ test('with autosuggest off, the search box is a plain input', async ({ page }) =
|
||||
|
||||
// ── Destination measures ───────────────────────────────────────────────────
|
||||
//
|
||||
// These journeys need marts.fact_ks4_destinations to be populated, which only
|
||||
// happens after the annual EES DAG runs. Until then the helper below fails the
|
||||
// suite loudly rather than skipping: a silent skip here would let a genuine
|
||||
// regression in the sections ride along unnoticed, which is exactly what the
|
||||
// distance journeys were changed to avoid.
|
||||
// Two failure modes have to be told apart here, and conflating them is how
|
||||
// this suite would either hide a regression or block the promotion pipeline:
|
||||
//
|
||||
// * the backend does not serve the `destinations` field at all — a code
|
||||
// regression, or a deploy that did not land. FAILS.
|
||||
// * the field is served but every school is empty — the annual EES DAG has
|
||||
// not run on this environment yet. SKIPS, loudly.
|
||||
//
|
||||
// The second is a data-load precondition, not a defect, and it is true for
|
||||
// every commit between this merging and the DAG being triggered. Failing on it
|
||||
// would redden the staging gate for unrelated work. This is not the quiet skip
|
||||
// 4f01fbd removed from the distance journeys: that one hid a broken feature
|
||||
// behind a flag check, whereas the assertion that the code is deployed and
|
||||
// correctly shaped still runs here on every commit.
|
||||
|
||||
async function secondaryWithDestinations(page: Page): Promise<{
|
||||
urn: string; destinations: any;
|
||||
@@ -2431,17 +2440,28 @@ async function secondaryWithDestinations(page: Page): Promise<{
|
||||
.filter((s: { phase?: string; attainment_8_score?: number | null }) =>
|
||||
s.phase === 'Secondary' && s.attainment_8_score != null)
|
||||
.map((s: { urn: number }) => String(s.urn));
|
||||
expect(urns.length).toBeGreaterThan(0);
|
||||
|
||||
let served = false;
|
||||
for (const urn of urns.slice(0, 25)) {
|
||||
const detail = await page.request.get(`/api/schools/${urn}`);
|
||||
if (!detail.ok()) continue;
|
||||
const data = await detail.json();
|
||||
// The key must exist, even as null. Its absence means the backend in front
|
||||
// of us does not know about destinations at all.
|
||||
if ('destinations' in data) served = true;
|
||||
if (data.destinations?.ks4) return { urn, destinations: data.destinations };
|
||||
}
|
||||
throw new Error(
|
||||
'No secondary school returned a destinations block. Either the annual EES '
|
||||
+ 'DAG has not run on this environment, or the destinations marts are empty.',
|
||||
);
|
||||
|
||||
expect(served,
|
||||
'GET /api/schools/{urn} served no `destinations` key at all — the backend '
|
||||
+ 'is missing this feature, not merely missing its data').toBeTruthy();
|
||||
|
||||
test.skip(true,
|
||||
'No school has destination data yet: the annual EES DAG has not run on '
|
||||
+ 'this environment. The API shape is correct, so this is a data-load '
|
||||
+ 'precondition rather than a regression.');
|
||||
throw new Error('unreachable');
|
||||
}
|
||||
|
||||
test('a secondary school page says where its Year 11 leavers went', async ({ page }) => {
|
||||
|
||||
@@ -22,7 +22,7 @@ const ALL_PUBLISHED = [
|
||||
|
||||
const fullPhase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: { all: { cohort: 180, categories: ALL_PUBLISHED, aggregates: {} } },
|
||||
groups: { all: { cohort: 180, categories: ALL_PUBLISHED } },
|
||||
};
|
||||
|
||||
const suppressedPhase: DestinationPhase = {
|
||||
@@ -36,7 +36,6 @@ const suppressedPhase: DestinationPhase = {
|
||||
cell('apprenticeship', 8), cell('employment', 6),
|
||||
cell('not_sustained', 5), cell('not_captured', 4),
|
||||
],
|
||||
aggregates: {},
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -90,3 +89,61 @@ describe('DestinationsSection', () => {
|
||||
expect(container.firstChild).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('the detail table keeps the three statuses apart', () => {
|
||||
// 'suppressed' and 'not_applicable' are different claims, and the mart, the
|
||||
// SQLAlchemy model and the serialiser all preserve the difference. The table
|
||||
// used to key its Share column off `percentage === null`, which is true for
|
||||
// both, so a category that simply does not apply was labelled "withheld" —
|
||||
// while the Pupils column beside it rendered blank.
|
||||
const mixedPhase: DestinationPhase = {
|
||||
cohort_year: '2022/23',
|
||||
groups: {
|
||||
all: {
|
||||
cohort: 180,
|
||||
categories: [
|
||||
cell('school_sixth_form', 75),
|
||||
cell('sixth_form_college', null, 'suppressed'),
|
||||
cell('further_education', null, 'suppressed'),
|
||||
cell('apprenticeship', null, 'not_applicable'),
|
||||
],
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const rowFor = (container: HTMLElement, category: string) =>
|
||||
Array.from(container.querySelectorAll('tbody tr'))
|
||||
.find(tr => tr.textContent?.includes(category));
|
||||
|
||||
it('never labels a not-applicable category as withheld', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'Apprenticeship');
|
||||
expect(row).toBeTruthy();
|
||||
expect(row!.textContent).not.toMatch(/withheld/i);
|
||||
});
|
||||
|
||||
it('labels a genuinely suppressed category as withheld in both columns', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'Sixth-form college');
|
||||
expect(row).toBeTruthy();
|
||||
expect(row!.querySelectorAll('td')).toHaveLength(2);
|
||||
Array.from(row!.querySelectorAll('td')).forEach(td =>
|
||||
expect(td.textContent).toMatch(/withheld/i));
|
||||
});
|
||||
|
||||
it('the two columns of a row never disagree about what the row is', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
Array.from(container.querySelectorAll('tbody tr')).forEach(tr => {
|
||||
const cells = Array.from(tr.querySelectorAll('td'))
|
||||
.map(td => /withheld/i.test(td.textContent ?? ''));
|
||||
expect(new Set(cells).size).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
it('shows a published category its real figures', () => {
|
||||
const { container } = render(<DestinationsSection destinations={mixedPhase} />);
|
||||
const row = rowFor(container, 'State-funded school sixth form');
|
||||
expect(row!.textContent).toMatch(/75/);
|
||||
expect(row!.textContent).toMatch(/42%/);
|
||||
});
|
||||
});
|
||||
@@ -14,7 +14,6 @@ const phase: DestinationPhase = {
|
||||
{ category: 'employment', pupils: 13, percentage: 13.5, status: 'published' },
|
||||
{ category: 'not_sustained', pupils: 6, percentage: 6.3, status: 'published' },
|
||||
],
|
||||
aggregates: {},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import {
|
||||
canAggregate, aggregateCells, canRenderPublishedAggregate,
|
||||
canAggregate, aggregateCells,
|
||||
canRenderBar, toBarSegments, CARD_GROUPS,
|
||||
type DestinationCell, type DestinationGroup, type DestinationCategory,
|
||||
} from '@/lib/destinations';
|
||||
@@ -19,7 +19,6 @@ const fullGroup = (): DestinationGroup => ({
|
||||
pub('apprenticeship', 8, 180), pub('employment', 6, 180),
|
||||
pub('not_sustained', 5, 180), pub('not_captured', 4, 180),
|
||||
],
|
||||
aggregates: {},
|
||||
});
|
||||
|
||||
describe('canAggregate — R2, computing from components', () => {
|
||||
@@ -47,26 +46,6 @@ describe('aggregateCells', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('canRenderPublishedAggregate — R2, a total DfE published itself', () => {
|
||||
it('allows it when no component is suppressed', () => {
|
||||
expect(canRenderPublishedAggregate([
|
||||
pub('school_sixth_form', 75, 180), pub('sixth_form_college', 21, 180),
|
||||
])).toBe(true);
|
||||
});
|
||||
|
||||
it('REFUSES it when exactly one component is suppressed — the aggregate identifies it', () => {
|
||||
expect(canRenderPublishedAggregate([
|
||||
pub('school_sixth_form', 75, 180), sup('sixth_form_college'),
|
||||
])).toBe(false);
|
||||
});
|
||||
|
||||
it('allows it when two or more components are suppressed', () => {
|
||||
expect(canRenderPublishedAggregate([
|
||||
sup('school_sixth_form'), sup('sixth_form_college'),
|
||||
])).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('canRenderBar — R1', () => {
|
||||
it('allows a bar when the whole group is published', () => {
|
||||
expect(canRenderBar(fullGroup())).toBe(true);
|
||||
|
||||
@@ -17,7 +17,6 @@ const phase = (categories = 1) => ({
|
||||
category: 'school_sixth_form' as const,
|
||||
pupils: 75, percentage: 41.7, status: 'published' as const,
|
||||
})),
|
||||
aggregates: {},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
@@ -39,7 +39,6 @@ function toGroup(payload: DestinationGroupPayload): DestinationGroup {
|
||||
return {
|
||||
cohort: payload.cohort ?? 0,
|
||||
cells: payload.categories,
|
||||
aggregates: payload.aggregates,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -48,6 +47,44 @@ function cellsFor(group: DestinationGroup, card: CardGroup): DestinationCell[] {
|
||||
return group.cells.filter(c => wanted.has(c.category));
|
||||
}
|
||||
|
||||
/**
|
||||
* One cell of the detail table.
|
||||
*
|
||||
* The three statuses are three different statements and the table has to keep
|
||||
* them apart, because the whole pipeline does — the mart, the SQLAlchemy model
|
||||
* and the serialiser all preserve the difference deliberately:
|
||||
*
|
||||
* published the figure
|
||||
* suppressed DfE withheld it to protect a small number of pupils
|
||||
* not_applicable this destination does not apply to this school at all
|
||||
*
|
||||
* An earlier version keyed the share column off `percentage === null`, which is
|
||||
* also true for not_applicable, so a category that simply does not apply was
|
||||
* labelled "withheld" — while the pupils column beside it rendered blank. Both
|
||||
* columns now derive from `status`, so they cannot disagree.
|
||||
*/
|
||||
function cellValue(
|
||||
cell: DestinationCell, cohort: number, kind: 'pupils' | 'share',
|
||||
) {
|
||||
if (cell.status === 'suppressed') {
|
||||
return <span className={styles.withheldMark}>withheld</span>;
|
||||
}
|
||||
const notApplicable = (
|
||||
<span className={styles.notApplicable} title="Does not apply to this school">
|
||||
—
|
||||
</span>
|
||||
);
|
||||
|
||||
if (cell.status !== 'published' || cell.pupils === null) return notApplicable;
|
||||
if (kind === 'pupils') return cell.pupils;
|
||||
|
||||
// Percentages come from the mart, but a published count with no published
|
||||
// percentage is recoverable from the cohort — both halves are published, so
|
||||
// nothing withheld is involved. Same derivation the bar widths use.
|
||||
const share = cell.percentage ?? (cohort > 0 ? (cell.pupils / cohort) * 100 : null);
|
||||
return share === null ? notApplicable : `${Math.round(share)}%`;
|
||||
}
|
||||
|
||||
export function DestinationsView({
|
||||
destinations, phase,
|
||||
}: { destinations: DestinationPhase; phase: 'ks4' | 'ks5' }) {
|
||||
@@ -194,27 +231,19 @@ export function DestinationsView({
|
||||
const cell = group.cells.find(c => c.category === category);
|
||||
if (!cell) return [];
|
||||
const card = cardGroupFor(category);
|
||||
const isWithheld = cell.status === 'suppressed';
|
||||
return [(
|
||||
<tr
|
||||
key={category}
|
||||
data-group={card ?? 'none'}
|
||||
data-status={cell.status}
|
||||
className={dimmed(card) ? styles.dim : ''}
|
||||
>
|
||||
<th scope="row" className={styles.rowName}>
|
||||
<span className={`${styles.swatch} ${styles[category]}`} />
|
||||
{CATEGORY_LABELS[category]}
|
||||
</th>
|
||||
<td>
|
||||
{isWithheld
|
||||
? <span className={styles.withheldMark}>withheld</span>
|
||||
: cell.pupils}
|
||||
</td>
|
||||
<td>
|
||||
{isWithheld || cell.percentage === null
|
||||
? <span className={styles.withheldMark}>withheld</span>
|
||||
: `${Math.round(cell.percentage)}%`}
|
||||
</td>
|
||||
<td>{cellValue(cell, group.cohort, 'pupils')}</td>
|
||||
<td>{cellValue(cell, group.cohort, 'share')}</td>
|
||||
</tr>
|
||||
)];
|
||||
})}
|
||||
|
||||
@@ -297,3 +297,11 @@
|
||||
font-size: var(--step--2);
|
||||
color: var(--text-muted);
|
||||
}
|
||||
|
||||
/* A destination that does not apply to this school. Deliberately not the
|
||||
withheld badge: "we are not told" and "there is nothing to tell" are
|
||||
different statements, and the rest of the pipeline keeps them apart. */
|
||||
.notApplicable {
|
||||
color: var(--text-muted);
|
||||
cursor: help;
|
||||
}
|
||||
@@ -5,9 +5,13 @@
|
||||
* DfE suppresses individual cells with `c`, and the destination categories sum
|
||||
* to the cohort. So subtracting the published cells from the cohort total
|
||||
* recovers a lone suppressed cell exactly — which is the case on 22% of
|
||||
* mainstream secondaries. The guards below are what stop this module's
|
||||
* consumers doing that by accident, and they are why a percentage is never
|
||||
* reconstructed from a partial sum.
|
||||
* mainstream secondaries.
|
||||
*
|
||||
* The guards here are the SECOND line of defence, not the first. Not drawing a
|
||||
* number does nothing to stop it being computed, so the real fix lives in
|
||||
* backend/data_loader.py::_mask_for_disclosure, which withholds a companion
|
||||
* cell before the figures ever leave the server. These functions keep the UI
|
||||
* honest about what it draws from an already-safe payload.
|
||||
*
|
||||
* See docs/superpowers/specs/2026-08-28-destination-measures-design.md.
|
||||
*/
|
||||
@@ -40,8 +44,6 @@ export interface DestinationCell {
|
||||
export interface DestinationGroup {
|
||||
cohort: number;
|
||||
cells: DestinationCell[];
|
||||
/** Aggregates DfE published itself, keyed by slug. */
|
||||
aggregates: Partial<Record<'sustained_education' | 'sustained_all', DestinationCell>>;
|
||||
}
|
||||
|
||||
/** Display order, which is also bar order: education, then work, then absence. */
|
||||
@@ -80,15 +82,6 @@ export function aggregateCells(
|
||||
return { pupils, percentage: (pupils / cohort) * 100 };
|
||||
}
|
||||
|
||||
/**
|
||||
* R2, the other direction: DfE published this total itself. Showing it beside
|
||||
* the components is safe only when it spans no suppressed component, or two or
|
||||
* more. Exactly one, and the total names the withheld figure.
|
||||
*/
|
||||
export function canRenderPublishedAggregate(components: DestinationCell[]): boolean {
|
||||
return suppressedCount(components) !== 1;
|
||||
}
|
||||
|
||||
/** R1: a bar is drawable only when nothing in the group is withheld. */
|
||||
export function canRenderBar(group: DestinationGroup): boolean {
|
||||
return group.cohort > 0 && group.cells.every(c => c.status === 'published');
|
||||
|
||||
@@ -616,8 +616,6 @@ import type { DestinationCell, PupilGroup } from './destinations';
|
||||
export interface DestinationGroupPayload {
|
||||
cohort: number | null;
|
||||
categories: DestinationCell[];
|
||||
/** Totals DfE published itself. Never computed here — see lib/destinations.ts. */
|
||||
aggregates: Partial<Record<'sustained_education' | 'sustained_all', DestinationCell>>;
|
||||
}
|
||||
|
||||
export interface DestinationPhase {
|
||||
|
||||
@@ -18,6 +18,7 @@ COPY plugins/ plugins/
|
||||
RUN pip install --no-cache-dir \
|
||||
./plugins/extractors/tap-uk-gias \
|
||||
./plugins/extractors/tap-uk-ees \
|
||||
./plugins/extractors/tap-uk-ees-destinations \
|
||||
./plugins/extractors/tap-uk-ofsted \
|
||||
./plugins/extractors/tap-uk-fbit \
|
||||
./plugins/extractors/tap-uk-idaci
|
||||
|
||||
@@ -190,7 +190,7 @@ with DAG(
|
||||
|
||||
dbt_build_ees = BashOperator(
|
||||
task_id="dbt_build",
|
||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_ees_ks2+ stg_legacy_ks2+ stg_ees_ks4+ stg_legacy_ks4+ stg_ees_census+ stg_ees_admissions+ stg_ees_ks2_national+ stg_ees_ks4_national+ stg_ees_ks4_destinations+ stg_ees_ks5_destinations+",
|
||||
bash_command=f"cd {PIPELINE_DIR}/transform && {DBT_BIN} build --profiles-dir . --target production --select stg_ees_ks2+ stg_legacy_ks2+ stg_ees_ks4+ stg_legacy_ks4+ stg_ees_census+ stg_ees_admissions+ stg_ees_ks2_national+ stg_ees_ks4_national+ stg_ees_ks4_destinations+ stg_ees_ks5_destinations+ stg_ees_ks4_destinations_national+ stg_ees_ks5_destinations_national+",
|
||||
)
|
||||
|
||||
sync_typesense_ees = BashOperator(
|
||||
|
||||
+74
-18
@@ -188,21 +188,25 @@ class DestinationsStream(Stream):
|
||||
_national_pinned: list[str] = []
|
||||
_indicators: dict[str, str] = {}
|
||||
|
||||
schema = th.PropertiesList(
|
||||
th.Property("urn", th.StringType),
|
||||
# School rows and the England reference are DIFFERENT GRAINS, so they are
|
||||
# different streams. Carrying both in one table meant a null `urn` inside
|
||||
# the primary key, which target-postgres turns into a NOT NULL constraint:
|
||||
# the first national row killed the loader mid-run, and the tap saw only a
|
||||
# BrokenPipeError on its stdout.
|
||||
_MEASURE_PROPERTIES = (
|
||||
th.Property("time_period", th.StringType),
|
||||
th.Property("pupil_group", th.StringType),
|
||||
th.Property("destination_measure", th.StringType),
|
||||
th.Property("cohort_pupils", th.StringType),
|
||||
th.Property("pupils_raw", th.StringType),
|
||||
th.Property("percentage_raw", th.StringType),
|
||||
).to_dict()
|
||||
)
|
||||
_MEASURE_KEYS = ["time_period", "pupil_group", "destination_measure"]
|
||||
|
||||
primary_keys = ["urn", "time_period", "pupil_group", "destination_measure"]
|
||||
replication_key = None
|
||||
|
||||
def _query(self, period: str, pinned: list[str], keep_level: str,
|
||||
urn_by_location: dict[str, str]):
|
||||
urn_by_location: dict[str, str]): # noqa: D401
|
||||
"""Page one period of one geographic level, yielding Singer records."""
|
||||
criteria = [
|
||||
{"filters": {"in": list(self._destination_slugs)}},
|
||||
@@ -248,19 +252,51 @@ class DestinationsStream(Stream):
|
||||
"%s: %d school locations, %d time periods",
|
||||
self.name, len(urn_by_location), len(periods),
|
||||
)
|
||||
|
||||
# Two queries per period, because school rows and the England reference
|
||||
# need different establishment pins — see KS4_NATIONAL_PINNED. The API
|
||||
# cannot filter by geographic level, so each pass keeps its own and
|
||||
# discards the local-authority, district, regional and constituency
|
||||
# rows that come with them.
|
||||
for period in periods:
|
||||
yield from self._query(period, self._pinned, "SCH", urn_by_location)
|
||||
yield from self._query(period, self._national_pinned, "NAT", urn_by_location)
|
||||
yield from self._query(
|
||||
period, self._pinned_for_level, self._level, urn_by_location,
|
||||
)
|
||||
|
||||
|
||||
class KS4DestinationsStream(DestinationsStream):
|
||||
name = "ees_ks4_destinations"
|
||||
class SchoolDestinationsStream(DestinationsStream):
|
||||
"""School-level rows. `urn` is part of the key and is never null."""
|
||||
|
||||
_level = "SCH"
|
||||
|
||||
@property
|
||||
def _pinned_for_level(self):
|
||||
return self._pinned
|
||||
|
||||
schema = th.PropertiesList(
|
||||
th.Property("urn", th.StringType),
|
||||
*DestinationsStream._MEASURE_PROPERTIES,
|
||||
).to_dict()
|
||||
|
||||
primary_keys = ["urn", *DestinationsStream._MEASURE_KEYS]
|
||||
|
||||
|
||||
class NationalDestinationsStream(DestinationsStream):
|
||||
"""The England reference. No `urn` column at all — a school identifier that
|
||||
is always null is not a column, it is a grain mismatch."""
|
||||
|
||||
_level = "NAT"
|
||||
|
||||
@property
|
||||
def _pinned_for_level(self):
|
||||
return self._national_pinned
|
||||
|
||||
schema = th.PropertiesList(*DestinationsStream._MEASURE_PROPERTIES).to_dict()
|
||||
|
||||
primary_keys = list(DestinationsStream._MEASURE_KEYS)
|
||||
|
||||
def post_process(self, row, context=None):
|
||||
# row_to_record emits urn=None for national rows; drop the key rather
|
||||
# than ship a column that is null in every row.
|
||||
row.pop("urn", None)
|
||||
return row
|
||||
|
||||
|
||||
class _KS4Config(DestinationsStream):
|
||||
_dataset_id = KS4_DATASET
|
||||
_destination_slugs = KS4_DESTINATION_SLUGS
|
||||
_pupil_group_slugs = KS4_PUPIL_GROUP_SLUGS
|
||||
@@ -269,8 +305,7 @@ class KS4DestinationsStream(DestinationsStream):
|
||||
_indicators = KS4_INDICATORS
|
||||
|
||||
|
||||
class KS5DestinationsStream(DestinationsStream):
|
||||
name = "ees_ks5_destinations"
|
||||
class _KS5Config(DestinationsStream):
|
||||
_dataset_id = KS5_DATASET
|
||||
_destination_slugs = KS5_DESTINATION_SLUGS
|
||||
_pupil_group_slugs = KS5_PUPIL_GROUP_SLUGS
|
||||
@@ -279,12 +314,33 @@ class KS5DestinationsStream(DestinationsStream):
|
||||
_indicators = KS5_INDICATORS
|
||||
|
||||
|
||||
class KS4DestinationsStream(SchoolDestinationsStream, _KS4Config):
|
||||
name = "ees_ks4_destinations"
|
||||
|
||||
|
||||
class KS5DestinationsStream(SchoolDestinationsStream, _KS5Config):
|
||||
name = "ees_ks5_destinations"
|
||||
|
||||
|
||||
class KS4NationalDestinationsStream(NationalDestinationsStream, _KS4Config):
|
||||
name = "ees_ks4_destinations_national"
|
||||
|
||||
|
||||
class KS5NationalDestinationsStream(NationalDestinationsStream, _KS5Config):
|
||||
name = "ees_ks5_destinations_national"
|
||||
|
||||
|
||||
class TapUKEESDestinations(Tap):
|
||||
name = "tap-uk-ees-destinations"
|
||||
config_jsonschema = th.PropertiesList().to_dict()
|
||||
|
||||
def discover_streams(self):
|
||||
return [KS4DestinationsStream(self), KS5DestinationsStream(self)]
|
||||
return [
|
||||
KS4DestinationsStream(self),
|
||||
KS5DestinationsStream(self),
|
||||
KS4NationalDestinationsStream(self),
|
||||
KS5NationalDestinationsStream(self),
|
||||
]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -132,3 +132,68 @@ def test_establishment_dimensions_are_not_pinned():
|
||||
establishment_totals = {"4369U", "rgHcN", "EfHQq", "S4ROV"}
|
||||
assert not set(KS4_PINNED) & establishment_totals
|
||||
assert not set(KS5_PINNED) & establishment_totals
|
||||
|
||||
|
||||
# ── Grain separation ────────────────────────────────────────────────────────
|
||||
#
|
||||
# School rows and the England reference were originally one stream with a
|
||||
# nullable `urn` in the primary key. target-postgres turns primary_keys into a
|
||||
# NOT NULL constraint, so the first national row killed the loader mid-run and
|
||||
# the tap saw only a BrokenPipeError on its stdout — a symptom several frames
|
||||
# away from the cause.
|
||||
|
||||
def _streams():
|
||||
from tap_uk_ees_destinations.tap import TapUKEESDestinations
|
||||
return TapUKEESDestinations(config={}, validate_config=False).discover_streams()
|
||||
|
||||
|
||||
def test_no_stream_has_a_nullable_primary_key_column():
|
||||
for stream in _streams():
|
||||
props = stream.schema["properties"]
|
||||
for key in stream.primary_keys:
|
||||
assert key in props, f"{stream.name}: key {key} is not in the schema"
|
||||
if "urn" in stream.primary_keys:
|
||||
assert stream._level == "SCH", (
|
||||
f"{stream.name} keys on urn but does not emit school rows"
|
||||
)
|
||||
|
||||
|
||||
def test_national_streams_carry_no_urn_column_at_all():
|
||||
for stream in _streams():
|
||||
if not stream.name.endswith("_national"):
|
||||
continue
|
||||
assert "urn" not in stream.schema["properties"], (
|
||||
"a school identifier that is null in every row is a grain "
|
||||
"mismatch, not a column"
|
||||
)
|
||||
assert "urn" not in stream.primary_keys
|
||||
|
||||
|
||||
def test_national_post_process_drops_the_null_urn():
|
||||
from tap_uk_ees_destinations.tap import KS4NationalDestinationsStream, TapUKEESDestinations
|
||||
tap = TapUKEESDestinations(config={}, validate_config=False)
|
||||
stream = KS4NationalDestinationsStream(tap)
|
||||
row = {"urn": None, "time_period": "202223", "pupil_group": "all",
|
||||
"destination_measure": "school_sixth_form", "cohort_pupils": "1",
|
||||
"pupils_raw": "1", "percentage_raw": "1"}
|
||||
assert "urn" not in stream.post_process(dict(row))
|
||||
|
||||
|
||||
def test_school_and_national_streams_exist_for_both_phases():
|
||||
names = {s.name for s in _streams()}
|
||||
assert names == {
|
||||
"ees_ks4_destinations", "ees_ks5_destinations",
|
||||
"ees_ks4_destinations_national", "ees_ks5_destinations_national",
|
||||
}
|
||||
|
||||
|
||||
def test_each_stream_queries_its_own_geographic_level_with_its_own_pins():
|
||||
"""The national pass needs establishment pinned to Total; the school pass
|
||||
must not pin it at all, or every school returns zero rows."""
|
||||
for stream in _streams():
|
||||
if stream.name.endswith("_national"):
|
||||
assert stream._level == "NAT"
|
||||
assert stream._pinned_for_level is stream._national_pinned
|
||||
else:
|
||||
assert stream._level == "SCH"
|
||||
assert stream._pinned_for_level is stream._pinned
|
||||
@@ -22,8 +22,7 @@ select
|
||||
pupils,
|
||||
percentage,
|
||||
status
|
||||
from {{ ref('stg_ees_ks4_destinations') }}
|
||||
where urn is null
|
||||
from {{ ref('stg_ees_ks4_destinations_national') }}
|
||||
|
||||
union all
|
||||
|
||||
@@ -36,5 +35,4 @@ select
|
||||
pupils,
|
||||
percentage,
|
||||
status
|
||||
from {{ ref('stg_ees_ks5_destinations') }}
|
||||
where urn is null
|
||||
from {{ ref('stg_ees_ks5_destinations_national') }}
|
||||
@@ -47,6 +47,17 @@ sources:
|
||||
16-18 study leavers destinations. Same grain, same suppression
|
||||
caveat, and only institutions with post-16 provision appear.
|
||||
|
||||
- name: ees_ks4_destinations_national
|
||||
description: >
|
||||
England KS4 destination measures by pupil group. A separate table
|
||||
from ees_ks4_destinations because it is a separate grain — no school,
|
||||
so no urn column. Same 'c' suppression caveat.
|
||||
|
||||
- name: ees_ks5_destinations_national
|
||||
description: >
|
||||
England 16-18 destination measures by pupil group. Same grain and
|
||||
caveat as ees_ks4_destinations_national.
|
||||
|
||||
- name: ees_ks4_performance
|
||||
description: KS4 performance tables (long format — one row per school × breakdown × sex)
|
||||
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
{{ config(materialized='table') }}
|
||||
|
||||
-- Staging model: KS4 leavers destinations, school level plus the England
|
||||
-- reference (which carries a null urn).
|
||||
-- Staging model: KS4 leavers destinations, school level.
|
||||
--
|
||||
-- School rows only. The England reference is a different grain and lives in
|
||||
-- stg_ees_ks4_destinations_national.
|
||||
--
|
||||
-- DELIBERATELY DOES NOT USE safe_numeric. That macro maps every EES sentinel
|
||||
-- (z, c, x, q, u) to NULL, which is right for attainment — there, "suppressed"
|
||||
@@ -14,13 +16,12 @@
|
||||
|
||||
with source as (
|
||||
select * from {{ source('raw', 'ees_ks4_destinations') }}
|
||||
-- National rows carry a null urn and feed fact_destination_national.
|
||||
where (urn is null or urn = '' or urn ~ '^[0-9]+$')
|
||||
where urn ~ '^[0-9]+$'
|
||||
and time_period ~ '^[0-9]+$'
|
||||
)
|
||||
|
||||
select
|
||||
case when urn ~ '^[0-9]+$' then cast(trim(urn) as integer) end as urn,
|
||||
cast(trim(urn) as integer) as urn,
|
||||
cast(trim(time_period) as integer) as year,
|
||||
trim(pupil_group) as pupil_group,
|
||||
trim(destination_measure) as destination_measure,
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
{{ config(materialized='table') }}
|
||||
|
||||
-- Staging model: England KS4 destination measures — the national
|
||||
-- reference the school sections compare against.
|
||||
--
|
||||
-- A separate model because it is a separate grain: there is no school here, and
|
||||
-- carrying these rows in the school table meant a null urn inside the primary
|
||||
-- key, which the Postgres loader rejects.
|
||||
--
|
||||
-- DELIBERATELY DOES NOT USE safe_numeric, for the same reason as the school
|
||||
-- model: 'suppressed' and 'not applicable' are different claims.
|
||||
|
||||
with source as (
|
||||
select * from {{ source('raw', 'ees_ks4_destinations_national') }}
|
||||
where time_period ~ '^[0-9]+$'
|
||||
)
|
||||
|
||||
select
|
||||
cast(trim(time_period) as integer) as year,
|
||||
trim(pupil_group) as pupil_group,
|
||||
trim(destination_measure) as destination_measure,
|
||||
|
||||
case when cohort_pupils ~ '^[0-9]+$'
|
||||
then cast(cohort_pupils as integer) end as cohort_pupils,
|
||||
|
||||
case when pupils_raw ~ '^[0-9]+$'
|
||||
then cast(pupils_raw as integer) end as pupils,
|
||||
|
||||
case when percentage_raw ~ '^-?[0-9]+(\.[0-9]+)?$'
|
||||
then cast(percentage_raw as numeric) end as percentage,
|
||||
|
||||
case
|
||||
when pupils_raw ~ '^[0-9]+$' then 'published'
|
||||
when lower(trim(pupils_raw)) = 'c' then 'suppressed'
|
||||
else 'not_applicable'
|
||||
end as status
|
||||
|
||||
from source
|
||||
@@ -1,8 +1,10 @@
|
||||
{{ config(materialized='table') }}
|
||||
|
||||
-- Staging model: 16-18 study leavers destinations, institution level plus the
|
||||
-- England reference (which carries a null urn). Only sixth forms and colleges
|
||||
-- appear here, so a secondary with no post-16 provision has no rows at all.
|
||||
-- Staging model: 16-18 study leavers destinations, institution level.
|
||||
--
|
||||
-- Only sixth forms and colleges appear, so a secondary with no post-16
|
||||
-- provision has no rows at all. The England reference is a different grain and
|
||||
-- lives in stg_ees_ks5_destinations_national.
|
||||
--
|
||||
-- DELIBERATELY DOES NOT USE safe_numeric. That macro maps every EES sentinel
|
||||
-- (z, c, x, q, u) to NULL, which is right for attainment — there, "suppressed"
|
||||
@@ -15,13 +17,12 @@
|
||||
|
||||
with source as (
|
||||
select * from {{ source('raw', 'ees_ks5_destinations') }}
|
||||
-- National rows carry a null urn and feed fact_destination_national.
|
||||
where (urn is null or urn = '' or urn ~ '^[0-9]+$')
|
||||
where urn ~ '^[0-9]+$'
|
||||
and time_period ~ '^[0-9]+$'
|
||||
)
|
||||
|
||||
select
|
||||
case when urn ~ '^[0-9]+$' then cast(trim(urn) as integer) end as urn,
|
||||
cast(trim(urn) as integer) as urn,
|
||||
cast(trim(time_period) as integer) as year,
|
||||
trim(pupil_group) as pupil_group,
|
||||
trim(destination_measure) as destination_measure,
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
{{ config(materialized='table') }}
|
||||
|
||||
-- Staging model: England KS5 destination measures — the national
|
||||
-- reference the school sections compare against.
|
||||
--
|
||||
-- A separate model because it is a separate grain: there is no school here, and
|
||||
-- carrying these rows in the school table meant a null urn inside the primary
|
||||
-- key, which the Postgres loader rejects.
|
||||
--
|
||||
-- DELIBERATELY DOES NOT USE safe_numeric, for the same reason as the school
|
||||
-- model: 'suppressed' and 'not applicable' are different claims.
|
||||
|
||||
with source as (
|
||||
select * from {{ source('raw', 'ees_ks5_destinations_national') }}
|
||||
where time_period ~ '^[0-9]+$'
|
||||
)
|
||||
|
||||
select
|
||||
cast(trim(time_period) as integer) as year,
|
||||
trim(pupil_group) as pupil_group,
|
||||
trim(destination_measure) as destination_measure,
|
||||
|
||||
case when cohort_pupils ~ '^[0-9]+$'
|
||||
then cast(cohort_pupils as integer) end as cohort_pupils,
|
||||
|
||||
case when pupils_raw ~ '^[0-9]+$'
|
||||
then cast(pupils_raw as integer) end as pupils,
|
||||
|
||||
case when percentage_raw ~ '^-?[0-9]+(\.[0-9]+)?$'
|
||||
then cast(percentage_raw as numeric) end as percentage,
|
||||
|
||||
case
|
||||
when pupils_raw ~ '^[0-9]+$' then 'published'
|
||||
when lower(trim(pupils_raw)) = 'c' then 'suppressed'
|
||||
else 'not_applicable'
|
||||
end as status
|
||||
|
||||
from source
|
||||
Reference in new issue
Block a user