Model/tests/domain/epc_prediction/test_component_accuracy_gate.py
Khalim Conn-Kowlessar 06a66b3dd9 feat(epc-prediction): coherent heating donor selection (#1225)
Heating sub-fields can't be field-moded without breaking system coherence,
so the whole SapHeating cluster is now copied as a unit from a single
coherent donor rather than inherited from the structural template: the
neighbour matching the cohort's modal heating signature (main fuel +
category + cylinder presence), most recent among the matches (recent cert =
current system). Including cylinder presence in the signature is load-bearing
— it protects has_hot_water_cylinder + cylinder_insulation (a bare fuel+cat
signature regressed them).

Corpus (150pc/514): heating_main_control 66.3 -> 73.9% (+7.6, the target),
main_fuel 92.8 -> 96.9, category 90.7 -> 95.7, water_fuel 92.8 -> 96.3,
water_code 88.5 -> 95.3, has_cylinder 81.1 -> 89.7, secondary 36.2 -> 42.0.
SAP MAE vs lodged 7.08 -> 6.00 (calculator floor 1.57). cylinder_insulation
-13.6 corpus (tiny-n) but +33pp on the fixture; AC requires control up +
fuel/category hold + SAP not worsened, all met.

Gate (36-target fixture): zero regression; ratcheted main_category
0.8889->0.9444, main_control 0.7500->0.8056, water_fuel 0.9167->0.9722,
water_code 0.8889->0.9444, cylinder_insulation_type 0.1667->0.5000. This is
the per-component heating method ([[feedback_per_component_best_method]]):
coherent donor, never field-mode.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-15 13:48:15 +00:00

109 lines
4.1 KiB
Python

"""Tier-1 ratcheting Component Accuracy gate (ADR-0030).
Runs the calculator-free leave-one-out scorer over the committed, anonymised
fixture and asserts each per-component classification rate / geometry residual is
no worse than a committed baseline. Because the prediction is deterministic and
the fixture is frozen, every run reproduces the same numbers exactly — so a
failure means a real *regression* in prediction quality, never sample noise.
The floors / ceilings are the currently-measured values and only ever **tighten**
(the repo's no-tolerance-widening ethos applied to an aggregate): when prediction
improves, ratchet the relevant floor up in the same change. The end-to-end
SAP / carbon / PE guards are deliberately *not* here — they need the calculator,
whose API-path residual is a separate workstream; the component floors are the
real gate (ADR-0030).
"""
from pathlib import Path
import pytest
from domain.epc_prediction.validation import (
ComponentAccuracy,
evaluate_component_accuracy,
)
from harness.epc_prediction_corpus import load_corpus
_FIXTURE = Path(__file__).parents[3] / "tests" / "fixtures" / "epc_prediction"
# Minimum classification hit-rate per component (ratchet floors). Tighten — never
# loosen — as prediction improves. Values are the measured rates over the frozen
# 36-target fixture; a 1e-3 tolerance absorbs float rounding only.
_RATE_FLOORS: dict[str, float] = {
"wall_construction": 0.8889,
"wall_insulation_type": 0.8333,
"construction_age_band": 0.6389,
"construction_age_band_pm1": 0.8333,
"roof_construction": 0.7222,
"floor_construction": 0.8125,
"heating_main_fuel": 0.9722,
"heating_main_category": 0.9444,
"heating_main_control": 0.8056,
"water_heating_fuel": 0.9722,
"water_heating_code": 0.9444,
"has_hot_water_cylinder": 0.8889,
"cylinder_insulation_type": 0.5000,
"secondary_heating_type": 0.0000,
"roof_insulation_thickness": 0.4118,
"floor_insulation": 0.9375,
"has_room_in_roof": 0.8333,
"modal_glazing_type": 0.5278,
"has_pv": 1.0000,
"solar_water_heating": 1.0000,
}
# Maximum mean absolute residual per numeric component (ratchet ceilings).
# window_count is deliberately excluded — it is cosmetic for SAP (issue #1222):
# the predicted picture clusters at a mapper-default 4 windows while actuals
# spread 1-21, yet total_window_area (the SAP-relevant signal) stays tight.
_RESIDUAL_CEILINGS: dict[str, float] = {
"floor_area": 11.8983,
"total_window_area": 4.4067,
"building_parts": 0.3333,
"door_count": 0.6389,
}
_TOLERANCE = 1e-3
@pytest.fixture(scope="module")
def accuracy() -> ComponentAccuracy:
if not (_FIXTURE / "_index.json").exists():
pytest.skip(f"no EPC Prediction fixture at {_FIXTURE}")
return evaluate_component_accuracy(load_corpus(_FIXTURE))
def test_fixture_yields_the_expected_target_count(
accuracy: ComponentAccuracy,
) -> None:
# The frozen fixture must still produce its full set of SAP-10.2 targets — a
# drop means the fixture or the target filter changed.
assert accuracy.targets >= 36
@pytest.mark.parametrize("component,floor", sorted(_RATE_FLOORS.items()))
def test_classification_rate_does_not_regress(
accuracy: ComponentAccuracy, component: str, floor: float
) -> None:
# Arrange / Act
rate = accuracy.rate(component)
# Assert — the component is still applicable and at or above its floor.
assert rate is not None, f"{component} had no applicable targets"
assert rate >= floor - _TOLERANCE, (
f"{component} classification regressed: {rate:.4f} < floor {floor:.4f}"
)
@pytest.mark.parametrize("component,ceiling", sorted(_RESIDUAL_CEILINGS.items()))
def test_residual_does_not_regress(
accuracy: ComponentAccuracy, component: str, ceiling: float
) -> None:
# Arrange / Act
mean_abs = accuracy.mean_abs_residual(component)
# Assert — the mean absolute residual is at or below its ceiling.
assert mean_abs is not None, f"{component} had no residuals"
assert mean_abs <= ceiling + _TOLERANCE, (
f"{component} residual regressed: {mean_abs:.4f} > ceiling {ceiling:.4f}"
)