mirror of
https://github.com/Hestia-Homes/Model.git
synced 2026-07-22 08:48:38 +00:00
ADR-0029 decision 5: survivors were treated equally; now each neighbour's vote in the cohort mode decays with its distance from the cohort's physical centre (floor area from the median, age band from the modal band), so the mode leans on the most representative neighbours instead of being swayed by size/era outliers. Scales (size 20 m^2, age weight 0.5) chosen on the validation corpus; the tight size kernel is load-bearing (looser scales regress floor_insulation on the fixture). Corpus (181 SAP-10.2 targets): wall_insulation 83.4->86.2%, roof_construction 86.2->87.3%, floor_construction 78.8->81.2%, floor_insulation 92.9->94.1%; net +7.5pp gained vs -1.1pp (two 1-cert dips, both held on the fixture). Geometry/residuals untouched (template unchanged). Gate (36-target fixture): zero regression across all 24 floors/ceilings; ratcheted wall_insulation_type 0.7778->0.8333, floor_construction 0.7500->0.8125, floor_insulation 0.9062->0.9375. Dead _mode/_int_mode removed (superseded by the weighted variants). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
317 lines
13 KiB
Python
317 lines
13 KiB
Python
"""EPC Prediction synthesis (ADR-0029).
|
|
|
|
`EpcPrediction.predict` turns the selected `ComparableProperties` into a
|
|
predicted `EpcPropertyData`: copy a coherent representative template's structure
|
|
(building parts, windows, geometry), set the homogeneous categoricals to the
|
|
recency-weighted cohort mode, then apply Landlord Overrides on top. Pure domain
|
|
logic — deterministic neighbour synthesis, not ML.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
import math
|
|
import statistics
|
|
from collections import Counter, defaultdict
|
|
from dataclasses import dataclass
|
|
from datetime import date
|
|
from typing import Iterable, Optional, Union
|
|
|
|
from datatypes.epc.domain.epc_property_data import (
|
|
EpcPropertyData,
|
|
SapBuildingPart,
|
|
)
|
|
from domain.epc_prediction.comparable_properties import (
|
|
Comparable,
|
|
ComparableProperties,
|
|
PredictionTarget,
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class PredictionConfidence:
|
|
"""A compute-only confidence signal for a prediction (ADR-0029 open item).
|
|
|
|
`cohort_size` is the number of Comparable Properties the prediction drew on;
|
|
`component_agreement` maps a homogeneous component to the cohort's *agreement*
|
|
— the modal value's share (0..1) of the neighbours that lodge one. A small or
|
|
split cohort flags a component downstream may want to treat cautiously (e.g.
|
|
the per-dwelling fields with a low accuracy ceiling). Surfacing / persisting
|
|
this is a separate HITL follow-up; here it is computed only.
|
|
"""
|
|
|
|
cohort_size: int
|
|
component_agreement: dict[str, float]
|
|
|
|
def agreement(self, component: str) -> Optional[float]:
|
|
"""The cohort's modal-value share for a component, or None when no
|
|
neighbour lodges one (it was not applicable)."""
|
|
return self.component_agreement.get(component)
|
|
|
|
|
|
class EpcPrediction:
|
|
"""Synthesises a predicted `EpcPropertyData` from Comparable Properties."""
|
|
|
|
def predict(
|
|
self, target: PredictionTarget, comparables: ComparableProperties
|
|
) -> EpcPropertyData:
|
|
"""Predict the target's EPC picture: copy a representative template's
|
|
structure (coherent for the calculator), then set the homogeneous
|
|
categoricals to the cohort mode."""
|
|
template: Comparable = self._template(comparables)
|
|
predicted: EpcPropertyData = copy.deepcopy(template.epc)
|
|
self._apply_categorical_modes(predicted, comparables)
|
|
self._apply_overrides(predicted, target)
|
|
return predicted
|
|
|
|
def confidence(
|
|
self, comparables: ComparableProperties
|
|
) -> PredictionConfidence:
|
|
"""Compute the per-prediction confidence from the cohort: its size plus,
|
|
for each homogeneous categorical, the modal value's share among the
|
|
neighbours that lodge one (ADR-0029). Compute-only — it never alters the
|
|
prediction, only annotates how much the cohort agreed."""
|
|
members: tuple[Comparable, ...] = comparables.members
|
|
agreement: dict[str, float] = {}
|
|
for attr in _MAIN_PART_CATEGORICALS:
|
|
share: Optional[float] = _modal_share(
|
|
_main_part_attr(c, attr) for c in members
|
|
)
|
|
if share is not None:
|
|
agreement[attr] = share
|
|
for attr in _FLOOR_DIM_CATEGORICALS:
|
|
floor_share: Optional[float] = _modal_share(
|
|
_main_floor_attr(c, attr) for c in members
|
|
)
|
|
if floor_share is not None:
|
|
agreement[attr] = floor_share
|
|
return PredictionConfidence(
|
|
cohort_size=len(members), component_agreement=agreement
|
|
)
|
|
|
|
@staticmethod
|
|
def _template(comparables: ComparableProperties) -> Comparable:
|
|
"""The representative comparable whose structure seeds the prediction:
|
|
the member whose floor area is closest to the cohort median. A single
|
|
neighbour's geometry is copied wholesale, so a size-representative
|
|
template keeps the prediction off the cohort's size outliers (ADR-0029
|
|
decision 4: closest on size)."""
|
|
members: tuple[Comparable, ...] = comparables.members
|
|
median_area: float = statistics.median(
|
|
c.epc.total_floor_area_m2 for c in members
|
|
)
|
|
return min(
|
|
members,
|
|
key=lambda c: abs(c.epc.total_floor_area_m2 - median_area),
|
|
)
|
|
|
|
@staticmethod
|
|
def _apply_categorical_modes(
|
|
predicted: EpcPropertyData, comparables: ComparableProperties
|
|
) -> None:
|
|
"""Override the predicted picture's homogeneous categoricals — wall /
|
|
roof / floor construction + insulation, age band — with the cohort mode
|
|
(robust to an atypical template, per ADR-0029 decision 4). The mode is
|
|
physically-similarity-weighted (decision 5): each neighbour's vote decays
|
|
with its distance from the cohort's physical centre, so the mode leans on
|
|
the most representative neighbours rather than treating every survivor
|
|
equally. The template still supplies the geometry; only the categorical
|
|
codes move to the mode. (Glazing type is deliberately left on the
|
|
template — moding it is marginal and noisy; revisit with a larger
|
|
corpus.)"""
|
|
if not predicted.sap_building_parts:
|
|
return
|
|
main: SapBuildingPart = predicted.sap_building_parts[0]
|
|
members = comparables.members
|
|
weights: list[float] = _similarity_weights(members)
|
|
for attr in _MAIN_PART_CATEGORICALS:
|
|
if attr in _RECENCY_WEIGHTED_CATEGORICALS:
|
|
mode = _recency_weighted_mode(members, attr)
|
|
else:
|
|
mode = _weighted_mode(
|
|
(_main_part_attr(c, attr) for c in members), weights
|
|
)
|
|
if mode is not None:
|
|
setattr(main, attr, mode)
|
|
floor_dims = main.sap_floor_dimensions
|
|
if floor_dims:
|
|
for attr in _FLOOR_DIM_CATEGORICALS:
|
|
floor_mode = _weighted_int_mode(
|
|
(_main_floor_attr(c, attr) for c in members), weights
|
|
)
|
|
if floor_mode is not None:
|
|
setattr(floor_dims[0], attr, floor_mode)
|
|
|
|
@staticmethod
|
|
def _apply_overrides(
|
|
predicted: EpcPropertyData, target: PredictionTarget
|
|
) -> None:
|
|
"""Apply the known Landlord Overrides on top of the estimate — a known
|
|
value always wins over the cohort mode (ADR-0029)."""
|
|
if not predicted.sap_building_parts:
|
|
return
|
|
if target.wall_construction is not None:
|
|
predicted.sap_building_parts[0].wall_construction = (
|
|
target.wall_construction
|
|
)
|
|
|
|
|
|
# The homogeneous categoricals carried directly on the main building part. Floor
|
|
# categoricals live on the main floor dimension and glazing on the windows; both
|
|
# are handled separately.
|
|
_MAIN_PART_CATEGORICALS: tuple[str, ...] = (
|
|
"wall_construction",
|
|
"wall_insulation_type",
|
|
"construction_age_band",
|
|
"roof_construction",
|
|
"roof_insulation_thickness",
|
|
)
|
|
|
|
# Integer-coded categoricals on the main building part's ground-floor dimension.
|
|
_FLOOR_DIM_CATEGORICALS: tuple[str, ...] = (
|
|
"floor_construction",
|
|
"floor_insulation",
|
|
)
|
|
|
|
# Categoricals whose physical value CHANGES over time (e.g. loft top-ups), so a
|
|
# recent neighbour reflects the current state better than an old one — these take
|
|
# a recency-WEIGHTED mode. Permanent categoricals (wall / age) take the plain
|
|
# mode: recency-weighting them was net-negative on the validation corpus (it
|
|
# discards data that is still valid). `_RECENCY_TAU_YEARS` is the exponential
|
|
# decay constant (≈2.8-year half-life), chosen on the corpus (roof insulation
|
|
# +4pp / +12pp on the fixture).
|
|
_RECENCY_WEIGHTED_CATEGORICALS: frozenset[str] = frozenset(
|
|
{"roof_insulation_thickness"}
|
|
)
|
|
_RECENCY_TAU_YEARS: float = 4.0
|
|
_DAYS_PER_YEAR: float = 365.0
|
|
|
|
# Physical-similarity weighting of the categorical mode (ADR-0029 decision 5): a
|
|
# comparable's vote decays exponentially with how far it sits from the cohort's
|
|
# physical centre — floor area from the median, construction age from the modal
|
|
# band — so an outlier-sized or outlier-era neighbour can't sway the mode. Scales
|
|
# chosen on the validation corpus (wall-insulation +2.8pp / roof +1.1pp /
|
|
# floor-construction +2.4pp / floor-insulation +1.2pp; gate-safe, no regression).
|
|
_SIMILARITY_SIZE_SCALE_M2: float = 20.0
|
|
_SIMILARITY_AGE_WEIGHT: float = 0.5
|
|
_AGE_BAND_ORDER: str = "ABCDEFGHIJKL"
|
|
|
|
|
|
def _main_part_attr(
|
|
comparable: Comparable, attr: str
|
|
) -> Optional[Union[int, str]]:
|
|
parts: list[SapBuildingPart] = comparable.epc.sap_building_parts
|
|
return getattr(parts[0], attr) if parts else None
|
|
|
|
|
|
def _main_floor_attr(comparable: Comparable, attr: str) -> Optional[int]:
|
|
parts: list[SapBuildingPart] = comparable.epc.sap_building_parts
|
|
if not parts:
|
|
return None
|
|
dims = parts[0].sap_floor_dimensions
|
|
value: Optional[int] = getattr(dims[0], attr) if dims else None
|
|
return value
|
|
|
|
|
|
def _age_band_index(comparable: Comparable) -> Optional[int]:
|
|
"""The main building part's construction-age-band position (A=0 … L=11), or
|
|
None when no recognisable band is lodged."""
|
|
band = _main_part_attr(comparable, "construction_age_band")
|
|
if isinstance(band, str) and band in _AGE_BAND_ORDER:
|
|
return _AGE_BAND_ORDER.index(band)
|
|
return None
|
|
|
|
|
|
def _similarity_weights(members: tuple[Comparable, ...]) -> list[float]:
|
|
"""A physical-similarity weight per comparable (ADR-0029 decision 5): the
|
|
product of an exponential decay in its floor-area distance from the cohort
|
|
median and in its age-band distance from the cohort's modal band. A neighbour
|
|
missing a size or age contributes a neutral weight on that axis, so it is
|
|
never penalised for absent data. Aligned with `members` index-for-index."""
|
|
if not members:
|
|
return []
|
|
median_area: float = statistics.median(
|
|
c.epc.total_floor_area_m2 for c in members
|
|
)
|
|
age_indices: list[Optional[int]] = [_age_band_index(c) for c in members]
|
|
present_ages: list[int] = [i for i in age_indices if i is not None]
|
|
modal_age: Optional[float] = (
|
|
statistics.median(present_ages) if present_ages else None
|
|
)
|
|
weights: list[float] = []
|
|
for comparable, age_index in zip(members, age_indices):
|
|
size_term: float = math.exp(
|
|
-abs(comparable.epc.total_floor_area_m2 - median_area)
|
|
/ _SIMILARITY_SIZE_SCALE_M2
|
|
)
|
|
age_term: float = (
|
|
math.exp(-_SIMILARITY_AGE_WEIGHT * abs(age_index - modal_age))
|
|
if modal_age is not None and age_index is not None
|
|
else 1.0
|
|
)
|
|
weights.append(size_term * age_term)
|
|
return weights
|
|
|
|
|
|
def _weighted_mode(
|
|
values: Iterable[Optional[Union[int, str]]], weights: list[float]
|
|
) -> Optional[Union[int, str]]:
|
|
"""The value with the greatest total similarity weight (ties broken by first
|
|
appearance, matching `_mode`), or None when no non-None value is present."""
|
|
totals: dict[Union[int, str], float] = defaultdict(float)
|
|
for value, weight in zip(values, weights):
|
|
if value is not None:
|
|
totals[value] += weight
|
|
if not totals:
|
|
return None
|
|
return max(totals, key=lambda value: totals[value])
|
|
|
|
|
|
def _weighted_int_mode(
|
|
values: Iterable[Optional[int]], weights: list[float]
|
|
) -> Optional[int]:
|
|
"""`_weighted_mode` narrowed to int-coded fields (keeps pyright strict happy
|
|
when the target attribute is typed `Optional[int]`)."""
|
|
totals: dict[int, float] = defaultdict(float)
|
|
for value, weight in zip(values, weights):
|
|
if value is not None:
|
|
totals[value] += weight
|
|
if not totals:
|
|
return None
|
|
return max(totals, key=lambda value: totals[value])
|
|
|
|
|
|
def _modal_share(
|
|
values: Iterable[Optional[Union[int, str]]],
|
|
) -> Optional[float]:
|
|
"""The most common value's share of the present (non-None) values — a 0..1
|
|
measure of how much the cohort agrees — or None when none are present."""
|
|
present = [v for v in values if v is not None]
|
|
if not present:
|
|
return None
|
|
modal_count: int = Counter(present).most_common(1)[0][1]
|
|
return modal_count / len(present)
|
|
|
|
|
|
def _recency_weighted_mode(
|
|
members: tuple[Comparable, ...], attr: str
|
|
) -> Optional[Union[int, str]]:
|
|
"""The cohort mode of a main-part attribute, weighting each comparable's vote
|
|
by recency — an exponential decay in the cert's age relative to the newest in
|
|
the cohort. Newer neighbours dominate, so a stale majority can't outvote the
|
|
current state. Falls back to a plain mode when no registration dates are
|
|
lodged (all ages 0 ⇒ equal weight)."""
|
|
newest: date = max(
|
|
(c.registration_date or date.min for c in members), default=date.min
|
|
)
|
|
weights: dict[Union[int, str], float] = defaultdict(float)
|
|
for comparable in members:
|
|
value = _main_part_attr(comparable, attr)
|
|
if value is None:
|
|
continue
|
|
lodged: date = comparable.registration_date or date.min
|
|
age_years: float = (newest - lodged).days / _DAYS_PER_YEAR
|
|
weights[value] += math.exp(-age_years / _RECENCY_TAU_YEARS)
|
|
if not weights:
|
|
return None
|
|
return max(weights, key=lambda value: weights[value])
|