mirror of
https://github.com/Hestia-Homes/Model.git
synced 2026-07-27 23:35:01 +00:00
Merge pull request #1545 from Hestia-Homes/backfill-epc-property-uprn
Script to backfill UPRN in epc_property for rows populated by sitenotes before the mapper included the UPRN
This commit is contained in:
commit
ff1302b514
1 changed files with 199 additions and 0 deletions
199
scripts/backfill_site_notes_epc_property_uprn.py
Normal file
199
scripts/backfill_site_notes_epc_property_uprn.py
Normal file
|
|
@ -0,0 +1,199 @@
|
|||
"""Backfill ``epc_property.uprn`` on PasHub site-notes rows written before the
|
||||
UPRN-persistence fix (#1538). Targets have ``uprn IS NULL`` and a non-null
|
||||
``uploaded_file_id`` (the site-notes signature; these rows carry no ``property_id``
|
||||
/ ``portfolio_id``).
|
||||
|
||||
The run is scoped to the rows whose deal (``uploaded_files.hubspot_deal_id`` →
|
||||
``hubspot_deal_data.deal_id``) carries ``PROJECT_CODE``. For each such target the
|
||||
UPRN is taken, in priority order, from (1) ``uploaded_files.uprn``, then when that's
|
||||
null (2) ``hubspot_deal_data.uprn`` (text, so numeric-guarded and cast). A row is
|
||||
skipped and reported when neither yields a UPRN, or when the deal offers several
|
||||
different UPRNs (ambiguous). ``uprn_source`` is left untouched, as in #1538.
|
||||
|
||||
Set ``PROJECT_CODE`` and ``DRY_RUN`` below, then run the file (VS Code ▷ Run, or
|
||||
``python -m scripts.backfill_site_notes_epc_property_uprn``). With ``DRY_RUN = True``
|
||||
it prints what it would change and writes nothing; set it to ``False`` to execute in
|
||||
a transaction. Idempotent — re-running is a no-op.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from sqlalchemy import Connection, text
|
||||
|
||||
from scripts.e2e_common import build_engine, load_env
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Set these before running.
|
||||
# ---------------------------------------------------------------------------
|
||||
PROJECT_CODE: str = (
|
||||
"[The Guinness Partnership GMCA Bid - WH:SF - 052026 - 1219680020683]" # TODO: hubspot_deal_data.project_code to scope to
|
||||
)
|
||||
DRY_RUN: bool = True # True = report only (writes nothing); False = apply the update
|
||||
|
||||
|
||||
# Target rows: site-notes-mapped epc_property rows still missing a UPRN, scoped to
|
||||
# the deals carrying PROJECT_CODE. The non-null uploaded_file_id is the site-notes
|
||||
# signature (predicted / lodged-EPC rows don't have one; the JOIN to uploaded_files
|
||||
# enforces it), and uprn IS NULL is the pre-#1538 gap. The EXISTS restricts to rows
|
||||
# whose deal (uploaded_files.hubspot_deal_id -> hubspot_deal_data.deal_id) has the
|
||||
# requested project_code.
|
||||
#
|
||||
# resolved: per target, the UPRN candidates — uploaded_files.uprn (single, via the
|
||||
# PK join) and the deal's uprn. hubspot_deal_data.uprn is text, so it's
|
||||
# numeric-guarded (^[0-9]+$) before casting; a deal_id can have several rows, so we
|
||||
# take the min and also count the distinct values to spot ambiguity.
|
||||
#
|
||||
# final: the value written is uploaded_files.uprn if present, else the deal's uprn.
|
||||
# via_uploaded_file / via_hubspot_deal record which source won. A row is ambiguous
|
||||
# (skipped) when it falls back to the deal and the deal offers >1 distinct UPRN.
|
||||
_RESOLVED_CTE = """
|
||||
WITH target AS (
|
||||
SELECT e.id AS epc_property_id, e.uploaded_file_id
|
||||
FROM epc_property e
|
||||
JOIN uploaded_files uf ON uf.id = e.uploaded_file_id
|
||||
WHERE e.uprn IS NULL
|
||||
AND EXISTS (
|
||||
SELECT 1 FROM hubspot_deal_data d
|
||||
WHERE d.deal_id = uf.hubspot_deal_id
|
||||
AND d.project_code = :project_code
|
||||
)
|
||||
),
|
||||
resolved AS (
|
||||
SELECT t.epc_property_id,
|
||||
uf.uprn AS uploaded_uprn,
|
||||
min(CASE WHEN btrim(hdd.uprn) ~ '^[0-9]+$'
|
||||
THEN btrim(hdd.uprn)::bigint END) AS deal_uprn,
|
||||
count(DISTINCT CASE WHEN btrim(hdd.uprn) ~ '^[0-9]+$'
|
||||
THEN btrim(hdd.uprn)::bigint END) AS distinct_deal_uprns
|
||||
FROM target t
|
||||
JOIN uploaded_files uf ON uf.id = t.uploaded_file_id
|
||||
LEFT JOIN hubspot_deal_data hdd ON hdd.deal_id = uf.hubspot_deal_id
|
||||
GROUP BY t.epc_property_id, uf.uprn
|
||||
),
|
||||
final AS (
|
||||
SELECT epc_property_id,
|
||||
COALESCE(uploaded_uprn, deal_uprn) AS resolved_uprn,
|
||||
uploaded_uprn IS NOT NULL AS via_uploaded_file,
|
||||
(uploaded_uprn IS NULL AND deal_uprn IS NOT NULL) AS via_hubspot_deal,
|
||||
(uploaded_uprn IS NULL AND distinct_deal_uprns > 1) AS ambiguous
|
||||
FROM resolved
|
||||
)
|
||||
"""
|
||||
|
||||
_REPORT = text(_RESOLVED_CTE + """
|
||||
SELECT
|
||||
(SELECT count(*) FROM target) AS targets,
|
||||
count(*) FILTER (WHERE resolved_uprn IS NOT NULL
|
||||
AND NOT ambiguous) AS resolvable,
|
||||
count(*) FILTER (WHERE resolved_uprn IS NOT NULL
|
||||
AND NOT ambiguous
|
||||
AND via_uploaded_file) AS via_uploaded_file,
|
||||
count(*) FILTER (WHERE resolved_uprn IS NOT NULL
|
||||
AND NOT ambiguous
|
||||
AND via_hubspot_deal) AS via_hubspot_deal,
|
||||
count(*) FILTER (WHERE ambiguous) AS ambiguous
|
||||
FROM final
|
||||
""")
|
||||
|
||||
_UPDATE = text(_RESOLVED_CTE + """
|
||||
UPDATE epc_property e
|
||||
SET uprn = f.resolved_uprn
|
||||
FROM final f
|
||||
WHERE e.id = f.epc_property_id
|
||||
AND f.resolved_uprn IS NOT NULL
|
||||
AND NOT f.ambiguous
|
||||
AND e.uprn IS NULL
|
||||
""")
|
||||
|
||||
# Addresses of the rows we can't backfill (no UPRN from either source), so they can
|
||||
# be chased up by hand before / after a live run.
|
||||
_UNRESOLVED = text(_RESOLVED_CTE + """
|
||||
SELECT e.id AS epc_property_id,
|
||||
e.address_line_1, e.address_line_2, e.post_town, e.postcode
|
||||
FROM final f
|
||||
JOIN epc_property e ON e.id = f.epc_property_id
|
||||
WHERE f.resolved_uprn IS NULL
|
||||
ORDER BY e.postcode, e.address_line_1
|
||||
""")
|
||||
|
||||
|
||||
def _unresolved_addresses(conn: Connection, *, project_code: str) -> list[str]:
|
||||
"""One formatted ``"epc_property <id>: <address>"`` line per target row that
|
||||
yielded no UPRN from either source."""
|
||||
lines: list[str] = []
|
||||
for r in conn.execute(_UNRESOLVED, {"project_code": project_code}):
|
||||
parts = [r.address_line_1, r.address_line_2, r.post_town, r.postcode]
|
||||
address = ", ".join(p for p in parts if p) or "(no address on epc_property)"
|
||||
lines.append(f"epc_property {r.epc_property_id}: {address}")
|
||||
return lines
|
||||
|
||||
|
||||
def backfill(conn: Connection, *, project_code: str, apply: bool) -> dict[str, int]:
|
||||
"""Report (and, when ``apply``, execute) the UPRN backfill for the deals
|
||||
carrying ``project_code``.
|
||||
|
||||
Returns the diagnostic counts: total ``targets`` (site-notes rows missing a
|
||||
UPRN), how many are ``resolvable`` (split by the source that resolved them —
|
||||
``via_uploaded_file`` then ``via_hubspot_deal``, in that priority order), how
|
||||
many are ``ambiguous`` (skipped), and — when applied — how many rows were
|
||||
``updated``."""
|
||||
params = {"project_code": project_code}
|
||||
row = conn.execute(_REPORT, params).one()
|
||||
counts = {
|
||||
"targets": row.targets,
|
||||
"resolvable": row.resolvable,
|
||||
"via_uploaded_file": row.via_uploaded_file,
|
||||
"via_hubspot_deal": row.via_hubspot_deal,
|
||||
"ambiguous": row.ambiguous,
|
||||
}
|
||||
if apply:
|
||||
result = conn.execute(_UPDATE, params)
|
||||
counts["updated"] = result.rowcount
|
||||
return counts
|
||||
|
||||
|
||||
def main() -> None:
|
||||
load_env()
|
||||
if not PROJECT_CODE:
|
||||
raise SystemExit(
|
||||
"Set PROJECT_CODE at the top of this script to the hubspot_deal_data "
|
||||
"project_code to scope to before running."
|
||||
)
|
||||
|
||||
engine = build_engine()
|
||||
with engine.begin() as conn:
|
||||
conn.execute(text("SET statement_timeout = 120000"))
|
||||
counts = backfill(conn, project_code=PROJECT_CODE, apply=not DRY_RUN)
|
||||
unresolved_addresses = _unresolved_addresses(conn, project_code=PROJECT_CODE)
|
||||
|
||||
unresolved = counts["targets"] - counts["resolvable"] - counts["ambiguous"]
|
||||
verb = "would backfill" if DRY_RUN else "backfilled"
|
||||
print(
|
||||
f"Project {PROJECT_CODE!r}: {counts['targets']} site-notes epc_property "
|
||||
f"row(s) missing a UPRN."
|
||||
)
|
||||
print(
|
||||
f" {verb} {counts['resolvable']} row(s) "
|
||||
f"({counts['via_uploaded_file']} via uploaded_files.uprn, "
|
||||
f"{counts['via_hubspot_deal']} via hubspot_deal_data.uprn)."
|
||||
)
|
||||
if counts["ambiguous"]:
|
||||
print(
|
||||
f" SKIPPED {counts['ambiguous']} row(s) whose deal offers more than one "
|
||||
"distinct UPRN (ambiguous — resolve by hand)."
|
||||
)
|
||||
if unresolved:
|
||||
print(
|
||||
f" SKIPPED {unresolved} row(s) with no UPRN available "
|
||||
"(no uploaded_files.uprn and no numeric deal uprn):"
|
||||
)
|
||||
for line in unresolved_addresses:
|
||||
print(f" {line}")
|
||||
if DRY_RUN:
|
||||
print("\nDRY-RUN — nothing written. Set DRY_RUN = False to execute.")
|
||||
else:
|
||||
print(f"\nApplied. {counts.get('updated', 0)} row(s) updated.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Reference in a new issue