"""The Bulk Document Download orchestrator (ADR-0060). Ties the pieces together for one run: gather a property set's uploaded files (by `landlord_property_id`), enrich each with its display address, lay them out with the Download Package plan, stream the chosen files into a single ZIP, upload it, mint a presigned URL, and email the requester. Best-effort — skipped rows are reported, not fatal; a wholly-empty selection is a recorded failure (ADR-0060). """ from __future__ import annotations import logging import os import tempfile import zipfile from dataclasses import dataclass from typing import Protocol, Sequence from domain.bulk_document_download.package_plan import ( ResolvedDocument, SkippedDocument, plan_download_package, ) from domain.bulk_document_download.property_document import PropertyDocument from domain.tasks.subtasks import SubTaskFailure from infrastructure.s3.s3_client import S3Client from repositories.email.email_sender import EmailSender logger = logging.getLogger(__name__) class UploadedFileReader(Protocol): def by_landlord_property_ids( self, landlord_property_ids: Sequence[str] ) -> list[PropertyDocument]: ... class AddressResolver(Protocol): def address_for(self, landlord_property_id: str) -> str: ... class DocumentDownloader(Protocol): def download(self, bucket: str, key: str, dest_path: str) -> None: ... @dataclass(frozen=True) class DownloadPackageResult: """What a completed run reports (lands in `sub_task.outputs`, ADR-0060).""" presigned_url: str package_s3_key: str included: int skipped: tuple[SkippedDocument, ...] class BulkDocumentDownloadOrchestrator: def __init__( self, *, uploaded_files: UploadedFileReader, addresses: AddressResolver, documents: DocumentDownloader, packages: S3Client, email: EmailSender, url_ttl_seconds: int, package_key_prefix: str = "bulk-downloads", ) -> None: self._uploaded_files = uploaded_files self._addresses = addresses self._documents = documents self._packages = packages self._email = email self._url_ttl_seconds = url_ttl_seconds self._package_key_prefix = package_key_prefix def run( self, *, landlord_property_ids: Sequence[str], recipient_email: str, package_name: str, ) -> DownloadPackageResult: documents = self._uploaded_files.by_landlord_property_ids( landlord_property_ids ) plan = plan_download_package([self._resolve(doc) for doc in documents]) skipped: list[SkippedDocument] = list(plan.skipped) logger.info( "bulk_document_download: selected=%d matched_documents=%d " "planned_entries=%d skipped=%d", len(landlord_property_ids), len(documents), len(plan.entries), len(skipped), ) if not plan.entries: raise self._empty_failure( selected=len(landlord_property_ids), matched=len(documents), skipped=skipped, ) # Stream to a /tmp file and multipart-upload it (ADR-0060): a Download # Package can be several GB, so it is never held whole in memory. Each # source file is streamed to its own temp file, added to the on-disk ZIP, # then deleted; the finished archive is streamed up from disk. Best-effort # (ADR-0060): a file that can't be read — deleted, or in a bucket the role # can't reach — becomes a SkippedDocument, it does not fail the package. key = f"{self._package_key_prefix}/{package_name}.zip" included = 0 with tempfile.TemporaryDirectory() as tmp_dir: zip_path = os.path.join(tmp_dir, "package.zip") with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as archive: for index, entry in enumerate(plan.entries): member_path = os.path.join(tmp_dir, f"member-{index}") try: self._documents.download( entry.s3_bucket, entry.s3_key, member_path ) except Exception: skipped.append( SkippedDocument( landlord_property_id=entry.landlord_property_id, s3_key=entry.s3_key, reason="unreadable", ) ) continue archive.write(member_path, arcname=entry.zip_path) os.remove(member_path) included += 1 if included == 0: # Every file that resolved failed to read — no package to send. raise self._empty_failure( selected=len(landlord_property_ids), matched=len(documents), skipped=skipped, ) self._packages.upload_file(zip_path, key) url = self._packages.generate_presigned_url(key, self._url_ttl_seconds) self._email.send( to=recipient_email, subject="Your document download is ready", body=( f"Your document download is ready. It contains " f"{included} document(s).\n\n" f"Download it here (link valid for " f"{self._url_ttl_seconds // 60} minutes):\n{url}" ), ) return DownloadPackageResult( presigned_url=url, package_s3_key=key, included=included, skipped=tuple(skipped), ) def _empty_failure( self, *, selected: int, matched: int, skipped: list[SkippedDocument] ) -> SubTaskFailure: """A recorded failure when the selection yields no readable documents (ADR-0060) — never an empty ZIP emailed to the user. The message carries the stage counts (so they surface in the worker's WARNING log) and the skip detail rides on the failure's ``details`` (→ ``sub_task.outputs``).""" return SubTaskFailure( f"no documents could be packaged for the selection " f"(selected {selected} properties, matched {matched} documents, " f"{len(skipped)} skipped)", details={ "selected_properties": selected, "matched_documents": matched, "planned_entries": 0, "skipped": [ { "landlord_property_id": s.landlord_property_id, "s3_key": s.s3_key, "reason": s.reason, } for s in skipped ], }, ) def _resolve(self, document: PropertyDocument) -> ResolvedDocument: return ResolvedDocument( landlord_property_id=document.landlord_property_id, address=self._addresses.address_for(document.landlord_property_id), document_type=document.document_type, s3_bucket=document.s3_bucket, s3_key=document.s3_key, uploaded_at=document.uploaded_at, )