Files
wyndham-ARR/company_reports/publishing.py

512 lines
19 KiB
Python

"""Artifact-tool build adapter and recoverable atomic publication."""
from __future__ import annotations
import hashlib
import json
import os
import shutil
import subprocess
import tempfile
import uuid
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Dict, Mapping, Optional, Tuple
from company_reports.contracts import CompanyReport, ErrorCode, RESULT_SCHEMA_VERSION
from company_reports.repository import (
FileMetadata,
ReportRepository,
RepositoryError,
ReservedReport,
)
XLSX_MIME = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
class BuildError(RuntimeError):
def __init__(self, code: str, safe_message: str):
super().__init__(safe_message)
self.code = code
self.safe_message = safe_message
class PublicationError(RuntimeError):
def __init__(self, code: str, safe_message: str):
super().__init__(safe_message)
self.code = code
self.safe_message = safe_message
@dataclass(frozen=True)
class BuiltWorkbook:
path: Path
sha256: str
byte_size: int
summary: Mapping[str, Any]
@dataclass(frozen=True)
class PublicationOutcome:
current_path: Path
archive_path: Path
result_path: Path
artifact: FileMetadata
result_json: FileMetadata
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _private_json(path: Path, payload: Mapping[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
try:
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
json.dump(payload, handle, ensure_ascii=False, sort_keys=True, indent=2)
handle.write("\n")
handle.flush()
os.fsync(handle.fileno())
except Exception:
try:
os.close(descriptor)
except OSError:
pass
raise
class ArtifactToolBuilder:
def __init__(
self,
builder_script: Path,
node_binary: Optional[str] = None,
artifact_tool_module: Optional[Path] = None,
timeout_seconds: int = 120,
) -> None:
configured = (node_binary or os.environ.get("COMPANY_REPORT_NODE_BINARY", "")).strip()
self._node_binary = configured or shutil.which("node") or ""
self._builder_script = builder_script.resolve()
self._artifact_tool_module = (
artifact_tool_module.resolve() if artifact_tool_module else None
)
self._timeout_seconds = timeout_seconds
def build(self, report: CompanyReport, work_dir: Path) -> BuiltWorkbook:
if not self._node_binary or not self._builder_script.is_file():
raise BuildError(
ErrorCode.OUTPUT_VALIDATION_FAILED,
"the XLSX builder runtime is unavailable",
)
work_dir.mkdir(parents=True, exist_ok=True, mode=0o700)
os.chmod(work_dir, 0o700)
payload_path = work_dir / "workbook-payload.json"
output_path = work_dir / report.filename
preview_dir = work_dir / "previews"
summary_path = work_dir / "workbook-summary.json"
_private_json(payload_path, report.to_workbook_payload())
try:
environment = os.environ.copy()
if self._artifact_tool_module is not None:
environment["COMPANY_REPORT_ARTIFACT_TOOL_MODULE"] = str(
self._artifact_tool_module
)
completed = subprocess.run(
[
self._node_binary,
str(self._builder_script),
str(payload_path),
str(output_path),
str(preview_dir),
str(summary_path),
],
cwd=self._builder_script.parent,
capture_output=True,
text=True,
timeout=self._timeout_seconds,
env=environment,
check=False,
)
except (OSError, subprocess.TimeoutExpired):
raise BuildError(
ErrorCode.OUTPUT_VALIDATION_FAILED,
"the XLSX builder did not complete",
) from None
if completed.returncode != 0:
raise BuildError(
ErrorCode.OUTPUT_VALIDATION_FAILED,
"the XLSX builder rejected the generated payload",
)
try:
summary = json.loads(summary_path.read_text(encoding="utf-8"))
expected_rows = [len(period.rows) for period in report.periods]
expected_names = [period.sheet_name for period in report.periods]
if (
summary.get("status") != "success"
or summary.get("company") != report.company
or summary.get("filename") != report.filename
or summary.get("sheet_names") != expected_names
or summary.get("row_counts") != expected_rows
or summary.get("formula_count") != 0
or not isinstance(summary.get("semantic_sha256"), str)
or len(summary.get("semantic_sha256")) != 64
or summary.get("preview_count") != 3
or not output_path.is_file()
or output_path.stat().st_size <= 0
):
raise ValueError("summary mismatch")
except (OSError, ValueError, TypeError, json.JSONDecodeError):
raise BuildError(
ErrorCode.OUTPUT_VALIDATION_FAILED,
"the XLSX builder result could not be validated",
) from None
os.chmod(output_path, 0o600)
return BuiltWorkbook(
path=output_path,
sha256=sha256_file(output_path),
byte_size=output_path.stat().st_size,
summary=summary,
)
class AtomicReportPublisher:
def __init__(self, project_root: Path, output_root: Path) -> None:
self._project_root = project_root.resolve()
self._output_root = output_root.resolve()
try:
self._output_root.relative_to(self._project_root)
except ValueError:
raise ValueError("output_root must be inside project_root") from None
@staticmethod
def _fsync_parent(path: Path) -> None:
descriptor = os.open(path.parent, os.O_RDONLY)
try:
os.fsync(descriptor)
finally:
os.close(descriptor)
@classmethod
def _atomic_copy(cls, source: Path, target: Path) -> None:
target.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
temporary = target.parent / f".{target.name}.tmp-{uuid.uuid4().hex}"
try:
shutil.copyfile(source, temporary)
os.chmod(temporary, 0o600)
with temporary.open("rb") as handle:
os.fsync(handle.fileno())
os.replace(temporary, target)
cls._fsync_parent(target)
finally:
temporary.unlink(missing_ok=True)
@classmethod
def _atomic_json(cls, target: Path, payload: Mapping[str, Any]) -> None:
target.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
temporary = target.parent / f".{target.name}.tmp-{uuid.uuid4().hex}"
try:
_private_json(temporary, payload)
os.replace(temporary, target)
cls._fsync_parent(target)
finally:
temporary.unlink(missing_ok=True)
@classmethod
def _install_once(cls, source: Path, target: Path, expected_sha256: str) -> bool:
if target.exists():
if sha256_file(target) != expected_sha256:
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"an immutable archive path already contains different content",
)
return False
cls._atomic_copy(source, target)
return True
def _storage_key(self, path: Path) -> str:
try:
return path.resolve().relative_to(self._project_root).as_posix()
except ValueError:
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"an artifact path escaped the controlled project root",
) from None
@staticmethod
def _is_regular_file(path: Path) -> bool:
return path.is_file() and not path.is_symlink()
@staticmethod
def _is_sha256(value: Any) -> bool:
return (
isinstance(value, str)
and len(value) == 64
and all(character in "0123456789abcdef" for character in value)
)
def _existing_publication_metadata(
self,
report: CompanyReport,
reservation: ReservedReport,
built: BuiltWorkbook,
archive_path: Path,
result_path: Path,
) -> Optional[Tuple[FileMetadata, FileMetadata]]:
archive_present = archive_path.exists() or archive_path.is_symlink()
result_present = result_path.exists() or result_path.is_symlink()
if not archive_present and not result_present:
return None
if not self._is_regular_file(archive_path) or not self._is_regular_file(
result_path
):
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"an existing company report publication is incomplete",
)
semantic_sha256 = built.summary.get("semantic_sha256")
if not self._is_sha256(semantic_sha256):
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"the generated company report semantic identity is invalid",
)
try:
result_payload = json.loads(result_path.read_text(encoding="utf-8"))
if not isinstance(result_payload, dict):
raise ValueError("result JSON must be an object")
artifact_payload = result_payload.get("artifact")
if not isinstance(artifact_payload, dict):
raise ValueError("result JSON artifact identity is missing")
existing_sha256 = artifact_payload.get("sha256")
if not self._is_sha256(existing_sha256):
raise ValueError("result JSON artifact hash is invalid")
archive_sha256 = sha256_file(archive_path)
if archive_sha256 != existing_sha256:
raise ValueError("archive and result JSON hashes differ")
artifact_storage_key = self._storage_key(archive_path)
expected_payload = self._success_result(
report,
reservation,
artifact_storage_key,
existing_sha256,
semantic_sha256,
)
if result_payload != expected_payload:
raise ValueError("result JSON identity differs")
artifact = FileMetadata(
file_kind="company_ten_day_xlsx",
original_filename=report.filename,
storage_key=artifact_storage_key,
sha256=existing_sha256,
byte_size=archive_path.stat().st_size,
mime_type=XLSX_MIME,
)
result_json = FileMetadata(
file_kind="result_json",
original_filename=result_path.name,
storage_key=self._storage_key(result_path),
sha256=sha256_file(result_path),
byte_size=result_path.stat().st_size,
mime_type="application/json",
)
artifact.validate()
result_json.validate()
return artifact, result_json
except (OSError, UnicodeError, TypeError, ValueError, RepositoryError):
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"an existing company report publication identity is invalid",
) from None
def _archive_untracked_current(self, current_path: Path, archive_root: Path) -> None:
if not current_path.is_file():
return
current_sha = sha256_file(current_path)
legacy_path = archive_root / "legacy" / current_sha[:16] / current_path.name
self._install_once(current_path, legacy_path, current_sha)
@staticmethod
def _success_result(
report: CompanyReport,
reservation: ReservedReport,
artifact_storage_key: str,
artifact_sha256: str,
semantic_sha256: str,
) -> Dict[str, Any]:
return {
"schema_version": RESULT_SCHEMA_VERSION,
"status": "success",
"company": report.company,
"report_year": report.report_year,
"report_month": report.report_month,
"as_of_date": report.as_of_date.isoformat(),
"report_version_id": reservation.report_version_id,
"version_no": reservation.version_no,
"artifact": {
"filename": report.filename,
"storage_key": artifact_storage_key,
"sha256": artifact_sha256,
"semantic_sha256": semantic_sha256,
},
"row_count": report.row_count,
"period_row_counts": {
period.key: len(period.rows) for period in report.periods
},
"daily_version_count": len(report.daily_versions),
"booking_version_count": len(report.booking_versions),
"warnings": [
{
"code": warning.code,
"stage": warning.stage,
"period": warning.period,
"record_ids": list(warning.record_ids),
}
for warning in report.warnings
],
"errors": [],
}
def publish(
self,
report: CompanyReport,
reservation: ReservedReport,
built: BuiltWorkbook,
repository: ReportRepository,
work_dir: Path,
) -> PublicationOutcome:
month_root = (
self._output_root
/ f"{report.report_year:04d}"
/ f"{report.report_month:02d}"
)
archive_root = month_root / "archive"
version_root = archive_root / f"v{reservation.version_no:04d}"
current_path = month_root / report.filename
archive_path = version_root / report.filename
result_path = version_root / f"{report.company}-v{reservation.version_no:04d}.result.json"
backup_path = work_dir / "previous-current.xlsx"
current_existed = current_path.is_file()
archive_created = False
result_created = False
current_replaced = False
try:
month_root.mkdir(parents=True, exist_ok=True, mode=0o700)
existing = self._existing_publication_metadata(
report,
reservation,
built,
archive_path,
result_path,
)
if existing is not None:
artifact, result_json = existing
if not current_existed:
self._atomic_copy(archive_path, current_path)
current_replaced = True
repository.activate_report(reservation, artifact, result_json)
return PublicationOutcome(
current_path=current_path,
archive_path=archive_path,
result_path=result_path,
artifact=artifact,
result_json=result_json,
)
self._archive_untracked_current(current_path, archive_root)
if current_existed:
shutil.copyfile(current_path, backup_path)
os.chmod(backup_path, 0o600)
archive_created = self._install_once(
built.path, archive_path, built.sha256
)
artifact_storage_key = self._storage_key(archive_path)
result_payload = self._success_result(
report,
reservation,
artifact_storage_key,
built.sha256,
str(built.summary.get("semantic_sha256", "")),
)
if result_path.exists():
expected = json.dumps(
result_payload,
ensure_ascii=False,
sort_keys=True,
indent=2,
) + "\n"
if result_path.read_text(encoding="utf-8") != expected:
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"an immutable result path already contains different content",
)
else:
self._atomic_json(result_path, result_payload)
result_created = True
self._atomic_copy(built.path, current_path)
current_replaced = True
artifact = FileMetadata(
file_kind="company_ten_day_xlsx",
original_filename=report.filename,
storage_key=artifact_storage_key,
sha256=built.sha256,
byte_size=built.byte_size,
mime_type=XLSX_MIME,
)
result_json = FileMetadata(
file_kind="result_json",
original_filename=result_path.name,
storage_key=self._storage_key(result_path),
sha256=sha256_file(result_path),
byte_size=result_path.stat().st_size,
mime_type="application/json",
)
repository.activate_report(reservation, artifact, result_json)
return PublicationOutcome(
current_path=current_path,
archive_path=archive_path,
result_path=result_path,
artifact=artifact,
result_json=result_json,
)
except Exception as error:
if current_replaced:
if current_existed and backup_path.is_file():
self._atomic_copy(backup_path, current_path)
else:
current_path.unlink(missing_ok=True)
if result_created:
result_path.unlink(missing_ok=True)
if archive_created:
archive_path.unlink(missing_ok=True)
code = getattr(error, "code", ErrorCode.PUBLISH_FAILED)
safe_message = getattr(error, "safe_message", "company report publication failed")
try:
repository.mark_failed(reservation, code, safe_message)
except Exception:
pass
if isinstance(error, PublicationError):
raise
if isinstance(error, RepositoryError):
raise PublicationError(error.code, error.safe_message) from None
raise PublicationError(
ErrorCode.PUBLISH_FAILED,
"company report publication failed",
) from None
def private_staging_directory(parent: Path):
parent.mkdir(parents=True, exist_ok=True, mode=0o700)
return tempfile.TemporaryDirectory(prefix="company-report-", dir=parent)