Files
ARR-2.0-0918/integrations/ohip/prepare_arr_source.py

221 lines
11 KiB
Python

"""Prepare all ARR field candidates and gaps from a complete pinned capture.
Offline only. This composes supported selectors without deciding report row
inclusion/order or unresolved display semantics. It does not emit SourceRow,
XML, a processor outcome or any business acceptance flag. All rows survive.
"""
from __future__ import annotations
import argparse
from collections import Counter
from decimal import Decimal
import hashlib
import json
from pathlib import Path
from . import audit_arr_day as audit
from . import audit_arr_named_day as named_audit
from . import collect_arr_source as capture
from . import source_facts as facts
from . import source_fields as fields
from .source_facts_contract import FIELDS, FLAGS, MAX_FACTS_BYTES, canonical
from .validate_source_facts import verify_source_facts
# v4 adds agreed room-type code selection to v3's typed BlockCode support.
# Previously written preparation/evidence is immutable and is not rewritten.
VERSION = "arr-source-preparation/v4"
UNRESOLVED = {
"COMPANY_NAME": "company_display_mapping_unresolved",
"PRODUCTS": "package_display_mapping_unresolved",
}
REPORT_GAPS = (
"report_inclusion", "report_order_and_row_granularity", "same_source_acceptance",
"name_display", "historical_room", "company_role_and_display", "note_scope_order",
"shared_room_and_guest_count_semantics", "optional_report_displays", "rate_interval_semantics",
)
def _candidate(value):
# Only already-selected values reach this encoder. Never normalize strings,
# coerce booleans, round amounts or manufacture an empty string for a gap.
if type(value) is tuple:
capture.require(all(type(item) is str for item in value), "preparation_invalid_note_value")
return {"state": "candidate", "kind": "text_list", "value": list(value)}
capture.require(type(value) is str, "preparation_invalid_text_value")
return {"state": "candidate", "kind": "text", "value": value}
def _gap(reason):
return {"state": "gap", "reason": reason}
def _attempt(select):
try:
return _candidate(select())
except capture.CollectionError as error:
# Selectors use fixed codes; arbitrary exceptions are fatal, never
# serialized as guest data or treated as field-level missing values.
return _gap(str(error))
def _at(document, pointer):
node = document
for key in pointer.split("/")[1:]:
key = key.replace("~1", "/").replace("~0", "~")
node = node[int(key)] if type(node) is list else node[key]
return node
def _raw(archive, reference):
raw = archive.read(reference["file"])
capture.strict_json(raw)
return _at(json.loads(raw, parse_float=Decimal), reference["pointer"])
def _name(assessment, named):
if not named:
return _gap("profile_source_not_acquired")
profile = assessment["profile"]
if "error" in profile:
return _gap(profile["error"])
return _candidate(profile["full_name"])
def _select(archive, record, named):
refs = record["sources"]
search, detail = _raw(archive, refs["search"]), _raw(archive, refs["detail"])
assessment = _raw(archive, refs["assessment"])
day = archive.options.from_date
# Both fields in a pair depend on the same strict selector, so an invalid
# pair cannot silently contribute one apparently valid date/count.
result = {
"BLOCK_CODE": _attempt(lambda: fields.agreed_block_code(search, detail, day)),
"ROOM_CATEGORY_LABEL": _attempt(lambda: fields.agreed_room_type(search, detail, day)),
"CONFIRMATION_NO": _attempt(lambda: fields.confirmation_no(detail)),
"ARRIVAL": _attempt(lambda: fields.stay_dates(detail, day).arrival.isoformat()),
"DEPARTURE": _attempt(lambda: fields.stay_dates(detail, day).departure.isoformat()),
"RATE_CODE": _attempt(lambda: fields.arrival_rate_code(detail, day)),
"DISP_ROOM_NO": _attempt(lambda: fields.agreed_room_number(search, detail, day)),
"ADULTS": _attempt(lambda: str(fields.agreed_guest_counts(detail, day).adults)),
"CHILDREN": _attempt(lambda: str(fields.agreed_guest_counts(detail, day).children)),
"NO_OF_ROOMS": _attempt(lambda: str(fields.arrival_number_of_units(detail, day))),
"RES_COMMENT": _attempt(lambda: fields.reservation_gen_notes(detail)),
"TRACE_TEXT": _candidate(fields.traces()),
"FULL_NAME": _name(assessment, named),
}
rate_ref = refs["rate"]
request = archive.document(rate_ref["request"]["file"])
response = json.loads(archive.read(rate_ref["file"]), parse_float=Decimal)
try:
rate = fields.effective_rate(detail, request, response, archive.options)
except capture.CollectionError as error:
result["EFFECTIVE_RATE_AMOUNT"] = _gap(str(error))
currency = None
else:
# Keep Decimal's compact exact representation: expanding 1E-1000000
# into fixed-point text would allocate a huge string before the budget.
result["EFFECTIVE_RATE_AMOUNT"] = _candidate(str(rate.amount))
currency = rate.currency
result.update({name: _gap(reason) for name, reason in UNRESOLVED.items()})
capture.require(set(result) == set(FIELDS), "preparation_field_set_mismatch")
return {"capture_sequence": record["capture_sequence"], "reservation_id": record["reservation_id"],
"fields": {name: result[name] for name in FIELDS}, "effective_rate_currency": currency}
def prepare(archive, evidence: bytes, *, name_supplements=()) -> bytes:
"""Reverify evidence before using any source references or selected values.
Evidence independently authenticates paths/identity/request provenance;
candidate selection composes the existing selectors. It is NOT a second
independent implementation of ARR business mapping.
"""
named = type(archive) is named_audit.VerifiedArchive
capture.require(named or type(archive) is audit.VerifiedArchive, "preparation_capture_type_invalid")
protocol = named_audit if named else audit
archive = protocol.VerifiedArchive(archive.directory, archive.pin)
verified = verify_source_facts(archive, evidence)
records = capture.strict_json(evidence)["records"]
result = {
"version": VERSION, "selector_version": fields.VERSION, "status": "unaccepted_source_candidates",
"capture_manifest_sha256": archive.pin, "field_evidence_sha256": verified["evidence_sha256"],
"field_contract_sha256": verified["contract_sha256"], "hotel_id": archive.options.hotel_id,
"from_date": archive.options.from_date, "to_date": archive.options.to_date,
"rate_date": archive.options.rate_date, "record_order": "capture_order_not_report_order",
"report_gaps": list(REPORT_GAPS), "complete_arr_output": False, **FLAGS,
"records": [_select(archive, record, named) for record in records],
}
if name_supplements:
from . import profile_supplement
capture.require(not named, "supplement_requires_v2_base")
names, _ = profile_supplement.verify_chain(archive, name_supplements)
capture.require(len(names) == len(result["records"]), "supplement_record_count_mismatch")
for record, name in zip(result["records"], names):
capture.require(record["reservation_id"] == name["reservation_id"], "supplement_record_identity_mismatch")
record["fields"]["FULL_NAME"] = _name(name, True)
result.update(name_supplement_pins=[pin for _, pin in name_supplements])
raw = canonical(result)
capture.require(len(raw) <= MAX_FACTS_BYTES, "preparation_byte_budget_exceeded")
return raw
def summary(raw: bytes):
document = capture.strict_json(raw)
counts = {name: {"candidates": 0, "gaps": 0, "gap_reasons": Counter()} for name in FIELDS}
gap_rows = 0
for record in document["records"]:
gap_rows += any(value["state"] == "gap" for value in record["fields"].values())
for name, value in record["fields"].items():
if value["state"] == "candidate":
counts[name]["candidates"] += 1
else:
counts[name]["gaps"] += 1
counts[name]["gap_reasons"][value["reason"]] += 1
return {"status": document["status"], "version": document["version"], "records": len(document["records"]),
"records_with_gaps": gap_rows, "fields": counts, "report_gaps": document["report_gaps"],
"preparation_sha256": hashlib.sha256(raw).hexdigest(),
"capture_manifest_sha256": document["capture_manifest_sha256"],
"field_evidence_sha256": document["field_evidence_sha256"],
"complete_arr_output": False, "network_calls": 0, **FLAGS}
def export(capture_dir: Path, capture_pin: str, output_dir: Path, *, capture_version="v2", name_supplements=()):
repository = Path(__file__).resolve().parents[2]
capture.require(not output_dir.resolve().is_relative_to(repository), "output_must_be_outside_repository")
capture.require(capture_version in ("v2", "v3"), "unsupported_preparation_capture_version")
protocol = named_audit if capture_version == "v3" else audit
archive = protocol.VerifiedArchive(capture_dir, capture_pin)
evidence = facts.build_source_facts(archive)
raw = prepare(archive, evidence, name_supplements=name_supplements)
result = summary(raw)
# No output directory exists until complete input integrity checks pass.
sink = capture.Archive(output_dir)
sink.write("source-facts.json", evidence)
sink.write("source-candidates.json", raw)
sink.write("preparation-summary.json", result)
return result
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--capture-dir", required=True, type=Path)
parser.add_argument("--capture-sha256", required=True)
parser.add_argument("--output-dir", required=True, type=Path)
parser.add_argument("--capture-version", choices=("v2", "v3"), default="v2")
parser.add_argument("--profile-supplement", nargs=2, action="append", default=[], metavar=("DIRECTORY", "SHA256"))
args = parser.parse_args(argv)
try:
result = export(args.capture_dir, args.capture_sha256, args.output_dir,
capture_version=args.capture_version, name_supplements=args.profile_supplement)
except capture.CollectionError as error:
result = {"status": "failed", "error": str(error), "complete_arr_output": False, **FLAGS}
except Exception:
result = {"status": "failed", "error": "source_preparation_failed", "complete_arr_output": False, **FLAGS}
print(json.dumps(result, ensure_ascii=False, sort_keys=True))
# Success means diagnostics are complete, even when every row has a gap.
return 0 if result["status"] == "unaccepted_source_candidates" else 1
if __name__ == "__main__":
raise SystemExit(main())