"""Prepare all ARR field candidates and gaps from a complete pinned capture. Offline only. This composes supported selectors without deciding report row inclusion/order or unresolved display semantics. It does not emit SourceRow, XML, a processor outcome or any business acceptance flag. All rows survive. """ from __future__ import annotations import argparse from collections import Counter from decimal import Decimal import hashlib import json from pathlib import Path from . import audit_arr_day as audit from . import audit_arr_named_day as named_audit from . import collect_arr_source as capture from . import source_facts as facts from . import source_fields as fields from .source_facts_contract import FIELDS, FLAGS, MAX_FACTS_BYTES, canonical from .validate_source_facts import verify_source_facts # v4 adds agreed room-type code selection to v3's typed BlockCode support. # Previously written preparation/evidence is immutable and is not rewritten. VERSION = "arr-source-preparation/v4" UNRESOLVED = { "COMPANY_NAME": "company_display_mapping_unresolved", "PRODUCTS": "package_display_mapping_unresolved", } REPORT_GAPS = ( "report_inclusion", "report_order_and_row_granularity", "same_source_acceptance", "name_display", "historical_room", "company_role_and_display", "note_scope_order", "shared_room_and_guest_count_semantics", "optional_report_displays", "rate_interval_semantics", ) def _candidate(value): # Only already-selected values reach this encoder. Never normalize strings, # coerce booleans, round amounts or manufacture an empty string for a gap. if type(value) is tuple: capture.require(all(type(item) is str for item in value), "preparation_invalid_note_value") return {"state": "candidate", "kind": "text_list", "value": list(value)} capture.require(type(value) is str, "preparation_invalid_text_value") return {"state": "candidate", "kind": "text", "value": value} def _gap(reason): return {"state": "gap", "reason": reason} def _attempt(select): try: return _candidate(select()) except capture.CollectionError as error: # Selectors use fixed codes; arbitrary exceptions are fatal, never # serialized as guest data or treated as field-level missing values. return _gap(str(error)) def _at(document, pointer): node = document for key in pointer.split("/")[1:]: key = key.replace("~1", "/").replace("~0", "~") node = node[int(key)] if type(node) is list else node[key] return node def _raw(archive, reference): raw = archive.read(reference["file"]) capture.strict_json(raw) return _at(json.loads(raw, parse_float=Decimal), reference["pointer"]) def _name(assessment, named): if not named: return _gap("profile_source_not_acquired") profile = assessment["profile"] if "error" in profile: return _gap(profile["error"]) return _candidate(profile["full_name"]) def _select(archive, record, named): refs = record["sources"] search, detail = _raw(archive, refs["search"]), _raw(archive, refs["detail"]) assessment = _raw(archive, refs["assessment"]) day = archive.options.from_date # Both fields in a pair depend on the same strict selector, so an invalid # pair cannot silently contribute one apparently valid date/count. result = { "BLOCK_CODE": _attempt(lambda: fields.agreed_block_code(search, detail, day)), "ROOM_CATEGORY_LABEL": _attempt(lambda: fields.agreed_room_type(search, detail, day)), "CONFIRMATION_NO": _attempt(lambda: fields.confirmation_no(detail)), "ARRIVAL": _attempt(lambda: fields.stay_dates(detail, day).arrival.isoformat()), "DEPARTURE": _attempt(lambda: fields.stay_dates(detail, day).departure.isoformat()), "RATE_CODE": _attempt(lambda: fields.arrival_rate_code(detail, day)), "DISP_ROOM_NO": _attempt(lambda: fields.agreed_room_number(search, detail, day)), "ADULTS": _attempt(lambda: str(fields.agreed_guest_counts(detail, day).adults)), "CHILDREN": _attempt(lambda: str(fields.agreed_guest_counts(detail, day).children)), "NO_OF_ROOMS": _attempt(lambda: str(fields.arrival_number_of_units(detail, day))), "RES_COMMENT": _attempt(lambda: fields.reservation_gen_notes(detail)), "TRACE_TEXT": _candidate(fields.traces()), "FULL_NAME": _name(assessment, named), } rate_ref = refs["rate"] request = archive.document(rate_ref["request"]["file"]) response = json.loads(archive.read(rate_ref["file"]), parse_float=Decimal) try: rate = fields.effective_rate(detail, request, response, archive.options) except capture.CollectionError as error: result["EFFECTIVE_RATE_AMOUNT"] = _gap(str(error)) currency = None else: # Keep Decimal's compact exact representation: expanding 1E-1000000 # into fixed-point text would allocate a huge string before the budget. result["EFFECTIVE_RATE_AMOUNT"] = _candidate(str(rate.amount)) currency = rate.currency result.update({name: _gap(reason) for name, reason in UNRESOLVED.items()}) capture.require(set(result) == set(FIELDS), "preparation_field_set_mismatch") return {"capture_sequence": record["capture_sequence"], "reservation_id": record["reservation_id"], "fields": {name: result[name] for name in FIELDS}, "effective_rate_currency": currency} def prepare(archive, evidence: bytes, *, name_supplements=()) -> bytes: """Reverify evidence before using any source references or selected values. Evidence independently authenticates paths/identity/request provenance; candidate selection composes the existing selectors. It is NOT a second independent implementation of ARR business mapping. """ named = type(archive) is named_audit.VerifiedArchive capture.require(named or type(archive) is audit.VerifiedArchive, "preparation_capture_type_invalid") protocol = named_audit if named else audit archive = protocol.VerifiedArchive(archive.directory, archive.pin) verified = verify_source_facts(archive, evidence) records = capture.strict_json(evidence)["records"] result = { "version": VERSION, "selector_version": fields.VERSION, "status": "unaccepted_source_candidates", "capture_manifest_sha256": archive.pin, "field_evidence_sha256": verified["evidence_sha256"], "field_contract_sha256": verified["contract_sha256"], "hotel_id": archive.options.hotel_id, "from_date": archive.options.from_date, "to_date": archive.options.to_date, "rate_date": archive.options.rate_date, "record_order": "capture_order_not_report_order", "report_gaps": list(REPORT_GAPS), "complete_arr_output": False, **FLAGS, "records": [_select(archive, record, named) for record in records], } if name_supplements: from . import profile_supplement capture.require(not named, "supplement_requires_v2_base") names, _ = profile_supplement.verify_chain(archive, name_supplements) capture.require(len(names) == len(result["records"]), "supplement_record_count_mismatch") for record, name in zip(result["records"], names): capture.require(record["reservation_id"] == name["reservation_id"], "supplement_record_identity_mismatch") record["fields"]["FULL_NAME"] = _name(name, True) result.update(name_supplement_pins=[pin for _, pin in name_supplements]) raw = canonical(result) capture.require(len(raw) <= MAX_FACTS_BYTES, "preparation_byte_budget_exceeded") return raw def summary(raw: bytes): document = capture.strict_json(raw) counts = {name: {"candidates": 0, "gaps": 0, "gap_reasons": Counter()} for name in FIELDS} gap_rows = 0 for record in document["records"]: gap_rows += any(value["state"] == "gap" for value in record["fields"].values()) for name, value in record["fields"].items(): if value["state"] == "candidate": counts[name]["candidates"] += 1 else: counts[name]["gaps"] += 1 counts[name]["gap_reasons"][value["reason"]] += 1 return {"status": document["status"], "version": document["version"], "records": len(document["records"]), "records_with_gaps": gap_rows, "fields": counts, "report_gaps": document["report_gaps"], "preparation_sha256": hashlib.sha256(raw).hexdigest(), "capture_manifest_sha256": document["capture_manifest_sha256"], "field_evidence_sha256": document["field_evidence_sha256"], "complete_arr_output": False, "network_calls": 0, **FLAGS} def export(capture_dir: Path, capture_pin: str, output_dir: Path, *, capture_version="v2", name_supplements=()): repository = Path(__file__).resolve().parents[2] capture.require(not output_dir.resolve().is_relative_to(repository), "output_must_be_outside_repository") capture.require(capture_version in ("v2", "v3"), "unsupported_preparation_capture_version") protocol = named_audit if capture_version == "v3" else audit archive = protocol.VerifiedArchive(capture_dir, capture_pin) evidence = facts.build_source_facts(archive) raw = prepare(archive, evidence, name_supplements=name_supplements) result = summary(raw) # No output directory exists until complete input integrity checks pass. sink = capture.Archive(output_dir) sink.write("source-facts.json", evidence) sink.write("source-candidates.json", raw) sink.write("preparation-summary.json", result) return result def main(argv=None): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--capture-dir", required=True, type=Path) parser.add_argument("--capture-sha256", required=True) parser.add_argument("--output-dir", required=True, type=Path) parser.add_argument("--capture-version", choices=("v2", "v3"), default="v2") parser.add_argument("--profile-supplement", nargs=2, action="append", default=[], metavar=("DIRECTORY", "SHA256")) args = parser.parse_args(argv) try: result = export(args.capture_dir, args.capture_sha256, args.output_dir, capture_version=args.capture_version, name_supplements=args.profile_supplement) except capture.CollectionError as error: result = {"status": "failed", "error": str(error), "complete_arr_output": False, **FLAGS} except Exception: result = {"status": "failed", "error": "source_preparation_failed", "complete_arr_output": False, **FLAGS} print(json.dumps(result, ensure_ascii=False, sort_keys=True)) # Success means diagnostics are complete, even when every row has a gap. return 0 if result["status"] == "unaccepted_source_candidates" else 1 if __name__ == "__main__": raise SystemExit(main())