"""Read-only integrity replay and aggregate field audit of an ARR API capture. No network, credentials, source projection, pricing, or Finance ingestion. Field presence and candidate agreement do not establish report equivalence. """ from __future__ import annotations import argparse from collections import Counter from datetime import date import hashlib import json import os from pathlib import Path import re import stat import sys import urllib.error if __package__: from . import collect_arr_source as source else: import collect_arr_source as source require = source.require MAX_MANIFEST_BYTES = source.MAX_MANIFEST_BYTES MAX_ARCHIVE_BYTES = source.MAX_ARCHIVE_BYTES FILE_PATTERN = re.compile(r"(?:capture\.json|replay-provenance\.json|request-[0-9]{6}\.(?:json|meta\.json|response\.bin))") # Paths are candidate observations, not an adopted transformation contract. # No block-code candidate exists in the currently selected detail schema. FIELD_PATHS = { "BLOCK_CODE": (), "ADULTS": ("detail.roomStay.guestCounts.adults",), "CHILDREN": ("detail.roomStay.guestCounts.children",), "COMPANY_NAME": ("detail.reservationProfiles.reservationProfile[].profile.company.companyName", "detail.roomStay.roomRates[].stayProfiles[].profile.company.companyName"), "CONFIRMATION_NO": (), # Typed identity list, handled separately. "DISP_ROOM_NO": ("detail.roomStay.currentRoomInfo.roomId", "detail.roomStay.roomRates[].roomId"), "EFFECTIVE_RATE_AMOUNT": ("detail.roomStay.roomRates[].rates.rate[].effectiveRate.amountBeforeTax",), "FULL_NAME": ("search.reservationGuest.fullName",), "RES_COMMENT": ("detail.comments[].comment.text.value",), "TRACE_TEXT": ("detail.traces[].traceText",), "NO_OF_ROOMS": ("detail.roomStay.roomRates[].numberOfUnits",), "PRODUCTS": ("detail.reservationPackages[].packageCode",), "RATE_CODE": ("detail.roomStay.roomRates[].ratePlanCode",), "ROOM_CATEGORY_LABEL": ("detail.roomStay.roomRates[].roomType",), "ARRIVAL": ("detail.roomStay.arrivalDate",), "DEPARTURE": ("detail.roomStay.departureDate",), } OPTIONAL_FIELDS = {"BLOCK_CODE", "PRODUCTS", "ROOM_CATEGORY_LABEL", "RES_COMMENT", "TRACE_TEXT"} def protected_read(path: Path, maximum: int) -> bytes: fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) with os.fdopen(fd, "rb") as handle: info = os.fstat(handle.fileno()) require(stat.S_ISREG(info.st_mode) and info.st_uid == os.getuid() and stat.S_IMODE(info.st_mode) == 0o600, "unsafe_archive_file") require(info.st_size <= maximum, "archive_file_too_large") raw = handle.read(maximum + 1) require(len(raw) <= maximum, "archive_file_too_large") return raw class VerifiedArchive: version = "arr-api-capture/v1" file_pattern = FILE_PATTERN operations = [source.SEARCH, source.DETAIL] options_type = source.Options max_inventory_files = 92000 def __init__(self, directory: Path, expected_manifest_sha256: str): require(bool(re.fullmatch(r"[0-9a-f]{64}", expected_manifest_sha256)), "invalid_manifest_pin") info = directory.lstat() require(stat.S_ISDIR(info.st_mode) and info.st_uid == os.getuid() and stat.S_IMODE(info.st_mode) == 0o700, "unsafe_archive_directory") self.directory = directory raw = protected_read(directory / "result.json", MAX_MANIFEST_BYTES) require(hashlib.sha256(raw).hexdigest() == expected_manifest_sha256, "manifest_hash_mismatch") self.result = source.strict_json(raw) require(self.result.get("version") == self.version and self.result.get("status") == "complete_candidate_capture" and self.result.get("candidate_capture_complete") is True, "incomplete_capture") require(all(self.result.get(key) is False for key in ("finance_ready", "report_equivalence_verified", "atomic_snapshot")), "unsupported_capture_claim") items = self.result.get("files") require(isinstance(items, list) and 1 <= len(items) <= self.max_inventory_files, "invalid_file_inventory") self.inventory = {} total = 0 for item in items: require(isinstance(item, dict) and set(item) == {"name", "bytes", "sha256"}, "invalid_file_entry") name = item["name"] require(isinstance(name, str) and bool(self.file_pattern.fullmatch(name)), "unsafe_inventory_name") require(name not in self.inventory, "duplicate_inventory_name") require(type(item["bytes"]) is int and 0 <= item["bytes"] <= source.MAX_RESPONSE_BYTES + 1, "invalid_inventory_size") require(isinstance(item["sha256"], str) and bool(re.fullmatch(r"[0-9a-f]{64}", item["sha256"])), "invalid_inventory_hash") self.inventory[name] = item total += item["bytes"] require(total <= MAX_ARCHIVE_BYTES, "archive_byte_budget_exceeded") require({path.name for path in directory.iterdir()} == set(self.inventory) | {"result.json"}, "archive_file_set_mismatch") for name in self.inventory: self.read(name) self.capture = self.document("capture.json") require(self.capture.get("version") == self.version and self.capture.get("service_url") == source.SERVICE and self.capture.get("application_id") == source.APPLICATION and self.capture.get("operations") == self.operations and self.capture.get("fetch_instructions") == list(source.FETCH), "capture_contract_mismatch") self.options = self.options_type(**self.capture["options"]) self.options.validate() for document in (self.capture, self.result): require(document.get("hotel_id") == self.options.hotel_id and document.get("arrival_date") == self.options.arrival_date, "capture_context_mismatch") self.pin = expected_manifest_sha256 def read(self, name: str) -> bytes: require(name in self.inventory, "missing_archive_file") item = self.inventory[name] raw = protected_read(self.directory / name, item["bytes"]) require(len(raw) == item["bytes"] and hashlib.sha256(raw).hexdigest() == item["sha256"], "archive_hash_mismatch") return raw def document(self, name: str) -> dict: return source.strict_json(self.read(name)) class MemorySink: """Satisfy the collector archive protocol without any filesystem writes.""" def __init__(self): self.files = [] self.latest_request = None def write(self, name, value): if re.fullmatch(r"request-[0-9]{6}\.json", name): self.latest_request = value raw = value if isinstance(value, bytes) else source.json_bytes(value) self.files.append({"name": name, "bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()}) def replay(archive: VerifiedArchive) -> tuple[list[dict], list[dict]]: """Replay request construction and protocol validation; never instantiate HTTPTransport.""" calls = 0 consumed = {"capture.json"} if "replay-provenance.json" in archive.inventory: consumed.add("replay-provenance.json") initial_rows, details = [], [] initial_search_done = False sink = MemorySink() def transport(method, path, body): nonlocal calls, initial_search_done calls += 1 label = f"request-{calls:06d}" request_name, meta_name, response_name = label + ".json", label + ".meta.json", label + ".response.bin" request, meta = archive.document(request_name), archive.document(meta_name) consumed.update((request_name, meta_name)) expected_body = source.strict_json(body) if body is not None else None operation = source.SEARCH if method == "POST" else source.DETAIL require(request.get("operation_id") == operation and request.get("method") == method and request.get("path") == path and request.get("body") == expected_body, "archived_request_mismatch") require(type(request.get("attempt")) is int and request["attempt"] == sink.latest_request["attempt"], "archived_attempt_mismatch") if meta.get("error") == "transport_failure": require(response_name not in archive.inventory, "unexpected_transport_response") raise urllib.error.URLError("archived_transport_failure") raw = archive.read(response_name) consumed.add(response_name) require(type(meta.get("http_status")) is int and meta.get("oversized") is False, "invalid_response_metadata") if meta["http_status"] == 200: document = source.strict_json(raw) page = document.get("data", {}).get("reservations", {}) if operation == source.SEARCH and not initial_search_done: initial_rows.extend(page.get("reservationInfo", [])) if page.get("hasMore", False) is False: initial_search_done = True elif operation == source.DETAIL: details.extend(page.get("reservation", [])) # Retry timing is not reconstructed: v1 archives intentionally omit response headers. return meta["http_status"], {}, raw reader = source.Reader(sink, archive.options.hotel_id, transport, sleep=lambda _: None) result = source.collect(archive.options, sink, reader) require(result.get("candidate_capture_complete") is True, "capture_protocol_replay_failed") require(consumed == set(archive.inventory), "unconsumed_archive_files") for name in ("http_attempts", "search_records", "verified_details", "search_recheck_equal"): require(type(result[name]) is type(archive.result.get(name)) and result[name] == archive.result[name], "capture_summary_mismatch") require(len(initial_rows) == len(details) == result["search_records"], "capture_pair_count_mismatch") return initial_rows, details def values_at(value, path): values = [value] for step in path.split("."): array = step.endswith("[]") key = step[:-2] if array else step found = [] for parent in values: if not isinstance(parent, dict) or key not in parent: continue child = parent[key] if array: if isinstance(child, list): found.extend(child) else: found.append(child) values = found return values def populated(value): return type(value) in (str, int, float) and (not isinstance(value, str) or bool(value.strip())) def iso_date(value): try: return date.fromisoformat(value) if isinstance(value, str) and date.fromisoformat(value).isoformat() == value else None except ValueError: return None def analyze_fields(search_rows: list[dict], details: list[dict]) -> dict: require(len(search_rows) == len(details) > 0, "invalid_audit_pairs") fields = {name: {"records_with_candidate": 0, "records_without_candidate": 0, "records_with_distinct_candidates": 0, "required_for_whitelist_candidate": name not in OPTIONAL_FIELDS, "required_for_all_source_rows": name == "RATE_CODE", "report_mapping_verified": False} for name in FIELD_PATHS} scenarios = Counter() rooms = Counter() roles = Counter() for search, detail in zip(search_rows, details): require(source.reservation_id(search) == source.reservation_id(detail), "audit_pair_identity_mismatch") pair = {"search": search, "detail": detail} for name, paths in FIELD_PATHS.items(): values = [value for path in paths for value in values_at(pair, path) if populated(value)] if name == "CONFIRMATION_NO": values = [item.get("id") for item in detail["reservationIdList"] if item.get("type") == "Confirmation" and populated(item.get("id"))] distinct = {source.json_bytes(value) for value in values} fields[name]["records_with_candidate"] += bool(values) fields[name]["records_without_candidate"] += not values fields[name]["records_with_distinct_candidates"] += len(distinct) > 1 stay = detail.get("roomStay", {}) arrival, departure = iso_date(stay.get("arrivalDate")), iso_date(stay.get("departureDate")) scenarios["invalid_stay_dates"] += not arrival or not departure or departure < arrival scenarios["zero_night_records"] += bool(arrival and departure and arrival == departure) room = stay.get("currentRoomInfo", {}).get("roomId") scenarios["missing_current_room"] += not populated(room) if populated(room): rooms[(str(room).strip(), str(stay.get("arrivalDate")))] += 1 segments = stay.get("roomRates", []) scenarios["multiple_rate_segments"] += len(segments) > 1 scenarios["rate_segments_total"] += len(segments) scenarios["single_day_rate_segments"] += sum(bool(iso_date(seg.get("start"))) and seg.get("start") == seg.get("end") for seg in segments) scenarios["multi_room_records"] += any(type(seg.get("numberOfUnits")) is int and seg["numberOfUnits"] > 1 for seg in segments) # Research hypothesis only; do not select a source value using this interval. arrival_segments = [seg for seg in segments if iso_date(seg.get("start")) and iso_date(seg.get("end")) and arrival and iso_date(seg["start"]) <= arrival <= iso_date(seg["end"])] scenarios["arrival_interval_zero_candidates"] += len(arrival_segments) == 0 scenarios["arrival_interval_one_candidate"] += len(arrival_segments) == 1 scenarios["arrival_interval_multiple_candidates"] += len(arrival_segments) > 1 current_names = {str(seg["roomId"]) for seg in arrival_segments if populated(seg.get("roomId"))} scenarios["current_vs_arrival_segment_room_disagreement"] += bool(populated(room) and current_names and current_names != {str(room)}) scenarios["current_room_without_arrival_segment_room"] += bool(populated(room) and not current_names) gen = [value for value in values_at(detail, "comments[].comment") if isinstance(value, dict) and value.get("type") == "GEN" and value.get("notificationLocation") == "RESERVATION"] nonempty_gen = [value for value in gen if any(populated(x) for x in values_at(value, "text.value"))] scenarios["records_with_gen_reservation_note"] += bool(nonempty_gen) scenarios["records_with_multiple_nonempty_gen_notes"] += len(nonempty_gen) > 1 scenarios["records_with_internal_gen_note"] += any(value.get("internal") is True for value in nonempty_gen) scenarios["records_with_internal_and_external_gen_notes"] += (any(value.get("internal") is True for value in nonempty_gen) and any(value.get("internal") is False for value in nonempty_gen)) scenarios["records_with_multiple_nonempty_traces"] += sum(populated(x) for x in values_at(detail, "traces[].traceText")) > 1 guests = [guest for guest in detail.get("reservationGuests", []) if guest.get("primary") is True] scenarios["records_without_unique_primary_guest"] += len(guests) != 1 names = [name for guest in guests for name in values_at(guest, "profileInfo.profile.customer.personName[]") if isinstance(name, dict) and name.get("nameType") == "Primary"] scenarios["records_without_unique_primary_name"] += len(names) != 1 for component in ("surname", "givenName", "middleName", "nameTitle"): scenarios["records_with_primary_" + component] += len(names) == 1 and populated(names[0].get(component)) scenarios["records_with_name_components_but_no_display_name"] += (len(names) == 1 and populated(names[0].get("surname")) and not any(populated(value) for value in values_at(search, "reservationGuest.fullName"))) profiles = values_at(detail, "reservationProfiles.reservationProfile[]") for profile in profiles: role = profile.get("reservationProfileType") roles[role if role in {"Company", "TravelAgent", "Source", "Group"} else "Other"] += 1 scenarios["duplicate_nonblank_room_arrival_groups"] = sum(count > 1 for count in rooms.values()) scenarios["records_in_duplicate_nonblank_room_arrival_groups"] = sum(count for count in rooms.values() if count > 1) return {"source_records": len(details), "fields": fields, "scenarios": dict(scenarios), "profile_roles": dict(roles), "candidate_scope": "all returned segments/roles/notes; no chosen report row or field projection", "field_notes": {"FULL_NAME": "Primary-name components are counted separately; no display-name concatenation", "BLOCK_CODE": "No candidate code in the selected detail schema; IDs/names are not substituted", "EFFECTIVE_RATE_AMOUNT": "Observe existing effectiveRate.amountBeforeTax only; no base/total fallback", "candidate_counts": "Nonempty scalar presence only, not full integer/range/currency/display validity; observations before report selection, not processor outcomes", "duplicate_groups": "All-source current-room candidates, trimmed with leading zeros preserved; not post-whitelist/validation duplicate outcomes or deletion counts"}, "arrival_interval_hypothesis": "start <= arrival <= end; diagnostic only, not an adopted mapping", "blocked_business_decisions": ["effective_rate_for_arrival_date", "report_inclusion_and_eta", "source_row_granularity_and_order", "company_role_and_display", "note_type_and_order_mapping", "historical_room_and_rate_segment", "same_environment_report_equivalence"]} def audit_capture(directory: Path, expected_manifest_sha256: str) -> dict: try: archive = VerifiedArchive(directory, expected_manifest_sha256) search, details = replay(archive) analysis = analyze_fields(search, details) return {"version": "arr-api-capture-audit/v1", "status": "verified_capture_audited", "manifest_sha256": archive.pin, "verified_files": len(archive.inventory) + 1, "capture_protocol_replay_passed": True, "finance_ready": False, "report_equivalence_verified": False, "network_calls": 0, "analysis": analysis} except source.CollectionError as error: return {"status": "invalid_capture", "error": str(error), "finance_ready": False} except Exception: return {"status": "invalid_capture", "error": "archive_or_shape_invalid", "finance_ready": False} def main(argv=None): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--capture-dir", required=True, type=Path) parser.add_argument("--manifest-sha256", required=True, help="Previously recorded SHA-256 of this capture's result.json") args = parser.parse_args(argv) result = audit_capture(args.capture_dir, args.manifest_sha256) print(json.dumps(result, ensure_ascii=False, allow_nan=False)) return 0 if result["status"] == "verified_capture_audited" else 1 if __name__ == "__main__": sys.exit(main())