Files
ARR-2.0-0918/integrations/ohip/audit_arr_capture.py
T

334 lines
19 KiB
Python

"""Read-only integrity replay and aggregate field audit of an ARR API capture.
No network, credentials, source projection, pricing, or Finance ingestion.
Field presence and candidate agreement do not establish report equivalence.
"""
from __future__ import annotations
import argparse
from collections import Counter
from datetime import date
import hashlib
import json
import os
from pathlib import Path
import re
import stat
import sys
import urllib.error
if __package__:
from . import collect_arr_source as source
else:
import collect_arr_source as source
require = source.require
MAX_MANIFEST_BYTES = source.MAX_MANIFEST_BYTES
MAX_ARCHIVE_BYTES = source.MAX_ARCHIVE_BYTES
FILE_PATTERN = re.compile(r"(?:capture\.json|replay-provenance\.json|request-[0-9]{6}\.(?:json|meta\.json|response\.bin))")
# Paths are candidate observations, not an adopted transformation contract.
# No block-code candidate exists in the currently selected detail schema.
FIELD_PATHS = {
"BLOCK_CODE": (),
"ADULTS": ("detail.roomStay.guestCounts.adults",),
"CHILDREN": ("detail.roomStay.guestCounts.children",),
"COMPANY_NAME": ("detail.reservationProfiles.reservationProfile[].profile.company.companyName",
"detail.roomStay.roomRates[].stayProfiles[].profile.company.companyName"),
"CONFIRMATION_NO": (), # Typed identity list, handled separately.
"DISP_ROOM_NO": ("detail.roomStay.currentRoomInfo.roomId", "detail.roomStay.roomRates[].roomId"),
"EFFECTIVE_RATE_AMOUNT": ("detail.roomStay.roomRates[].rates.rate[].effectiveRate.amountBeforeTax",),
"FULL_NAME": ("search.reservationGuest.fullName",),
"RES_COMMENT": ("detail.comments[].comment.text.value",),
"TRACE_TEXT": ("detail.traces[].traceText",),
"NO_OF_ROOMS": ("detail.roomStay.roomRates[].numberOfUnits",),
"PRODUCTS": ("detail.reservationPackages[].packageCode",),
"RATE_CODE": ("detail.roomStay.roomRates[].ratePlanCode",),
"ROOM_CATEGORY_LABEL": ("detail.roomStay.roomRates[].roomType",),
"ARRIVAL": ("detail.roomStay.arrivalDate",),
"DEPARTURE": ("detail.roomStay.departureDate",),
}
OPTIONAL_FIELDS = {"BLOCK_CODE", "PRODUCTS", "ROOM_CATEGORY_LABEL", "RES_COMMENT", "TRACE_TEXT"}
def protected_read(path: Path, maximum: int) -> bytes:
fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK)
with os.fdopen(fd, "rb") as handle:
info = os.fstat(handle.fileno())
require(stat.S_ISREG(info.st_mode) and info.st_uid == os.getuid()
and stat.S_IMODE(info.st_mode) == 0o600, "unsafe_archive_file")
require(info.st_size <= maximum, "archive_file_too_large")
raw = handle.read(maximum + 1)
require(len(raw) <= maximum, "archive_file_too_large")
return raw
class VerifiedArchive:
version = "arr-api-capture/v1"
file_pattern = FILE_PATTERN
operations = [source.SEARCH, source.DETAIL]
options_type = source.Options
max_inventory_files = 92000
def __init__(self, directory: Path, expected_manifest_sha256: str):
require(bool(re.fullmatch(r"[0-9a-f]{64}", expected_manifest_sha256)), "invalid_manifest_pin")
info = directory.lstat()
require(stat.S_ISDIR(info.st_mode) and info.st_uid == os.getuid()
and stat.S_IMODE(info.st_mode) == 0o700, "unsafe_archive_directory")
self.directory = directory
raw = protected_read(directory / "result.json", MAX_MANIFEST_BYTES)
require(hashlib.sha256(raw).hexdigest() == expected_manifest_sha256, "manifest_hash_mismatch")
self.result = source.strict_json(raw)
require(self.result.get("version") == self.version
and self.result.get("status") == "complete_candidate_capture"
and self.result.get("candidate_capture_complete") is True, "incomplete_capture")
require(all(self.result.get(key) is False for key in
("finance_ready", "report_equivalence_verified", "atomic_snapshot")), "unsupported_capture_claim")
items = self.result.get("files")
require(isinstance(items, list) and 1 <= len(items) <= self.max_inventory_files, "invalid_file_inventory")
self.inventory = {}
total = 0
for item in items:
require(isinstance(item, dict) and set(item) == {"name", "bytes", "sha256"}, "invalid_file_entry")
name = item["name"]
require(isinstance(name, str) and bool(self.file_pattern.fullmatch(name)), "unsafe_inventory_name")
require(name not in self.inventory, "duplicate_inventory_name")
require(type(item["bytes"]) is int and 0 <= item["bytes"] <= source.MAX_RESPONSE_BYTES + 1,
"invalid_inventory_size")
require(isinstance(item["sha256"], str) and bool(re.fullmatch(r"[0-9a-f]{64}", item["sha256"])),
"invalid_inventory_hash")
self.inventory[name] = item
total += item["bytes"]
require(total <= MAX_ARCHIVE_BYTES, "archive_byte_budget_exceeded")
require({path.name for path in directory.iterdir()} == set(self.inventory) | {"result.json"},
"archive_file_set_mismatch")
for name in self.inventory:
self.read(name)
self.capture = self.document("capture.json")
require(self.capture.get("version") == self.version
and self.capture.get("service_url") == source.SERVICE
and self.capture.get("application_id") == source.APPLICATION
and self.capture.get("operations") == self.operations
and self.capture.get("fetch_instructions") == list(source.FETCH), "capture_contract_mismatch")
self.options = self.options_type(**self.capture["options"])
self.options.validate()
for document in (self.capture, self.result):
require(document.get("hotel_id") == self.options.hotel_id
and document.get("arrival_date") == self.options.arrival_date, "capture_context_mismatch")
self.pin = expected_manifest_sha256
def read(self, name: str) -> bytes:
require(name in self.inventory, "missing_archive_file")
item = self.inventory[name]
raw = protected_read(self.directory / name, item["bytes"])
require(len(raw) == item["bytes"] and hashlib.sha256(raw).hexdigest() == item["sha256"], "archive_hash_mismatch")
return raw
def document(self, name: str) -> dict:
return source.strict_json(self.read(name))
class MemorySink:
"""Satisfy the collector archive protocol without any filesystem writes."""
def __init__(self):
self.files = []
self.latest_request = None
def write(self, name, value):
if re.fullmatch(r"request-[0-9]{6}\.json", name):
self.latest_request = value
raw = value if isinstance(value, bytes) else source.json_bytes(value)
self.files.append({"name": name, "bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()})
def replay(archive: VerifiedArchive) -> tuple[list[dict], list[dict]]:
"""Replay request construction and protocol validation; never instantiate HTTPTransport."""
calls = 0
consumed = {"capture.json"}
if "replay-provenance.json" in archive.inventory:
consumed.add("replay-provenance.json")
initial_rows, details = [], []
initial_search_done = False
sink = MemorySink()
def transport(method, path, body):
nonlocal calls, initial_search_done
calls += 1
label = f"request-{calls:06d}"
request_name, meta_name, response_name = label + ".json", label + ".meta.json", label + ".response.bin"
request, meta = archive.document(request_name), archive.document(meta_name)
consumed.update((request_name, meta_name))
expected_body = source.strict_json(body) if body is not None else None
operation = source.SEARCH if method == "POST" else source.DETAIL
require(request.get("operation_id") == operation and request.get("method") == method
and request.get("path") == path and request.get("body") == expected_body, "archived_request_mismatch")
require(type(request.get("attempt")) is int and request["attempt"] == sink.latest_request["attempt"],
"archived_attempt_mismatch")
if meta.get("error") == "transport_failure":
require(response_name not in archive.inventory, "unexpected_transport_response")
raise urllib.error.URLError("archived_transport_failure")
raw = archive.read(response_name)
consumed.add(response_name)
require(type(meta.get("http_status")) is int and meta.get("oversized") is False, "invalid_response_metadata")
if meta["http_status"] == 200:
document = source.strict_json(raw)
page = document.get("data", {}).get("reservations", {})
if operation == source.SEARCH and not initial_search_done:
initial_rows.extend(page.get("reservationInfo", []))
if page.get("hasMore", False) is False:
initial_search_done = True
elif operation == source.DETAIL:
details.extend(page.get("reservation", []))
# Retry timing is not reconstructed: v1 archives intentionally omit response headers.
return meta["http_status"], {}, raw
reader = source.Reader(sink, archive.options.hotel_id, transport, sleep=lambda _: None)
result = source.collect(archive.options, sink, reader)
require(result.get("candidate_capture_complete") is True, "capture_protocol_replay_failed")
require(consumed == set(archive.inventory), "unconsumed_archive_files")
for name in ("http_attempts", "search_records", "verified_details", "search_recheck_equal"):
require(type(result[name]) is type(archive.result.get(name)) and result[name] == archive.result[name],
"capture_summary_mismatch")
require(len(initial_rows) == len(details) == result["search_records"], "capture_pair_count_mismatch")
return initial_rows, details
def values_at(value, path):
values = [value]
for step in path.split("."):
array = step.endswith("[]")
key = step[:-2] if array else step
found = []
for parent in values:
if not isinstance(parent, dict) or key not in parent:
continue
child = parent[key]
if array:
if isinstance(child, list):
found.extend(child)
else:
found.append(child)
values = found
return values
def populated(value):
return type(value) in (str, int, float) and (not isinstance(value, str) or bool(value.strip()))
def iso_date(value):
try:
return date.fromisoformat(value) if isinstance(value, str) and date.fromisoformat(value).isoformat() == value else None
except ValueError:
return None
def analyze_fields(search_rows: list[dict], details: list[dict]) -> dict:
require(len(search_rows) == len(details) > 0, "invalid_audit_pairs")
fields = {name: {"records_with_candidate": 0, "records_without_candidate": 0,
"records_with_distinct_candidates": 0, "required_for_whitelist_candidate": name not in OPTIONAL_FIELDS,
"required_for_all_source_rows": name == "RATE_CODE",
"report_mapping_verified": False} for name in FIELD_PATHS}
scenarios = Counter()
rooms = Counter()
roles = Counter()
for search, detail in zip(search_rows, details):
require(source.reservation_id(search) == source.reservation_id(detail), "audit_pair_identity_mismatch")
pair = {"search": search, "detail": detail}
for name, paths in FIELD_PATHS.items():
values = [value for path in paths for value in values_at(pair, path) if populated(value)]
if name == "CONFIRMATION_NO":
values = [item.get("id") for item in detail["reservationIdList"]
if item.get("type") == "Confirmation" and populated(item.get("id"))]
distinct = {source.json_bytes(value) for value in values}
fields[name]["records_with_candidate"] += bool(values)
fields[name]["records_without_candidate"] += not values
fields[name]["records_with_distinct_candidates"] += len(distinct) > 1
stay = detail.get("roomStay", {})
arrival, departure = iso_date(stay.get("arrivalDate")), iso_date(stay.get("departureDate"))
scenarios["invalid_stay_dates"] += not arrival or not departure or departure < arrival
scenarios["zero_night_records"] += bool(arrival and departure and arrival == departure)
room = stay.get("currentRoomInfo", {}).get("roomId")
scenarios["missing_current_room"] += not populated(room)
if populated(room):
rooms[(str(room).strip(), str(stay.get("arrivalDate")))] += 1
segments = stay.get("roomRates", [])
scenarios["multiple_rate_segments"] += len(segments) > 1
scenarios["rate_segments_total"] += len(segments)
scenarios["single_day_rate_segments"] += sum(bool(iso_date(seg.get("start"))) and seg.get("start") == seg.get("end") for seg in segments)
scenarios["multi_room_records"] += any(type(seg.get("numberOfUnits")) is int and seg["numberOfUnits"] > 1 for seg in segments)
# Research hypothesis only; do not select a source value using this interval.
arrival_segments = [seg for seg in segments if iso_date(seg.get("start")) and iso_date(seg.get("end"))
and arrival and iso_date(seg["start"]) <= arrival <= iso_date(seg["end"])]
scenarios["arrival_interval_zero_candidates"] += len(arrival_segments) == 0
scenarios["arrival_interval_one_candidate"] += len(arrival_segments) == 1
scenarios["arrival_interval_multiple_candidates"] += len(arrival_segments) > 1
current_names = {str(seg["roomId"]) for seg in arrival_segments if populated(seg.get("roomId"))}
scenarios["current_vs_arrival_segment_room_disagreement"] += bool(populated(room) and current_names and current_names != {str(room)})
scenarios["current_room_without_arrival_segment_room"] += bool(populated(room) and not current_names)
gen = [value for value in values_at(detail, "comments[].comment") if isinstance(value, dict)
and value.get("type") == "GEN" and value.get("notificationLocation") == "RESERVATION"]
nonempty_gen = [value for value in gen if any(populated(x) for x in values_at(value, "text.value"))]
scenarios["records_with_gen_reservation_note"] += bool(nonempty_gen)
scenarios["records_with_multiple_nonempty_gen_notes"] += len(nonempty_gen) > 1
scenarios["records_with_internal_gen_note"] += any(value.get("internal") is True for value in nonempty_gen)
scenarios["records_with_internal_and_external_gen_notes"] += (any(value.get("internal") is True for value in nonempty_gen)
and any(value.get("internal") is False for value in nonempty_gen))
scenarios["records_with_multiple_nonempty_traces"] += sum(populated(x) for x in values_at(detail, "traces[].traceText")) > 1
guests = [guest for guest in detail.get("reservationGuests", []) if guest.get("primary") is True]
scenarios["records_without_unique_primary_guest"] += len(guests) != 1
names = [name for guest in guests for name in values_at(guest, "profileInfo.profile.customer.personName[]")
if isinstance(name, dict) and name.get("nameType") == "Primary"]
scenarios["records_without_unique_primary_name"] += len(names) != 1
for component in ("surname", "givenName", "middleName", "nameTitle"):
scenarios["records_with_primary_" + component] += len(names) == 1 and populated(names[0].get(component))
scenarios["records_with_name_components_but_no_display_name"] += (len(names) == 1 and populated(names[0].get("surname"))
and not any(populated(value) for value in values_at(search, "reservationGuest.fullName")))
profiles = values_at(detail, "reservationProfiles.reservationProfile[]")
for profile in profiles:
role = profile.get("reservationProfileType")
roles[role if role in {"Company", "TravelAgent", "Source", "Group"} else "Other"] += 1
scenarios["duplicate_nonblank_room_arrival_groups"] = sum(count > 1 for count in rooms.values())
scenarios["records_in_duplicate_nonblank_room_arrival_groups"] = sum(count for count in rooms.values() if count > 1)
return {"source_records": len(details), "fields": fields, "scenarios": dict(scenarios), "profile_roles": dict(roles),
"candidate_scope": "all returned segments/roles/notes; no chosen report row or field projection",
"field_notes": {"FULL_NAME": "Primary-name components are counted separately; no display-name concatenation",
"BLOCK_CODE": "No candidate code in the selected detail schema; IDs/names are not substituted",
"EFFECTIVE_RATE_AMOUNT": "Observe existing effectiveRate.amountBeforeTax only; no base/total fallback",
"candidate_counts": "Nonempty scalar presence only, not full integer/range/currency/display validity; observations before report selection, not processor outcomes",
"duplicate_groups": "All-source current-room candidates, trimmed with leading zeros preserved; not post-whitelist/validation duplicate outcomes or deletion counts"},
"arrival_interval_hypothesis": "start <= arrival <= end; diagnostic only, not an adopted mapping",
"blocked_business_decisions": ["effective_rate_for_arrival_date", "report_inclusion_and_eta",
"source_row_granularity_and_order", "company_role_and_display", "note_type_and_order_mapping",
"historical_room_and_rate_segment", "same_environment_report_equivalence"]}
def audit_capture(directory: Path, expected_manifest_sha256: str) -> dict:
try:
archive = VerifiedArchive(directory, expected_manifest_sha256)
search, details = replay(archive)
analysis = analyze_fields(search, details)
return {"version": "arr-api-capture-audit/v1", "status": "verified_capture_audited",
"manifest_sha256": archive.pin, "verified_files": len(archive.inventory) + 1,
"capture_protocol_replay_passed": True, "finance_ready": False,
"report_equivalence_verified": False, "network_calls": 0, "analysis": analysis}
except source.CollectionError as error:
return {"status": "invalid_capture", "error": str(error), "finance_ready": False}
except Exception:
return {"status": "invalid_capture", "error": "archive_or_shape_invalid", "finance_ready": False}
def main(argv=None):
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--capture-dir", required=True, type=Path)
parser.add_argument("--manifest-sha256", required=True, help="Previously recorded SHA-256 of this capture's result.json")
args = parser.parse_args(argv)
result = audit_capture(args.capture_dir, args.manifest_sha256)
print(json.dumps(result, ensure_ascii=False, allow_nan=False))
return 0 if result["status"] == "verified_capture_audited" else 1
if __name__ == "__main__":
sys.exit(main())