Files
ARR-2.0-0918/integrations/ohip/compare_arr_sources.py
T

466 lines
26 KiB
Python

"""Offline observations comparing a pinned v2/v3 API capture to native ARR XML.
This is a research/acceptance aid, not a source adapter or an ingestion gate.
Hypotheses remain separate: matching a candidate never adopts a business rule.
Output contains aggregate counts and bounded row positions, not guest values.
"""
from __future__ import annotations
import argparse
from collections import Counter
from dataclasses import dataclass
from datetime import datetime
from decimal import Decimal, InvalidOperation
import hashlib
import json
from pathlib import Path
import re
from . import audit_arr_day as audit
from . import audit_arr_named_day as named_audit
from . import collect_arr_source as source
from . import compare_report_xml as native
LIMIT = 20
ROLES = ("Company", "TravelAgent", "Source", "Group")
ROLE_PREFIX_HYPOTHESES = {"Company": "C- ", "TravelAgent": "T- ", "Source": "S- ", "Group": "G- "}
API_STATUSES = {"InHouse", "CheckedOut", "Reserved", "Cancelled", "NoShow", "Waitlist", "InSession", "Prospect"}
XML_STATUSES = {"CKIN", "CKOT", "CXL", "NS", "RES", "RSV", "CHECKED IN", "CHECKED OUT", "CANCELLED", "NO SHOW", "RESERVED"}
@dataclass(frozen=True)
class Candidate:
value: object = None
issue: str | None = None
def scalar(value) -> Candidate:
if not isinstance(value, str):
return Candidate(issue="missing_or_invalid_text")
return Candidate(value.strip())
def literal_text(value) -> Candidate:
return Candidate(value) if type(value) is str else Candidate(issue="missing_or_invalid_text")
def number(value, *, integer=False, api_integer=False) -> Candidate:
if api_integer and type(value) is not int:
return Candidate(issue="missing_or_invalid_number")
if type(value) not in {str, int, Decimal}:
return Candidate(issue="missing_or_invalid_number")
text = str(value).strip()
if not re.fullmatch(r"[+-]?[0-9]+(?:\.[0-9]+)?(?:[Ee][+-]?[0-9]+)?", text):
return Candidate(issue="missing_or_invalid_number")
try:
result = Decimal(text)
except InvalidOperation:
return Candidate(issue="missing_or_invalid_number")
if not result.is_finite() or result < 0 or (integer and result != result.to_integral_value()):
return Candidate(issue="missing_or_invalid_number")
return Candidate(result)
def report_date(value: str | None) -> Candidate:
if not isinstance(value, str):
return Candidate(issue="missing_or_invalid_date")
# Match the existing XML processor's accepted formats, independently of
# the strict API date contract. Do not infer a format from the hotel.
for fmt in ("%d-%b-%y", "%d-%m-%y", "%Y%m%d", "%Y-%m-%d", "%m/%d/%Y"):
try:
return Candidate(datetime.strptime(value.strip(), fmt).date().isoformat())
except ValueError:
pass
return Candidate(issue="missing_or_invalid_date")
def xml_number(value, *, integer=False) -> Candidate:
if not isinstance(value, str) or not value.strip():
return Candidate(issue="missing_or_invalid_number")
try:
# The processor removes commas before Decimal conversion. Keep exact
# precision here; normalization must not round a comparison value.
result = Decimal(value.replace(",", "").strip())
except InvalidOperation:
return Candidate(issue="missing_or_invalid_number")
if not result.is_finite() or result < 0 or (integer and result != result.to_integral_value()):
return Candidate(issue="missing_or_invalid_number")
return Candidate(result)
def api_date(value) -> Candidate:
if not isinstance(value, str) or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", value):
return Candidate(issue="missing_or_invalid_date")
return report_date(value)
def xml_value(row, field: str, convert=scalar) -> Candidate:
values = row.findall(field)
if len(values) != 1 or len(values[0]):
return Candidate(issue="missing_or_ambiguous_xml_field")
return convert(values[0].text or "")
def first_nonempty(values: list[str]) -> str:
return next((value for value in values if value), "")
def value_at(value, *keys):
for key in keys:
if not isinstance(value, dict):
return None
value = value.get(key)
return value
def xml_texts(row, path: str) -> Candidate:
container_path = "/".join(path.split("/")[:-2])
if len(row.findall(container_path)) > 1:
return Candidate(issue="ambiguous_xml_text_list")
# Missing optional XML containers mean no text to the existing processor.
# Missing API arrays remain unavailable; they have a different contract.
values = row.findall(path)
if any(len(value) for value in values):
return Candidate(issue="invalid_xml_text_array")
return Candidate([text for value in values if (text := (value.text or "").strip())])
def confirmation(detail: dict) -> Candidate:
values = [item.get("id") for item in detail["reservationIdList"] if item.get("type") == "Confirmation"]
return scalar(values[0]) if len(values) == 1 else Candidate(issue="missing_or_ambiguous_confirmation")
def single_day_segment(detail: dict, day: str) -> Candidate:
segments = detail.get("roomStay", {}).get("roomRates")
if not isinstance(segments, list) or not segments or any(not isinstance(s, dict) for s in segments):
return Candidate(issue="missing_or_invalid_rate_segments")
# No general inclusive/exclusive interval rule is adopted. The observed
# start=end shape can be tested without guessing a boundary convention.
if any(api_date(s.get("start")).issue or api_date(s.get("end")).issue
or s.get("start") != s.get("end") for s in segments):
return Candidate(issue="unsupported_or_invalid_rate_interval")
selected = [segment for segment in segments if segment["start"] == day]
return Candidate(selected[0]) if len(selected) == 1 else Candidate(issue="missing_or_ambiguous_day_segment")
def segment_value(segment: Candidate, name: str, convert=scalar) -> Candidate:
return Candidate(issue=segment.issue) if segment.issue else convert(segment.value.get(name))
def company_name(profiles, role: str) -> Candidate:
if profiles is None:
return Candidate(issue="profiles_not_returned")
if not isinstance(profiles, list) or any(not isinstance(p, dict) for p in profiles):
return Candidate(issue="invalid_profile_array")
selected = [profile for profile in profiles if profile.get("reservationProfileType") == role]
if not selected:
return Candidate("")
if len(selected) != 1:
return Candidate(issue="ambiguous_company_role")
profile = selected[0].get("profile")
company = profile.get("company") if isinstance(profile, dict) else None
return scalar(company.get("companyName") if isinstance(company, dict) else None)
def prefixed_company(candidate: Candidate, role: str) -> Candidate:
# This tests a display hypothesis, not Oracle's role priority or a mapping
# rule. Keep the raw-name observation separate, including legitimate blanks.
if candidate.issue or candidate.value == "":
return candidate
if "\n" in candidate.value or "\r" in candidate.value:
return Candidate(issue="multiline_company_display_unresolved")
return Candidate(ROLE_PREFIX_HYPOTHESES[role] + candidate.value)
def primary_name(detail: dict) -> Candidate:
guests = detail.get("reservationGuests")
if not isinstance(guests, list) or any(not isinstance(g, dict) for g in guests):
return Candidate(issue="missing_or_invalid_guests")
primary = [guest for guest in guests if guest.get("primary") is True]
if len(primary) != 1:
return Candidate(issue="missing_or_ambiguous_primary_guest")
info = value_at(primary[0], "profileInfo", "profile", "customer", "personName")
if not isinstance(info, list) or any(not isinstance(name, dict) for name in info):
return Candidate(issue="invalid_person_names")
names = [name for name in info if name.get("nameType") == "Primary"]
return Candidate(names[0]) if len(names) == 1 else Candidate(issue="missing_or_ambiguous_primary_name")
def name_hypotheses(detail: dict) -> dict[str, Candidate]:
name = primary_name(detail)
keys = ("FULL_NAME_SURNAME_GIVEN", "FULL_NAME_GIVEN_SURNAME", "FULL_NAME_SURNAME_COMMA_GIVEN")
if name.issue:
return dict.fromkeys(keys, name)
surname, given = scalar(name.value.get("surname")), scalar(name.value.get("givenName", ""))
if surname.issue or not surname.value or given.issue:
return dict.fromkeys(keys, Candidate(issue="missing_or_invalid_name_components"))
# Only observations of these simple formats; titles, middle names, aliases
# and localization remain unresolved. Never use these to populate a source.
return {keys[0]: Candidate(" ".join(filter(None, (surname.value, given.value)))),
keys[1]: Candidate(" ".join(filter(None, (given.value, surname.value)))),
keys[2]: Candidate(surname.value + (", " + given.value if given.value else ""))}
def note_texts(detail: dict, *, selected_gen: bool) -> Candidate:
comments = detail.get("comments")
if comments is None:
return Candidate(issue="comments_not_returned")
if not isinstance(comments, list):
return Candidate(issue="invalid_comments")
values = []
for item in comments:
comment = item.get("comment") if isinstance(item, dict) else None
if not isinstance(comment, dict):
return Candidate(issue="invalid_comment")
if selected_gen and not (comment.get("type") == "GEN" and comment.get("notificationLocation") == "RESERVATION"):
continue
text = comment.get("text")
if not isinstance(text, dict) or not isinstance(text.get("value"), str):
return Candidate(issue="missing_or_invalid_note_text")
if text["value"].strip():
values.append(text["value"].strip())
return Candidate(values)
def trace_texts(detail: dict, *, without_resolution_markers=False) -> Candidate:
traces = detail.get("traces")
if traces is None:
return Candidate(issue="traces_not_returned")
if not isinstance(traces, list) or any(not isinstance(t, dict) or not isinstance(t.get("traceText"), str) for t in traces):
return Candidate(issue="missing_or_invalid_trace_text")
values = []
for trace in traces:
if without_resolution_markers:
info = trace.get("resolveInfo", {})
if not isinstance(info, dict) or any(not isinstance(info.get(key, ""), str)
for key in ("resolvedOn", "resolvedBy")):
return Candidate(issue="invalid_trace_resolution_info")
if any(info.get(key, "").strip() for key in ("resolvedOn", "resolvedBy")):
continue
if trace["traceText"].strip():
values.append(trace["traceText"].strip())
# Absence of resolution markers is only a hypothesis, not proof the trace
# is open. Report department/date filters and ordering also remain unknown.
return Candidate(values)
def observations(row, search: dict, detail: dict, assessment: dict, day: str, *, named=False) -> dict[str, tuple[Candidate, Candidate]]:
stay = detail.get("roomStay", {})
segment = single_day_segment(detail, day)
rate = (Candidate(issue=assessment["error"]) if "error" in assessment
else number(assessment.get("effective_rate")))
currency = Candidate(issue=assessment["error"]) if "error" in assessment else scalar(assessment.get("currency"))
pairs = {
"CONFIRMATION_NO": (xml_value(row, "CONFIRMATION_NO"), confirmation(detail)),
"ARRIVAL": (xml_value(row, "TRUNC_BEGIN", report_date), api_date(stay.get("arrivalDate"))),
"DEPARTURE": (xml_value(row, "TRUNC_END", report_date), api_date(stay.get("departureDate"))),
"ADULTS_STAY": (xml_value(row, "ADULTS", lambda x: xml_number(x, integer=True)),
number(value_at(stay, "guestCounts", "adults"), integer=True, api_integer=True)),
"CHILDREN_STAY": (xml_value(row, "CF_CHILDREN", lambda x: xml_number(x, integer=True)),
number(value_at(stay, "guestCounts", "children"), integer=True, api_integer=True)),
"ADULTS_SINGLE_DAY_SEGMENT": (xml_value(row, "ADULTS", lambda x: xml_number(x, integer=True)),
Candidate(issue=segment.issue) if segment.issue else
number(value_at(segment.value, "guestCounts", "adults"), integer=True, api_integer=True)),
"CHILDREN_SINGLE_DAY_SEGMENT": (xml_value(row, "CF_CHILDREN", lambda x: xml_number(x, integer=True)),
Candidate(issue=segment.issue) if segment.issue else
number(value_at(segment.value, "guestCounts", "children"), integer=True, api_integer=True)),
"EFFECTIVE_RATE_AMOUNT": (xml_value(row, "EFFECTIVE_RATE_AMOUNT", xml_number), rate),
"CURRENCY_CODE": (xml_value(row, "CURRENCY_CODE"), currency),
"ROOM_CURRENT": (xml_value(row, "DISP_ROOM_NO"), scalar(value_at(stay, "currentRoomInfo", "roomId"))),
"ROOM_SINGLE_DAY_SEGMENT": (xml_value(row, "DISP_ROOM_NO"), segment_value(segment, "roomId")),
"RATE_CODE_SINGLE_DAY_SEGMENT": (xml_value(row, "RATE_CODE"), segment_value(segment, "ratePlanCode")),
"ROOM_TYPE_CODE_SINGLE_DAY_SEGMENT": (xml_value(row, "ROOM_CATEGORY_LABEL"), segment_value(segment, "roomType")),
"ROOM_COUNT_SINGLE_DAY_SEGMENT": (xml_value(row, "NO_OF_ROOMS", lambda x: xml_number(x, integer=True)),
segment_value(segment, "numberOfUnits", lambda x: number(x, integer=True, api_integer=True))),
"FULL_NAME_SEARCH": (xml_value(row, "FULL_NAME"), scalar(value_at(search, "reservationGuest", "fullName"))),
}
if named:
profile = assessment["profile"] # Rebuilt from pinned originals by v3 protocol replay.
exact = (Candidate(issue="primary_summary_name_unavailable") if "error" in profile
else literal_text(profile.get("full_name")))
trimmed = Candidate(issue=exact.issue) if exact.issue else scalar(exact.value)
pairs["FULL_NAME_PRIMARY_SUMMARY_EXACT"] = (xml_value(row, "FULL_NAME", literal_text), exact)
pairs["FULL_NAME_PRIMARY_SUMMARY_TRIMMED"] = (xml_value(row, "FULL_NAME"), trimmed)
for key, candidate in name_hypotheses(detail).items():
pairs[key] = (xml_value(row, "FULL_NAME"), candidate)
for role in ROLES:
profiles = value_at(detail, "reservationProfiles", "reservationProfile")
candidate = company_name(profiles, role)
pairs["COMPANY_RESERVATION_" + role.upper()] = (xml_value(row, "COMPANY_NAME"), candidate)
pairs["COMPANY_RESERVATION_" + role.upper() + "_PREFIXED"] = (
xml_value(row, "COMPANY_NAME"), prefixed_company(candidate, role))
candidate = (Candidate(issue=segment.issue) if segment.issue
else company_name(segment.value.get("stayProfiles"), role))
pairs["COMPANY_SINGLE_DAY_" + role.upper()] = (xml_value(row, "COMPANY_NAME"), candidate)
pairs["COMPANY_SINGLE_DAY_" + role.upper() + "_PREFIXED"] = (
xml_value(row, "COMPANY_NAME"), prefixed_company(candidate, role))
report_notes = xml_texts(row, "./LIST_G_COMMENT_RESV_NAME_ID/G_COMMENT_RESV_NAME_ID/RES_COMMENT")
for name, candidate in (("GEN_RESERVATION", note_texts(detail, selected_gen=True)),
("ALL_RESERVATION_NOTES", note_texts(detail, selected_gen=False)),
("ALL_TRACES", trace_texts(detail)),
("TRACES_WITHOUT_RESOLUTION_MARKERS", trace_texts(detail, without_resolution_markers=True))):
target = (xml_texts(row, "./LIST_G_DEPT_ID/G_DEPT_ID/TRACE_TEXT") if "TRACES" in name else report_notes)
pairs[name + "_ORDERED_TEXTS"] = (target, candidate)
pairs[name + "_FIRST_NONEMPTY"] = (
Candidate(issue=target.issue) if target.issue else Candidate(first_nonempty(target.value)),
Candidate(issue=candidate.issue) if candidate.issue else Candidate(first_nonempty(candidate.value)))
return pairs
def safe_status(value, allowed: set[str]) -> str:
return value if isinstance(value, str) and value in allowed else "OTHER_OR_MISSING"
def native_observations(report: native.Report) -> dict:
"""Privacy-safe, single-input observations, even when no API pair exists.
Fixed buckets only: no arbitrary XML value becomes an output key. Lexical
monotonicity of this sample never establishes a configured sort or tie rule.
"""
companies, note_types, note_descriptions = Counter(), Counter(), Counter()
note_rows = trace_rows = notes = traces = comma_names = no_share_equal = rownum_equal = 0
lexical = {field: [] for field in ("DISP_ROOM_NO", "FULL_NAME", "CONFIRMATION_NO")}
for position, row in enumerate(report.rows, 1):
company = xml_value(row, "COMPANY_NAME")
if company.issue:
companies["missing_or_ambiguous"] += 1
elif not company.value:
companies["blank"] += 1
else:
prefix = next((p for p in ROLE_PREFIX_HYPOTHESES.values() if company.value.startswith(p)), None)
companies[(prefix[:2] if prefix else "other_nonblank")] += 1
if "\n" in company.value or "\r" in company.value:
companies["multiline"] += 1
name, plain = xml_value(row, "FULL_NAME"), xml_value(row, "FULL_NAME_NO_SHR_IND")
comma_names += not name.issue and "," in name.value
no_share_equal += not name.issue and not plain.issue and bool(name.value) and name.value == plain.value
rownum_equal += xml_value(row, "ROWNUM").value == str(position)
for field, values in lexical.items():
value = xml_value(row, field)
values.append(value.value.casefold() if not value.issue and value.value else None)
row_notes = row.findall("./LIST_G_COMMENT_RESV_NAME_ID/G_COMMENT_RESV_NAME_ID")
row_traces = row.findall("./LIST_G_DEPT_ID/G_DEPT_ID")
note_rows += bool(row_notes)
trace_rows += bool(row_traces)
notes += len(row_notes)
traces += len(row_traces)
for note in row_notes:
note_types[safe_status(xml_value(note, "RES_COMMENT_TYPE").value, {"CAS", "GEN"})] += 1
note_descriptions[safe_status(xml_value(note, "RES_COMMENT_DESCRIPTION").value, {"GENERAL"})] += 1
return {"scope": "native_input_only_not_mapping_acceptance", "rows": len(report.rows),
"company_display_buckets": dict(companies), "names_containing_comma": comma_names,
"nonblank_full_name_equals_no_share_name": no_share_equal, "rownum_matches_physical_position": rownum_equal,
"lexical_casefold_ascending_hypotheses": {
field: (all(a <= b for a, b in zip(values, values[1:]))
if len(values) > 1 and all(v is not None for v in values) else None)
for field, values in lexical.items()},
"rows_with_reservation_notes": note_rows, "reservation_note_elements": notes,
"note_type_buckets": dict(note_types), "note_description_buckets": dict(note_descriptions),
"rows_with_traces": trace_rows, "trace_elements": traces,
"configured_sort_verified": False, "company_role_priority_verified": False,
"note_type_mapping_verified": False}
def compare_report(archive: audit.VerifiedArchive, report: native.Report) -> dict:
named = type(archive) is named_audit.VerifiedArchive
protocol = named_audit if named else audit
result = {"version": "arr-api-native-comparison/v2" if named else "arr-api-native-comparison/v1", "finance_ready": False,
"report_equivalence_verified": False, "source_mapping_verified": False,
"capture_manifest_sha256": archive.pin, "report_sha256": report.raw_sha256,
"api_hotel": archive.options.hotel_id, "report_hotel": report.hotel,
"api_date": archive.options.from_date, "report_date": report.arrival_date,
"network_calls": 0, "native_observations": native_observations(report)}
if (archive.options.hotel_id, archive.options.from_date) != (report.hotel, report.arrival_date):
return {**result, "status": "not_comparable", "reason": "hotel_or_date_differs"}
for row in report.rows:
arrival = xml_value(row, "TRUNC_BEGIN", report_date)
if arrival.issue or arrival.value != report.arrival_date:
return {**result, "status": "not_comparable", "reason": "report_row_date_missing_or_different"}
rows, details, assessments = protocol.replay(archive)
counts = Counter(report.identities)
api_ids = [source.reservation_id(detail) for detail in details]
pairs = {identity: (search, detail, assessment) for identity, search, detail, assessment
in zip(api_ids, rows, details, assessments["records"])}
unique = {identity for identity in pairs if counts[identity] == 1}
api_only = set(pairs) - counts.keys()
report_only = counts.keys() - pairs.keys()
fields, status_pairs = {}, Counter()
for position, (identity, row) in enumerate(zip(report.identities, report.rows), 1):
if identity not in unique:
continue # Repeated IDs are ambiguous; never pair them by row order.
search, detail, assessment = pairs[identity]
api_status = safe_status(detail.get("reservationStatus"), API_STATUSES)
xml_status = xml_value(row, "SHORT_RESV_STATUS").value
status_pairs[(api_status, safe_status(xml_status, XML_STATUSES))] += 1
for name, (expected, actual) in observations(row, search, detail, assessment, archive.options.from_date, named=named).items():
field = fields.setdefault(name, {"compared_rows": 0, "equal_nonblank": 0, "both_blank": 0,
"different": 0, "unavailable": 0, "reasons": Counter(), "problem_report_positions": []})
field["compared_rows"] += 1
if expected.issue or actual.issue:
field["unavailable"] += 1
for side, issue in (("xml", expected.issue), ("api", actual.issue)):
if issue:
field["reasons"][side + ":" + issue] += 1
elif expected.value == actual.value:
field["both_blank" if expected.value in ("", []) else "equal_nonblank"] += 1
continue
else:
field["different"] += 1
if len(field["problem_report_positions"]) < LIMIT:
field["problem_report_positions"].append(position)
return {**result, "status": "compared_observations", "comparison_scope": "candidate_hypotheses_only",
"api_records": len(api_ids), "report_records": len(report.rows), "matched_unique_records": len(unique),
"api_only_records": len(api_only), "report_only_distinct_ids": len(report_only),
"report_only_rows": sum(counts[identity] for identity in report_only),
"ambiguous_report_ids": sum(n > 1 for n in counts.values()),
"ambiguous_report_rows": sum(n for n in counts.values() if n > 1),
"identity_sequence_equal": api_ids == report.identities,
"matched_identity_order_equal": ([identity for identity in api_ids if identity in unique]
== [identity for identity in report.identities if identity in unique]) if unique else None,
"api_only_statuses": dict(Counter(safe_status(pairs[i][1].get("reservationStatus"), API_STATUSES) for i in api_only)),
"paired_status_counts": [{"api": key[0], "xml": key[1], "records": count}
for key, count in sorted(status_pairs.items())],
"observations": fields, "position_samples_limit": LIMIT,
"unresolved": ["report_inclusion_rules", "source_order_and_row_granularity", "company_role_priority",
"display_name_format", "note_type_and_order", *([] if named else ["trace_scope_and_order"]), "general_rate_interval_boundaries",
"block_code_and_product_display", "historical_room_selection", "shared_and_component_rooms",
"hidden_rate_report_behavior", "source_time_alignment"]}
def compare_files(capture_dir: Path, capture_pin: str, report_path: Path, report_pin: str, *, capture_version="v2") -> tuple[dict, int]:
try:
source.require(isinstance(report_pin, str) and bool(re.fullmatch(r"[0-9a-f]{64}", report_pin)), "invalid_report_pin")
source.require(capture_version in ("v2", "v3"), "unsupported_comparison_capture_version")
protocol = named_audit if capture_version == "v3" else audit
archive = protocol.VerifiedArchive(capture_dir, capture_pin)
raw = native._read(report_path)
source.require(hashlib.sha256(raw).hexdigest() == report_pin, "report_hash_mismatch")
report = native._parse(raw)
result = compare_report(archive, report)
return result, 0 if result["status"] == "compared_observations" else 2
except (source.CollectionError, native.InputError) as error:
return {"status": "invalid_input", "error": str(error), "finance_ready": False,
"report_equivalence_verified": False, "source_mapping_verified": False}, 3
except Exception:
return {"status": "invalid_input", "error": "archive_or_report_invalid", "finance_ready": False,
"report_equivalence_verified": False, "source_mapping_verified": False}, 3
def main(argv=None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--capture-dir", required=True, type=Path)
parser.add_argument("--capture-sha256", required=True)
parser.add_argument("--report", required=True, type=Path)
parser.add_argument("--report-sha256", required=True)
parser.add_argument("--capture-version", choices=("v2", "v3"), default="v2")
args = parser.parse_args(argv)
result, code = compare_files(args.capture_dir, args.capture_sha256, args.report, args.report_sha256,
capture_version=args.capture_version)
print(json.dumps(result, ensure_ascii=False, indent=2))
return code
if __name__ == "__main__":
raise SystemExit(main())