"""Offline observations comparing a pinned v2/v3 API capture to native ARR XML. This is a research/acceptance aid, not a source adapter or an ingestion gate. Hypotheses remain separate: matching a candidate never adopts a business rule. Output contains aggregate counts and bounded row positions, not guest values. """ from __future__ import annotations import argparse from collections import Counter from dataclasses import dataclass from datetime import datetime from decimal import Decimal, InvalidOperation import hashlib import json from pathlib import Path import re from . import audit_arr_day as audit from . import audit_arr_named_day as named_audit from . import collect_arr_source as source from . import compare_report_xml as native LIMIT = 20 ROLES = ("Company", "TravelAgent", "Source", "Group") ROLE_PREFIX_HYPOTHESES = {"Company": "C- ", "TravelAgent": "T- ", "Source": "S- ", "Group": "G- "} API_STATUSES = {"InHouse", "CheckedOut", "Reserved", "Cancelled", "NoShow", "Waitlist", "InSession", "Prospect"} XML_STATUSES = {"CKIN", "CKOT", "CXL", "NS", "RES", "RSV", "CHECKED IN", "CHECKED OUT", "CANCELLED", "NO SHOW", "RESERVED"} @dataclass(frozen=True) class Candidate: value: object = None issue: str | None = None def scalar(value) -> Candidate: if not isinstance(value, str): return Candidate(issue="missing_or_invalid_text") return Candidate(value.strip()) def literal_text(value) -> Candidate: return Candidate(value) if type(value) is str else Candidate(issue="missing_or_invalid_text") def number(value, *, integer=False, api_integer=False) -> Candidate: if api_integer and type(value) is not int: return Candidate(issue="missing_or_invalid_number") if type(value) not in {str, int, Decimal}: return Candidate(issue="missing_or_invalid_number") text = str(value).strip() if not re.fullmatch(r"[+-]?[0-9]+(?:\.[0-9]+)?(?:[Ee][+-]?[0-9]+)?", text): return Candidate(issue="missing_or_invalid_number") try: result = Decimal(text) except InvalidOperation: return Candidate(issue="missing_or_invalid_number") if not result.is_finite() or result < 0 or (integer and result != result.to_integral_value()): return Candidate(issue="missing_or_invalid_number") return Candidate(result) def report_date(value: str | None) -> Candidate: if not isinstance(value, str): return Candidate(issue="missing_or_invalid_date") # Match the existing XML processor's accepted formats, independently of # the strict API date contract. Do not infer a format from the hotel. for fmt in ("%d-%b-%y", "%d-%m-%y", "%Y%m%d", "%Y-%m-%d", "%m/%d/%Y"): try: return Candidate(datetime.strptime(value.strip(), fmt).date().isoformat()) except ValueError: pass return Candidate(issue="missing_or_invalid_date") def xml_number(value, *, integer=False) -> Candidate: if not isinstance(value, str) or not value.strip(): return Candidate(issue="missing_or_invalid_number") try: # The processor removes commas before Decimal conversion. Keep exact # precision here; normalization must not round a comparison value. result = Decimal(value.replace(",", "").strip()) except InvalidOperation: return Candidate(issue="missing_or_invalid_number") if not result.is_finite() or result < 0 or (integer and result != result.to_integral_value()): return Candidate(issue="missing_or_invalid_number") return Candidate(result) def api_date(value) -> Candidate: if not isinstance(value, str) or not re.fullmatch(r"\d{4}-\d{2}-\d{2}", value): return Candidate(issue="missing_or_invalid_date") return report_date(value) def xml_value(row, field: str, convert=scalar) -> Candidate: values = row.findall(field) if len(values) != 1 or len(values[0]): return Candidate(issue="missing_or_ambiguous_xml_field") return convert(values[0].text or "") def first_nonempty(values: list[str]) -> str: return next((value for value in values if value), "") def value_at(value, *keys): for key in keys: if not isinstance(value, dict): return None value = value.get(key) return value def xml_texts(row, path: str) -> Candidate: container_path = "/".join(path.split("/")[:-2]) if len(row.findall(container_path)) > 1: return Candidate(issue="ambiguous_xml_text_list") # Missing optional XML containers mean no text to the existing processor. # Missing API arrays remain unavailable; they have a different contract. values = row.findall(path) if any(len(value) for value in values): return Candidate(issue="invalid_xml_text_array") return Candidate([text for value in values if (text := (value.text or "").strip())]) def confirmation(detail: dict) -> Candidate: values = [item.get("id") for item in detail["reservationIdList"] if item.get("type") == "Confirmation"] return scalar(values[0]) if len(values) == 1 else Candidate(issue="missing_or_ambiguous_confirmation") def single_day_segment(detail: dict, day: str) -> Candidate: segments = detail.get("roomStay", {}).get("roomRates") if not isinstance(segments, list) or not segments or any(not isinstance(s, dict) for s in segments): return Candidate(issue="missing_or_invalid_rate_segments") # No general inclusive/exclusive interval rule is adopted. The observed # start=end shape can be tested without guessing a boundary convention. if any(api_date(s.get("start")).issue or api_date(s.get("end")).issue or s.get("start") != s.get("end") for s in segments): return Candidate(issue="unsupported_or_invalid_rate_interval") selected = [segment for segment in segments if segment["start"] == day] return Candidate(selected[0]) if len(selected) == 1 else Candidate(issue="missing_or_ambiguous_day_segment") def segment_value(segment: Candidate, name: str, convert=scalar) -> Candidate: return Candidate(issue=segment.issue) if segment.issue else convert(segment.value.get(name)) def company_name(profiles, role: str) -> Candidate: if profiles is None: return Candidate(issue="profiles_not_returned") if not isinstance(profiles, list) or any(not isinstance(p, dict) for p in profiles): return Candidate(issue="invalid_profile_array") selected = [profile for profile in profiles if profile.get("reservationProfileType") == role] if not selected: return Candidate("") if len(selected) != 1: return Candidate(issue="ambiguous_company_role") profile = selected[0].get("profile") company = profile.get("company") if isinstance(profile, dict) else None return scalar(company.get("companyName") if isinstance(company, dict) else None) def prefixed_company(candidate: Candidate, role: str) -> Candidate: # This tests a display hypothesis, not Oracle's role priority or a mapping # rule. Keep the raw-name observation separate, including legitimate blanks. if candidate.issue or candidate.value == "": return candidate if "\n" in candidate.value or "\r" in candidate.value: return Candidate(issue="multiline_company_display_unresolved") return Candidate(ROLE_PREFIX_HYPOTHESES[role] + candidate.value) def primary_name(detail: dict) -> Candidate: guests = detail.get("reservationGuests") if not isinstance(guests, list) or any(not isinstance(g, dict) for g in guests): return Candidate(issue="missing_or_invalid_guests") primary = [guest for guest in guests if guest.get("primary") is True] if len(primary) != 1: return Candidate(issue="missing_or_ambiguous_primary_guest") info = value_at(primary[0], "profileInfo", "profile", "customer", "personName") if not isinstance(info, list) or any(not isinstance(name, dict) for name in info): return Candidate(issue="invalid_person_names") names = [name for name in info if name.get("nameType") == "Primary"] return Candidate(names[0]) if len(names) == 1 else Candidate(issue="missing_or_ambiguous_primary_name") def name_hypotheses(detail: dict) -> dict[str, Candidate]: name = primary_name(detail) keys = ("FULL_NAME_SURNAME_GIVEN", "FULL_NAME_GIVEN_SURNAME", "FULL_NAME_SURNAME_COMMA_GIVEN") if name.issue: return dict.fromkeys(keys, name) surname, given = scalar(name.value.get("surname")), scalar(name.value.get("givenName", "")) if surname.issue or not surname.value or given.issue: return dict.fromkeys(keys, Candidate(issue="missing_or_invalid_name_components")) # Only observations of these simple formats; titles, middle names, aliases # and localization remain unresolved. Never use these to populate a source. return {keys[0]: Candidate(" ".join(filter(None, (surname.value, given.value)))), keys[1]: Candidate(" ".join(filter(None, (given.value, surname.value)))), keys[2]: Candidate(surname.value + (", " + given.value if given.value else ""))} def note_texts(detail: dict, *, selected_gen: bool) -> Candidate: comments = detail.get("comments") if comments is None: return Candidate(issue="comments_not_returned") if not isinstance(comments, list): return Candidate(issue="invalid_comments") values = [] for item in comments: comment = item.get("comment") if isinstance(item, dict) else None if not isinstance(comment, dict): return Candidate(issue="invalid_comment") if selected_gen and not (comment.get("type") == "GEN" and comment.get("notificationLocation") == "RESERVATION"): continue text = comment.get("text") if not isinstance(text, dict) or not isinstance(text.get("value"), str): return Candidate(issue="missing_or_invalid_note_text") if text["value"].strip(): values.append(text["value"].strip()) return Candidate(values) def trace_texts(detail: dict, *, without_resolution_markers=False) -> Candidate: traces = detail.get("traces") if traces is None: return Candidate(issue="traces_not_returned") if not isinstance(traces, list) or any(not isinstance(t, dict) or not isinstance(t.get("traceText"), str) for t in traces): return Candidate(issue="missing_or_invalid_trace_text") values = [] for trace in traces: if without_resolution_markers: info = trace.get("resolveInfo", {}) if not isinstance(info, dict) or any(not isinstance(info.get(key, ""), str) for key in ("resolvedOn", "resolvedBy")): return Candidate(issue="invalid_trace_resolution_info") if any(info.get(key, "").strip() for key in ("resolvedOn", "resolvedBy")): continue if trace["traceText"].strip(): values.append(trace["traceText"].strip()) # Absence of resolution markers is only a hypothesis, not proof the trace # is open. Report department/date filters and ordering also remain unknown. return Candidate(values) def observations(row, search: dict, detail: dict, assessment: dict, day: str, *, named=False) -> dict[str, tuple[Candidate, Candidate]]: stay = detail.get("roomStay", {}) segment = single_day_segment(detail, day) rate = (Candidate(issue=assessment["error"]) if "error" in assessment else number(assessment.get("effective_rate"))) currency = Candidate(issue=assessment["error"]) if "error" in assessment else scalar(assessment.get("currency")) pairs = { "CONFIRMATION_NO": (xml_value(row, "CONFIRMATION_NO"), confirmation(detail)), "ARRIVAL": (xml_value(row, "TRUNC_BEGIN", report_date), api_date(stay.get("arrivalDate"))), "DEPARTURE": (xml_value(row, "TRUNC_END", report_date), api_date(stay.get("departureDate"))), "ADULTS_STAY": (xml_value(row, "ADULTS", lambda x: xml_number(x, integer=True)), number(value_at(stay, "guestCounts", "adults"), integer=True, api_integer=True)), "CHILDREN_STAY": (xml_value(row, "CF_CHILDREN", lambda x: xml_number(x, integer=True)), number(value_at(stay, "guestCounts", "children"), integer=True, api_integer=True)), "ADULTS_SINGLE_DAY_SEGMENT": (xml_value(row, "ADULTS", lambda x: xml_number(x, integer=True)), Candidate(issue=segment.issue) if segment.issue else number(value_at(segment.value, "guestCounts", "adults"), integer=True, api_integer=True)), "CHILDREN_SINGLE_DAY_SEGMENT": (xml_value(row, "CF_CHILDREN", lambda x: xml_number(x, integer=True)), Candidate(issue=segment.issue) if segment.issue else number(value_at(segment.value, "guestCounts", "children"), integer=True, api_integer=True)), "EFFECTIVE_RATE_AMOUNT": (xml_value(row, "EFFECTIVE_RATE_AMOUNT", xml_number), rate), "CURRENCY_CODE": (xml_value(row, "CURRENCY_CODE"), currency), "ROOM_CURRENT": (xml_value(row, "DISP_ROOM_NO"), scalar(value_at(stay, "currentRoomInfo", "roomId"))), "ROOM_SINGLE_DAY_SEGMENT": (xml_value(row, "DISP_ROOM_NO"), segment_value(segment, "roomId")), "RATE_CODE_SINGLE_DAY_SEGMENT": (xml_value(row, "RATE_CODE"), segment_value(segment, "ratePlanCode")), "ROOM_TYPE_CODE_SINGLE_DAY_SEGMENT": (xml_value(row, "ROOM_CATEGORY_LABEL"), segment_value(segment, "roomType")), "ROOM_COUNT_SINGLE_DAY_SEGMENT": (xml_value(row, "NO_OF_ROOMS", lambda x: xml_number(x, integer=True)), segment_value(segment, "numberOfUnits", lambda x: number(x, integer=True, api_integer=True))), "FULL_NAME_SEARCH": (xml_value(row, "FULL_NAME"), scalar(value_at(search, "reservationGuest", "fullName"))), } if named: profile = assessment["profile"] # Rebuilt from pinned originals by v3 protocol replay. exact = (Candidate(issue="primary_summary_name_unavailable") if "error" in profile else literal_text(profile.get("full_name"))) trimmed = Candidate(issue=exact.issue) if exact.issue else scalar(exact.value) pairs["FULL_NAME_PRIMARY_SUMMARY_EXACT"] = (xml_value(row, "FULL_NAME", literal_text), exact) pairs["FULL_NAME_PRIMARY_SUMMARY_TRIMMED"] = (xml_value(row, "FULL_NAME"), trimmed) for key, candidate in name_hypotheses(detail).items(): pairs[key] = (xml_value(row, "FULL_NAME"), candidate) for role in ROLES: profiles = value_at(detail, "reservationProfiles", "reservationProfile") candidate = company_name(profiles, role) pairs["COMPANY_RESERVATION_" + role.upper()] = (xml_value(row, "COMPANY_NAME"), candidate) pairs["COMPANY_RESERVATION_" + role.upper() + "_PREFIXED"] = ( xml_value(row, "COMPANY_NAME"), prefixed_company(candidate, role)) candidate = (Candidate(issue=segment.issue) if segment.issue else company_name(segment.value.get("stayProfiles"), role)) pairs["COMPANY_SINGLE_DAY_" + role.upper()] = (xml_value(row, "COMPANY_NAME"), candidate) pairs["COMPANY_SINGLE_DAY_" + role.upper() + "_PREFIXED"] = ( xml_value(row, "COMPANY_NAME"), prefixed_company(candidate, role)) report_notes = xml_texts(row, "./LIST_G_COMMENT_RESV_NAME_ID/G_COMMENT_RESV_NAME_ID/RES_COMMENT") for name, candidate in (("GEN_RESERVATION", note_texts(detail, selected_gen=True)), ("ALL_RESERVATION_NOTES", note_texts(detail, selected_gen=False)), ("ALL_TRACES", trace_texts(detail)), ("TRACES_WITHOUT_RESOLUTION_MARKERS", trace_texts(detail, without_resolution_markers=True))): target = (xml_texts(row, "./LIST_G_DEPT_ID/G_DEPT_ID/TRACE_TEXT") if "TRACES" in name else report_notes) pairs[name + "_ORDERED_TEXTS"] = (target, candidate) pairs[name + "_FIRST_NONEMPTY"] = ( Candidate(issue=target.issue) if target.issue else Candidate(first_nonempty(target.value)), Candidate(issue=candidate.issue) if candidate.issue else Candidate(first_nonempty(candidate.value))) return pairs def safe_status(value, allowed: set[str]) -> str: return value if isinstance(value, str) and value in allowed else "OTHER_OR_MISSING" def native_observations(report: native.Report) -> dict: """Privacy-safe, single-input observations, even when no API pair exists. Fixed buckets only: no arbitrary XML value becomes an output key. Lexical monotonicity of this sample never establishes a configured sort or tie rule. """ companies, note_types, note_descriptions = Counter(), Counter(), Counter() note_rows = trace_rows = notes = traces = comma_names = no_share_equal = rownum_equal = 0 lexical = {field: [] for field in ("DISP_ROOM_NO", "FULL_NAME", "CONFIRMATION_NO")} for position, row in enumerate(report.rows, 1): company = xml_value(row, "COMPANY_NAME") if company.issue: companies["missing_or_ambiguous"] += 1 elif not company.value: companies["blank"] += 1 else: prefix = next((p for p in ROLE_PREFIX_HYPOTHESES.values() if company.value.startswith(p)), None) companies[(prefix[:2] if prefix else "other_nonblank")] += 1 if "\n" in company.value or "\r" in company.value: companies["multiline"] += 1 name, plain = xml_value(row, "FULL_NAME"), xml_value(row, "FULL_NAME_NO_SHR_IND") comma_names += not name.issue and "," in name.value no_share_equal += not name.issue and not plain.issue and bool(name.value) and name.value == plain.value rownum_equal += xml_value(row, "ROWNUM").value == str(position) for field, values in lexical.items(): value = xml_value(row, field) values.append(value.value.casefold() if not value.issue and value.value else None) row_notes = row.findall("./LIST_G_COMMENT_RESV_NAME_ID/G_COMMENT_RESV_NAME_ID") row_traces = row.findall("./LIST_G_DEPT_ID/G_DEPT_ID") note_rows += bool(row_notes) trace_rows += bool(row_traces) notes += len(row_notes) traces += len(row_traces) for note in row_notes: note_types[safe_status(xml_value(note, "RES_COMMENT_TYPE").value, {"CAS", "GEN"})] += 1 note_descriptions[safe_status(xml_value(note, "RES_COMMENT_DESCRIPTION").value, {"GENERAL"})] += 1 return {"scope": "native_input_only_not_mapping_acceptance", "rows": len(report.rows), "company_display_buckets": dict(companies), "names_containing_comma": comma_names, "nonblank_full_name_equals_no_share_name": no_share_equal, "rownum_matches_physical_position": rownum_equal, "lexical_casefold_ascending_hypotheses": { field: (all(a <= b for a, b in zip(values, values[1:])) if len(values) > 1 and all(v is not None for v in values) else None) for field, values in lexical.items()}, "rows_with_reservation_notes": note_rows, "reservation_note_elements": notes, "note_type_buckets": dict(note_types), "note_description_buckets": dict(note_descriptions), "rows_with_traces": trace_rows, "trace_elements": traces, "configured_sort_verified": False, "company_role_priority_verified": False, "note_type_mapping_verified": False} def compare_report(archive: audit.VerifiedArchive, report: native.Report) -> dict: named = type(archive) is named_audit.VerifiedArchive protocol = named_audit if named else audit result = {"version": "arr-api-native-comparison/v2" if named else "arr-api-native-comparison/v1", "finance_ready": False, "report_equivalence_verified": False, "source_mapping_verified": False, "capture_manifest_sha256": archive.pin, "report_sha256": report.raw_sha256, "api_hotel": archive.options.hotel_id, "report_hotel": report.hotel, "api_date": archive.options.from_date, "report_date": report.arrival_date, "network_calls": 0, "native_observations": native_observations(report)} if (archive.options.hotel_id, archive.options.from_date) != (report.hotel, report.arrival_date): return {**result, "status": "not_comparable", "reason": "hotel_or_date_differs"} for row in report.rows: arrival = xml_value(row, "TRUNC_BEGIN", report_date) if arrival.issue or arrival.value != report.arrival_date: return {**result, "status": "not_comparable", "reason": "report_row_date_missing_or_different"} rows, details, assessments = protocol.replay(archive) counts = Counter(report.identities) api_ids = [source.reservation_id(detail) for detail in details] pairs = {identity: (search, detail, assessment) for identity, search, detail, assessment in zip(api_ids, rows, details, assessments["records"])} unique = {identity for identity in pairs if counts[identity] == 1} api_only = set(pairs) - counts.keys() report_only = counts.keys() - pairs.keys() fields, status_pairs = {}, Counter() for position, (identity, row) in enumerate(zip(report.identities, report.rows), 1): if identity not in unique: continue # Repeated IDs are ambiguous; never pair them by row order. search, detail, assessment = pairs[identity] api_status = safe_status(detail.get("reservationStatus"), API_STATUSES) xml_status = xml_value(row, "SHORT_RESV_STATUS").value status_pairs[(api_status, safe_status(xml_status, XML_STATUSES))] += 1 for name, (expected, actual) in observations(row, search, detail, assessment, archive.options.from_date, named=named).items(): field = fields.setdefault(name, {"compared_rows": 0, "equal_nonblank": 0, "both_blank": 0, "different": 0, "unavailable": 0, "reasons": Counter(), "problem_report_positions": []}) field["compared_rows"] += 1 if expected.issue or actual.issue: field["unavailable"] += 1 for side, issue in (("xml", expected.issue), ("api", actual.issue)): if issue: field["reasons"][side + ":" + issue] += 1 elif expected.value == actual.value: field["both_blank" if expected.value in ("", []) else "equal_nonblank"] += 1 continue else: field["different"] += 1 if len(field["problem_report_positions"]) < LIMIT: field["problem_report_positions"].append(position) return {**result, "status": "compared_observations", "comparison_scope": "candidate_hypotheses_only", "api_records": len(api_ids), "report_records": len(report.rows), "matched_unique_records": len(unique), "api_only_records": len(api_only), "report_only_distinct_ids": len(report_only), "report_only_rows": sum(counts[identity] for identity in report_only), "ambiguous_report_ids": sum(n > 1 for n in counts.values()), "ambiguous_report_rows": sum(n for n in counts.values() if n > 1), "identity_sequence_equal": api_ids == report.identities, "matched_identity_order_equal": ([identity for identity in api_ids if identity in unique] == [identity for identity in report.identities if identity in unique]) if unique else None, "api_only_statuses": dict(Counter(safe_status(pairs[i][1].get("reservationStatus"), API_STATUSES) for i in api_only)), "paired_status_counts": [{"api": key[0], "xml": key[1], "records": count} for key, count in sorted(status_pairs.items())], "observations": fields, "position_samples_limit": LIMIT, "unresolved": ["report_inclusion_rules", "source_order_and_row_granularity", "company_role_priority", "display_name_format", "note_type_and_order", *([] if named else ["trace_scope_and_order"]), "general_rate_interval_boundaries", "block_code_and_product_display", "historical_room_selection", "shared_and_component_rooms", "hidden_rate_report_behavior", "source_time_alignment"]} def compare_files(capture_dir: Path, capture_pin: str, report_path: Path, report_pin: str, *, capture_version="v2") -> tuple[dict, int]: try: source.require(isinstance(report_pin, str) and bool(re.fullmatch(r"[0-9a-f]{64}", report_pin)), "invalid_report_pin") source.require(capture_version in ("v2", "v3"), "unsupported_comparison_capture_version") protocol = named_audit if capture_version == "v3" else audit archive = protocol.VerifiedArchive(capture_dir, capture_pin) raw = native._read(report_path) source.require(hashlib.sha256(raw).hexdigest() == report_pin, "report_hash_mismatch") report = native._parse(raw) result = compare_report(archive, report) return result, 0 if result["status"] == "compared_observations" else 2 except (source.CollectionError, native.InputError) as error: return {"status": "invalid_input", "error": str(error), "finance_ready": False, "report_equivalence_verified": False, "source_mapping_verified": False}, 3 except Exception: return {"status": "invalid_input", "error": "archive_or_report_invalid", "finance_ready": False, "report_equivalence_verified": False, "source_mapping_verified": False}, 3 def main(argv=None) -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--capture-dir", required=True, type=Path) parser.add_argument("--capture-sha256", required=True) parser.add_argument("--report", required=True, type=Path) parser.add_argument("--report-sha256", required=True) parser.add_argument("--capture-version", choices=("v2", "v3"), default="v2") args = parser.parse_args(argv) result, code = compare_files(args.capture_dir, args.capture_sha256, args.report, args.report_sha256, capture_version=args.capture_version) print(json.dumps(result, ensure_ascii=False, indent=2)) return code if __name__ == "__main__": raise SystemExit(main())