#!/usr/bin/env python3 """Generate the deployable 0812 Rate-package source mapping from the final XLSX. The generated catalog is the query layer for: directory company + explicit FIT/GROUP type + email room label + price -> base RATE CODE + breakfast The workbook's Roomtype column remains source provenance only. Runtime Roomtype resolution uses the independent user-confirmed authority; changing that authority must not rewrite or change the semantic fingerprint of the Rate-package projection. """ from __future__ import annotations import argparse import hashlib import json import re import sys import zipfile from collections import defaultdict from pathlib import Path, PurePosixPath from typing import Any from xml.etree import ElementTree as ET EFFECTIVE_MAPPING_SCHEMA_VERSION = 1 EFFECTIVE_MAPPING_VERSION = "rate-room-effective-mapping-v20260812.1" GENERATOR_CONTRACT_VERSION = "rate-room-effective-mapping-generator-v1" FINGERPRINT_CONTRACT_VERSION = "canonical-json-v1" MANIFEST_SCHEMA_VERSION = 1 SOURCE_FILE_NAME = "room&rate 0812.xlsx" SOURCE_SHA256 = "7fdf2b5a771e69466db2b1bdc7911f978f490905c42ee67d5ae19200343b04d3" SOURCE_SHEET_NAME = "映射主表" SOURCE_RANGE = "A1:H214" SOURCE_HEADERS = [ "公司 / Company", "规则集类型 / Type", "邮件内房型 / TYPE OF ROOM", "实际房型 / Roomtype", "基础 RATE CODE", "早餐 / Breakfast", "含早餐(公式)", "价格 / Price", ] EXPECTED_MAPPING_ROWS = 213 EXPECTED_SCENARIOS = 163 EXPECTED_COMPANIES = 8 EXPECTED_TYPES = {"Group&FIT", "Group", "FIT"} EXPECTED_MULTI_CANDIDATE_SCENARIOS = 27 EXPECTED_MULTI_ROOMTYPE_SCENARIOS = 19 EXPECTED_MULTI_RATE_CODE_SCENARIOS = 10 EXPECTED_MULTI_BREAKFAST_SCENARIOS = 6 EXPECTED_DIRECTORY_VERSION = "rate-room-v20260810" EXPECTED_CORRECTION_VERSION = "rate-room-knowledge-corrections-v20260812" EXPECTED_NORMALIZATION_CONTRACT = "source-room-exact-label-v3" MAIN_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" NS = {"main": MAIN_NS, "rel": REL_NS, "package": PACKAGE_REL_NS} def fail(message: str) -> None: raise ValueError(message) def require(condition: bool, message: str) -> None: if not condition: fail(message) def sha256_bytes(value: bytes) -> str: return hashlib.sha256(value).hexdigest() def sha256(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as source: for chunk in iter(lambda: source.read(1024 * 1024), b""): digest.update(chunk) return digest.hexdigest() def canonical_json_bytes(value: Any) -> bytes: return json.dumps( value, ensure_ascii=False, sort_keys=True, separators=(",", ":"), ).encode("utf-8") def semantic_fingerprint(value: Any) -> str: return sha256_bytes(canonical_json_bytes(value)) def normalize_source_room(value: str) -> str: normalized = " ".join(value.replace("\u00a0", " ").strip().split()).upper() return re.sub(r"\s*-\s*", "-", normalized) def xml_text(element: ET.Element | None) -> str: if element is None: return "" return "".join(element.itertext()) def column_index(cell_reference: str) -> int: match = re.match(r"([A-Z]+)", cell_reference or "") if not match: fail(f"Invalid Excel cell reference: {cell_reference!r}") result = 0 for character in match.group(1): result = result * 26 + ord(character) - ord("A") + 1 return result - 1 def load_shared_strings(workbook: zipfile.ZipFile) -> list[str]: if "xl/sharedStrings.xml" not in workbook.namelist(): return [] root = ET.fromstring(workbook.read("xl/sharedStrings.xml")) return [xml_text(item) for item in root.findall("main:si", NS)] def resolve_sheet_path(workbook: zipfile.ZipFile, sheet_name: str) -> str: workbook_xml = ET.fromstring(workbook.read("xl/workbook.xml")) sheet = next( ( item for item in workbook_xml.findall("main:sheets/main:sheet", NS) if item.attrib.get("name") == sheet_name ), None, ) if sheet is None: fail(f"Expected worksheet {sheet_name!r} was not found") relation_id = sheet.attrib.get(f"{{{REL_NS}}}id") relations = ET.fromstring(workbook.read("xl/_rels/workbook.xml.rels")) relation = next( ( item for item in relations.findall("package:Relationship", NS) if item.attrib.get("Id") == relation_id ), None, ) if relation is None: fail(f"Worksheet relation {relation_id!r} was not found") target = relation.attrib["Target"] if target.startswith("/"): return target.lstrip("/") return str(PurePosixPath("xl") / target) def cell_value(cell: ET.Element, shared_strings: list[str]) -> str: cell_type = cell.attrib.get("t") if cell_type == "inlineStr": return xml_text(cell.find("main:is", NS)).strip() value = cell.findtext("main:v", default="", namespaces=NS) if cell_type == "s": try: return shared_strings[int(value)].strip() except (IndexError, ValueError) as error: fail(f"Invalid shared string index {value!r}: {error}") return value.strip() def read_sheet(source: Path) -> dict[int, dict[int, tuple[str, str | None]]]: with zipfile.ZipFile(source) as workbook: shared_strings = load_shared_strings(workbook) root = ET.fromstring(workbook.read(resolve_sheet_path(workbook, SOURCE_SHEET_NAME))) rows: dict[int, dict[int, tuple[str, str | None]]] = {} for row in root.findall("main:sheetData/main:row", NS): row_number = int(row.attrib["r"]) cells: dict[int, tuple[str, str | None]] = {} for cell in row.findall("main:c", NS): column = column_index(cell.attrib.get("r", "")) formula = cell.findtext("main:f", default=None, namespaces=NS) cells[column] = (cell_value(cell, shared_strings), formula) rows[row_number] = cells return rows def parse_price(value: str, workbook_row: int) -> int: normalized = value.replace(",", "").replace(" ", "") if re.fullmatch(r"\d+\.0+", normalized): normalized = normalized.split(".", 1)[0] require(bool(re.fullmatch(r"\d+", normalized)), f"Workbook row {workbook_row} has invalid price") price = int(normalized) require(price > 0, f"Workbook row {workbook_row} price must be positive") return price def parse_workbook(source: Path) -> list[dict[str, Any]]: require(source.name == SOURCE_FILE_NAME, "Effective mapping workbook file name is not frozen") require(sha256(source) == SOURCE_SHA256, "Effective mapping workbook SHA-256 does not match 0812 authority") sheet = read_sheet(source) headers = [sheet.get(1, {}).get(index, ("", None))[0] for index in range(8)] require(headers == SOURCE_HEADERS, "Effective mapping workbook A1:H1 headers do not match") require( all( not any(value for column, (value, _formula) in cells.items() if column < 8) for row_number, cells in sheet.items() if row_number > EXPECTED_MAPPING_ROWS + 1 ), "Effective mapping workbook contains unexpected A:H data after row 214", ) mappings: list[dict[str, Any]] = [] seen_rows: set[tuple[Any, ...]] = set() for workbook_row in range(2, EXPECTED_MAPPING_ROWS + 2): cells = sheet.get(workbook_row) require(cells is not None, f"Effective mapping workbook row {workbook_row} is missing") values = [cells.get(index, ("", None))[0] for index in range(8)] require(all(value != "" for value in values), f"Effective mapping workbook row {workbook_row} is incomplete") company, mapping_type, email_room_type, room_type, rate_code, breakfast, included, price_text = values require(mapping_type in EXPECTED_TYPES, f"Workbook row {workbook_row} has invalid Type") require(breakfast in {"BUALUANG", "LEELA", "RO"}, f"Workbook row {workbook_row} has invalid Breakfast") expected_included = breakfast != "RO" require(included == ("是" if expected_included else "否"), f"Workbook row {workbook_row} breakfast cache is stale") expected_formula = f'IF(F{workbook_row}="RO","否","是")' formula = cells.get(6, ("", None))[1] require(formula == expected_formula, f"Workbook row {workbook_row} breakfast formula is stale") normalized_room = normalize_source_room(email_room_type) row = { "source_workbook_row": workbook_row, "company": company, "type": mapping_type, "email_room_type": email_room_type, "normalized_email_room_type": normalized_room, "price": parse_price(price_text, workbook_row), "room_type": room_type, "base_rate_code": rate_code, "breakfast": breakfast, "breakfast_included": expected_included, } semantic_key = tuple(row[field] for field in ( "company", "type", "normalized_email_room_type", "price", "room_type", "base_rate_code", "breakfast", "breakfast_included", )) require(semantic_key not in seen_rows, f"Workbook contains duplicate effective mapping at row {workbook_row}") seen_rows.add(semantic_key) mappings.append(row) require(len(mappings) == EXPECTED_MAPPING_ROWS, "Effective mapping row count is not 213") return mappings def load_json(path: Path, label: str) -> dict[str, Any]: try: value = json.loads(path.read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: fail(f"Cannot load {label}: {error}") require(isinstance(value, dict), f"{label} root must be an object") return value def effective_type(company: str, source_type: str) -> str: return "Group&FIT" if company == "Q.B.D" else source_type def expected_rows(directory: dict[str, Any], correction: dict[str, Any]) -> list[dict[str, Any]]: require(directory.get("directory_version") == EXPECTED_DIRECTORY_VERSION, "Base directory version mismatch") require(correction.get("correction_set_version") == EXPECTED_CORRECTION_VERSION, "Correction version mismatch") active_labels = { rule["normalized_literal"] for rule in correction.get("room_type_source_label_rules", []) } rate_codes = {item["code"]: item for item in directory.get("rate_codes", [])} grouped: dict[tuple[Any, ...], list[dict[str, Any]]] = defaultdict(list) for source_row in directory.get("mapping_rows", []): normalized = normalize_source_room(source_row["email_room_type"]) if normalized not in active_labels: continue key = ( source_row["company"], effective_type(source_row["company"], source_row["segment"]), normalized, source_row["price"], source_row["base_rate_code"], source_row.get("breakfast_restaurant_code") or "RO", ) grouped[key].append(source_row) projected: list[tuple[int, dict[str, Any]]] = [] for grouped_rows in grouped.values(): grouped_rows.sort(key=lambda item: item["source_row"]) template = grouped_rows[0] normalized = normalize_source_room(template["email_room_type"]) rate_definition = rate_codes.get(template["base_rate_code"]) require(rate_definition is not None, "Base directory mapping references an unknown Rate Code") breakfast = template.get("breakfast_restaurant_code") or "RO" projected.append(( min(item["source_row"] for item in grouped_rows), { "company": template["company"], "type": effective_type(template["company"], template["segment"]), "normalized_email_room_type": normalized, "price": template["price"], "base_rate_code": template["base_rate_code"], "breakfast": breakfast, "breakfast_included": bool(rate_definition["breakfast_included"]), }, )) projected.sort(key=lambda item: item[0]) return [item[1] for item in projected] def rate_projection(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: """Strip workbook-only Roomtype and row identity, preserving first-seen deterministic order.""" fields = ( "company", "type", "normalized_email_room_type", "price", "base_rate_code", "breakfast", "breakfast_included", ) seen: set[tuple[Any, ...]] = set() result: list[dict[str, Any]] = [] for row in rows: key = tuple(row[field] for field in fields) if key in seen: continue seen.add(key) result.append({field: row[field] for field in fields}) return result def scenario_summaries(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: groups: dict[tuple[Any, ...], list[dict[str, Any]]] = defaultdict(list) for row in rows: key = ( row["company"], row["type"], row["normalized_email_room_type"], row["price"], ) groups[key].append(row) summaries: list[dict[str, Any]] = [] for (company, mapping_type, normalized_room, price), candidates in groups.items(): def unique(field: str) -> list[Any]: return list(dict.fromkeys(item[field] for item in candidates)) summaries.append({ "company": company, "type": mapping_type, "normalized_email_room_type": normalized_room, "price": price, "email_room_type_labels": unique("email_room_type"), "source_workbook_rows": [item["source_workbook_row"] for item in candidates], "candidate_room_types": unique("room_type"), "candidate_base_rate_codes": unique("base_rate_code"), "candidate_breakfasts": unique("breakfast"), "candidate_breakfast_included": unique("breakfast_included"), "resolution": "UNIQUE" if len(candidates) == 1 else "MULTIPLE_CANDIDATES", }) summaries.sort(key=lambda item: min(item["source_workbook_rows"])) return summaries def mapping_counts(rows: list[dict[str, Any]], scenarios: list[dict[str, Any]]) -> dict[str, int]: return { "mapping_rows": len(rows), "scenarios": len(scenarios), "companies": len({row["company"] for row in rows}), "types": len({row["type"] for row in rows}), "multi_candidate_scenarios": sum(item["resolution"] == "MULTIPLE_CANDIDATES" for item in scenarios), "multi_roomtype_scenarios": sum(len(item["candidate_room_types"]) > 1 for item in scenarios), "multi_rate_code_scenarios": sum(len(item["candidate_base_rate_codes"]) > 1 for item in scenarios), "multi_breakfast_scenarios": sum(len(item["candidate_breakfasts"]) > 1 for item in scenarios), } def validate_counts(counts: dict[str, int]) -> None: expected = { "mapping_rows": EXPECTED_MAPPING_ROWS, "scenarios": EXPECTED_SCENARIOS, "companies": EXPECTED_COMPANIES, "types": len(EXPECTED_TYPES), "multi_candidate_scenarios": EXPECTED_MULTI_CANDIDATE_SCENARIOS, "multi_roomtype_scenarios": EXPECTED_MULTI_ROOMTYPE_SCENARIOS, "multi_rate_code_scenarios": EXPECTED_MULTI_RATE_CODE_SCENARIOS, "multi_breakfast_scenarios": EXPECTED_MULTI_BREAKFAST_SCENARIOS, } require(counts == expected, f"Effective mapping scenario counts are stale: {counts!r}") def build_catalog( source: Path, directory_path: Path, correction_path: Path, ) -> dict[str, Any]: directory = load_json(directory_path, "base Rate/Room directory") correction = load_json(correction_path, "0812 Roomtype correction") rows = parse_workbook(source) expected = expected_rows(directory, correction) require( rate_projection(rows) == expected, "Final workbook Rate projection is not the exact active-label projection of the base directory", ) canonical_room_types = {item["code"] for item in directory["room_types"]} rate_codes = {item["code"]: item for item in directory["rate_codes"]} for row in rows: require(row["room_type"] in canonical_room_types, "Effective mapping references unknown Roomtype") rate_definition = rate_codes.get(row["base_rate_code"]) require(rate_definition is not None, "Effective mapping references unknown Rate Code") expected_breakfast = rate_definition.get("breakfast_restaurant_code") or "RO" require(row["breakfast"] == expected_breakfast, "Effective mapping breakfast disagrees with Rate Code") require(row["breakfast_included"] == bool(rate_definition["breakfast_included"]), "Breakfast formula result disagrees with Rate Code") require((row["company"] == "Q.B.D") == (row["type"] == "Group&FIT"), "Group&FIT is reserved for Q.B.D") scenarios = scenario_summaries(rows) counts = mapping_counts(rows, scenarios) validate_counts(counts) require(correction["normalization_contract"]["contract_version"] == EXPECTED_NORMALIZATION_CONTRACT, "Normalization contract mismatch") applies_to = { "directory_version": directory["directory_version"], "directory_sha256": sha256(directory_path), "knowledge_correction_set_version": correction["correction_set_version"], "knowledge_mapping_content_sha256": correction["mapping_content_sha256"], } catalog = { "effective_mapping_schema_version": EFFECTIVE_MAPPING_SCHEMA_VERSION, "effective_mapping_version": EFFECTIVE_MAPPING_VERSION, "generator_contract_version": GENERATOR_CONTRACT_VERSION, "fingerprint_contract_version": FINGERPRINT_CONTRACT_VERSION, "source": { "file_name": SOURCE_FILE_NAME, "sha256": SOURCE_SHA256, "sheet_name": SOURCE_SHEET_NAME, "range": SOURCE_RANGE, "headers": SOURCE_HEADERS, "mapping_count": len(rows), "formula_count": len(rows), }, "applies_to": applies_to, "input_contract": { "company_source": "EMAIL", "booking_type_source": "EXPLICIT_LAYER4_RATE_OPTION", "fit_max_inclusive": 4, "group_min_inclusive": 5, "company_type_overrides": {"Q.B.D": "Group&FIT"}, "fit_directory_company_overrides": {"LIAN TAI": "LIANTAI ONLINE/DY"}, "email_room_type_source": "EMAIL", "price_source": "EMAIL_ROOM_TYPE_OR_EXPLICIT_PRICE", "source_room_normalization_contract": EXPECTED_NORMALIZATION_CONTRACT, "lookup_key_fields": ["company", "type", "normalized_email_room_type", "price"], "optional_disambiguation_fields": ["breakfast_restaurant", "same_booking_main_room_base_rate_code"], "missing_restaurant_default": "BUALUANG", "suite_inherit_main_room_base_rate_code_companies": ["Q.B.D", "LIAN TAI"], }, "output_contract": { "fields": ["room_type", "base_rate_code", "breakfast", "breakfast_included"], "base_rate_code_only": True, "breakfast_included_rule": "breakfast != RO", "zero_candidates": "REVIEW_REQUIRED", "multiple_candidates": "REVIEW_REQUIRED", }, "counts": counts, "mapping_content_sha256": semantic_fingerprint(rows), "rate_projection_sha256": semantic_fingerprint(rate_projection(rows)), "scenario_content_sha256": semantic_fingerprint(scenarios), "mapping_rows": rows, "scenario_summaries": scenarios, } return catalog def build_manifest(catalog: dict[str, Any], output: Path, output_bytes: bytes) -> dict[str, Any]: return { "manifest_schema_version": MANIFEST_SCHEMA_VERSION, "effective_mapping_schema_version": EFFECTIVE_MAPPING_SCHEMA_VERSION, "effective_mapping_version": EFFECTIVE_MAPPING_VERSION, "generator_contract_version": GENERATOR_CONTRACT_VERSION, "fingerprint_contract_version": FINGERPRINT_CONTRACT_VERSION, "effective_mapping": { "file_name": output.name, "sha256": sha256_bytes(output_bytes), "counts": catalog["counts"], }, "source": catalog["source"], "applies_to": catalog["applies_to"], "generator": { "file_name": Path(__file__).name, "sha256": sha256(Path(__file__).resolve()), }, "semantic_fingerprints": { "mapping_rows": catalog["mapping_content_sha256"], "rate_projection": catalog["rate_projection_sha256"], "scenario_summaries": catalog["scenario_content_sha256"], }, } def validate_manifest(catalog: dict[str, Any], output_bytes: bytes, manifest: dict[str, Any]) -> None: require(manifest["manifest_schema_version"] == MANIFEST_SCHEMA_VERSION, "Manifest schema mismatch") require(manifest["effective_mapping_version"] == EFFECTIVE_MAPPING_VERSION, "Manifest version mismatch") require(manifest["effective_mapping"]["sha256"] == sha256_bytes(output_bytes), "Manifest artifact SHA mismatch") require(manifest["effective_mapping"]["counts"] == catalog["counts"], "Manifest counts are stale") require(manifest["source"] == catalog["source"], "Manifest source provenance is stale") require(manifest["applies_to"] == catalog["applies_to"], "Manifest binding is stale") require( manifest["semantic_fingerprints"] == { "mapping_rows": semantic_fingerprint(catalog["mapping_rows"]), "rate_projection": semantic_fingerprint(rate_projection(catalog["mapping_rows"])), "scenario_summaries": semantic_fingerprint(catalog["scenario_summaries"]), }, "Manifest semantic fingerprints are stale", ) def write_or_check(path: Path, value: str, check: bool) -> None: expected = value.encode("utf-8") if check: require(path.is_file() and path.read_bytes() == expected, f"Generated artifact is stale or missing: {path}") return path.parent.mkdir(parents=True, exist_ok=True) path.write_bytes(expected) def parse_arguments() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--source", type=Path, required=True, help="Final room&rate 0812 workbook") parser.add_argument("--directory", type=Path, required=True, help="Frozen 207-row base directory JSON") parser.add_argument("--correction", type=Path, required=True, help="0812 Roomtype correction JSON") parser.add_argument("--output", type=Path, required=True, help="Generated effective mapping JSON") parser.add_argument("--manifest-output", type=Path, required=True, help="Generated manifest JSON") parser.add_argument("--check", action="store_true", help="Verify generated artifacts are current") return parser.parse_args() def main() -> int: args = parse_arguments() try: catalog = build_catalog(args.source, args.directory, args.correction) rendered_catalog = json.dumps(catalog, ensure_ascii=False, indent=2, sort_keys=False) + "\n" output_bytes = rendered_catalog.encode("utf-8") manifest = build_manifest(catalog, args.output, output_bytes) validate_manifest(catalog, output_bytes, manifest) rendered_manifest = json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=False) + "\n" write_or_check(args.output, rendered_catalog, args.check) write_or_check(args.manifest_output, rendered_manifest, args.check) except (KeyError, OSError, ValueError, ET.ParseError, zipfile.BadZipFile) as error: print(f"effective Rate/Room mapping generation failed: {error}", file=sys.stderr) return 1 return 0 if __name__ == "__main__": raise SystemExit(main())