Files
th-hotel-simple/scripts/generate_rate_room_directory.py
T
鲨鱼辣椒 694c4317a3 checkpoint: complete recoverable V2 pre-separation baseline
Complete the selective V2 checkpoint with its minimal AgentBus, object-storage, replay persistence, and validated-workbench shared dependency closure.
2026-08-20 17:09:00 +08:00

1183 lines
49 KiB
Python

#!/usr/bin/env python3
"""Generate the v20260810 rate/room directory from the frozen Excel workbook.
The source workbook is intentionally outside the repository. This script records only
its identity and row provenance in generated artifacts, so the directory can be reviewed
and reproduced without committing hotel source files.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import re
import sys
import zipfile
from collections import defaultdict
from pathlib import Path
from typing import Any
from xml.etree import ElementTree as ET
DIRECTORY_VERSION = "rate-room-v20260810"
SCHEMA_VERSION = 1
GENERATOR_CONTRACT_VERSION = "rate-room-directory-generator-v1"
FINGERPRINT_CONTRACT_VERSION = "canonical-json-v1"
MANIFEST_SCHEMA_VERSION = 1
SOURCE_VERSION = "v20260810"
SOURCE_FILE_NAME = "REVISE- RATE CODE_8.7 更新 .xlsx"
SOURCE_SHA256 = "514ad0bf7ef71566fc0995561b5982b855e672fe214071d8f3bc65b5cbe776ca"
SOURCE_SHEET_NAME = "RATECODE-待处理-已处理"
WORD_SOURCE_FILE_NAME = "ประเด็นเกี่ยวกับ RATE CODE(1).docx"
WORD_SOURCE_SHA256 = "78d2ce374214b97188ce9dc7431df8516880db0e3098393402e9463d002fac87"
WORD_CORRECTIONS = (
(173, "HONGYUN", "Group", 4800, "GARDEN", "POOL", "TRP"),
(186, "HONGYUN", "FIT", 5100, "GARDEN", "POOL", "TRP"),
(202, "HANA TOUR", "Group", 3300, "POOL", "GARDEN", "TWN"),
(204, "HANA TOUR", "Group", 3600, "GARDEN", "POOL", "DBL"),
(206, "HANA TOUR", "Group", 4700, "GARDEN", "POOL", "TRP"),
)
EXPECTED_WORD_CORRECTIONS = len(WORD_CORRECTIONS)
EXPECTED_HEADERS = [
"COMPANY'S NAME",
"Type",
"TYPE OF ROOM",
"Roomtype",
"RATE CODE",
"Breakfast",
"Price",
]
EXPECTED_MAPPING_COUNT = 207
EXPECTED_DUPLICATE_CANDIDATE_SCENARIOS = 26
EXPECTED_ROOM_TYPES = {
"RM2",
"RM3",
"RM4",
"SU1",
"SU2",
"SU3",
"SU6",
"UG1",
"UG2",
"UG3",
}
EXPECTED_RATE_CODES = {
"GL2000LT",
"GL2100B",
"GL2100CN",
"GL2200KR",
"GLSPCB",
"GRP1",
"GRPA1",
"GRPA2",
"GRPA3",
"GRPA4",
"LBKB",
"LBLS",
"LBLT",
"LBMS",
"LBSM",
"LBW1",
"LTLT",
"WHKR2100B",
"WHO1",
"WHO2",
"WHO3",
}
# These values were not present in the source workbook. Retain the capacities that
# existed in the previous catalog and leave the newly introduced types unspecified.
ROOM_TYPE_ADULT_CAPACITY = {
"RM2": 2,
"RM3": 2,
"RM4": 4,
"SU1": 2,
"SU2": 2,
"SU3": 4,
}
# Historical V25 codes remain readable in existing tasks, but an incoming value is
# normalized to its current base rate code before validation or persistence.
LEGACY_RATE_CODE_ALIASES = {
"GRPA2-850UP": "GRPA2",
"GRPA2-1275": "GRPA2",
"GRPA2-1400": "GRPA2",
"GRPA2-1800": "GRPA2",
"GRPA2-2400": "GRPA2",
"GRPA1-900": "GRPA1",
"GRPA1-1400": "GRPA1",
"GRPA1-1300": "GRPA1",
"GRPA1-1150": "GRPA1",
"GRPA1-1725": "GRPA1",
"GRPA1-2300": "GRPA1",
"GRPA3-1200*B'FAST BUALUANG": "GRPA3",
"GRPA3-2000*B'FAST BUALUANG": "GRPA3",
"GRPA3-1400*B'FAST BUALUANG": "GRPA3",
"GRPA3-1800*B'FAST BUALUANG": "GRPA3",
"GRPA3-2400*B'FAST BUALUANG": "GRPA3",
"GRPA4-1200*B'FAST LEELA": "GRPA4",
"GRPA4-2000*B'FAST LEELA": "GRPA4",
"GRPA4-1400*B'FAST LEELA": "GRPA4",
"GRPA4-1800*B'FAST LEELA": "GRPA4",
"GRPA4-2400*B'FAST LEELA": "GRPA4",
"WHO1-850UP": "WHO1",
"WHO1-1275": "WHO1",
"WHO1-1400": "WHO1",
"WHO1-1800": "WHO1",
"WHO1-2400": "WHO1",
"GRP1-900": "GRP1",
"GRP1-1400": "GRP1",
"GRP1-1300": "GRP1",
"GRP1-1800": "GRP1",
"GRP1-2400": "GRP1",
"WHO2-1100": "WHO2",
"WHO2-1600": "WHO2",
"WHO2-1400": "WHO2",
"WHO2-1800": "WHO2",
"WHO2-2400": "WHO2",
"WHO3-1200": "WHO3",
"WHO3-1800": "WHO3",
"WHO3-1400": "WHO3",
"WHO3-2400": "WHO3",
}
MAIN_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
NS = {"main": MAIN_NS, "rel": REL_NS, "package": PACKAGE_REL_NS}
def fail(message: str) -> None:
raise ValueError(message)
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def sha256_bytes(value: bytes) -> str:
"""Return a lowercase SHA-256 digest for deterministic generated bytes."""
return hashlib.sha256(value).hexdigest()
def canonical_json_bytes(value: Any) -> bytes:
"""Serialize a semantic value with the versioned canonical-json-v1 contract."""
return json.dumps(
value,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
).encode("utf-8")
def semantic_fingerprint(value: Any) -> str:
return sha256_bytes(canonical_json_bytes(value))
def normalized_code_key(value: Any) -> str:
if not isinstance(value, str):
return ""
return "".join(value.replace("\u00a0", " ").upper().split())
def require(condition: bool, message: str) -> None:
if not condition:
fail(message)
def xml_text(element: ET.Element | None) -> str:
if element is None:
return ""
return "".join(element.itertext())
def column_index(cell_reference: str) -> int:
match = re.match(r"([A-Z]+)", cell_reference or "")
if not match:
fail(f"Invalid Excel cell reference: {cell_reference!r}")
value = 0
for character in match.group(1):
value = value * 26 + ord(character) - ord("A") + 1
return value - 1
def load_shared_strings(workbook: zipfile.ZipFile) -> list[str]:
if "xl/sharedStrings.xml" not in workbook.namelist():
return []
root = ET.fromstring(workbook.read("xl/sharedStrings.xml"))
return [xml_text(item) for item in root.findall("main:si", NS)]
def resolve_sheet_path(workbook: zipfile.ZipFile, sheet_name: str) -> str:
workbook_xml = ET.fromstring(workbook.read("xl/workbook.xml"))
sheet = next(
(item for item in workbook_xml.findall("main:sheets/main:sheet", NS) if item.attrib.get("name") == sheet_name),
None,
)
if sheet is None:
fail(f"Expected worksheet {sheet_name!r} was not found")
relation_id = sheet.attrib.get(f"{{{REL_NS}}}id")
relations = ET.fromstring(workbook.read("xl/_rels/workbook.xml.rels"))
relation = next(
(item for item in relations.findall("package:Relationship", NS) if item.attrib.get("Id") == relation_id),
None,
)
if relation is None:
fail(f"Worksheet relation {relation_id!r} was not found")
target = relation.attrib["Target"]
if target.startswith("/"):
return target.lstrip("/")
return str(Path("xl") / target)
def cell_value(cell: ET.Element, shared_strings: list[str]) -> str:
cell_type = cell.attrib.get("t")
if cell_type == "inlineStr":
return xml_text(cell.find("main:is", NS)).strip()
value = cell.findtext("main:v", default="", namespaces=NS)
if cell_type == "s":
try:
return shared_strings[int(value)].strip()
except (IndexError, ValueError) as error:
fail(f"Invalid shared string index {value!r}: {error}")
return value.strip()
def read_worksheet_rows(source: Path) -> list[tuple[int, list[str]]]:
with zipfile.ZipFile(source) as workbook:
shared_strings = load_shared_strings(workbook)
sheet_path = resolve_sheet_path(workbook, SOURCE_SHEET_NAME)
root = ET.fromstring(workbook.read(sheet_path))
rows: list[tuple[int, list[str]]] = []
for row in root.findall("main:sheetData/main:row", NS):
values: dict[int, str] = {}
for cell in row.findall("main:c", NS):
values[column_index(cell.attrib.get("r", ""))] = cell_value(cell, shared_strings)
width = max(values.keys(), default=-1) + 1
rows.append((int(row.attrib["r"]), [values.get(index, "") for index in range(width)]))
return rows
def parse_price(value: str, source_row: int) -> int:
normalized = value.replace(",", "").replace(" ", "")
if re.fullmatch(r"\d+\.0+", normalized):
normalized = normalized.split(".", 1)[0]
if not re.fullmatch(r"\d+", normalized):
fail(f"Source row {source_row} has a non-integer price: {value!r}")
return int(normalized)
def normalize_label(value: str) -> str:
return " ".join(value.strip().split())
def room_semantic(label: str) -> dict[str, Any]:
normalized = normalize_label(label).upper()
bed_signals: list[str] = []
if re.search(r"\b(?:DBL|DOUBLE|KING)\b", normalized):
bed_signals.append("DBL")
if re.search(r"\b(?:TWN|TWIN)\b", normalized):
bed_signals.append("TWN")
if re.search(r"\b(?:TRP|TRIPLE)\b", normalized):
bed_signals.append("TRP")
room_category = "SUITE" if "SUITE" in normalized else "ROOM"
suite_layout = None
if "TWO-BEDROOM" in normalized or "TWO BEDROOM" in normalized:
suite_layout = "TWO_BEDROOM"
elif "ONE-BEDROOM" in normalized or "ONE BEDROOM" in normalized:
suite_layout = "ONE_BEDROOM"
view = "POOL" if "POOL" in normalized else "GARDEN" if "GARDEN" in normalized else None
meal = "RO" if re.search(r"\b(?:RO|NOBF)\b", normalized) else "BREAKFAST" if "BF" in normalized else None
return {
"normalized": normalized,
"bed_signals": bed_signals,
"room_category": room_category,
"suite_layout": suite_layout,
"view": view,
"meal": meal,
"upgrade": normalized.startswith("U-") or " U-" in normalized,
}
def candidate_scenarios(mappings: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Return source-distinct candidate groups the resolver must not silently collapse."""
groups: dict[tuple[str, str, str, int], list[dict[str, Any]]] = defaultdict(list)
for mapping in mappings:
groups[(
mapping["company"],
mapping["segment"],
mapping["email_room_type"],
mapping["price"],
)].append(mapping)
scenarios = []
for (company, segment, email_room_type, price), rows in groups.items():
if len(rows) < 2:
continue
scenarios.append({
"company": company,
"segment": segment,
"email_room_type": email_room_type,
"price": price,
"source_rows": sorted(row["source_row"] for row in rows),
"candidate_room_types": sorted({row["room_type"] for row in rows}),
"candidate_base_rate_codes": sorted({row["base_rate_code"] for row in rows}),
"candidate_breakfast_restaurant_codes": sorted({
"RO" if row["breakfast_restaurant_code"] is None else row["breakfast_restaurant_code"]
for row in rows
}),
})
return sorted(
scenarios,
key=lambda item: (
item["company"],
item["segment"],
item["email_room_type"],
item["price"],
),
)
def mapping_semantic_value(mapping: dict[str, Any]) -> dict[str, Any]:
"""Return business content without source provenance for true duplicate detection."""
return {
"company": mapping.get("company"),
"segment": mapping.get("segment"),
"email_room_type": mapping.get("email_room_type"),
"room_type": mapping.get("room_type"),
"base_rate_code": mapping.get("base_rate_code"),
"breakfast_restaurant_code": mapping.get("breakfast_restaurant_code"),
"price": mapping.get("price"),
"room_semantic": mapping.get("room_semantic"),
}
def semantic_duplicate_rows(mappings: list[dict[str, Any]]) -> list[list[int]]:
duplicates: dict[bytes, list[int]] = defaultdict(list)
for mapping in mappings:
duplicates[canonical_json_bytes(mapping_semantic_value(mapping))].append(mapping.get("source_row"))
return sorted(
(sorted(rows) for rows in duplicates.values() if len(rows) > 1),
key=lambda rows: rows[0],
)
def canonical_candidate_scenarios(scenarios: list[dict[str, Any]]) -> list[dict[str, Any]]:
return sorted(
scenarios,
key=lambda item: (
item.get("company"),
item.get("segment"),
item.get("email_room_type"),
item.get("price"),
),
)
def normalize_source_room_label(value: str) -> str:
"""Canonicalize a TYPE OF ROOM label for exact authority lookup."""
normalized = normalize_label(value).upper()
return re.sub(r"\s*-\s*", "-", normalized)
def room_type_mapping_projection(directory: dict[str, Any]) -> list[dict[str, Any]]:
"""Project source Roomtype facts without Rate Code fields or provenance."""
grouped: dict[str, dict[str, set[str]]] = {}
for row in directory["mapping_rows"]:
normalized_label = normalize_source_room_label(row["email_room_type"])
item = grouped.setdefault(
normalized_label,
{"source_labels": set(), "candidate_room_types": set()},
)
item["source_labels"].add(row["email_room_type"])
item["candidate_room_types"].add(row["room_type"])
return [
{
"normalized_email_room_type": normalized_label,
"source_labels": sorted(values["source_labels"], key=lambda value: (value.upper(), value)),
"candidate_room_types": sorted(values["candidate_room_types"]),
}
for normalized_label, values in sorted(grouped.items())
]
def rate_code_mapping_projection(directory: dict[str, Any]) -> list[dict[str, Any]]:
"""Project Rate Code decisions without canonical Roomtype as an input."""
grouped: dict[tuple[str, str, str, int], dict[str, set[str | None]]] = {}
for row in directory["mapping_rows"]:
key = (row["company"], row["segment"], row["email_room_type"], row["price"])
item = grouped.setdefault(
key,
{"candidate_base_rate_codes": set(), "candidate_breakfast_restaurant_codes": set()},
)
item["candidate_base_rate_codes"].add(row["base_rate_code"])
item["candidate_breakfast_restaurant_codes"].add(row["breakfast_restaurant_code"])
return [
{
"company": company,
"segment": segment,
"email_room_type": email_room_type,
"price": price,
"candidate_base_rate_codes": sorted(values["candidate_base_rate_codes"]),
"candidate_breakfast_restaurant_codes": sorted(
values["candidate_breakfast_restaurant_codes"],
key=lambda value: "" if value is None else value,
),
}
for (company, segment, email_room_type, price), values in sorted(grouped.items())
]
def semantic_fingerprints(directory: dict[str, Any]) -> dict[str, str]:
"""Fingerprint stable business sections independently from JSON formatting/provenance."""
mapping_values = [mapping_semantic_value(row) for row in directory["mapping_rows"]]
mapping_values.sort(key=canonical_json_bytes)
return {
"room_types": semantic_fingerprint(sorted(directory["room_types"], key=lambda item: item["code"])),
"rate_codes": semantic_fingerprint(sorted(directory["rate_codes"], key=lambda item: item["code"])),
"mapping_rows": semantic_fingerprint(mapping_values),
"room_type_mapping_projection": semantic_fingerprint(room_type_mapping_projection(directory)),
"rate_code_mapping_projection": semantic_fingerprint(rate_code_mapping_projection(directory)),
"candidate_scenarios": semantic_fingerprint(
canonical_candidate_scenarios(directory["candidate_scenarios"])
),
"policies": semantic_fingerprint(directory["policies"]),
"legacy_rate_code_aliases": semantic_fingerprint(
sorted(directory["legacy_rate_code_aliases"], key=lambda item: item["legacy_code"])
),
}
def apply_word_corrections(rows: list[dict[str, Any]]) -> None:
for source_row, company, segment, price, before, after, bed_signal in WORD_CORRECTIONS:
candidates = [
row
for row in rows
if row["source_row"] == source_row
and row["company"] == company
and row["segment"] == segment
and row["price"] == price
and "ONE-BEDROOM-SUITE" in row["email_room_type"].upper()
and before in row["email_room_type"].upper()
]
if len(candidates) != 1:
fail(
f"Word correction for source row {source_row} / {company} / {price} "
f"expected one row, found {len(candidates)}"
)
candidate = candidates[0]
candidate["email_room_type"] = re.sub(
before,
after,
candidate["email_room_type"],
count=1,
flags=re.IGNORECASE,
)
candidate["word_correction"] = f"{company} {price} {bed_signal} {before} -> {after}"
def build_directory(source: Path, word_source: Path) -> dict[str, Any]:
actual_sha256 = sha256(source)
if actual_sha256 != SOURCE_SHA256:
fail(
"Unexpected Excel SHA-256: "
f"expected {SOURCE_SHA256}, received {actual_sha256}."
)
actual_word_sha256 = sha256(word_source)
if actual_word_sha256 != WORD_SOURCE_SHA256:
fail(
"Unexpected Word SHA-256: "
f"expected {WORD_SOURCE_SHA256}, received {actual_word_sha256}."
)
worksheet_rows = read_worksheet_rows(source)
if not worksheet_rows:
fail("The source worksheet is empty")
header_row, headers = worksheet_rows[0]
if headers[: len(EXPECTED_HEADERS)] != EXPECTED_HEADERS:
fail(f"Unexpected headers at row {header_row}: {headers!r}")
mappings: list[dict[str, Any]] = []
for source_row, values in worksheet_rows[1:]:
values = values[: len(EXPECTED_HEADERS)] + [""] * max(0, len(EXPECTED_HEADERS) - len(values))
if not any(value.strip() for value in values):
continue
if any(not value.strip() for value in values):
fail(f"Source row {source_row} has an incomplete mapping: {values!r}")
raw_segment = values[1].strip()
segment = {"GRP": "Group", "Group": "Group", "FIT": "FIT"}.get(raw_segment)
if segment is None:
fail(f"Source row {source_row} has an unsupported Type value: {raw_segment!r}")
breakfast_source = values[5].strip().upper()
if breakfast_source not in {"BUALUANG", "LEELA", "RO"}:
fail(f"Source row {source_row} has an unsupported Breakfast value: {breakfast_source!r}")
mappings.append(
{
"company": values[0].strip(),
"segment": segment,
"email_room_type": normalize_label(values[2]),
"room_type": values[3].strip().upper(),
"base_rate_code": values[4].strip().upper(),
"breakfast_restaurant_code": None if breakfast_source == "RO" else breakfast_source,
"price": parse_price(values[6], source_row),
"source_row": source_row,
"source_version": SOURCE_VERSION,
}
)
if len(mappings) != EXPECTED_MAPPING_COUNT:
fail(f"Expected {EXPECTED_MAPPING_COUNT} mappings, found {len(mappings)}")
apply_word_corrections(mappings)
for mapping in mappings:
mapping["room_semantic"] = room_semantic(mapping["email_room_type"])
mapping.pop("word_correction", None)
duplicate_rows = semantic_duplicate_rows(mappings)
if duplicate_rows:
fail(f"Duplicate source mappings found at rows: {duplicate_rows!r}")
room_types = {mapping["room_type"] for mapping in mappings}
rate_code_restaurants: dict[str, set[str | None]] = defaultdict(set)
for mapping in mappings:
rate_code_restaurants[mapping["base_rate_code"]].add(mapping["breakfast_restaurant_code"])
rate_codes = set(rate_code_restaurants)
if room_types != EXPECTED_ROOM_TYPES:
fail(f"Unexpected Roomtype set: {sorted(room_types)!r}")
if rate_codes != EXPECTED_RATE_CODES:
fail(f"Unexpected base RATE CODE set: {sorted(rate_codes)!r}")
inconsistent_breakfast = {
code: sorted("RO" if restaurant is None else restaurant for restaurant in restaurants)
for code, restaurants in rate_code_restaurants.items()
if len(restaurants) != 1
}
if inconsistent_breakfast:
fail(f"Each base RATE CODE must have one breakfast state: {inconsistent_breakfast!r}")
ordered_room_types = sorted(room_types)
ordered_rate_codes = sorted(rate_codes)
duplicate_candidate_scenarios = candidate_scenarios(mappings)
if len(duplicate_candidate_scenarios) != EXPECTED_DUPLICATE_CANDIDATE_SCENARIOS:
fail(
"Unexpected duplicate candidate scenario count: "
f"{len(duplicate_candidate_scenarios)}"
)
directory = {
"schema_version": SCHEMA_VERSION,
"generator_contract_version": GENERATOR_CONTRACT_VERSION,
"directory_version": DIRECTORY_VERSION,
"source": {
"source_version": SOURCE_VERSION,
"file_name": SOURCE_FILE_NAME,
"sha256": actual_sha256,
"sheet_name": SOURCE_SHEET_NAME,
"headers": EXPECTED_HEADERS,
"mapping_count": EXPECTED_MAPPING_COUNT,
"word_corrections_applied": EXPECTED_WORD_CORRECTIONS,
"word": {
"file_name": WORD_SOURCE_FILE_NAME,
"sha256": actual_word_sha256,
"corrections_applied": EXPECTED_WORD_CORRECTIONS,
},
},
"room_types": [
{
"code": code,
"display_name": code,
"sort_order": (index + 1) * 10,
"adult_capacity": ROOM_TYPE_ADULT_CAPACITY.get(code),
}
for index, code in enumerate(ordered_room_types)
],
"rate_codes": [
{
"code": code,
"display_name": code,
"sort_order": (index + 1) * 10,
"breakfast_included": rate_code_restaurants[code] != {None},
"breakfast_restaurant_code": next(iter(rate_code_restaurants[code])),
}
for index, code in enumerate(ordered_rate_codes)
],
"mapping_rows": mappings,
"candidate_scenarios": duplicate_candidate_scenarios,
"policies": {
"segment": {
"canonicalization": {"GRP": "Group", "Group": "Group", "FIT": "FIT"},
"fixed_by_company": {},
"default_by_room_quantity": {"fit_max_inclusive": 4, "group_min_inclusive": 5},
},
"price": {
"allowed_sources": ["email_room_type", "explicit_price_field"],
"upgrade_marker_decimal_tenths_as_hundreds": True,
"quantity_after_room_label_is_not_price": True,
"examples": {"U-TWN8.5": 850, "U-TWN8.5 12": 850},
},
"room_type": {
"fallback": {"DBL_OR_KING": "RM2", "TWN": "RM3", "UNSPECIFIED_BED": "RM2"},
"company_specific": {
"FENGRUN": {
"SU2_SU6": {"DBL": "SU2", "TWN": "SU6", "UNSPECIFIED": "MANUAL_REVIEW"},
"TRP": "MANUAL_REVIEW",
},
"HONGTAI": {
"SU2_SU6": {"DBL": "SU2", "TWN": "SU6", "UNSPECIFIED": "MANUAL_REVIEW"},
"TRP": "MANUAL_REVIEW",
},
},
"multiple_or_zero_candidates": "MANUAL_REVIEW",
},
"rate_code": {
"output": "base_rate_code",
"priority": ["explicit_restaurant", "explicit_ro", "default_bualuang_when_only_restaurant_differs"],
"qbd_u1200": {"default": "GRPA3", "leela": "GRPA4"},
"suite_rate": {"companies": ["Q.B.D", "LIAN TAI"], "inherit_main_room": True, "suite_only": "MANUAL_REVIEW"},
"missing_price_default_base_rate_code": "GRPA1",
},
},
"legacy_rate_code_aliases": [
{"legacy_code": legacy_code, "base_rate_code": base_rate_code}
for legacy_code, base_rate_code in sorted(LEGACY_RATE_CODE_ALIASES.items())
],
}
validate_directory(directory)
return directory
def validate_directory(directory: dict[str, Any]) -> None:
require(isinstance(directory, dict), "Generated directory root must be an object")
require(directory.get("schema_version") == SCHEMA_VERSION, "Unexpected directory schema version")
require(
directory.get("generator_contract_version") == GENERATOR_CONTRACT_VERSION,
"Unexpected generator contract version",
)
require(directory.get("directory_version") == DIRECTORY_VERSION, "Unexpected directory version")
source = directory.get("source")
require(isinstance(source, dict), "Generated directory source must be an object")
require(source.get("source_version") == SOURCE_VERSION, "Generated directory lost the source version")
require(source.get("file_name") == SOURCE_FILE_NAME, "Generated directory lost the frozen Excel file name")
require(source.get("sha256") == SOURCE_SHA256, "Generated directory lost the frozen Excel hash")
require(source.get("sheet_name") == SOURCE_SHEET_NAME, "Generated directory lost the worksheet name")
require(source.get("headers") == EXPECTED_HEADERS, "Generated directory lost the expected headers")
require(
source.get("mapping_count") == EXPECTED_MAPPING_COUNT,
"Generated directory source metadata has an unexpected mapping count",
)
require(
source.get("word_corrections_applied") == EXPECTED_WORD_CORRECTIONS,
"Generated directory source metadata has an unexpected Word correction count",
)
word_source = source.get("word")
require(isinstance(word_source, dict), "Generated directory lost the frozen Word provenance")
require(word_source.get("file_name") == WORD_SOURCE_FILE_NAME, "Unexpected frozen Word file name")
require(word_source.get("sha256") == WORD_SOURCE_SHA256, "Unexpected frozen Word SHA-256")
require(
word_source.get("corrections_applied") == EXPECTED_WORD_CORRECTIONS,
"Unexpected frozen Word correction count",
)
room_types = directory.get("room_types")
rate_codes = directory.get("rate_codes")
rows = directory.get("mapping_rows")
scenarios = directory.get("candidate_scenarios")
aliases = directory.get("legacy_rate_code_aliases")
policies = directory.get("policies")
require(isinstance(room_types, list), "Generated directory room_types must be an array")
require(isinstance(rate_codes, list), "Generated directory rate_codes must be an array")
require(isinstance(rows, list), "Generated directory mapping_rows must be an array")
require(isinstance(scenarios, list), "Generated directory candidate_scenarios must be an array")
require(isinstance(aliases, list), "Generated directory aliases must be an array")
require(isinstance(policies, dict), "Generated directory policies must be an object")
require(len(room_types) == len(EXPECTED_ROOM_TYPES), "Generated directory must contain 10 Roomtypes")
require(len(rate_codes) == len(EXPECTED_RATE_CODES), "Generated directory must contain 21 base RATE CODEs")
require(len(rows) == EXPECTED_MAPPING_COUNT, "Generated directory has an unexpected mapping row count")
require(
len(scenarios) == EXPECTED_DUPLICATE_CANDIDATE_SCENARIOS,
"Generated directory has an unexpected duplicate candidate scenario count",
)
require(len(aliases) == len(LEGACY_RATE_CODE_ALIASES), "Legacy aliases must cover all 40 prior codes")
room_type_by_code: dict[str, dict[str, Any]] = {}
room_type_keys: set[str] = set()
room_sort_orders: set[int] = set()
for room_type in room_types:
require(isinstance(room_type, dict), "Roomtype definition must be an object")
code = room_type.get("code")
require(isinstance(code, str) and code == code.strip().upper() and code, "Invalid canonical Roomtype code")
code_key = normalized_code_key(code)
require(code_key and code_key not in room_type_keys, f"Duplicate normalized Roomtype code: {code!r}")
room_type_keys.add(code_key)
room_type_by_code[code] = room_type
require(
isinstance(room_type.get("display_name"), str) and room_type["display_name"].strip(),
f"Roomtype {code} has no display name",
)
sort_order = room_type.get("sort_order")
require(
isinstance(sort_order, int) and not isinstance(sort_order, bool) and sort_order > 0,
f"Roomtype {code} has an invalid sort order",
)
require(sort_order not in room_sort_orders, f"Duplicate Roomtype sort order: {sort_order}")
room_sort_orders.add(sort_order)
capacity = room_type.get("adult_capacity")
require(
capacity is None or (isinstance(capacity, int) and not isinstance(capacity, bool) and capacity > 0),
f"Roomtype {code} has an invalid adult capacity",
)
require(set(room_type_by_code) == EXPECTED_ROOM_TYPES, "Generated directory has an unexpected Roomtype set")
rate_code_by_code: dict[str, dict[str, Any]] = {}
rate_code_keys: set[str] = set()
rate_sort_orders: set[int] = set()
for rate_code in rate_codes:
require(isinstance(rate_code, dict), "Rate Code definition must be an object")
code = rate_code.get("code")
require(isinstance(code, str) and code == code.strip().upper() and code, "Invalid canonical Rate Code")
code_key = normalized_code_key(code)
require(code_key and code_key not in rate_code_keys, f"Duplicate normalized Rate Code: {code!r}")
rate_code_keys.add(code_key)
rate_code_by_code[code] = rate_code
require(
isinstance(rate_code.get("display_name"), str) and rate_code["display_name"].strip(),
f"Rate Code {code} has no display name",
)
sort_order = rate_code.get("sort_order")
require(
isinstance(sort_order, int) and not isinstance(sort_order, bool) and sort_order > 0,
f"Rate Code {code} has an invalid sort order",
)
require(sort_order not in rate_sort_orders, f"Duplicate Rate Code sort order: {sort_order}")
rate_sort_orders.add(sort_order)
included = rate_code.get("breakfast_included")
restaurant = rate_code.get("breakfast_restaurant_code")
require(isinstance(included, bool), f"Rate Code {code} has an invalid breakfast flag")
require(restaurant in {None, "BUALUANG", "LEELA"}, f"Rate Code {code} has an invalid restaurant")
require(included == (restaurant is not None), f"Breakfast state is inconsistent for {code}")
require(set(rate_code_by_code) == EXPECTED_RATE_CODES, "Generated directory has an unexpected base RATE CODE set")
source_rows: set[int] = set()
for row in rows:
require(isinstance(row, dict), "Mapping row must be an object")
source_row = row.get("source_row")
require(
isinstance(source_row, int) and not isinstance(source_row, bool) and source_row > 0,
"Mapping row has an invalid source_row",
)
require(source_row not in source_rows, f"Duplicate source_row: {source_row}")
source_rows.add(source_row)
require(row.get("source_version") == SOURCE_VERSION, f"Source row {source_row} has an invalid source version")
for field_name in ("company", "email_room_type", "room_type", "base_rate_code"):
require(
isinstance(row.get(field_name), str) and row[field_name].strip(),
f"Source row {source_row} has an empty {field_name}",
)
require(row.get("segment") in {"Group", "FIT"}, f"Source row {source_row} has an invalid segment")
price = row.get("price")
require(
isinstance(price, int) and not isinstance(price, bool) and price > 0,
f"Source row {source_row} has an invalid price",
)
room_type = row["room_type"]
base_rate_code = row["base_rate_code"]
require(room_type in room_type_by_code, f"Source row {source_row} references unknown Roomtype {room_type}")
require(
base_rate_code in rate_code_by_code,
f"Source row {source_row} references unknown Rate Code {base_rate_code}",
)
require(
row.get("breakfast_restaurant_code")
== rate_code_by_code[base_rate_code]["breakfast_restaurant_code"],
f"Source row {source_row} breakfast state differs from Rate Code {base_rate_code}",
)
require(
row.get("room_semantic") == room_semantic(row["email_room_type"]),
f"Source row {source_row} has stale room semantics",
)
duplicate_rows = semantic_duplicate_rows(rows)
require(not duplicate_rows, f"Duplicate semantic source mappings found at rows: {duplicate_rows!r}")
require({row["room_type"] for row in rows} == EXPECTED_ROOM_TYPES, "Mappings do not cover all Roomtypes")
require({row["base_rate_code"] for row in rows} == EXPECTED_RATE_CODES, "Mappings do not cover all Rate Codes")
row_by_source = {row["source_row"]: row for row in rows}
for source_row, company, segment, price, before, after, bed_signal in WORD_CORRECTIONS:
corrected = row_by_source.get(source_row)
require(corrected is not None, f"Word correction source row {source_row} is missing")
require(
corrected["company"] == company
and corrected["segment"] == segment
and corrected["price"] == price
and before not in corrected["email_room_type"].upper()
and after in corrected["email_room_type"].upper()
and bed_signal in corrected["room_semantic"]["bed_signals"]
and corrected["room_semantic"]["view"] == after,
f"Word correction validation failed for source row {source_row}",
)
computed_scenarios = candidate_scenarios(rows)
require(
canonical_candidate_scenarios(scenarios) == computed_scenarios,
"Candidate scenarios are stale, incomplete, or contain silent collapses",
)
alias_keys: set[str] = set()
for alias in aliases:
require(isinstance(alias, dict), "Legacy Rate Code alias must be an object")
legacy_code = alias.get("legacy_code")
base_rate_code = alias.get("base_rate_code")
require(isinstance(legacy_code, str) and legacy_code.strip(), "Legacy Rate Code alias is empty")
alias_key = normalized_code_key(legacy_code)
require(alias_key and alias_key not in alias_keys, f"Duplicate normalized legacy alias: {legacy_code}")
require(alias_key not in rate_code_keys, f"Legacy alias conflicts with canonical Rate Code: {legacy_code}")
alias_keys.add(alias_key)
require(
base_rate_code in rate_code_by_code,
f"Legacy alias {legacy_code} points outside the canonical Rate Code set",
)
expected_segment_policy = {
"canonicalization": {"GRP": "Group", "Group": "Group", "FIT": "FIT"},
"fixed_by_company": {},
"default_by_room_quantity": {"fit_max_inclusive": 4, "group_min_inclusive": 5},
}
expected_price_policy = {
"allowed_sources": ["email_room_type", "explicit_price_field"],
"upgrade_marker_decimal_tenths_as_hundreds": True,
"quantity_after_room_label_is_not_price": True,
"examples": {"U-TWN8.5": 850, "U-TWN8.5 12": 850},
}
expected_room_type_policy = {
"fallback": {"DBL_OR_KING": "RM2", "TWN": "RM3", "UNSPECIFIED_BED": "RM2"},
"company_specific": {
"FENGRUN": {
"SU2_SU6": {"DBL": "SU2", "TWN": "SU6", "UNSPECIFIED": "MANUAL_REVIEW"},
"TRP": "MANUAL_REVIEW",
},
"HONGTAI": {
"SU2_SU6": {"DBL": "SU2", "TWN": "SU6", "UNSPECIFIED": "MANUAL_REVIEW"},
"TRP": "MANUAL_REVIEW",
},
},
"multiple_or_zero_candidates": "MANUAL_REVIEW",
}
require(policies.get("segment") == expected_segment_policy, "Unexpected segment policy")
require(policies.get("price") == expected_price_policy, "Unexpected price policy")
require(policies.get("room_type") == expected_room_type_policy, "Unexpected Roomtype policy")
rate_code_policy = policies.get("rate_code")
require(isinstance(rate_code_policy, dict), "Rate Code policy must be an object")
require(rate_code_policy.get("output") == "base_rate_code", "Rate Code output must be canonical base code")
require(
rate_code_policy.get("priority")
== ["explicit_restaurant", "explicit_ro", "default_bualuang_when_only_restaurant_differs"],
"Unexpected Rate Code priority policy",
)
require(
rate_code_policy.get("qbd_u1200") == {"default": "GRPA3", "leela": "GRPA4"},
"Unexpected Q.B.D U1200 policy",
)
require(
rate_code_policy.get("suite_rate")
== {"companies": ["Q.B.D", "LIAN TAI"], "inherit_main_room": True, "suite_only": "MANUAL_REVIEW"},
"Unexpected suite Rate Code policy",
)
missing_price_default = rate_code_policy.get("missing_price_default_base_rate_code")
require(
missing_price_default in rate_code_by_code,
"Missing-price default must reference a canonical base Rate Code",
)
require(missing_price_default == "GRPA1", "Missing-price default must remain GRPA1")
def sql_literal(value: str | int | None) -> str:
if value is None:
return "NULL"
if isinstance(value, int):
return str(value)
return "'" + value.replace("'", "''") + "'"
def generated_migration(directory: dict[str, Any]) -> str:
seed_rows: list[tuple[int, str, str, str, str | None]] = []
for room_type in directory["room_types"]:
metadata: dict[str, Any] = {}
if room_type["adult_capacity"] is not None:
metadata["adult_capacity"] = room_type["adult_capacity"]
seed_rows.append((
room_type["sort_order"],
"ROOM_TYPE",
room_type["code"],
room_type["display_name"],
json.dumps(metadata, separators=(",", ":"), ensure_ascii=False) if metadata else None,
))
for rate_code in directory["rate_codes"]:
metadata = {
"pricing_available": False,
"breakfast_included": rate_code["breakfast_included"],
"breakfast_restaurant_code": rate_code["breakfast_restaurant_code"],
}
seed_rows.append((
100 + rate_code["sort_order"],
"RATE_CODE",
rate_code["code"],
rate_code["display_name"],
json.dumps(metadata, separators=(",", ":"), ensure_ascii=False),
))
seed_sql = "\n UNION ALL ".join(
"SELECT "
f"{sort_order} AS sort_order, {sql_literal(catalog_type)} AS catalog_type, "
f"{sql_literal(code)} AS code, {sql_literal(display_name)} AS display_name, "
f"{sql_literal(metadata_json)} AS metadata_json"
for sort_order, catalog_type, code, display_name, metadata_json in seed_rows
)
return f"""-- M012 房型与基础 RATE CODE 目录:由 scripts/generate_rate_room_directory.py 从冻结 Excel 生成;不修改历史迁移。
-- 仅替换旧的固定种子目录。既有任务保存的组合 RATE CODE 值不被改写,读取时由应用层兼容解析。
DELETE FROM workflow_reservation_catalog_code
WHERE catalog_type IN ('ROOM_TYPE', 'RATE_CODE')
AND source_system = 'FIXED_SEED_IMPORT'
AND (
hotel_id IN ('HOTEL-TEST', 'HOTEL-DEV')
OR hotel_id IN (
SELECT hotel_id
FROM platform_hotel
WHERE hotel_status = 'ACTIVE'
)
);
INSERT INTO workflow_reservation_catalog_code (
id, hotel_id, catalog_type, code, display_name, status, source_system, external_id, sort_order,
catalog_version, last_synced_at, metadata_json, version, created_at, updated_at, logic_deleted_at,
logic_deleted_reason
)
SELECT
2700000000000000 + ROW_NUMBER() OVER (ORDER BY rate_room_hotels.hotel_id, seed.sort_order),
rate_room_hotels.hotel_id,
seed.catalog_type,
seed.code,
seed.display_name,
'ACTIVE',
'SYSTEM_MANAGED',
NULL,
seed.sort_order,
'{DIRECTORY_VERSION}',
NULL,
seed.metadata_json,
0,
'2026-08-10 00:00:00.000000',
'2026-08-10 00:00:00.000000',
NULL,
NULL
FROM (
SELECT hotel_id
FROM (
SELECT 'HOTEL-TEST' AS hotel_id
UNION ALL SELECT 'HOTEL-DEV'
UNION ALL
SELECT hotel_id
FROM platform_hotel
WHERE hotel_status = 'ACTIVE'
) raw_hotels
WHERE hotel_id IS NOT NULL
AND hotel_id <> ''
GROUP BY hotel_id
) rate_room_hotels
JOIN (
{seed_sql}
) seed ON 1 = 1
WHERE NOT EXISTS (
SELECT 1
FROM workflow_reservation_catalog_code existing
WHERE existing.hotel_id = rate_room_hotels.hotel_id
AND existing.catalog_type = seed.catalog_type
AND existing.code = seed.code
);
"""
def write_or_check(path: Path, rendered: str, check: bool) -> None:
expected = rendered.encode("utf-8")
if check:
if not path.is_file() or path.read_bytes() != expected:
fail(f"Generated artifact is stale or missing: {path}")
return
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(expected)
def build_manifest(
directory: dict[str, Any],
directory_path: Path,
directory_bytes: bytes,
migration_path: Path,
migration_bytes: bytes,
skill_output: Path | None,
) -> dict[str, Any]:
"""Build a deterministic sidecar manifest without embedding machine-specific paths."""
source = directory["source"]
derived_artifacts: dict[str, dict[str, str]] = {
"migration": {
"file_name": migration_path.name,
"sha256": sha256_bytes(migration_bytes),
}
}
if skill_output is not None:
derived_artifacts["skill_directory_copy"] = {
"file_name": skill_output.name,
"sha256": sha256_bytes(directory_bytes),
}
return {
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"schema_version": directory["schema_version"],
"directory_version": directory["directory_version"],
"generator_contract_version": directory["generator_contract_version"],
"fingerprint_contract_version": FINGERPRINT_CONTRACT_VERSION,
"directory": {
"file_name": directory_path.name,
"sha256": sha256_bytes(directory_bytes),
"counts": {
"mapping_rows": len(directory["mapping_rows"]),
"room_types": len(directory["room_types"]),
"rate_codes": len(directory["rate_codes"]),
"legacy_rate_code_aliases": len(directory["legacy_rate_code_aliases"]),
"candidate_scenarios": len(directory["candidate_scenarios"]),
"word_corrections": source["word"]["corrections_applied"],
},
},
"sources": {
"excel": {
"file_name": source["file_name"],
"sha256": source["sha256"],
"sheet_name": source["sheet_name"],
},
"word": {
"file_name": source["word"]["file_name"],
"sha256": source["word"]["sha256"],
},
},
"generator": {
"file_name": Path(__file__).name,
"sha256": sha256(Path(__file__).resolve()),
},
"semantic_fingerprints": semantic_fingerprints(directory),
"derived_artifacts": derived_artifacts,
}
def validate_manifest(directory_bytes: bytes, manifest: dict[str, Any]) -> None:
"""Validate deployment-critical manifest facts independently from file writes."""
require(manifest.get("manifest_schema_version") == MANIFEST_SCHEMA_VERSION, "Unexpected manifest schema")
require(manifest.get("schema_version") == SCHEMA_VERSION, "Manifest directory schema mismatch")
require(manifest.get("directory_version") == DIRECTORY_VERSION, "Manifest directory version mismatch")
require(
manifest.get("generator_contract_version") == GENERATOR_CONTRACT_VERSION,
"Manifest generator contract mismatch",
)
require(
manifest.get("fingerprint_contract_version") == FINGERPRINT_CONTRACT_VERSION,
"Manifest fingerprint contract mismatch",
)
directory_entry = manifest.get("directory")
require(isinstance(directory_entry, dict), "Manifest directory entry is missing")
require(directory_entry.get("sha256") == sha256_bytes(directory_bytes), "Manifest directory SHA-256 mismatch")
try:
directory = json.loads(directory_bytes.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as error:
fail(f"Manifest directory bytes are not valid UTF-8 JSON: {error}")
validate_directory(directory)
expected_counts = {
"mapping_rows": len(directory["mapping_rows"]),
"room_types": len(directory["room_types"]),
"rate_codes": len(directory["rate_codes"]),
"legacy_rate_code_aliases": len(directory["legacy_rate_code_aliases"]),
"candidate_scenarios": len(directory["candidate_scenarios"]),
"word_corrections": directory["source"]["word"]["corrections_applied"],
}
require(directory_entry.get("counts") == expected_counts, "Manifest directory counts are stale")
require(
manifest.get("sources")
== {
"excel": {
"file_name": directory["source"]["file_name"],
"sha256": directory["source"]["sha256"],
"sheet_name": directory["source"]["sheet_name"],
},
"word": {
"file_name": directory["source"]["word"]["file_name"],
"sha256": directory["source"]["word"]["sha256"],
},
},
"Manifest source provenance is stale",
)
fingerprints = manifest.get("semantic_fingerprints")
require(
isinstance(fingerprints, dict)
and set(fingerprints)
== {
"room_types",
"rate_codes",
"mapping_rows",
"room_type_mapping_projection",
"rate_code_mapping_projection",
"candidate_scenarios",
"policies",
"legacy_rate_code_aliases",
},
"Manifest semantic fingerprints are incomplete",
)
require(
all(isinstance(value, str) and re.fullmatch(r"[0-9a-f]{64}", value) for value in fingerprints.values()),
"Manifest contains an invalid semantic fingerprint",
)
require(fingerprints == semantic_fingerprints(directory), "Manifest semantic fingerprints are stale")
generator = manifest.get("generator")
require(isinstance(generator, dict), "Manifest generator entry is missing")
require(generator.get("file_name") == Path(__file__).name, "Manifest generator file name is stale")
require(generator.get("sha256") == sha256(Path(__file__).resolve()), "Manifest generator SHA-256 is stale")
def parse_arguments() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--source", type=Path, required=True, help="Frozen updated Excel workbook")
parser.add_argument("--word-source", type=Path, required=True, help="Frozen Word clarification source")
parser.add_argument("--output", type=Path, required=True, help="Generated JSON directory path")
parser.add_argument("--manifest-output", type=Path, required=True, help="Generated directory manifest path")
parser.add_argument("--migration-output", type=Path, required=True, help="Generated Flyway migration path")
parser.add_argument("--skill-output", type=Path, help="Optional generated JSON copy in the editable Booking Skill")
parser.add_argument("--check", action="store_true", help="Verify artifacts are already generated and current")
return parser.parse_args()
def main() -> int:
args = parse_arguments()
try:
directory = build_directory(args.source, args.word_source)
rendered_directory = json.dumps(directory, ensure_ascii=False, indent=2, sort_keys=False) + "\n"
directory_bytes = rendered_directory.encode("utf-8")
rendered_migration = generated_migration(directory)
migration_bytes = rendered_migration.encode("utf-8")
manifest = build_manifest(
directory,
args.output,
directory_bytes,
args.migration_output,
migration_bytes,
args.skill_output,
)
validate_manifest(directory_bytes, manifest)
rendered_manifest = json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=False) + "\n"
write_or_check(args.output, rendered_directory, args.check)
write_or_check(args.migration_output, rendered_migration, args.check)
if args.skill_output is not None:
write_or_check(args.skill_output, rendered_directory, args.check)
write_or_check(args.manifest_output, rendered_manifest, args.check)
except (OSError, ValueError, ET.ParseError, zipfile.BadZipFile) as error:
print(f"rate-room directory generation failed: {error}", file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())