Files
th-hotel-simple/scripts/generate_rate_room_effective_mapping.py
鲨鱼辣椒 694c4317a3 checkpoint: complete recoverable V2 pre-separation baseline
Complete the selective V2 checkpoint with its minimal AgentBus, object-storage, replay persistence, and validated-workbench shared dependency closure.
2026-08-20 17:09:00 +08:00

564 lines
24 KiB
Python

#!/usr/bin/env python3
"""Generate the deployable 0812 Rate-package source mapping from the final XLSX.
The generated catalog is the query layer for:
directory company + explicit FIT/GROUP type + email room label + price
-> base RATE CODE + breakfast
The workbook's Roomtype column remains source provenance only. Runtime Roomtype resolution
uses the independent user-confirmed authority; changing that authority must not rewrite or
change the semantic fingerprint of the Rate-package projection.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import re
import sys
import zipfile
from collections import defaultdict
from pathlib import Path, PurePosixPath
from typing import Any
from xml.etree import ElementTree as ET
EFFECTIVE_MAPPING_SCHEMA_VERSION = 1
EFFECTIVE_MAPPING_VERSION = "rate-room-effective-mapping-v20260812.1"
GENERATOR_CONTRACT_VERSION = "rate-room-effective-mapping-generator-v1"
FINGERPRINT_CONTRACT_VERSION = "canonical-json-v1"
MANIFEST_SCHEMA_VERSION = 1
SOURCE_FILE_NAME = "room&rate 0812.xlsx"
SOURCE_SHA256 = "7fdf2b5a771e69466db2b1bdc7911f978f490905c42ee67d5ae19200343b04d3"
SOURCE_SHEET_NAME = "映射主表"
SOURCE_RANGE = "A1:H214"
SOURCE_HEADERS = [
"公司 / Company",
"规则集类型 / Type",
"邮件内房型 / TYPE OF ROOM",
"实际房型 / Roomtype",
"基础 RATE CODE",
"早餐 / Breakfast",
"含早餐(公式)",
"价格 / Price",
]
EXPECTED_MAPPING_ROWS = 213
EXPECTED_SCENARIOS = 163
EXPECTED_COMPANIES = 8
EXPECTED_TYPES = {"Group&FIT", "Group", "FIT"}
EXPECTED_MULTI_CANDIDATE_SCENARIOS = 27
EXPECTED_MULTI_ROOMTYPE_SCENARIOS = 19
EXPECTED_MULTI_RATE_CODE_SCENARIOS = 10
EXPECTED_MULTI_BREAKFAST_SCENARIOS = 6
EXPECTED_DIRECTORY_VERSION = "rate-room-v20260810"
EXPECTED_CORRECTION_VERSION = "rate-room-knowledge-corrections-v20260812"
EXPECTED_NORMALIZATION_CONTRACT = "source-room-exact-label-v3"
MAIN_NS = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
NS = {"main": MAIN_NS, "rel": REL_NS, "package": PACKAGE_REL_NS}
def fail(message: str) -> None:
raise ValueError(message)
def require(condition: bool, message: str) -> None:
if not condition:
fail(message)
def sha256_bytes(value: bytes) -> str:
return hashlib.sha256(value).hexdigest()
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def canonical_json_bytes(value: Any) -> bytes:
return json.dumps(
value,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
).encode("utf-8")
def semantic_fingerprint(value: Any) -> str:
return sha256_bytes(canonical_json_bytes(value))
def normalize_source_room(value: str) -> str:
normalized = " ".join(value.replace("\u00a0", " ").strip().split()).upper()
return re.sub(r"\s*-\s*", "-", normalized)
def xml_text(element: ET.Element | None) -> str:
if element is None:
return ""
return "".join(element.itertext())
def column_index(cell_reference: str) -> int:
match = re.match(r"([A-Z]+)", cell_reference or "")
if not match:
fail(f"Invalid Excel cell reference: {cell_reference!r}")
result = 0
for character in match.group(1):
result = result * 26 + ord(character) - ord("A") + 1
return result - 1
def load_shared_strings(workbook: zipfile.ZipFile) -> list[str]:
if "xl/sharedStrings.xml" not in workbook.namelist():
return []
root = ET.fromstring(workbook.read("xl/sharedStrings.xml"))
return [xml_text(item) for item in root.findall("main:si", NS)]
def resolve_sheet_path(workbook: zipfile.ZipFile, sheet_name: str) -> str:
workbook_xml = ET.fromstring(workbook.read("xl/workbook.xml"))
sheet = next(
(
item
for item in workbook_xml.findall("main:sheets/main:sheet", NS)
if item.attrib.get("name") == sheet_name
),
None,
)
if sheet is None:
fail(f"Expected worksheet {sheet_name!r} was not found")
relation_id = sheet.attrib.get(f"{{{REL_NS}}}id")
relations = ET.fromstring(workbook.read("xl/_rels/workbook.xml.rels"))
relation = next(
(
item
for item in relations.findall("package:Relationship", NS)
if item.attrib.get("Id") == relation_id
),
None,
)
if relation is None:
fail(f"Worksheet relation {relation_id!r} was not found")
target = relation.attrib["Target"]
if target.startswith("/"):
return target.lstrip("/")
return str(PurePosixPath("xl") / target)
def cell_value(cell: ET.Element, shared_strings: list[str]) -> str:
cell_type = cell.attrib.get("t")
if cell_type == "inlineStr":
return xml_text(cell.find("main:is", NS)).strip()
value = cell.findtext("main:v", default="", namespaces=NS)
if cell_type == "s":
try:
return shared_strings[int(value)].strip()
except (IndexError, ValueError) as error:
fail(f"Invalid shared string index {value!r}: {error}")
return value.strip()
def read_sheet(source: Path) -> dict[int, dict[int, tuple[str, str | None]]]:
with zipfile.ZipFile(source) as workbook:
shared_strings = load_shared_strings(workbook)
root = ET.fromstring(workbook.read(resolve_sheet_path(workbook, SOURCE_SHEET_NAME)))
rows: dict[int, dict[int, tuple[str, str | None]]] = {}
for row in root.findall("main:sheetData/main:row", NS):
row_number = int(row.attrib["r"])
cells: dict[int, tuple[str, str | None]] = {}
for cell in row.findall("main:c", NS):
column = column_index(cell.attrib.get("r", ""))
formula = cell.findtext("main:f", default=None, namespaces=NS)
cells[column] = (cell_value(cell, shared_strings), formula)
rows[row_number] = cells
return rows
def parse_price(value: str, workbook_row: int) -> int:
normalized = value.replace(",", "").replace(" ", "")
if re.fullmatch(r"\d+\.0+", normalized):
normalized = normalized.split(".", 1)[0]
require(bool(re.fullmatch(r"\d+", normalized)), f"Workbook row {workbook_row} has invalid price")
price = int(normalized)
require(price > 0, f"Workbook row {workbook_row} price must be positive")
return price
def parse_workbook(source: Path) -> list[dict[str, Any]]:
require(source.name == SOURCE_FILE_NAME, "Effective mapping workbook file name is not frozen")
require(sha256(source) == SOURCE_SHA256, "Effective mapping workbook SHA-256 does not match 0812 authority")
sheet = read_sheet(source)
headers = [sheet.get(1, {}).get(index, ("", None))[0] for index in range(8)]
require(headers == SOURCE_HEADERS, "Effective mapping workbook A1:H1 headers do not match")
require(
all(
not any(value for column, (value, _formula) in cells.items() if column < 8)
for row_number, cells in sheet.items()
if row_number > EXPECTED_MAPPING_ROWS + 1
),
"Effective mapping workbook contains unexpected A:H data after row 214",
)
mappings: list[dict[str, Any]] = []
seen_rows: set[tuple[Any, ...]] = set()
for workbook_row in range(2, EXPECTED_MAPPING_ROWS + 2):
cells = sheet.get(workbook_row)
require(cells is not None, f"Effective mapping workbook row {workbook_row} is missing")
values = [cells.get(index, ("", None))[0] for index in range(8)]
require(all(value != "" for value in values), f"Effective mapping workbook row {workbook_row} is incomplete")
company, mapping_type, email_room_type, room_type, rate_code, breakfast, included, price_text = values
require(mapping_type in EXPECTED_TYPES, f"Workbook row {workbook_row} has invalid Type")
require(breakfast in {"BUALUANG", "LEELA", "RO"}, f"Workbook row {workbook_row} has invalid Breakfast")
expected_included = breakfast != "RO"
require(included == ("是" if expected_included else "否"), f"Workbook row {workbook_row} breakfast cache is stale")
expected_formula = f'IF(F{workbook_row}="RO","否","是")'
formula = cells.get(6, ("", None))[1]
require(formula == expected_formula, f"Workbook row {workbook_row} breakfast formula is stale")
normalized_room = normalize_source_room(email_room_type)
row = {
"source_workbook_row": workbook_row,
"company": company,
"type": mapping_type,
"email_room_type": email_room_type,
"normalized_email_room_type": normalized_room,
"price": parse_price(price_text, workbook_row),
"room_type": room_type,
"base_rate_code": rate_code,
"breakfast": breakfast,
"breakfast_included": expected_included,
}
semantic_key = tuple(row[field] for field in (
"company",
"type",
"normalized_email_room_type",
"price",
"room_type",
"base_rate_code",
"breakfast",
"breakfast_included",
))
require(semantic_key not in seen_rows, f"Workbook contains duplicate effective mapping at row {workbook_row}")
seen_rows.add(semantic_key)
mappings.append(row)
require(len(mappings) == EXPECTED_MAPPING_ROWS, "Effective mapping row count is not 213")
return mappings
def load_json(path: Path, label: str) -> dict[str, Any]:
try:
value = json.loads(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error:
fail(f"Cannot load {label}: {error}")
require(isinstance(value, dict), f"{label} root must be an object")
return value
def effective_type(company: str, source_type: str) -> str:
return "Group&FIT" if company == "Q.B.D" else source_type
def expected_rows(directory: dict[str, Any], correction: dict[str, Any]) -> list[dict[str, Any]]:
require(directory.get("directory_version") == EXPECTED_DIRECTORY_VERSION, "Base directory version mismatch")
require(correction.get("correction_set_version") == EXPECTED_CORRECTION_VERSION, "Correction version mismatch")
active_labels = {
rule["normalized_literal"]
for rule in correction.get("room_type_source_label_rules", [])
}
rate_codes = {item["code"]: item for item in directory.get("rate_codes", [])}
grouped: dict[tuple[Any, ...], list[dict[str, Any]]] = defaultdict(list)
for source_row in directory.get("mapping_rows", []):
normalized = normalize_source_room(source_row["email_room_type"])
if normalized not in active_labels:
continue
key = (
source_row["company"],
effective_type(source_row["company"], source_row["segment"]),
normalized,
source_row["price"],
source_row["base_rate_code"],
source_row.get("breakfast_restaurant_code") or "RO",
)
grouped[key].append(source_row)
projected: list[tuple[int, dict[str, Any]]] = []
for grouped_rows in grouped.values():
grouped_rows.sort(key=lambda item: item["source_row"])
template = grouped_rows[0]
normalized = normalize_source_room(template["email_room_type"])
rate_definition = rate_codes.get(template["base_rate_code"])
require(rate_definition is not None, "Base directory mapping references an unknown Rate Code")
breakfast = template.get("breakfast_restaurant_code") or "RO"
projected.append((
min(item["source_row"] for item in grouped_rows),
{
"company": template["company"],
"type": effective_type(template["company"], template["segment"]),
"normalized_email_room_type": normalized,
"price": template["price"],
"base_rate_code": template["base_rate_code"],
"breakfast": breakfast,
"breakfast_included": bool(rate_definition["breakfast_included"]),
},
))
projected.sort(key=lambda item: item[0])
return [item[1] for item in projected]
def rate_projection(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Strip workbook-only Roomtype and row identity, preserving first-seen deterministic order."""
fields = (
"company",
"type",
"normalized_email_room_type",
"price",
"base_rate_code",
"breakfast",
"breakfast_included",
)
seen: set[tuple[Any, ...]] = set()
result: list[dict[str, Any]] = []
for row in rows:
key = tuple(row[field] for field in fields)
if key in seen:
continue
seen.add(key)
result.append({field: row[field] for field in fields})
return result
def scenario_summaries(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
groups: dict[tuple[Any, ...], list[dict[str, Any]]] = defaultdict(list)
for row in rows:
key = (
row["company"],
row["type"],
row["normalized_email_room_type"],
row["price"],
)
groups[key].append(row)
summaries: list[dict[str, Any]] = []
for (company, mapping_type, normalized_room, price), candidates in groups.items():
def unique(field: str) -> list[Any]:
return list(dict.fromkeys(item[field] for item in candidates))
summaries.append({
"company": company,
"type": mapping_type,
"normalized_email_room_type": normalized_room,
"price": price,
"email_room_type_labels": unique("email_room_type"),
"source_workbook_rows": [item["source_workbook_row"] for item in candidates],
"candidate_room_types": unique("room_type"),
"candidate_base_rate_codes": unique("base_rate_code"),
"candidate_breakfasts": unique("breakfast"),
"candidate_breakfast_included": unique("breakfast_included"),
"resolution": "UNIQUE" if len(candidates) == 1 else "MULTIPLE_CANDIDATES",
})
summaries.sort(key=lambda item: min(item["source_workbook_rows"]))
return summaries
def mapping_counts(rows: list[dict[str, Any]], scenarios: list[dict[str, Any]]) -> dict[str, int]:
return {
"mapping_rows": len(rows),
"scenarios": len(scenarios),
"companies": len({row["company"] for row in rows}),
"types": len({row["type"] for row in rows}),
"multi_candidate_scenarios": sum(item["resolution"] == "MULTIPLE_CANDIDATES" for item in scenarios),
"multi_roomtype_scenarios": sum(len(item["candidate_room_types"]) > 1 for item in scenarios),
"multi_rate_code_scenarios": sum(len(item["candidate_base_rate_codes"]) > 1 for item in scenarios),
"multi_breakfast_scenarios": sum(len(item["candidate_breakfasts"]) > 1 for item in scenarios),
}
def validate_counts(counts: dict[str, int]) -> None:
expected = {
"mapping_rows": EXPECTED_MAPPING_ROWS,
"scenarios": EXPECTED_SCENARIOS,
"companies": EXPECTED_COMPANIES,
"types": len(EXPECTED_TYPES),
"multi_candidate_scenarios": EXPECTED_MULTI_CANDIDATE_SCENARIOS,
"multi_roomtype_scenarios": EXPECTED_MULTI_ROOMTYPE_SCENARIOS,
"multi_rate_code_scenarios": EXPECTED_MULTI_RATE_CODE_SCENARIOS,
"multi_breakfast_scenarios": EXPECTED_MULTI_BREAKFAST_SCENARIOS,
}
require(counts == expected, f"Effective mapping scenario counts are stale: {counts!r}")
def build_catalog(
source: Path,
directory_path: Path,
correction_path: Path,
) -> dict[str, Any]:
directory = load_json(directory_path, "base Rate/Room directory")
correction = load_json(correction_path, "0812 Roomtype correction")
rows = parse_workbook(source)
expected = expected_rows(directory, correction)
require(
rate_projection(rows) == expected,
"Final workbook Rate projection is not the exact active-label projection of the base directory",
)
canonical_room_types = {item["code"] for item in directory["room_types"]}
rate_codes = {item["code"]: item for item in directory["rate_codes"]}
for row in rows:
require(row["room_type"] in canonical_room_types, "Effective mapping references unknown Roomtype")
rate_definition = rate_codes.get(row["base_rate_code"])
require(rate_definition is not None, "Effective mapping references unknown Rate Code")
expected_breakfast = rate_definition.get("breakfast_restaurant_code") or "RO"
require(row["breakfast"] == expected_breakfast, "Effective mapping breakfast disagrees with Rate Code")
require(row["breakfast_included"] == bool(rate_definition["breakfast_included"]), "Breakfast formula result disagrees with Rate Code")
require((row["company"] == "Q.B.D") == (row["type"] == "Group&FIT"), "Group&FIT is reserved for Q.B.D")
scenarios = scenario_summaries(rows)
counts = mapping_counts(rows, scenarios)
validate_counts(counts)
require(correction["normalization_contract"]["contract_version"] == EXPECTED_NORMALIZATION_CONTRACT, "Normalization contract mismatch")
applies_to = {
"directory_version": directory["directory_version"],
"directory_sha256": sha256(directory_path),
"knowledge_correction_set_version": correction["correction_set_version"],
"knowledge_mapping_content_sha256": correction["mapping_content_sha256"],
}
catalog = {
"effective_mapping_schema_version": EFFECTIVE_MAPPING_SCHEMA_VERSION,
"effective_mapping_version": EFFECTIVE_MAPPING_VERSION,
"generator_contract_version": GENERATOR_CONTRACT_VERSION,
"fingerprint_contract_version": FINGERPRINT_CONTRACT_VERSION,
"source": {
"file_name": SOURCE_FILE_NAME,
"sha256": SOURCE_SHA256,
"sheet_name": SOURCE_SHEET_NAME,
"range": SOURCE_RANGE,
"headers": SOURCE_HEADERS,
"mapping_count": len(rows),
"formula_count": len(rows),
},
"applies_to": applies_to,
"input_contract": {
"company_source": "EMAIL",
"booking_type_source": "EXPLICIT_LAYER4_RATE_OPTION",
"fit_max_inclusive": 4,
"group_min_inclusive": 5,
"company_type_overrides": {"Q.B.D": "Group&FIT"},
"fit_directory_company_overrides": {"LIAN TAI": "LIANTAI ONLINE/DY"},
"email_room_type_source": "EMAIL",
"price_source": "EMAIL_ROOM_TYPE_OR_EXPLICIT_PRICE",
"source_room_normalization_contract": EXPECTED_NORMALIZATION_CONTRACT,
"lookup_key_fields": ["company", "type", "normalized_email_room_type", "price"],
"optional_disambiguation_fields": ["breakfast_restaurant", "same_booking_main_room_base_rate_code"],
"missing_restaurant_default": "BUALUANG",
"suite_inherit_main_room_base_rate_code_companies": ["Q.B.D", "LIAN TAI"],
},
"output_contract": {
"fields": ["room_type", "base_rate_code", "breakfast", "breakfast_included"],
"base_rate_code_only": True,
"breakfast_included_rule": "breakfast != RO",
"zero_candidates": "REVIEW_REQUIRED",
"multiple_candidates": "REVIEW_REQUIRED",
},
"counts": counts,
"mapping_content_sha256": semantic_fingerprint(rows),
"rate_projection_sha256": semantic_fingerprint(rate_projection(rows)),
"scenario_content_sha256": semantic_fingerprint(scenarios),
"mapping_rows": rows,
"scenario_summaries": scenarios,
}
return catalog
def build_manifest(catalog: dict[str, Any], output: Path, output_bytes: bytes) -> dict[str, Any]:
return {
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"effective_mapping_schema_version": EFFECTIVE_MAPPING_SCHEMA_VERSION,
"effective_mapping_version": EFFECTIVE_MAPPING_VERSION,
"generator_contract_version": GENERATOR_CONTRACT_VERSION,
"fingerprint_contract_version": FINGERPRINT_CONTRACT_VERSION,
"effective_mapping": {
"file_name": output.name,
"sha256": sha256_bytes(output_bytes),
"counts": catalog["counts"],
},
"source": catalog["source"],
"applies_to": catalog["applies_to"],
"generator": {
"file_name": Path(__file__).name,
"sha256": sha256(Path(__file__).resolve()),
},
"semantic_fingerprints": {
"mapping_rows": catalog["mapping_content_sha256"],
"rate_projection": catalog["rate_projection_sha256"],
"scenario_summaries": catalog["scenario_content_sha256"],
},
}
def validate_manifest(catalog: dict[str, Any], output_bytes: bytes, manifest: dict[str, Any]) -> None:
require(manifest["manifest_schema_version"] == MANIFEST_SCHEMA_VERSION, "Manifest schema mismatch")
require(manifest["effective_mapping_version"] == EFFECTIVE_MAPPING_VERSION, "Manifest version mismatch")
require(manifest["effective_mapping"]["sha256"] == sha256_bytes(output_bytes), "Manifest artifact SHA mismatch")
require(manifest["effective_mapping"]["counts"] == catalog["counts"], "Manifest counts are stale")
require(manifest["source"] == catalog["source"], "Manifest source provenance is stale")
require(manifest["applies_to"] == catalog["applies_to"], "Manifest binding is stale")
require(
manifest["semantic_fingerprints"]
== {
"mapping_rows": semantic_fingerprint(catalog["mapping_rows"]),
"rate_projection": semantic_fingerprint(rate_projection(catalog["mapping_rows"])),
"scenario_summaries": semantic_fingerprint(catalog["scenario_summaries"]),
},
"Manifest semantic fingerprints are stale",
)
def write_or_check(path: Path, value: str, check: bool) -> None:
expected = value.encode("utf-8")
if check:
require(path.is_file() and path.read_bytes() == expected, f"Generated artifact is stale or missing: {path}")
return
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(expected)
def parse_arguments() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--source", type=Path, required=True, help="Final room&rate 0812 workbook")
parser.add_argument("--directory", type=Path, required=True, help="Frozen 207-row base directory JSON")
parser.add_argument("--correction", type=Path, required=True, help="0812 Roomtype correction JSON")
parser.add_argument("--output", type=Path, required=True, help="Generated effective mapping JSON")
parser.add_argument("--manifest-output", type=Path, required=True, help="Generated manifest JSON")
parser.add_argument("--check", action="store_true", help="Verify generated artifacts are current")
return parser.parse_args()
def main() -> int:
args = parse_arguments()
try:
catalog = build_catalog(args.source, args.directory, args.correction)
rendered_catalog = json.dumps(catalog, ensure_ascii=False, indent=2, sort_keys=False) + "\n"
output_bytes = rendered_catalog.encode("utf-8")
manifest = build_manifest(catalog, args.output, output_bytes)
validate_manifest(catalog, output_bytes, manifest)
rendered_manifest = json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=False) + "\n"
write_or_check(args.output, rendered_catalog, args.check)
write_or_check(args.manifest_output, rendered_manifest, args.check)
except (KeyError, OSError, ValueError, ET.ParseError, zipfile.BadZipFile) as error:
print(f"effective Rate/Room mapping generation failed: {error}", file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())