Files
wyndham-ARR/arr-opera-daily-ingest/scripts/validate_daily.py
2026-08-06 22:40:18 +08:00

1353 lines
48 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""Independently replay and validate one Opera daily processing result."""
from __future__ import annotations
import argparse
import copy
import json
import sys
import traceback
from datetime import date, datetime
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
from openpyxl import load_workbook
from openpyxl.utils import get_column_letter
import process_daily as core
TEXT_FIELDS = {
"BLOCK_CODE",
"COMPANY_NAME",
"CONFIRMATION_NO",
"DISP_ROOM_NO",
"FULL_NAME",
"RES_COMMENT",
"TRACE_TEXT",
"PRODUCTS",
"RATE_CODE",
"ROOM_CATEGORY_LABEL",
}
DATE_FIELDS = {"ARRIVAL", "DEPARTURE"}
DAILY_NUMERIC_FIELDS = {
"ADULTS",
"CHILDREN",
"EFFECTIVE_RATE_AMOUNT",
"NO_OF_ROOMS",
"NIGHTS",
"REAL PRICE",
"TOTAL PRICE",
}
REQUIRED_TEXT_FIELDS = {
"COMPANY_NAME",
"CONFIRMATION_NO",
"DISP_ROOM_NO",
"FULL_NAME",
"RATE_CODE",
}
def validation_error(
code: str,
message: str,
source_location: Optional[str] = None,
record: Optional[Dict[str, Any]] = None,
) -> core.ErrorItem:
record = record or {}
amount = record.get("EFFECTIVE_RATE_AMOUNT")
try:
normalized_amount = core.parse_decimal(amount) if amount is not None else None
except ValueError:
normalized_amount = None
return core.ErrorItem(
code=code,
stage="output",
message=message,
source_location=source_location,
company_name=core.text_or_blank(record.get("COMPANY_NAME")) or None,
rate_code=core.text_or_blank(record.get("RATE_CODE")) or None,
effective_rate_amount=normalized_amount,
confirmation_no=core.text_or_blank(record.get("CONFIRMATION_NO")) or None,
)
def actual_date(value: Any) -> Optional[date]:
if isinstance(value, datetime):
return value.date()
if isinstance(value, date):
return value
return None
def is_number(value: Any) -> bool:
return isinstance(value, (int, float)) and not isinstance(value, bool)
def comparable(value: Any, field: str) -> Any:
if field in DATE_FIELDS:
return actual_date(value)
if field in DAILY_NUMERIC_FIELDS:
try:
return core.parse_decimal(value)
except ValueError:
return value
if field in TEXT_FIELDS:
return core.text_or_blank(value)
return value
def workbook_rows(
sheet: Any,
headers: Sequence[str],
workbook_name: str,
errors: List[core.ErrorItem],
) -> List[Tuple[int, Dict[str, Any]]]:
rows: List[Tuple[int, Dict[str, Any]]] = []
last_column = get_column_letter(len(headers))
for row_number in range(2, sheet.max_row + 1):
values = [sheet.cell(row_number, col).value for col in range(1, len(headers) + 1)]
if all(value is None or core.text_or_blank(value) == "" for value in values):
continue
location = f"{workbook_name}!{sheet.title}!A{row_number}:{last_column}{row_number}"
for col, value in enumerate(values, 1):
if isinstance(value, str) and value.startswith("=") and sheet.cell(row_number, col).data_type == "f":
errors.append(
validation_error(
"OUTPUT_FORMULA_FORBIDDEN",
"输出数据不得包含公式",
f"{workbook_name}!{sheet.title}!{sheet.cell(row_number, col).coordinate}",
)
)
rows.append((row_number, dict(zip(headers, values))))
for row in sheet.iter_rows(min_row=1):
for cell in row:
if cell.value is not None and cell.data_type == "f":
location = f"{workbook_name}!{sheet.title}!{cell.coordinate}"
if not any(error.source_location == location for error in errors):
errors.append(
validation_error(
"OUTPUT_FORMULA_FORBIDDEN", "输出工作簿不得包含公式", location
)
)
return rows
def validate_headers(
sheet: Any,
expected: Sequence[str],
workbook_name: str,
errors: List[core.ErrorItem],
) -> bool:
last_column = get_column_letter(len(expected))
actual = [core.text_or_blank(sheet.cell(1, col).value) for col in range(1, len(expected) + 1)]
if actual != list(expected):
errors.append(
validation_error(
"OUTPUT_HEADER_MISMATCH",
f"{len(expected)}列表头不符合契约;应为 {list(expected)}",
f"{workbook_name}!{sheet.title}!A1:{last_column}1",
)
)
return False
for col in range(len(expected) + 1, sheet.max_column + 1):
if sheet.cell(1, col).value not in (None, ""):
errors.append(
validation_error(
"OUTPUT_HEADER_MISMATCH",
f"{len(expected)}列之后不得出现额外表头",
f"{workbook_name}!{sheet.title}!{sheet.cell(1, col).coordinate}",
)
)
return False
return True
def validate_row_types(
row: Dict[str, Any],
numeric_fields: Iterable[str],
location: str,
errors: List[core.ErrorItem],
) -> None:
for field in TEXT_FIELDS:
value = row.get(field)
if value is not None and not isinstance(value, str):
errors.append(
validation_error(
"OUTPUT_TEXT_TYPE_MISMATCH",
f"{field} 必须以Excel文本类型写入",
location,
row,
)
)
if field in REQUIRED_TEXT_FIELDS and core.text_or_blank(value) == "":
errors.append(
validation_error(
"OUTPUT_REQUIRED_TEXT_MISSING", f"{field} 不能为空", location, row
)
)
for field in DATE_FIELDS:
if actual_date(row.get(field)) is None:
errors.append(
validation_error(
"OUTPUT_DATE_TYPE_MISMATCH",
f"{field} 必须是Excel真实日期而非文本",
location,
row,
)
)
for field in numeric_fields:
if not is_number(row.get(field)):
errors.append(
validation_error(
"OUTPUT_NUMBER_TYPE_MISMATCH",
f"{field} 必须是静态Excel数字",
location,
row,
)
)
def compare_rows(
actual_rows: Sequence[Tuple[int, Dict[str, Any]]],
expected_rows: Sequence[Dict[str, Any]],
headers: Sequence[str],
workbook_name: str,
sheet_name: str,
errors: List[core.ErrorItem],
) -> None:
if len(actual_rows) != len(expected_rows):
errors.append(
validation_error(
"OUTPUT_ROW_COUNT_MISMATCH",
f"应有 {len(expected_rows)} 行,实际 {len(actual_rows)}",
f"{workbook_name}!{sheet_name}",
)
)
for offset, (actual_pair, expected) in enumerate(zip(actual_rows, expected_rows), 1):
row_number, actual = actual_pair
for field in headers:
if comparable(actual.get(field), field) != comparable(expected.get(field), field):
errors.append(
validation_error(
"OUTPUT_VALUE_OR_ORDER_MISMATCH",
f"{offset} 条记录的 {field} 与XML推导结果不一致",
f"{workbook_name}!{sheet_name}!row[{row_number}]/{field}",
actual,
)
)
break
def expected_from_xml(
xml_path: Path, price_path: Path
) -> Tuple[date, int, int, int, List[Dict[str, Any]]]:
business_date, reservations = core.read_xml(xml_path)
records, removed_by_rate, removed_duplicates = core.filter_and_deduplicate(
reservations, business_date
)
price_map = core.load_price_map(price_path)
core.apply_prices(records, price_map)
return business_date, len(reservations), removed_by_rate, removed_duplicates, records
def replay_classified_from_xml(
xml_path: Path,
price_path: Path,
manual_prices: Optional[Dict[Tuple[str, str, core.Decimal], core.Decimal]] = None,
) -> Tuple[
date,
int,
int,
int,
List[Dict[str, Any]],
List[Dict[str, Any]],
List[core.ErrorItem],
]:
"""Freshly derive all source records, including non-output outcomes, from XML."""
business_date, reservations = core.read_xml(xml_path)
all_records, records, removed_by_rate, removed_duplicates, classification_errors = (
core.classify_source_records(reservations, business_date)
)
if classification_errors:
raise core.ProcessingFailure(classification_errors)
price_map = core.load_price_map(price_path)
pricing_errors = core.apply_prices_classified(records, price_map, manual_prices)
return (
business_date,
len(reservations),
removed_by_rate,
removed_duplicates,
all_records,
records,
pricing_errors,
)
def validate_result_contract(
payload: Dict[str, Any],
business_date: date,
source_rows: int,
removed_by_rate: int,
removed_duplicates: int,
expected_records: Sequence[Dict[str, Any]],
daily_path: Optional[Path],
structured_result_path: Path,
expected_channels: Sequence[Dict[str, Any]],
errors: List[core.ErrorItem],
*,
status: str = "success",
candidate_rows: int = 0,
review_required_rows: int = 0,
review_issue_count: int = 0,
expected_errors: Sequence[core.ErrorItem] = (),
) -> None:
required = {
"version",
"status",
"business_date",
"message",
"metrics",
"outputs",
"errors",
}
if set(payload) != required:
errors.append(
validation_error(
"OUTPUT_RESULT_CONTRACT_MISMATCH", "result.json 顶层字段不符合固定Schema"
)
)
return
if payload.get("version") != core.RESULT_VERSION or payload.get("status") != status:
errors.append(
validation_error(
"OUTPUT_RESULT_CONTRACT_MISMATCH", "独立校验时result版本或status不正确"
)
)
if payload.get("business_date") != business_date.isoformat():
errors.append(
validation_error(
"OUTPUT_RESULT_DATE_MISMATCH", "result业务日期与XML不一致"
)
)
metrics = payload.get("metrics")
expected_metrics = {
"source_rows": source_rows,
"removed_by_rate_code": removed_by_rate,
"removed_as_duplicates": removed_duplicates,
"output_rows": len(expected_records),
"candidate_rows": candidate_rows,
"review_required_rows": review_required_rows,
"review_issue_count": review_issue_count,
}
if not isinstance(metrics, dict):
errors.append(validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "metrics必须是对象"))
else:
for key, value in expected_metrics.items():
if metrics.get(key) != value:
errors.append(
validation_error(
"OUTPUT_RESULT_METRIC_MISMATCH",
f"metrics.{key} 应为 {value},实际为 {metrics.get(key)}",
)
)
if not isinstance(metrics.get("channels"), list):
errors.append(
validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "metrics.channels必须是数组")
)
elif metrics.get("channels") != list(expected_channels):
errors.append(
validation_error(
"OUTPUT_RESULT_CHANNEL_MISMATCH",
"metrics.channels 与XML确定性渠道路由不一致",
)
)
outputs = payload.get("outputs")
if not isinstance(outputs, dict):
errors.append(validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "outputs必须是对象"))
else:
if daily_path is not None and outputs.get("daily_report") != daily_path.name:
errors.append(validation_error("OUTPUT_RESULT_FILENAME_MISMATCH", "result日报文件名不一致"))
if daily_path is None and outputs.get("daily_report") is not None:
errors.append(validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "待复核结果不得包含日报文件名"))
if outputs.get("structured_result") != structured_result_path.name:
errors.append(
validation_error(
"OUTPUT_RESULT_FILENAME_MISMATCH",
"result结构化结果文件名不一致",
)
)
if outputs.get("exception_report") is not None:
errors.append(
validation_error(
"OUTPUT_RESULT_CONTRACT_MISMATCH", "成功结果不得包含异常清单文件名"
)
)
for field in ("structured_result",):
value = outputs.get(field)
if not isinstance(value, str) or Path(value).name != value:
errors.append(
validation_error(
"OUTPUT_RESULT_PATH_FORBIDDEN", f"outputs.{field} 必须是相对文件名"
)
)
if daily_path is not None:
value = outputs.get("daily_report")
if not isinstance(value, str) or Path(value).name != value:
errors.append(
validation_error(
"OUTPUT_RESULT_PATH_FORBIDDEN", "outputs.daily_report 必须是相对文件名"
)
)
if payload.get("errors") != [item.to_dict() for item in expected_errors]:
errors.append(
validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "result的errors与独立重放不一致")
)
def validate_artifact(
actual: Any,
expected_path: Path,
file_kind: str,
mime_type: str,
label: str,
errors: List[core.ErrorItem],
) -> None:
expected = core.artifact_object(expected_path, file_kind, mime_type)
if actual != expected:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_ARTIFACT_MISMATCH",
f"structured-result.json 的 {label} 哈希或文件元数据不一致",
expected_path.name,
)
)
def validate_structured_result_contract(
payload: Dict[str, Any],
xml_path: Path,
daily_path: Path,
result_json: Path,
business_date: date,
source_rows: int,
removed_by_rate: int,
removed_duplicates: int,
expected_records: Sequence[Dict[str, Any]],
result_payload: Dict[str, Any],
errors: List[core.ErrorItem],
*,
manual_manifest: Optional[core.ManualOverrideManifest] = None,
manual_override_path: Optional[Path] = None,
) -> None:
required = {
"result_schema_version",
"status",
"activation_eligible",
"ingestion_mode",
"business_date",
"processor_version",
"rule_set_sha256",
"source_rows",
"removed_by_rate_code",
"removed_as_duplicates",
"output_rows",
"outcome_counts",
"candidate_rows",
"review_required_rows",
"review_issue_count",
"review_issues",
"review_case_id",
"manual_override_sha256",
"manually_priced_rows",
"channels",
"artifacts",
"records",
"errors",
}
if set(payload) != required:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_CONTRACT_MISMATCH",
"structured-result.json 顶层字段不符合固定Schema",
)
)
return
try:
core.validate_structured_completeness(payload)
except core.ProcessingFailure as exc:
for item in exc.errors:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_CONTRACT_MISMATCH",
item.message,
item.source_location,
)
)
expected_scalars = {
"result_schema_version": core.STRUCTURED_RESULT_SCHEMA_VERSION,
"status": "success",
"activation_eligible": True,
"ingestion_mode": "opera_xml",
"business_date": business_date.isoformat(),
"processor_version": core.PROCESSOR_VERSION,
"rule_set_sha256": core.rule_set_sha256(),
"source_rows": source_rows,
"removed_by_rate_code": removed_by_rate,
"removed_as_duplicates": removed_duplicates,
"output_rows": len(expected_records),
"candidate_rows": 0,
"review_required_rows": 0,
"review_issue_count": 0,
"review_issues": [],
"review_case_id": manual_manifest.review_case_id if manual_manifest else None,
"manual_override_sha256": manual_manifest.sha256 if manual_manifest else None,
"manually_priced_rows": sum(
1 for record in expected_records if record.get("_PRICING_METHOD") == "manual_review"
),
"channels": result_payload.get("metrics", {}).get("channels"),
"errors": [],
}
for field, expected in expected_scalars.items():
if payload.get(field) != expected:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_VALUE_MISMATCH",
f"structured-result.json 的 {field} 应为 {expected!r}",
)
)
expected_outcome_counts = {
"duplicate": removed_duplicates,
"excluded_rate_code": removed_by_rate,
"candidate": 0,
"price_unmatched": 0,
"retained": len(expected_records),
"validation_failed": 0,
}
if payload.get("outcome_counts") != expected_outcome_counts:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_OUTCOME_MISMATCH",
f"structured outcome计数应为 {expected_outcome_counts}",
)
)
artifacts = payload.get("artifacts")
if isinstance(artifacts, dict):
validate_artifact(artifacts.get("source_xml"), xml_path, "opera_xml", "application/xml", "source_xml", errors)
validate_artifact(
artifacts.get("daily_report"),
daily_path,
"daily_xlsx",
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"daily_report",
errors,
)
validate_artifact(
artifacts.get("result_json"),
result_json,
"result_json",
"application/json",
"result_json",
errors,
)
if artifacts.get("exception_report") is not None:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_ARTIFACT_MISMATCH",
"成功structured-result.json不得引用异常清单",
)
)
if manual_manifest is not None and manual_override_path is not None:
validate_artifact(
artifacts.get("manual_override_json"),
manual_override_path,
"manual_override_json",
"application/json",
"manual_override_json",
errors,
)
if payload.get("manual_override_sha256") != core.sha256_file(manual_override_path):
errors.append(
validation_error(
"OUTPUT_STRUCTURED_ARTIFACT_MISMATCH",
"人工价格清单哈希与清单文件不一致",
)
)
elif artifacts.get("manual_override_json") is not None:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_ARTIFACT_MISMATCH",
"非人工定价成功结果不得引用人工价格清单",
)
)
else:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_CONTRACT_MISMATCH", "structured artifacts必须是对象"
)
)
_date, reservations = core.read_xml(xml_path)
all_records, retained, _removed_rate, _removed_duplicates, classification_errors = (
core.classify_source_records(reservations, business_date)
)
if classification_errors:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_SOURCE_REPLAY_FAILED",
"成功批次的XML独立重放不应出现行级校验错误",
)
)
return
price_map = core.load_price_map(Path(core.PRICE_REFERENCE).resolve())
pricing_errors = core.apply_prices_classified(
retained,
price_map,
manual_manifest.prices if manual_manifest is not None else None,
)
if pricing_errors:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_SOURCE_REPLAY_FAILED",
"成功批次的XML独立重放不应出现定价错误",
)
)
return
core.assign_channels(retained)
actual_records = payload.get("records")
if not isinstance(actual_records, list) or len(actual_records) != len(all_records):
errors.append(
validation_error(
"OUTPUT_STRUCTURED_RECORD_COUNT_MISMATCH",
f"structured records应保留全部 {len(all_records)} 条XML源记录",
)
)
return
actual_by_sequence = {
item.get("source_sequence"): item for item in actual_records if isinstance(item, dict)
}
for expected_record in all_records:
sequence = expected_record["_SOURCE_INDEX"]
actual = actual_by_sequence.get(sequence)
if actual is None:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_SOURCE_SEQUENCE_MISSING",
f"structured records缺少source_sequence={sequence}",
)
)
continue
if expected_record["_OUTCOME"] == "pending":
expected_record["_OUTCOME"] = "retained"
expected = core.structured_record(expected_record)
if actual != expected:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_RECORD_MISMATCH",
f"source_sequence={sequence}的结构化字段与XML确定性推导不一致",
f"reservation[{sequence}]",
)
)
def validate_review_structured_result_contract(
payload: Dict[str, Any],
xml_path: Path,
result_json: Path,
business_date: date,
source_rows: int,
removed_by_rate: int,
removed_duplicates: int,
all_records: Sequence[Dict[str, Any]],
records: Sequence[Dict[str, Any]],
pricing_errors: Sequence[core.ErrorItem],
price_map: Dict[Tuple[str, str, core.Decimal], core.Decimal],
errors: List[core.ErrorItem],
) -> None:
"""Validate the initial no-XLSX review result against a second XML replay."""
required = {
"result_schema_version",
"status",
"activation_eligible",
"ingestion_mode",
"business_date",
"processor_version",
"rule_set_sha256",
"source_rows",
"removed_by_rate_code",
"removed_as_duplicates",
"output_rows",
"outcome_counts",
"candidate_rows",
"review_required_rows",
"review_issue_count",
"review_issues",
"review_case_id",
"manual_override_sha256",
"manually_priced_rows",
"channels",
"artifacts",
"records",
"errors",
}
if set(payload) != required:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_CONTRACT_MISMATCH",
"review structured-result.json 顶层字段不符合固定Schema",
)
)
return
try:
core.validate_structured_completeness(payload)
except core.ProcessingFailure as exc:
errors.extend(
validation_error("OUTPUT_STRUCTURED_CONTRACT_MISMATCH", item.message, item.source_location)
for item in exc.errors
)
core.assign_channels(records)
expected_issues = core.review_issues(records, price_map)
core.finalize_review_outcomes(all_records)
candidate_rows = sum(1 for record in all_records if record.get("_OUTCOME") == "candidate")
review_rows = sum(1 for record in all_records if record.get("_OUTCOME") == "price_unmatched")
expected_scalars = {
"result_schema_version": core.STRUCTURED_RESULT_SCHEMA_VERSION,
"status": "review_required",
"activation_eligible": False,
"ingestion_mode": "opera_xml",
"business_date": business_date.isoformat(),
"processor_version": core.PROCESSOR_VERSION,
"rule_set_sha256": core.rule_set_sha256(),
"source_rows": source_rows,
"removed_by_rate_code": removed_by_rate,
"removed_as_duplicates": removed_duplicates,
"output_rows": 0,
"candidate_rows": candidate_rows,
"review_required_rows": review_rows,
"review_issue_count": len(expected_issues),
"review_issues": expected_issues,
"review_case_id": None,
"manual_override_sha256": None,
"manually_priced_rows": 0,
"channels": [],
"errors": [item.to_dict() for item in pricing_errors],
}
for field, expected in expected_scalars.items():
if payload.get(field) != expected:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_VALUE_MISMATCH",
f"review structured-result.json 的 {field} 应为 {expected!r}",
)
)
expected_outcome_counts = {
"candidate": candidate_rows,
"duplicate": removed_duplicates,
"excluded_rate_code": removed_by_rate,
"price_unmatched": review_rows,
"retained": 0,
"validation_failed": 0,
}
if payload.get("outcome_counts") != expected_outcome_counts:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_OUTCOME_MISMATCH",
f"review structured outcome计数应为 {expected_outcome_counts}",
)
)
artifacts = payload.get("artifacts")
if not isinstance(artifacts, dict):
errors.append(validation_error("OUTPUT_STRUCTURED_CONTRACT_MISMATCH", "review artifacts必须是对象"))
else:
validate_artifact(artifacts.get("source_xml"), xml_path, "opera_xml", "application/xml", "source_xml", errors)
validate_artifact(
artifacts.get("result_json"), result_json, "result_json", "application/json", "result_json", errors
)
for name in ("daily_report", "exception_report", "manual_override_json"):
if artifacts.get(name) is not None:
errors.append(
validation_error(
"OUTPUT_STRUCTURED_ARTIFACT_MISMATCH",
f"review structured-result.json不得引用{name}",
)
)
actual_records = payload.get("records")
if not isinstance(actual_records, list) or len(actual_records) != len(all_records):
errors.append(
validation_error(
"OUTPUT_STRUCTURED_RECORD_COUNT_MISMATCH",
f"review structured records应保留全部 {len(all_records)} 条XML源记录",
)
)
return
actual_by_sequence = {
item.get("source_sequence"): item for item in actual_records if isinstance(item, dict)
}
for expected_record in all_records:
sequence = expected_record["_SOURCE_INDEX"]
actual = actual_by_sequence.get(sequence)
if actual != core.structured_record(expected_record):
errors.append(
validation_error(
"OUTPUT_STRUCTURED_RECORD_MISMATCH",
f"review source_sequence={sequence}的结构化字段与XML确定性推导不一致",
f"reservation[{sequence}]",
)
)
def validate_legacy_direct_success_contracts(
result_payload: Dict[str, Any],
structured_payload: Dict[str, Any],
xml_path: Path,
daily_path: Path,
result_json: Path,
structured_result_json: Path,
business_date: date,
source_rows: int,
removed_by_rate: int,
removed_duplicates: int,
expected_records: Sequence[Dict[str, Any]],
expected_channels: Sequence[Dict[str, Any]],
errors: List[core.ErrorItem],
) -> None:
"""Validate the retired v3 direct-MCP success contract through v4 replay.
The active replay logic remains the source of truth for XML, workbook and
row calculations. This adapter first rejects any field outside the frozen
v3 shape, then adds only zero-valued v4 review fields in memory so the
shared independent checks can replay the same deterministic business facts.
"""
result_fields = {
"version",
"status",
"business_date",
"message",
"metrics",
"outputs",
"errors",
}
structured_fields = {
"result_schema_version",
"status",
"activation_eligible",
"ingestion_mode",
"business_date",
"processor_version",
"rule_set_sha256",
"source_rows",
"removed_by_rate_code",
"removed_as_duplicates",
"output_rows",
"outcome_counts",
"channels",
"artifacts",
"records",
"errors",
}
result_metrics = {
"source_rows",
"removed_by_rate_code",
"removed_as_duplicates",
"output_rows",
"channels",
}
outcome_fields = {
"duplicate",
"excluded_rate_code",
"price_unmatched",
"retained",
"validation_failed",
}
artifact_fields = {
"source_xml",
"daily_report",
"result_json",
"exception_report",
}
if set(result_payload) != result_fields or set(structured_payload) != structured_fields:
errors.append(
validation_error(
"OUTPUT_LEGACY_DIRECT_CONTRACT_MISMATCH",
"旧direct_mcp结果字段集不符合冻结v3契约",
)
)
return
if (
result_payload.get("version") != core.LEGACY_DIRECT_RESULT_VERSION
or result_payload.get("status") != "success"
or not isinstance(result_payload.get("message"), str)
or result_payload.get("business_date") != business_date.isoformat()
or result_payload.get("errors") != []
):
errors.append(
validation_error(
"OUTPUT_LEGACY_DIRECT_CONTRACT_MISMATCH",
"旧direct_mcp result.json身份或成功状态无效",
)
)
return
metrics = result_payload.get("metrics")
outputs = result_payload.get("outputs")
if (
not isinstance(metrics, dict)
or set(metrics) != result_metrics
or not isinstance(outputs, dict)
or set(outputs) != {"daily_report", "structured_result", "exception_report"}
):
errors.append(
validation_error(
"OUTPUT_LEGACY_DIRECT_CONTRACT_MISMATCH",
"旧direct_mcp result指标或输出字段不符合冻结v3契约",
)
)
return
artifacts = structured_payload.get("artifacts")
outcomes = structured_payload.get("outcome_counts")
if (
not isinstance(artifacts, dict)
or set(artifacts) != artifact_fields
or not isinstance(outcomes, dict)
or set(outcomes) != outcome_fields
or structured_payload.get("result_schema_version") != core.LEGACY_DIRECT_RESULT_VERSION
or structured_payload.get("status") != "success"
or structured_payload.get("activation_eligible") is not True
or structured_payload.get("ingestion_mode") != "opera_xml"
or structured_payload.get("business_date") != business_date.isoformat()
or structured_payload.get("processor_version") != core.LEGACY_DIRECT_PROCESSOR_VERSION
or structured_payload.get("rule_set_sha256") != core.LEGACY_DIRECT_RULE_SET_SHA256
or structured_payload.get("errors") != []
):
errors.append(
validation_error(
"OUTPUT_LEGACY_DIRECT_CONTRACT_MISMATCH",
"旧direct_mcp structured-result身份或字段无效",
)
)
return
replay_result = copy.deepcopy(result_payload)
replay_result["version"] = core.RESULT_VERSION
replay_metrics = dict(metrics)
replay_metrics.update(
{
"candidate_rows": 0,
"review_required_rows": 0,
"review_issue_count": 0,
}
)
replay_result["metrics"] = replay_metrics
replay_structured = copy.deepcopy(structured_payload)
replay_structured.update(
{
"result_schema_version": core.STRUCTURED_RESULT_SCHEMA_VERSION,
"processor_version": core.PROCESSOR_VERSION,
"rule_set_sha256": core.rule_set_sha256(),
"candidate_rows": 0,
"review_required_rows": 0,
"review_issue_count": 0,
"review_issues": [],
"review_case_id": None,
"manual_override_sha256": None,
"manually_priced_rows": 0,
}
)
replay_outcomes = dict(outcomes)
replay_outcomes["candidate"] = 0
replay_structured["outcome_counts"] = replay_outcomes
replay_artifacts = dict(artifacts)
replay_artifacts["manual_override_json"] = None
replay_structured["artifacts"] = replay_artifacts
validate_result_contract(
replay_result,
business_date,
source_rows,
removed_by_rate,
removed_duplicates,
expected_records,
daily_path,
structured_result_json,
expected_channels,
errors,
)
validate_structured_result_contract(
replay_structured,
xml_path,
daily_path,
result_json,
business_date,
source_rows,
removed_by_rate,
removed_duplicates,
expected_records,
replay_result,
errors,
)
def validate_daily(
daily_path: Path,
business_date: date,
expected_records: Sequence[Dict[str, Any]],
errors: List[core.ErrorItem],
) -> None:
if not daily_path.is_file():
errors.append(validation_error("OUTPUT_DAILY_MISSING", "找不到候选日报", daily_path.name))
return
try:
workbook = load_workbook(daily_path, data_only=False)
except Exception as exc:
errors.append(
validation_error(
"OUTPUT_DAILY_UNREADABLE", f"候选日报无法读取:{exc}", daily_path.name
)
)
return
try:
if len(workbook.worksheets) != 1:
errors.append(
validation_error("OUTPUT_DAILY_SHEET_COUNT", "日报必须且只能包含一个工作表")
)
sheet = workbook.active
expected_title = f"{business_date.month}.{business_date.day}"
if sheet.title != expected_title:
errors.append(
validation_error(
"OUTPUT_DAILY_SHEET_NAME",
f"日报工作表名应为 {expected_title}",
f"{daily_path.name}!{sheet.title}",
)
)
if not validate_headers(sheet, core.DAILY_HEADERS, daily_path.name, errors):
return
rows = workbook_rows(sheet, core.DAILY_HEADERS, daily_path.name, errors)
seen: set = set()
for row_number, row in rows:
last_column = get_column_letter(len(core.DAILY_HEADERS))
location = f"{daily_path.name}!{sheet.title}!A{row_number}:{last_column}{row_number}"
validate_row_types(row, DAILY_NUMERIC_FIELDS, location, errors)
arrival = actual_date(row.get("ARRIVAL"))
departure = actual_date(row.get("DEPARTURE"))
if arrival and departure and is_number(row.get("NIGHTS")):
if departure < arrival or int(row["NIGHTS"]) != (departure - arrival).days:
errors.append(
validation_error(
"OUTPUT_NIGHTS_MISMATCH", "日报晚数与日期不一致", location, row
)
)
if arrival != business_date:
errors.append(
validation_error(
"OUTPUT_BUSINESS_DATE_MISMATCH", "日报ARRIVAL与XML业务日期不一致", location, row
)
)
if core.text_or_blank(row.get("RATE_CODE")).upper() not in core.RATE_WHITELIST:
errors.append(
validation_error(
"OUTPUT_RATE_NOT_WHITELISTED", "日报包含费率白名单外的记录", location, row
)
)
if all(
is_number(row.get(field))
for field in ("REAL PRICE", "NO_OF_ROOMS", "NIGHTS", "TOTAL PRICE")
):
expected_total = row["REAL PRICE"] * row["NO_OF_ROOMS"] * row["NIGHTS"]
if core.parse_decimal(row["TOTAL PRICE"]) != core.parse_decimal(expected_total):
errors.append(
validation_error(
"OUTPUT_DAILY_TOTAL_PRICE_MISMATCH",
"日报TOTAL PRICE不等于REAL PRICE×NO_OF_ROOMS×NIGHTS",
location,
row,
)
)
key = (core.text_or_blank(row.get("DISP_ROOM_NO")), arrival)
if key in seen:
errors.append(
validation_error(
"OUTPUT_DUPLICATE_KEY", "日报包含重复的房号+ARRIVAL", location, row
)
)
seen.add(key)
compare_rows(
rows,
expected_records,
core.DAILY_HEADERS,
daily_path.name,
sheet.title,
errors,
)
finally:
workbook.close()
def validate(args: argparse.Namespace) -> List[core.ErrorItem]:
xml_path = Path(args.xml)
result_json = Path(args.result_json)
structured_result_json = Path(args.structured_result_json)
price_path = Path(args.price_reference)
review_only = bool(getattr(args, "review_only", False))
legacy_v3_output = getattr(args, "legacy_v3_output", False)
daily_arg = getattr(args, "daily", None)
daily_path = Path(daily_arg) if daily_arg else None
manual_override_arg = getattr(args, "manual_override_json", None)
review_job_id = getattr(args, "review_job_id", None)
review_case_id = getattr(args, "review_case_id", None)
manual_override_sha256 = getattr(args, "manual_override_sha256", None)
manual_values = (manual_override_arg, review_job_id, review_case_id, manual_override_sha256)
if not isinstance(legacy_v3_output, bool):
return [
validation_error(
"OUTPUT_VALIDATOR_INPUT_INVALID", "旧direct_mcp兼容开关必须是布尔值"
)
]
if legacy_v3_output and (review_only or any(value is not None for value in manual_values)):
return [
validation_error(
"OUTPUT_VALIDATOR_INPUT_INVALID",
"旧direct_mcp兼容校验不支持待复核或人工价格清单",
)
]
required_paths = [
(xml_path, ".xml", "XML"),
(result_json, ".json", "result.json"),
(structured_result_json, ".json", "structured-result.json"),
(price_path, ".xlsx", "价格对照"),
]
if not review_only:
if daily_path is None:
return [validation_error("OUTPUT_VALIDATOR_INPUT_INVALID", "成功重放必须提供日报路径")]
required_paths.append((daily_path, ".xlsx", "日报"))
elif daily_path is not None:
return [validation_error("OUTPUT_VALIDATOR_INPUT_INVALID", "待复核重放不得提供日报路径")]
for path, suffix, label in required_paths:
if not path.is_absolute() or not path.is_file() or path.suffix.lower() != suffix:
return [
validation_error(
"OUTPUT_VALIDATOR_INPUT_INVALID",
f"独立校验器的{label}路径必须是存在的绝对{suffix}文件",
str(path),
)
]
if any(value is not None for value in manual_values) and not all(
isinstance(value, str) and value for value in manual_values
):
return [
validation_error(
"OUTPUT_VALIDATOR_INPUT_INVALID", "人工价格重放必须同时提供清单、任务、case和哈希"
)
]
if review_only and any(value is not None for value in manual_values):
return [validation_error("OUTPUT_VALIDATOR_INPUT_INVALID", "待复核校验不得提供人工价格清单")]
try:
payload = json.loads(result_json.read_text(encoding="utf-8"))
except Exception as exc:
return [
validation_error(
"OUTPUT_RESULT_UNREADABLE", f"result.json无法读取{exc}", result_json.name
)
]
if not isinstance(payload, dict):
return [validation_error("OUTPUT_RESULT_CONTRACT_MISMATCH", "result.json必须是对象")]
try:
structured_payload = json.loads(structured_result_json.read_text(encoding="utf-8"))
except Exception as exc:
return [
validation_error(
"OUTPUT_STRUCTURED_RESULT_UNREADABLE",
f"structured-result.json无法读取{exc}",
structured_result_json.name,
)
]
if not isinstance(structured_payload, dict):
return [
validation_error(
"OUTPUT_STRUCTURED_CONTRACT_MISMATCH", "structured-result.json必须是对象"
)
]
manual_manifest: Optional[core.ManualOverrideManifest] = None
manual_override_path = Path(manual_override_arg) if manual_override_arg else None
try:
business_date, reservations = core.read_xml(xml_path)
all_records, records, removed_rate, removed_duplicates, classification_errors = (
core.classify_source_records(reservations, business_date)
)
if classification_errors:
raise core.ProcessingFailure(classification_errors)
price_map = core.load_price_map(price_path)
if manual_override_path is not None:
manual_manifest = core.load_manual_override_manifest(
manual_override_path,
job_id=str(review_job_id),
review_case_id=str(review_case_id),
expected_sha256=str(manual_override_sha256),
xml_path=xml_path,
business_date=business_date,
)
if core.missing_price_keys(records, price_map) != set(manual_manifest.prices):
raise core.ProcessingFailure(
[
validation_error(
"OUTPUT_MANUAL_OVERRIDE_KEYSET_MISMATCH",
"人工价格清单问题键集合与XML独立重放不一致",
)
]
)
pricing_errors = core.apply_prices_classified(
records,
price_map,
manual_manifest.prices if manual_manifest is not None else None,
)
except core.ProcessingFailure as exc:
return list(exc.errors)
source_rows = len(reservations)
errors: List[core.ErrorItem] = []
if review_only:
if not pricing_errors or not all(item.code == "PRICE_UNMATCHED" for item in pricing_errors):
return [
validation_error(
"OUTPUT_REVIEW_SOURCE_REPLAY_FAILED",
"待复核结果只能来自纯PRICE_UNMATCHED的XML重放",
)
]
candidate_rows = len(records) - len(pricing_errors)
validate_result_contract(
payload,
business_date,
source_rows,
removed_rate,
removed_duplicates,
[],
None,
structured_result_json,
[],
errors,
status="review_required",
candidate_rows=candidate_rows,
review_required_rows=len(pricing_errors),
review_issue_count=len(core.review_issues(records, price_map)),
expected_errors=pricing_errors,
)
validate_review_structured_result_contract(
structured_payload,
xml_path,
result_json,
business_date,
source_rows,
removed_rate,
removed_duplicates,
all_records,
records,
pricing_errors,
price_map,
errors,
)
return errors
if pricing_errors:
return [
validation_error(
"OUTPUT_STRUCTURED_SOURCE_REPLAY_FAILED",
"成功批次的XML独立重放不应出现定价错误",
)
]
expected_channels = core.channel_metrics(core.assign_channels(records))
assert daily_path is not None
if legacy_v3_output:
validate_legacy_direct_success_contracts(
payload,
structured_payload,
xml_path,
daily_path,
result_json,
structured_result_json,
business_date,
source_rows,
removed_rate,
removed_duplicates,
records,
expected_channels,
errors,
)
validate_daily(daily_path, business_date, records, errors)
return errors
validate_result_contract(
payload,
business_date,
source_rows,
removed_rate,
removed_duplicates,
records,
daily_path,
structured_result_json,
expected_channels,
errors,
)
validate_structured_result_contract(
structured_payload,
xml_path,
daily_path,
result_json,
business_date,
source_rows,
removed_rate,
removed_duplicates,
records,
payload,
errors,
manual_manifest=manual_manifest,
manual_override_path=manual_override_path,
)
validate_daily(daily_path, business_date, records, errors)
return errors
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--xml", required=True)
parser.add_argument("--daily")
parser.add_argument("--result-json", required=True)
parser.add_argument("--structured-result-json", required=True)
parser.add_argument("--price-reference", required=True)
parser.add_argument("--review-only", action="store_true")
parser.add_argument("--manual-override-json")
parser.add_argument("--review-job-id")
parser.add_argument("--review-case-id")
parser.add_argument("--manual-override-sha256")
parser.add_argument("--legacy-v3-output", action="store_true", help=argparse.SUPPRESS)
return parser
def main() -> int:
try:
errors = validate(build_parser().parse_args())
if errors:
payload = {"status": "failed", "errors": [error.to_dict() for error in errors]}
print(json.dumps(payload, ensure_ascii=False))
return 2
print(json.dumps({"status": "success", "errors": []}, ensure_ascii=False))
return 0
except core.ProcessingFailure as exc:
payload = {"status": "failed", "errors": [error.to_dict() for error in exc.errors]}
print(json.dumps(payload, ensure_ascii=False))
return exc.exit_code if exc.exit_code in {2, 3} else 2
except Exception as exc:
error = validation_error(
"INTERNAL_ERROR", f"独立校验器内部错误:{type(exc).__name__}: {exc}"
)
print(json.dumps({"status": "failed", "errors": [error.to_dict()]}, ensure_ascii=False))
traceback.print_exc(file=sys.stderr)
return 4
if __name__ == "__main__":
raise SystemExit(main())