Files
Cloud-Tour-to-Libo/scripts/build_travel_graph_existing_product_project.py
T

2898 lines
140 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import csv
import hashlib
import json
import math
import re
import subprocess
from collections import Counter, defaultdict
from datetime import datetime
from pathlib import Path
from typing import Any
import pandas as pd
import psycopg
from falkordb import FalkorDB
from psycopg.rows import dict_row
from psycopg.types.json import Jsonb
from common_paths import PROJECT_ROOT, TRAVEL_AGENCY_SOURCE_ROOT, TRAVEL_KG_EXPORT_ROOT
SOURCE_ROOT = TRAVEL_AGENCY_SOURCE_ROOT
ROUTE_MD_DIR = SOURCE_ROOT / "2026年新行程打包_md整理"
ROUTE_MD_PRODUCTS = ROUTE_MD_DIR / "products"
SCHEMA_SRC = PROJECT_ROOT / "schema搭建/travel_agency_business/travel_agency_existing_product_schema.v1.json"
SCHEMA_OUT_DIR = PROJECT_ROOT / "schema搭建/travel_graph_existing_product"
OUT_DIR = TRAVEL_KG_EXPORT_ROOT / "travel_graph_旅行社线路制定"
AMAP_CACHE_PATH = OUT_DIR / "amap_poi_enrichment_cache.json"
AMAP_DRIVING_CACHE_PATH = OUT_DIR / "amap_driving_distance_cache.json"
DB_URL = "postgresql://admin:password@localhost:5433/kg_admin"
DB_SCHEMA = "kg_admin_new2"
TENANT_ID = "travel_agency"
PROJECT_ID = "travel_graph"
PROJECT_NAME = "旅行社线路制定"
GRAPH_NAME = "travel_graph"
SCHEMA_NAMESPACE = "travel_agency_existing_product"
SCHEMA_VERSION = 4
TEMPLATE_ID = "travel_graph_existing_product_v4"
ATTRACTION_SEEDS = [
("黄果树", ["黄果树", "黄果树瀑布", "黄果树大瀑布", "黄果树风景名胜区"], "安顺", "瀑布/5A", "贵州龙头景区,瀑布群核心卖点。"),
("天星桥", ["天星桥", "天星桥景区"], "安顺", "喀斯特/黄果树景区", "水上石林、天然盆景。"),
("陡坡塘瀑布", ["陡坡塘", "陡坡塘瀑布"], "安顺", "瀑布/黄果树景区", "瀑面宽,西游记取景。"),
("荔波小七孔", ["小七孔", "荔波小七孔", "小七孔景区"], "黔南", "山水/5A", "世界自然遗产,水上森林、卧龙潭等。"),
("西江千户苗寨", ["西江", "西江苗寨", "西江千户苗寨"], "黔东南", "民族村寨/4A", "苗寨夜景、长桌宴、吊脚楼。"),
("镇远古城", ["镇远", "镇远古镇", "镇远古城"], "黔东南", "古城/5A", "古城夜景、舞阳河沿岸住宿。"),
("梵净山", ["梵净山"], "铜仁", "山岳/5A", "弥勒道场、蘑菇石、金顶。"),
("青岩古镇", ["青岩", "青岩古镇"], "贵阳", "古镇/5A", "卤猪脚、小吃、送机前半日游。"),
("百里杜鹃", ["百里杜鹃"], "毕节", "赏花", "3-4月花期主题。"),
("平坝樱花", ["平坝樱花", "平坝农场"], "安顺", "赏花", "春季樱花主题。"),
("织金洞", ["织金洞"], "毕节", "溶洞/5A", "大型喀斯特溶洞。"),
("中国天眼", ["天眼", "中国天眼", "FAST"], "黔南", "科技研学", "天文研学卖点。"),
("茅台镇", ["茅台", "茅台镇"], "遵义", "酒文化", "酱酒文化体验。"),
("遵义会议会址", ["遵义会址", "遵义会议会址"], "遵义", "红色文化", "红色研学路线核心。"),
("兴义万峰林", ["万峰林", "兴义万峰林"], "黔西南", "峰林", "黔西南山水。"),
("万峰湖", ["万峰湖"], "黔西南", "湖泊", "兴义水上体验。"),
("马岭河峡谷", ["马岭河", "马岭河峡谷"], "黔西南", "峡谷", "兴义峡谷景观。"),
("花江大桥", ["花江大桥"], "安顺/黔西南", "桥梁景观", "桥见贵州特色线路。"),
("龙宫", ["龙宫"], "安顺", "溶洞/5A", "安顺秘境类产品。"),
("天河潭", ["天河潭"], "贵阳", "山水", "贵阳近郊半日/首日。"),
("甲秀楼", ["甲秀楼"], "贵阳", "城市地标", "贵阳市区地标。"),
("黔灵山公园", ["黔灵公园", "黔灵山"], "贵阳", "城市公园", "贵阳市区轻量游。"),
("乌江寨", ["乌江寨"], "遵义", "度假街区", "夜游/住宿度假。"),
("野洞河", ["野洞河"], "黔东南", "漂流", "漂流体验。"),
("安顺古城", ["安顺古城"], "安顺", "城市/夜游", "安顺中转游览。"),
("中南门古城", ["中南门", "中南门古城"], "铜仁", "古城/夜游", "铜仁夜游街区。"),
]
LOCATION_RESOURCE_LABELS = {"ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"}
SCENIC_AREA_DEFINITIONS = {
"黄果树旅游景区": {
"aliases": ["黄果树景区", "黄果树风景名胜区", "黄果树瀑布景区"],
"city": "安顺",
"admin_hint": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"area_type": "国家级风景名胜区/5A景区",
"route_anchor": True,
"members": {
"黄果树": "核心瀑布游览点",
"天星桥": "黄果树景区子景点/游览点",
"陡坡塘瀑布": "黄果树景区子景点/游览点",
},
"note": "黄果树片区作为资源池锚点;天星桥、陡坡塘、大瀑布不作为平级主目的地推荐。",
},
}
SCENIC_MEMBER_PARENT = {
member: (area_name, role)
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
for member, role in spec["members"].items()
}
ATTRACTION_ADMIN_HINTS = {
"黄果树": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"天星桥": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"陡坡塘瀑布": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"荔波小七孔": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
"西江千户苗寨": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
"镇远古城": ("贵州省", "黔东南苗族侗族自治州", "镇远县"),
"梵净山": ("贵州省", "铜仁市", "江口县"),
"青岩古镇": ("贵州省", "贵阳市", "花溪区"),
"百里杜鹃": ("贵州省", "毕节市", "百里杜鹃管理区"),
"平坝樱花": ("贵州省", "安顺市", "平坝区"),
"织金洞": ("贵州省", "毕节市", "织金县"),
"中国天眼": ("贵州省", "黔南布依族苗族自治州", "平塘县"),
"茅台镇": ("贵州省", "遵义市", "仁怀市"),
"遵义会议会址": ("贵州省", "遵义市", "红花岗区"),
"兴义万峰林": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"万峰湖": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"马岭河峡谷": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"花江大桥": ("贵州省", "安顺市", "关岭布依族苗族自治县"),
"龙宫": ("贵州省", "安顺市", "西秀区"),
"天河潭": ("贵州省", "贵阳市", "花溪区"),
"甲秀楼": ("贵州省", "贵阳市", "南明区"),
"黔灵山公园": ("贵州省", "贵阳市", "云岩区"),
"乌江寨": ("贵州省", "遵义市", "播州区"),
"野洞河": ("贵州省", "黔东南苗族侗族自治州", ""),
"安顺古城": ("贵州省", "安顺市", "西秀区"),
"中南门古城": ("贵州省", "铜仁市", "碧江区"),
}
def scenic_area_name_for_attraction(name: str) -> str:
if name in {"茅台镇"}:
return f"{name}酒文化片区"
if name in {"花江大桥"}:
return f"{name}观景片区"
if name in {"甲秀楼", "中南门古城", "安顺古城"}:
return f"{name}游览片区"
return f"{name}景区"
def expand_scenic_area_definitions() -> None:
"""Every route destination needs a scenic-area/resource-pool anchor.
黄果树 is a compound scenic area with child visit points. Other core route
destinations currently have one representative ScenicAttraction, but still
need a ScenicArea anchor so the browser can show:
行政区 -> 景区/片区 -> 酒店/餐饮/路线产品。
"""
existing_members = {
member
for spec in SCENIC_AREA_DEFINITIONS.values()
for member in (spec.get("members") or {})
}
for name, aliases, city, attraction_type, point in ATTRACTION_SEEDS:
if name in existing_members:
continue
area_name = scenic_area_name_for_attraction(name)
if area_name in SCENIC_AREA_DEFINITIONS:
continue
admin_hint = ATTRACTION_ADMIN_HINTS.get(name, ("贵州省", city, ""))
SCENIC_AREA_DEFINITIONS[area_name] = {
"aliases": sorted(set([area_name, name, *aliases])),
"city": city,
"admin_hint": admin_hint,
"area_type": attraction_type,
"route_anchor": True,
"standalone_area": True,
"representative_member": name,
"members": {name: "核心游览点/主目的地"},
"note": f"{area_name}作为{city}线路资源池锚点;酒店/餐饮候选以景区锚点高德车程计算。",
}
expand_scenic_area_definitions()
SCENIC_MEMBER_PARENT = {
member: (area_name, role)
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
for member, role in (spec.get("members") or {}).items()
}
SCENIC_MEMBER_IS_CHILD = {
member: not bool(spec.get("standalone_area"))
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
for member in (spec.get("members") or {})
}
REGION_TEXT_HINTS = {
"贵阳": ("贵州省", "贵阳市", ""),
"龙洞堡": ("贵州省", "贵阳市", "南明区"),
"双龙": ("贵州省", "贵阳市", ""),
"青岩": ("贵州省", "贵阳市", "花溪区"),
"花溪": ("贵州省", "贵阳市", "花溪区"),
"甲秀": ("贵州省", "贵阳市", "南明区"),
"黔灵": ("贵州省", "贵阳市", "云岩区"),
"安顺": ("贵州省", "安顺市", ""),
"黄果树": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"天星桥": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"陡坡塘": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
"龙宫": ("贵州省", "安顺市", "西秀区"),
"西江": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
"苗寨": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
"镇远": ("贵州省", "黔东南苗族侗族自治州", "镇远县"),
"梵净山": ("贵州省", "铜仁市", "江口县"),
"江口": ("贵州省", "铜仁市", "江口县"),
"铜仁": ("贵州省", "铜仁市", ""),
"荔波": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
"小七孔": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
"织金": ("贵州省", "毕节市", "织金县"),
"毕节": ("贵州省", "毕节市", ""),
"百里杜鹃": ("贵州省", "毕节市", "百里杜鹃管理区"),
"开阳": ("贵州省", "贵阳市", "开阳县"),
"猴耳天坑": ("贵州省", "贵阳市", "开阳县"),
"遵义": ("贵州省", "遵义市", ""),
"乌江寨": ("贵州省", "遵义市", "播州区"),
"茅台": ("贵州省", "遵义市", "仁怀市"),
"遵义会址": ("贵州省", "遵义市", "红花岗区"),
"习水": ("贵州省", "遵义市", "习水县"),
"四渡赤水": ("贵州省", "遵义市", "习水县"),
"独山": ("贵州省", "黔南布依族苗族自治州", "独山县"),
"都匀": ("贵州省", "黔南布依族苗族自治州", "都匀市"),
"龙里": ("贵州省", "黔南布依族苗族自治州", "龙里县"),
"肇兴": ("贵州省", "黔东南苗族侗族自治州", "黎平县"),
"从江": ("贵州省", "黔东南苗族侗族自治州", "从江县"),
"丹寨": ("贵州省", "黔东南苗族侗族自治州", "丹寨县"),
"兴义": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"万峰林": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"马岭河": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
"晴隆": ("贵州省", "黔西南布依族苗族自治州", "晴隆县"),
}
RESTAURANT_SECTION_REGION_ALIASES = {
"黔东南": "黔东南区域",
"黔北": "遵义区域",
"黔西": "黔西南区域",
"黔西南": "黔西南区域",
}
def clean(value: Any) -> str:
if value is None:
return ""
if isinstance(value, float) and pd.isna(value):
return ""
text = str(value).replace("\x00", "").replace("\u200b", "").replace("\u200f", "")
text = re.sub(r"[ \t]+", " ", text)
return re.sub(r"\n{3,}", "\n\n", text).strip()
def compact(value: Any) -> str:
return re.sub(r"\s+", " ", clean(value)).strip()
def norm(value: Any) -> str:
return re.sub(r"[\s()()《》【】、,。::/\\\\·++\-—_]+", "", compact(value)).lower()
def slug(text: str, prefix: str = "") -> str:
base = re.sub(r"[\s()()《》【】、,。::/\\\\]+", "_", compact(text))
base = re.sub(r"_+", "_", base).strip("_")
digest = hashlib.md5(compact(text).encode("utf-8")).hexdigest()[:8]
return f"{prefix}{base[:50]}_{digest}"
def digest(*parts: Any, length: int = 10) -> str:
raw = "||".join(compact(p) for p in parts)
return hashlib.md5(raw.encode("utf-8")).hexdigest()[:length].upper()
def money(value: Any) -> float | None:
m = re.search(r"-?\d+(?:\.\d+)?", compact(value))
return float(m.group()) if m else None
def number(value: Any) -> float | None:
try:
if value in (None, ""):
return None
return float(value)
except Exception:
m = re.search(r"-?\d+(?:\.\d+)?", compact(value))
return float(m.group()) if m else None
def haversine_km(lng1: float, lat1: float, lng2: float, lat2: float) -> float:
radius = 6371.0088
phi1 = math.radians(lat1)
phi2 = math.radians(lat2)
d_phi = math.radians(lat2 - lat1)
d_lam = math.radians(lng2 - lng1)
a = math.sin(d_phi / 2) ** 2 + math.cos(phi1) * math.cos(phi2) * math.sin(d_lam / 2) ** 2
return radius * (2 * math.atan2(math.sqrt(a), math.sqrt(1 - a)))
def split_items(text: str, limit: int = 40) -> list[str]:
parts = re.split(r"[、,,;/;\n]+", compact(text))
out: list[str] = []
seen: set[str] = set()
for part in parts:
part = part.strip()
if not part or part.lower() == "nan" or part in seen:
continue
seen.add(part)
out.append(part)
if len(out) >= limit:
break
return out
def clean_region_label(text: str) -> str:
value = compact(text)
value = re.sub(r"^\s*\d+(?:\.\d+)?\s*", "", value)
return value.strip()
def urls_from_text(text: Any) -> list[str]:
return re.findall(r"https?://[^\s,,;;]+", clean(text))
def read_office_text(path: Path) -> str:
proc = subprocess.run(["textutil", "-convert", "txt", "-stdout", str(path)], capture_output=True, text=True, check=False)
return proc.stdout if proc.returncode == 0 else ""
def duration_from_text(text: str) -> int | None:
m = re.search(r"(\d+)\s*(?:日游|天|日)", text)
if m:
return int(m.group(1))
cn = {"一": 1, "二": 2, "两": 2, "三": 3, "四": 4, "五": 5, "六": 6, "七": 7, "八": 8}
m = re.search(r"([一二两三四五六七八])日游", text)
return cn.get(m.group(1)) if m else None
def graph_safe_props(node: dict[str, Any]) -> dict[str, Any]:
props: dict[str, Any] = {}
for key, value in node.items():
if key == "label" or value is None:
continue
if isinstance(value, (dict, list)):
props[key] = json.dumps(value, ensure_ascii=False)
elif isinstance(value, (int, float, bool, str)):
props[key] = value
else:
props[key] = str(value)
return props
class KGBuilder:
def __init__(self) -> None:
self.nodes: dict[str, dict[str, Any]] = {}
self.relations: list[dict[str, Any]] = []
self._rel_seen: set[tuple[str, str, str, str]] = set()
def add_node(self, label: str, key: str, name: str, **props: Any) -> str:
if not key:
key = f"{label.lower()}:{slug(name)}"
payload = {
"label": label,
"natural_key": key,
"name": compact(name) or key,
**{k: v for k, v in props.items() if v not in (None, "", [], {})},
}
existing = self.nodes.get(key)
if existing:
merged = {**existing, **payload}
for field in (
"aliases", "source_files", "selling_points", "features", "applicable_products",
"signature_dishes", "meal_scene", "image_urls", "image_ids", "media_resource_ids",
"main_destinations", "service_scope",
):
vals: list[str] = []
for src in (existing.get(field), payload.get(field)):
if isinstance(src, list):
vals.extend(compact(x) for x in src if compact(x))
elif src:
vals.append(compact(src))
if vals:
merged[field] = sorted(set(vals))
self.nodes[key] = merged
else:
self.nodes[key] = payload
return key
def add_rel(self, rel_type: str, source: str, target: str, **props: Any) -> None:
if not source or not target or source == target:
return
payload = {k: v for k, v in props.items() if v not in (None, "", [], {})}
identity = (rel_type, source, target, json.dumps(payload, ensure_ascii=False, sort_keys=True))
if identity in self._rel_seen:
return
self._rel_seen.add(identity)
self.relations.append({"relation_type": rel_type, "source": source, "target": target, "properties": payload})
def load_schema() -> dict[str, Any]:
schema = json.loads(SCHEMA_SRC.read_text(encoding="utf-8"))
schema["namespace"] = SCHEMA_NAMESPACE
schema["version"] = "1.4"
schema["display_name"] = "旅行社线路制定知识图谱 Schema"
schema["purpose"] = "面向已有路线产品的客服问答和产品微调:固定路线骨架、可配置资源槽位、价格、规则、图片素材统一建模。"
schema["entity_types"]["PriceCatalogItem"] = {
"cn": "报价目录项",
"definition": "价格表中出现但未在路线文档中形成完整每日行程骨架的报价对象;用于保留报价,不作为完整路线推荐。",
"fields": [
"catalog_item_id", "name", "catalog_type", "duration_days", "product_direction",
"route_preview", "price_validity", "booking_notes", "source_file", "evidence_excerpt",
],
}
schema["entity_types"]["ResourceLibrary"] = {
"cn": "业务资料库",
"definition": "用于表达业务资料库/基础资源库/已有路线产品库等聚合目录,让图谱在业务视角下可浏览、可解释。",
"fields": ["library_id", "library_type", "name", "description", "scope", "source_file"],
}
schema["entity_types"]["ScenicArea"] = {
"cn": "景区/景点片区",
"definition": "面向路线推荐的景区级资源池锚点,例如黄果树旅游景区;用于把天星桥、陡坡塘等子游览点归到同一景区片区。",
"fields": [
"scenic_area_id", "name", "aliases", "city", "area_type", "route_anchor",
"admin_region_name", "admin_region_level", "resource_spatial_class",
"management_note", "source_files",
],
}
for entity in ["TicketFee", "FeeItem"]:
fields = schema["entity_types"][entity]["fields"]
for field in [
"price_text", "adult_price", "child_price", "currency", "consumer_group",
"is_free", "is_optional", "price_status", "child_price_status",
"scenic_target_name", "scenic_target_type", "source_product_name",
"evidence_excerpt", "data_quality_status", "extraction_rule",
]:
if field not in fields:
fields.append(field)
schema["entity_types"]["AdministrativeRegion"] = {
"cn": "行政区",
"definition": "贵州旅游资源的行政区底座,只为当前已入图景点及其相关资源建立省/市州/区县层级。",
"fields": [
"region_id", "name", "level", "province", "city", "county", "town",
"parent_id", "adcode", "center_lng", "center_lat", "geo_polygon",
"source_system", "source_file", "scope_note",
],
}
schema["entity_types"]["GeoPoint"] = {
"cn": "地理坐标点",
"definition": "资源的经纬度坐标,用于附近酒店/餐厅和距离推荐;车辆等服务资源不按附近关系建模。",
"fields": [
"geo_point_id", "lng", "lat", "address_text", "provider",
"precision_level", "source_system", "source_file",
],
}
image_fields = ["primary_image_url", "image_urls", "media_resource_ids", "media_reliability", "media_usage_note"]
for entity in ["ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource", "VehicleResource", "ResourceOptionGroup", "TourProduct"]:
fields = schema["entity_types"][entity]["fields"]
for field in image_fields:
if field not in fields:
fields.append(field)
amap_fields = [
"amap_poi_id", "amap_name", "amap_type", "amap_typecode", "amap_address",
"amap_location", "amap_lng", "amap_lat", "amap_pname", "amap_cityname",
"amap_adname", "amap_adcode", "amap_tel", "amap_rating", "amap_cost", "amap_open_time",
"amap_photo_urls", "amap_match_status", "amap_match_confidence",
"amap_match_reason", "data_completeness_note",
]
for entity in ["ScenicAttraction", "HotelResource", "RestaurantResource"]:
fields = schema["entity_types"][entity]["fields"]
for field in amap_fields:
if field not in fields:
fields.append(field)
location_fields = [
"resource_spatial_class", "location_lng", "location_lat", "admin_region_name",
"admin_region_level", "admin_region_source", "geo_match_status",
]
for entity in ["ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"]:
fields = schema["entity_types"][entity]["fields"]
for field in location_fields:
if field not in fields:
fields.append(field)
attraction_hierarchy_fields = [
"attraction_level", "parent_scenic_area_name", "is_independent_destination",
"route_role", "walk_intensity_hint", "recommendation_note",
]
fields = schema["entity_types"]["ScenicAttraction"]["fields"]
for field in attraction_hierarchy_fields:
if field not in fields:
fields.append(field)
route_region_fields = ["admin_region_name", "admin_region_level", "route_region_path", "origin_region_name", "destination_region_name"]
for entity in ["ProductDay", "RouteStop", "RouteSegment"]:
fields = schema["entity_types"][entity]["fields"]
for field in route_region_fields:
if field not in fields:
fields.append(field)
route_line_fields = [
"route_line_name", "route_line_id", "route_sequence_label", "route_display_name",
"route_stop_sequence", "route_day_summary", "route_display_text",
"global_stop_order", "day_stop_order", "station_like_name", "total_stop_count",
"previous_stop_name", "next_stop_name",
]
for entity in ["TourProduct", "ProductDay", "RouteStop"]:
fields = schema["entity_types"][entity]["fields"]
for field in route_line_fields:
if field not in fields:
fields.append(field)
vehicle_fields = ["resource_service_class", "service_region_mode", "suitable_group_size", "dispatch_basis"]
fields = schema["entity_types"]["VehicleResource"]["fields"]
for field in vehicle_fields:
if field not in fields:
fields.append(field)
schema["relation_types"]["PRICE_PACKAGE_HAS_FEE"] = ["ProductPricePackage", "FeeItem|TicketFee", "价格包包含费用项/门票小交通费用"]
schema["relation_types"]["CATALOG_ITEM_HAS_PRICE_PACKAGE"] = ["PriceCatalogItem", "ProductPricePackage", "报价目录项拥有价格包"]
schema["relation_types"]["CATALOG_ITEM_HAS_RULE"] = ["PriceCatalogItem", "BusinessRule", "报价目录项适用规则"]
schema["relation_types"]["CATALOG_ITEM_MATCHES_PRODUCT"] = ["PriceCatalogItem", "TourProduct", "报价目录项匹配已有路线产品"]
schema["relation_types"]["LIBRARY_HAS_SECTION"] = ["ResourceLibrary", "ResourceLibrary", "资料库包含子库"]
schema["relation_types"]["LIBRARY_CONTAINS_PRODUCT"] = ["ResourceLibrary", "TourProduct", "路线产品库包含已有路线产品"]
schema["relation_types"]["LIBRARY_CONTAINS_CATALOG_ITEM"] = ["ResourceLibrary", "PriceCatalogItem", "报价目录库包含报价项"]
schema["relation_types"]["LIBRARY_CONTAINS_ATTRACTION"] = ["ResourceLibrary", "ScenicAttraction", "景点资源库包含景点"]
schema["relation_types"]["LIBRARY_CONTAINS_SCENIC_AREA"] = ["ResourceLibrary", "ScenicArea", "资料库包含景区/片区锚点"]
schema["relation_types"]["LIBRARY_CONTAINS_HOTEL"] = ["ResourceLibrary", "HotelResource", "酒店资源库包含酒店"]
schema["relation_types"]["LIBRARY_CONTAINS_RESTAURANT"] = ["ResourceLibrary", "RestaurantResource", "餐饮资源库包含餐厅"]
schema["relation_types"]["LIBRARY_CONTAINS_VEHICLE"] = ["ResourceLibrary", "VehicleResource", "车辆资源库包含车辆"]
schema["relation_types"]["LIBRARY_CONTAINS_TRANSFER"] = ["ResourceLibrary", "TransferQuote", "接送报价库包含接送报价"]
schema["relation_types"]["LIBRARY_CONTAINS_OPTION_GROUP"] = ["ResourceLibrary", "ResourceOptionGroup", "基础资源库包含可选资源组"]
schema["relation_types"]["LIBRARY_CONTAINS_RULE"] = ["ResourceLibrary", "BusinessRule", "规则库包含业务规则"]
schema["relation_types"]["LIBRARY_CONTAINS_MEDIA"] = ["ResourceLibrary", "MediaResource", "素材库包含图片资源"]
schema["relation_types"]["GROUP_SERVES_ATTRACTION"] = ["ResourceOptionGroup", "ScenicAttraction", "资源组选项服务于景区/景点"]
schema["relation_types"]["HOTEL_SERVES_ATTRACTION"] = ["HotelResource", "ScenicAttraction", "酒店资源可服务于景区/景点住宿"]
schema["relation_types"]["RESTAURANT_SERVES_ATTRACTION"] = ["RestaurantResource", "ScenicAttraction", "餐厅资源可服务于景区/景点用餐"]
schema["relation_types"]["REGION_PARENT_OF"] = ["AdministrativeRegion", "AdministrativeRegion", "上级行政区包含下级行政区"]
schema["relation_types"]["SCENIC_AREA_HAS_ATTRACTION"] = ["ScenicArea", "ScenicAttraction", "景区包含具体游览点/子景点"]
schema["relation_types"]["ATTRACTION_PART_OF_SCENIC_AREA"] = ["ScenicAttraction", "ScenicArea", "具体游览点归属于景区片区"]
schema["relation_types"]["SCENIC_AREA_HAS_FEE"] = ["ScenicArea", "TicketFee|FeeItem", "景区片区拥有门票、小交通、保险或二次消费费用"]
schema["relation_types"]["ATTRACTION_HAS_FEE"] = ["ScenicAttraction", "TicketFee|FeeItem", "具体景点/游览点拥有门票、小交通、保险或二次消费费用"]
schema["relation_types"]["PRODUCT_HAS_FEE"] = ["TourProduct", "TicketFee|FeeItem", "产品涉及费用项目"]
schema["relation_types"]["LOCATED_IN_REGION"] = ["ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "AdministrativeRegion", "位置资源位于行政区"]
schema["relation_types"]["RESOURCE_HAS_GEOPOINT"] = ["ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "GeoPoint", "位置资源拥有经纬度坐标"]
schema["relation_types"]["NEARBY_LOCATION_RESOURCE"] = ["ScenicArea|ScenicAttraction", "HotelResource|RestaurantResource", "景区片区或独立景点附近位置资源,最终关系必须来自高德驾车距离/耗时;子景点不直接连酒店餐饮"]
schema["relation_types"]["DRIVING_ROUTE_METRIC"] = ["ScenicArea|ScenicAttraction", "ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "高德驾车距离与耗时指标,用于景区顺序、酒店/餐饮候选排序和行程可行性判断"]
schema["relation_types"]["VEHICLE_SUITABLE_FOR_PRODUCT"] = ["VehicleResource", "TourProduct", "车辆按人数/车型/服务能力适配产品,不按景点附近推荐"]
schema["relation_types"]["STOP_LOCATED_IN_REGION"] = ["RouteStop", "AdministrativeRegion", "停靠点落入行政区"]
schema["relation_types"]["STOP_VISITS_SCENIC_AREA"] = ["RouteStop", "ScenicArea", "路线停靠点访问景区片区"]
schema["relation_types"]["PRODUCT_HAS_ORDERED_STOP"] = ["TourProduct", "RouteStop", "产品线路按全程顺序包含停靠点,等价公交线路-站点序列"]
schema["relation_types"]["ROUTE_STOP_NEXT"] = ["RouteStop", "RouteStop", "同一产品线路中相邻停靠点的先后关系"]
schema["relation_types"]["DAY_NEXT_DAY"] = ["ProductDay", "ProductDay", "同一产品线路中每日行程的先后关系"]
schema["relation_types"]["DAY_COVERS_SCENIC_AREA"] = ["ProductDay", "ScenicArea", "每日行程覆盖景区片区"]
schema["relation_types"]["PRODUCT_COVERS_SCENIC_AREA"] = ["TourProduct", "ScenicArea", "产品覆盖景区片区"]
schema["relation_types"]["DAY_COVERS_REGION"] = ["ProductDay", "AdministrativeRegion", "每日行程覆盖行政区"]
schema["relation_types"]["SEGMENT_FROM_REGION"] = ["RouteSegment", "AdministrativeRegion", "移动段起点行政区"]
schema["relation_types"]["SEGMENT_TO_REGION"] = ["RouteSegment", "AdministrativeRegion", "移动段终点行政区"]
schema["relation_types"]["SLOT_CAN_USE_LOCATION_RESOURCE"] = ["ResourceSlot", "HotelResource|RestaurantResource", "住宿/餐饮槽位可用合作位置资源"]
schema["relation_types"]["SLOT_CAN_USE_SERVICE_RESOURCE"] = ["ResourceSlot", "VehicleResource", "车辆等服务槽位可用服务资源"]
return schema
def schema_to_dsl(schema: dict[str, Any]) -> str:
list_fields = {
"aliases", "source_files", "selling_points", "features", "applicable_products", "signature_dishes",
"meal_scene", "image_urls", "image_ids", "image_urls", "media_resource_ids", "main_destinations",
"image_urls", "image_ids", "image_urls", "service_scope", "amap_photo_urls",
"suitable_group_size",
}
lines = ["```text", f"namespace {schema['namespace']}", f"version {schema['version']}", ""]
for name, spec in schema["entity_types"].items():
lines.append(f"{name}({spec['cn']}): EntityType")
lines.append(" properties:")
for field in spec["fields"]:
typ = "TextList" if field in list_fields else "Text"
if any(token in field for token in ("count", "days", "nights", "price", "value", "order", "index", "min", "max", "score", "lng", "lat", "distance")):
typ = "Number"
if field in {"route_immutable", "required", "changeable", "customer_visible", "default_option", "confirmed", "confirmation_required", "refundable", "is_core_visit"}:
typ = "Boolean"
lines.append(f" {field}: {typ}")
lines.append("")
for rel, (start, end, desc) in schema["relation_types"].items():
lines.append(f"{rel}({desc}): RelationType")
lines.append(f" startNode: {start}")
lines.append(f" endNode: {end}")
lines.append("")
lines.append("```")
return "\n".join(lines)
class MediaIndex:
def __init__(self) -> None:
self.by_alias: dict[str, list[str]] = defaultdict(list)
self.by_id: dict[str, str] = {}
self.label_by_key: dict[str, str] = {}
self.accepted_vehicle_names: set[str] = set()
def add_alias(self, alias: str, media_key: str) -> None:
n = norm(alias)
if len(n) >= 2 and media_key not in self.by_alias[n]:
self.by_alias[n].append(media_key)
def find_for_name(self, name: str, aliases: list[str] | None = None, allowed_labels: set[str] | None = None) -> list[str]:
candidates: list[str] = []
names = [name] + (aliases or [])
normalized = [norm(x) for x in names if len(norm(x)) >= 2]
for n in normalized:
candidates.extend(self.by_alias.get(n, []))
if not candidates:
for alias_norm, media_keys in self.by_alias.items():
if any((n and len(n) >= 3 and (n in alias_norm or alias_norm in n)) for n in normalized):
candidates.extend(media_keys)
out: list[str] = []
for key in candidates:
if allowed_labels and self.label_by_key.get(key) not in allowed_labels:
continue
if key not in out:
out.append(key)
return out[:3]
def apply_media(
builder: KGBuilder,
media_index: MediaIndex,
node_key: str,
name: str,
aliases: list[str] | None = None,
allowed_labels: set[str] | None = None,
) -> None:
media_keys = media_index.find_for_name(name, aliases, allowed_labels)
if not media_keys:
return
node = builder.nodes[node_key]
image_urls: list[str] = []
image_ids: list[str] = []
reliabilities: list[str] = []
usage_notes: list[str] = []
for media_key in media_keys:
media = builder.nodes.get(media_key, {})
image_urls.extend(media.get("image_urls") or [])
image_ids.extend(media.get("image_ids") or [])
if media.get("reliability_level"):
reliabilities.append(media["reliability_level"])
if media.get("usage_note"):
usage_notes.append(media["usage_note"])
builder.add_rel("RESOURCE_HAS_MEDIA", node_key, media_key, match_rule="名称/别名匹配图片资源库")
if image_urls:
node["primary_image_url"] = image_urls[0]
node["image_urls"] = sorted(set(image_urls))
node["media_resource_ids"] = [builder.nodes[m].get("media_id") for m in media_keys if builder.nodes.get(m)]
node["media_reliability"] = ";".join(sorted(set(reliabilities)))
node["media_usage_note"] = ";".join(sorted(set(usage_notes)))[:500]
def load_amap_enrichment_cache() -> dict[str, dict[str, Any]]:
if not AMAP_CACHE_PATH.exists():
return {}
try:
payload = json.loads(AMAP_CACHE_PATH.read_text(encoding="utf-8"))
except Exception:
return {}
if isinstance(payload, dict) and isinstance(payload.get("items"), dict):
return payload["items"]
if isinstance(payload, dict):
return payload
return {}
def load_amap_driving_metric_cache() -> dict[str, dict[str, Any]]:
if not AMAP_DRIVING_CACHE_PATH.exists():
return {}
try:
payload = json.loads(AMAP_DRIVING_CACHE_PATH.read_text(encoding="utf-8"))
except Exception:
return {}
if isinstance(payload, dict) and isinstance(payload.get("items"), dict):
return payload["items"]
if isinstance(payload, dict):
return payload
return {}
def apply_amap_enrichment_layer(builder: KGBuilder, cache: dict[str, dict[str, Any]]) -> None:
if not cache:
return
target_labels = {"ScenicAttraction", "HotelResource", "RestaurantResource"}
for key, node in builder.nodes.items():
if node.get("label") not in target_labels:
continue
item = cache.get(key) or cache.get(node.get("natural_key", ""))
if not isinstance(item, dict):
continue
fields = item.get("fields") if isinstance(item.get("fields"), dict) else item
status = compact(fields.get("amap_match_status") or item.get("status"))
if status:
node["amap_match_status"] = status
for field in [
"amap_poi_id", "amap_name", "amap_type", "amap_typecode", "amap_address",
"amap_location", "amap_lng", "amap_lat", "amap_pname", "amap_cityname",
"amap_adname", "amap_adcode", "amap_tel", "amap_rating", "amap_cost", "amap_open_time",
"amap_photo_urls", "amap_match_confidence", "amap_match_reason",
"data_completeness_note",
]:
value = fields.get(field)
if value not in (None, "", [], {}):
node[field] = value
def upsert_relation_props(builder: KGBuilder, rel_type: str, source: str, target: str, **props: Any) -> None:
payload = {k: v for k, v in props.items() if v not in (None, "", [], {})}
if not payload:
return
for rel in builder.relations:
if rel["relation_type"] == rel_type and rel["source"] == source and rel["target"] == target:
rel.setdefault("properties", {}).update(payload)
return
builder.add_rel(rel_type, source, target, **payload)
def apply_driving_metric_layer(builder: KGBuilder, cache: dict[str, dict[str, Any]]) -> None:
if not cache:
return
for item in cache.values():
if not isinstance(item, dict):
continue
if item.get("status") != "matched":
continue
source_key = compact(item.get("source_key"))
target_key = compact(item.get("target_key"))
if source_key not in builder.nodes or target_key not in builder.nodes:
continue
source = builder.nodes[source_key]
target = builder.nodes[target_key]
source_label = source.get("label")
if source_label not in {"ScenicArea", "ScenicAttraction"}:
continue
if source_label == "ScenicAttraction" and source.get("is_independent_destination") is False:
continue
target_label = target.get("label")
if target_label not in {"ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"}:
continue
if target_label == "ScenicAttraction" and target.get("is_independent_destination") is False:
continue
distance_km = number(item.get("drive_distance_km"))
duration_min = number(item.get("drive_duration_min"))
metric_props = {
"metric_scope": item.get("metric_scope"),
"resource_type": item.get("target_resource_type"),
"drive_distance_km": distance_km,
"drive_duration_min": duration_min,
"amap_distance_m": item.get("amap_distance_m"),
"amap_duration_s": item.get("amap_duration_s"),
"provider": item.get("provider") or "amap",
"api": item.get("api") or "amap_distance",
"route_type": item.get("route_type") or "driving",
"region_match_level": item.get("region_match_level"),
"same_admin_region": item.get("same_admin_region"),
"source_region": item.get("source_region"),
"target_region": item.get("target_region"),
"origin_location": item.get("origin_location"),
"destination_location": item.get("destination_location"),
"updated_at": item.get("updated_at"),
"rule": "高德驾车距离/耗时;用于候选资源排序与行程可行性判断,价格和房态仍需二次确认",
}
builder.add_rel("DRIVING_ROUTE_METRIC", source_key, target_key, **metric_props)
if target_label in {"HotelResource", "RestaurantResource"}:
upsert_relation_props(
builder,
"NEARBY_LOCATION_RESOURCE",
source_key,
target_key,
resource_type="hotel" if target_label == "HotelResource" else "restaurant",
distance_km=item.get("straight_distance_km"),
drive_distance_km=distance_km,
drive_duration_min=duration_min,
metric_scope=item.get("metric_scope"),
region_match_level=item.get("region_match_level"),
candidate_basis="same_region_with_amap_driving_metric",
rule="同区县/同城可服务资源,并已补充高德驾车距离与耗时;推荐时优先按车程、房态、餐标过滤",
)
def load_media_resources(builder: KGBuilder) -> MediaIndex:
media_index = MediaIndex()
path = SOURCE_ROOT / "图片资源库_全品类别名索引.xlsx"
alias_df = pd.read_excel(path, sheet_name="别名索引")
group_df = pd.read_excel(path, sheet_name="资源分组")
alias_map: dict[str, list[str]] = defaultdict(list)
no_match_notes: dict[str, str] = {}
for _, row in alias_df.iterrows():
resource_id = compact(row.get("资源ID"))
alias = compact(row.get("别名"))
if resource_id and alias:
alias_map[resource_id].append(alias)
if compact(row.get("禁止误匹配")):
no_match_notes[resource_id] = compact(row.get("禁止误匹配"))
for _, row in group_df.iterrows():
resource_id = compact(row.get("资源ID"))
label = compact(row.get("标签"))
title = compact(row.get("主标题"))
reliability = compact(row.get("匹配级别"))
urls = urls_from_text(row.get("图片链接列表"))
if not resource_id or not title or not urls:
continue
if label == "待确认图片" or reliability in {"低", "无图/禁止冒充"}:
continue
aliases = split_items(row.get("别名列表"), limit=80) + alias_map.get(resource_id, [])
label_for_match = label
if label == "景点" and any(token in title for token in ["酒店", "客栈", "民宿", "宾馆", "维也纳", "住宿"]):
label_for_match = "酒店"
key = builder.add_node(
"MediaResource",
f"media:{resource_id}",
title,
media_id=resource_id,
title=title,
media_type="image_group",
resource_label=label_for_match,
original_resource_label=label,
reliability_level=reliability,
image_ids=split_items(row.get("图片编号列表"), limit=80),
image_urls=urls,
primary_image_url=urls[0],
aliases=sorted(set(a for a in aliases if a)),
usage_note=compact(row.get("备注")) or compact(row.get("说明")) or ("禁止误匹配:" + no_match_notes.get(resource_id, "") if no_match_notes.get(resource_id) else ""),
source_file=str(path),
)
media_index.by_id[resource_id] = key
media_index.label_by_key[key] = label_for_match
media_index.add_alias(title, key)
for alias in aliases:
media_index.add_alias(alias, key)
if label_for_match == "车辆" and reliability in {"高", "中"} and "待确认" not in title and "缺少可靠" not in title:
media_index.accepted_vehicle_names.add(title)
return media_index
def seed_attractions(builder: KGBuilder, media_index: MediaIndex) -> dict[str, str]:
alias_to_key: dict[str, str] = {}
scenic_area_keys: dict[str, str] = {}
for area_name, spec in SCENIC_AREA_DEFINITIONS.items():
key = builder.add_node(
"ScenicArea",
f"scenic_area:{slug(area_name)}",
area_name,
scenic_area_id=f"SA-{digest(area_name, length=8)}",
aliases=spec.get("aliases") or [],
city=spec.get("city"),
area_type=spec.get("area_type"),
route_anchor=bool(spec.get("route_anchor")),
resource_spatial_class="location_resource",
management_note=spec.get("note"),
source_files=["种子景区层级+产品路线+高德POI"],
)
scenic_area_keys[area_name] = key
for name, aliases, city, attraction_type, point in ATTRACTION_SEEDS:
area_key = builder.add_node("Area", f"area:{slug(city)}", city, area_id=f"AREA-{digest(city, length=8)}", area_type="目的地区域")
parent_area_name, member_role = SCENIC_MEMBER_PARENT.get(name, ("", ""))
is_sub_spot = bool(parent_area_name) and SCENIC_MEMBER_IS_CHILD.get(name, False)
key = builder.add_node(
"ScenicAttraction",
f"attraction:{slug(name)}",
name,
attraction_id=f"ATTR-{digest(name, length=8)}",
aliases=aliases,
city=city,
attraction_type=attraction_type,
attraction_level="scenic_spot" if is_sub_spot else ("representative_attraction" if parent_area_name else "standalone_attraction"),
parent_scenic_area_name=parent_area_name,
is_independent_destination=not is_sub_spot,
route_role=member_role or "独立线路目的地",
walk_intensity_hint="子景点步行强度需结合游客体力与景区交通确认" if is_sub_spot else "按产品行程安排确认",
recommendation_note=(
f"{name}是{parent_area_name}下的具体游览点,不作为独立主目的地;用于计算车程、游览顺序和客户偏好。"
if is_sub_spot else "可作为产品线路中的主目的地或独立景点资源。"
),
resource_spatial_class="location_resource",
selling_points=[point],
source_files=["种子景点+产品路线+图片资源库"],
)
builder.add_rel("LOCATED_IN", key, area_key)
if parent_area_name and parent_area_name in scenic_area_keys:
scenic_area_key = scenic_area_keys[parent_area_name]
builder.add_rel(
"SCENIC_AREA_HAS_ATTRACTION",
scenic_area_key,
key,
member_role=member_role,
independent_destination=not is_sub_spot,
relation_note=(
"景区片区包含具体子游览点,避免子景点在推荐中被误认为平级主目的地"
if is_sub_spot
else "景区片区锚点连接代表游览点,用于行政区资源池、路线覆盖和车程候选"
),
)
builder.add_rel(
"ATTRACTION_PART_OF_SCENIC_AREA",
key,
scenic_area_key,
member_role=member_role,
independent_destination=not is_sub_spot,
)
apply_media(builder, media_index, key, name, aliases, {"景点"})
for alias in aliases:
alias_to_key[alias] = key
return alias_to_key
def seed_vehicles_from_media(builder: KGBuilder, media_index: MediaIndex) -> tuple[dict[str, str], str]:
out: dict[str, str] = {}
group_key = builder.add_node(
"ResourceOptionGroup",
"option_group:vehicle:reliable_reference",
"可靠车辆资源组",
option_group_id="OG-VEHICLE-RELIABLE",
resource_type="vehicle",
city_or_area="贵州/全省",
grade_or_level="图片资源库高/中可靠车辆",
option_policy="车辆名称仅采用图片资源库中可靠车辆分组;实际派车按人数、行李、档期二次确认。",
default_option=False,
source_file=str(SOURCE_ROOT / "图片资源库_全品类别名索引.xlsx"),
)
for media_key, media in list(builder.nodes.items()):
if media.get("label") != "MediaResource" or media.get("resource_label") != "车辆":
continue
title = compact(media.get("title"))
if title not in media_index.accepted_vehicle_names:
continue
aliases = media.get("aliases") or []
seat_count = None
joined = f"{title} {' '.join(aliases)}"
m = re.search(r"(\d+)\s*座", joined)
if m:
seat_count = int(m.group(1))
elif "七座" in joined:
seat_count = 7
key = builder.add_node(
"VehicleResource",
f"vehicle:{slug(title)}",
title,
vehicle_id=f"VEH-{digest(title, length=8)}",
vehicle_type=title,
seat_count=seat_count,
seat_layout="参考图",
comfort_level="按实际派车确认",
capacity_min=None,
capacity_max=seat_count,
service_scope=["产品用车参考图", "接送/小团车型参考"],
aliases=aliases,
resource_service_class="service_resource",
service_region_mode="贵州全省调度/按产品确认",
suitable_group_size=f"建议不超过{seat_count}人" if seat_count else "按车型与行李二次确认",
dispatch_basis="按人数、行李、预算、车型等级和司机档期推荐;不按景点附近推荐。",
primary_image_url=media.get("primary_image_url"),
image_urls=media.get("image_urls"),
media_resource_ids=[media.get("media_id")],
media_reliability=media.get("reliability_level"),
media_usage_note=media.get("usage_note"),
)
builder.add_rel("RESOURCE_HAS_MEDIA", key, media_key, match_rule="车辆资源来自图片资源库资源分组")
builder.add_rel("GROUP_CONTAINS_VEHICLE", group_key, key)
out[title] = key
return out, group_key
def parse_resource_workbooks(builder: KGBuilder, media_index: MediaIndex) -> tuple[dict[str, str], dict[str, str]]:
hotel_groups: dict[str, str] = {}
restaurant_groups: dict[str, str] = {}
hotel_path = SOURCE_ROOT / "住宿资源库(四钻及以上).xlsx"
df = pd.read_excel(hotel_path, header=None)
region = ""
for _, row in df.iterrows():
values = [compact(x) for x in row.tolist()]
if values[0] and "区域" in values[0] and not values[1]:
region = clean_region_label(values[0])
group_key = builder.add_node(
"ResourceOptionGroup",
f"option_group:hotel:{slug(region)}",
f"{region}酒店资源组",
option_group_id=f"OG-HOTEL-{digest(region, length=8)}",
resource_type="hotel",
city_or_area=region,
grade_or_level="四钻及以上",
option_policy="可作为产品住宿槽位的同级/升级参考,具体房态和差价需二次确认。",
default_option=False,
source_file=str(hotel_path),
)
apply_media(builder, media_index, group_key, f"{region}参考酒店", [region, f"{region}4钻参考酒店"], {"酒店"})
hotel_groups[region] = group_key
continue
if values[0] in {"酒店名称", ""} or not values[0]:
continue
name = values[0]
key = builder.add_node(
"HotelResource",
f"hotel:{slug(name)}",
name,
hotel_id=f"HOTEL-{digest(name, length=8)}",
hotel_grade=values[1],
city_or_area=region,
address=values[2],
resource_spatial_class="location_resource",
contact_name=values[3],
features=split_items(values[4]),
listed_price_text=values[5],
off_season_price_text=values[6],
peak_season_price_text=values[7],
applicable_products=split_items(values[8]),
source_file=str(hotel_path),
)
apply_media(builder, media_index, key, name, [region, f"{region}4钻参考酒店"], {"酒店"})
if region:
area_key = builder.add_node("Area", f"area:{slug(region)}", region, area_id=f"AREA-{digest(region, length=8)}", area_type="酒店区域")
builder.add_rel("LOCATED_IN", key, area_key)
group_key = hotel_groups.get(region)
if group_key:
builder.add_rel("GROUP_CONTAINS_HOTEL", group_key, key)
rest_path = SOURCE_ROOT / "餐厅资源库.xlsx"
df = pd.read_excel(rest_path, header=None)
region = ""
for _, row in df.iterrows():
values = [compact(x) for x in row.tolist()]
section_region = RESTAURANT_SECTION_REGION_ALIASES.get(values[0]) if values[0] and not any(values[1:]) else ""
if values[0] and ("区域" in values[0] and not values[1] or section_region):
region = clean_region_label(section_region or values[0])
group_key = builder.add_node(
"ResourceOptionGroup",
f"option_group:restaurant:{slug(region)}",
f"{region}餐饮资源组",
option_group_id=f"OG-REST-{digest(region, length=8)}",
resource_type="restaurant",
city_or_area=region,
grade_or_level="按人均/特色菜选择",
option_policy="可作为产品餐饮槽位的团队餐或特色餐参考,需按人数、餐标、桌数确认。",
default_option=False,
source_file=str(rest_path),
)
restaurant_groups[region] = group_key
continue
if values[0] in {"餐厅名称", ""} or not values[0]:
continue
name = values[0]
key = builder.add_node(
"RestaurantResource",
f"restaurant:{slug(name)}",
name,
restaurant_id=f"REST-{digest(name, length=8)}",
city_or_area=region,
address=values[1] or values[5],
resource_spatial_class="location_resource",
per_capita_price_text=values[2],
signature_dishes=split_items(values[3]),
contact_name=values[4],
meal_scene=split_items(values[6]),
source_file=str(rest_path),
)
apply_media(builder, media_index, key, name, [region], {"餐饮"})
if region:
area_key = builder.add_node("Area", f"area:{slug(region)}", region, area_id=f"AREA-{digest(region, length=8)}", area_type="餐饮区域")
builder.add_rel("LOCATED_IN", key, area_key)
group_key = restaurant_groups.get(region)
if group_key:
builder.add_rel("GROUP_CONTAINS_RESTAURANT", group_key, key)
return hotel_groups, restaurant_groups
def group_member_keys(builder: KGBuilder, group_key: str, rel_type: str) -> list[str]:
return [rel["target"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["source"] == group_key]
def rel_targets(builder: KGBuilder, source_key: str, rel_type: str) -> list[str]:
return [rel["target"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["source"] == source_key]
def rel_sources(builder: KGBuilder, rel_type: str, target_key: str) -> list[str]:
return [rel["source"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["target"] == target_key]
def link_resource_groups_to_attractions(
builder: KGBuilder,
alias_to_key: dict[str, str],
hotel_groups: dict[str, str],
restaurant_groups: dict[str, str],
) -> None:
for region, group_key in hotel_groups.items():
attraction_names = RESOURCE_GROUP_ATTRACTION_MAP.get(clean_region_label(region), [])
for attraction_name in attraction_names:
attraction_key = find_attraction(attraction_name, alias_to_key)
if not attraction_key:
continue
builder.add_rel("GROUP_SERVES_ATTRACTION", group_key, attraction_key, basis="酒店区域与景区服务半径匹配")
for hotel_key in group_member_keys(builder, group_key, "GROUP_CONTAINS_HOTEL"):
builder.add_rel("HOTEL_SERVES_ATTRACTION", hotel_key, attraction_key, basis="继承酒店资源组服务景区")
for region, group_key in restaurant_groups.items():
attraction_names = RESOURCE_GROUP_ATTRACTION_MAP.get(clean_region_label(region), [])
for attraction_name in attraction_names:
attraction_key = find_attraction(attraction_name, alias_to_key)
if not attraction_key:
continue
builder.add_rel("GROUP_SERVES_ATTRACTION", group_key, attraction_key, basis="餐饮区域与景区服务半径匹配")
for restaurant_key in group_member_keys(builder, group_key, "GROUP_CONTAINS_RESTAURANT"):
builder.add_rel("RESTAURANT_SERVES_ATTRACTION", restaurant_key, attraction_key, basis="继承餐饮资源组服务景区")
def region_node_key(name: str, level: str, parent_key: str = "") -> str:
return f"admin_region:{level}:{slug(parent_key + ':' + name if parent_key else name)}"
def add_admin_region(
builder: KGBuilder,
name: str,
level: str,
province: str,
city: str = "",
county: str = "",
town: str = "",
parent_key: str = "",
adcode: str = "",
source_system: str = "",
scope_note: str = "",
) -> str:
key = region_node_key(name, level, parent_key)
builder.add_node(
"AdministrativeRegion",
key,
name,
region_id=f"REG-{digest(level, name, parent_key, length=10)}",
level=level,
province=province,
city=city,
county=county,
town=town,
parent_id=parent_key,
adcode=adcode,
source_system=source_system,
scope_note=scope_note or "仅为当前旅行社路线涉及资源建立行政区节点",
)
if parent_key:
builder.add_rel("REGION_PARENT_OF", parent_key, key)
return key
def add_region_hierarchy(
builder: KGBuilder,
province: str,
city: str = "",
county: str = "",
town: str = "",
adcode: str = "",
source_system: str = "",
) -> str:
province = province or "贵州省"
province_key = add_admin_region(builder, province, "province", province, source_system=source_system)
parent = province_key
most_specific = province_key
if city:
city_key = add_admin_region(
builder, city, "city", province, city=city, parent_key=parent,
source_system=source_system,
)
parent = city_key
most_specific = city_key
if county and county != city:
county_key = add_admin_region(
builder, county, "county", province, city=city, county=county, parent_key=parent,
adcode=adcode, source_system=source_system,
)
parent = county_key
most_specific = county_key
if town:
most_specific = add_admin_region(
builder, town, "town", province, city=city, county=county, town=town, parent_key=parent,
source_system=source_system,
)
return most_specific
def region_hint_for_text(text: str) -> tuple[str, str, str] | None:
n = compact(text)
for keyword, region in REGION_TEXT_HINTS.items():
if keyword in n:
return region
return None
def region_for_location_node(node: dict[str, Any]) -> tuple[str, str, str, str, str]:
province = compact(node.get("amap_pname")) or "贵州省"
city = compact(node.get("amap_cityname"))
county = compact(node.get("amap_adname"))
adcode = compact(node.get("amap_adcode"))
if city or county:
return province, city, county, adcode, "amap"
if node.get("label") == "ScenicArea":
area_spec = SCENIC_AREA_DEFINITIONS.get(compact(node.get("name"))) or {}
hint = area_spec.get("admin_hint")
if hint:
return hint[0], hint[1], hint[2], "", "scenic_area_admin_hint"
if node.get("label") == "ScenicAttraction":
hint = ATTRACTION_ADMIN_HINTS.get(compact(node.get("name")))
if hint:
return hint[0], hint[1], hint[2], "", "seed_admin_hint"
hint = region_hint_for_text(" ".join([
compact(node.get("name")),
compact(node.get("city_or_area")),
compact(node.get("address")),
compact(node.get("city")),
]))
if hint:
return hint[0], hint[1], hint[2], "", "business_region_hint"
return "贵州省", "", "", "", "province_fallback"
def add_geopoint_for_node(builder: KGBuilder, node_key: str, node: dict[str, Any]) -> None:
lng = number(node.get("amap_lng") or node.get("location_lng"))
lat = number(node.get("amap_lat") or node.get("location_lat"))
if lng is None or lat is None:
node["geo_match_status"] = node.get("geo_match_status") or "no_coordinate"
return
node["location_lng"] = lng
node["location_lat"] = lat
node["geo_match_status"] = node.get("amap_match_status") or "matched"
gp_key = f"geopoint:{node_key}"
builder.add_node(
"GeoPoint",
gp_key,
f"{node.get('name')}坐标",
geo_point_id=f"GEO-{digest(node_key, lng, lat, length=10)}",
lng=lng,
lat=lat,
address_text=node.get("amap_address") or node.get("address"),
provider="amap" if node.get("amap_poi_id") else "business_hint",
precision_level="poi" if node.get("amap_poi_id") else "region_hint",
source_system="amap" if node.get("amap_poi_id") else "business_source",
)
builder.add_rel("RESOURCE_HAS_GEOPOINT", node_key, gp_key)
def derive_scenic_area_coordinates(builder: KGBuilder) -> None:
"""Use the representative member POI as the scenic-area anchor coordinate.
Resource recommendation distances should start from scenic areas, not every
child spot. For 黄果树 this means the resource pool is anchored at 黄果树旅游景区,
while 天星桥/陡坡塘 remain child visit points.
"""
for area_key, area in list(builder.nodes.items()):
if area.get("label") != "ScenicArea":
continue
member_keys = rel_targets(builder, area_key, "SCENIC_AREA_HAS_ATTRACTION")
if not member_keys:
continue
preferred = ""
for member_key in member_keys:
member = builder.nodes.get(member_key, {})
if compact(member.get("name")) in {"黄果树", "黄果树瀑布"}:
preferred = member_key
break
preferred = preferred or member_keys[0]
member = builder.nodes.get(preferred, {})
lng = number(member.get("amap_lng") or member.get("location_lng"))
lat = number(member.get("amap_lat") or member.get("location_lat"))
if lng is None or lat is None:
continue
area["location_lng"] = lng
area["location_lat"] = lat
area["geo_match_status"] = "derived_from_representative_attraction"
area["data_completeness_note"] = f"景区片区坐标派生自代表游览点:{member.get('name')}"
def add_administrative_region_layer(builder: KGBuilder) -> None:
for node_key, node in list(builder.nodes.items()):
if node.get("label") not in LOCATION_RESOURCE_LABELS:
continue
province, city, county, adcode, source = region_for_location_node(node)
region_key = add_region_hierarchy(builder, province, city, county, adcode=adcode, source_system=source)
region_node = builder.nodes.get(region_key, {})
node["admin_region_name"] = region_node.get("name")
node["admin_region_level"] = region_node.get("level")
node["admin_region_source"] = source
node["resource_spatial_class"] = "location_resource"
builder.add_rel("LOCATED_IN_REGION", node_key, region_key, region_source=source)
add_geopoint_for_node(builder, node_key, node)
def same_region_enough(a: dict[str, Any], b: dict[str, Any]) -> bool:
ar = compact(a.get("admin_region_name"))
br = compact(b.get("admin_region_name"))
if ar and br and ar == br:
return True
ac = compact(a.get("amap_cityname") or a.get("city"))
bc = compact(b.get("amap_cityname") or b.get("city_or_area"))
return bool(ac and bc and ac in bc)
def add_nearby_location_relations(builder: KGBuilder) -> None:
sources = [
(k, n)
for k, n in builder.nodes.items()
if n.get("label") == "ScenicArea"
or (n.get("label") == "ScenicAttraction" and n.get("is_independent_destination") is not False)
]
hotels = [(k, n) for k, n in builder.nodes.items() if n.get("label") == "HotelResource"]
restaurants = [(k, n) for k, n in builder.nodes.items() if n.get("label") == "RestaurantResource"]
for source_key, source in sources:
a_lng = number(source.get("location_lng"))
a_lat = number(source.get("location_lat"))
if a_lng is None or a_lat is None:
continue
for target_key, target in hotels + restaurants:
t_lng = number(target.get("location_lng"))
t_lat = number(target.get("location_lat"))
if t_lng is None or t_lat is None:
continue
distance = haversine_km(a_lng, a_lat, t_lng, t_lat)
threshold = 35.0 if target.get("label") == "HotelResource" else 15.0
if distance <= threshold or (distance <= threshold * 1.6 and same_region_enough(source, target)):
builder.add_rel(
"NEARBY_LOCATION_RESOURCE",
source_key,
target_key,
resource_type="hotel" if target.get("label") == "HotelResource" else "restaurant",
distance_km=round(distance, 2),
source_scope="scenic_area_anchor" if source.get("label") == "ScenicArea" else "independent_attraction",
rule=f"以景区片区或独立景点为起点,直线距离不超过{threshold:g}km;子景点不直接连接酒店餐饮,实际推荐仍需结合车程、房态和餐标二次确认",
)
def route_stop_region_keys(builder: KGBuilder, stop_key: str) -> list[str]:
region_keys = rel_targets(builder, stop_key, "STOP_LOCATED_IN_REGION")
if region_keys:
return region_keys
attr_keys = rel_targets(builder, stop_key, "STOP_VISITS_ATTRACTION")
for attr_key in attr_keys:
region_keys.extend(rel_targets(builder, attr_key, "LOCATED_IN_REGION"))
if region_keys:
return list(dict.fromkeys(region_keys))
stop = builder.nodes.get(stop_key, {})
hint = region_hint_for_text(" ".join([compact(stop.get("name")), compact(stop.get("city_or_area"))]))
if not hint:
return []
return [add_region_hierarchy(builder, hint[0], hint[1], hint[2], source_system="route_stop_hint")]
def add_route_region_layer(builder: KGBuilder) -> None:
for stop_key, stop in list(builder.nodes.items()):
if stop.get("label") != "RouteStop":
continue
region_keys = route_stop_region_keys(builder, stop_key)
if not region_keys:
continue
first_region = builder.nodes.get(region_keys[0], {})
stop["admin_region_name"] = first_region.get("name")
stop["admin_region_level"] = first_region.get("level")
for region_key in region_keys:
builder.add_rel("STOP_LOCATED_IN_REGION", stop_key, region_key)
for day_key, day in list(builder.nodes.items()):
if day.get("label") != "ProductDay":
continue
stop_keys = rel_targets(builder, day_key, "DAY_HAS_STOP")
region_path: list[str] = []
for stop_key in stop_keys:
for region_key in rel_targets(builder, stop_key, "STOP_LOCATED_IN_REGION"):
region = builder.nodes.get(region_key, {})
region_name = compact(region.get("name"))
if region_name and region_name not in region_path:
region_path.append(region_name)
builder.add_rel("DAY_COVERS_REGION", day_key, region_key)
if region_path:
day["route_region_path"] = " -> ".join(region_path)
day["admin_region_name"] = region_path[-1]
for segment_key, segment in list(builder.nodes.items()):
if segment.get("label") != "RouteSegment":
continue
from_stops = rel_targets(builder, segment_key, "SEGMENT_FROM_STOP")
to_stops = rel_targets(builder, segment_key, "SEGMENT_TO_STOP")
from_regions = route_stop_region_keys(builder, from_stops[0]) if from_stops else []
to_regions = route_stop_region_keys(builder, to_stops[0]) if to_stops else []
if from_regions:
builder.add_rel("SEGMENT_FROM_REGION", segment_key, from_regions[0])
segment["origin_region_name"] = builder.nodes.get(from_regions[0], {}).get("name")
if to_regions:
builder.add_rel("SEGMENT_TO_REGION", segment_key, to_regions[0])
segment["destination_region_name"] = builder.nodes.get(to_regions[0], {}).get("name")
def add_route_scenic_area_layer(builder: KGBuilder) -> None:
for rel in list(builder.relations):
if rel["relation_type"] != "STOP_VISITS_ATTRACTION":
continue
stop_key = rel["source"]
attr_key = rel["target"]
area_keys = rel_targets(builder, attr_key, "ATTRACTION_PART_OF_SCENIC_AREA")
if not area_keys:
continue
for area_key in area_keys:
builder.add_rel(
"STOP_VISITS_SCENIC_AREA",
stop_key,
area_key,
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
relation_note="停靠点访问景区片区;具体游览点仍保留用于车程和游览强度判断",
)
for day_key in rel_sources(builder, "DAY_HAS_STOP", stop_key):
builder.add_rel(
"DAY_COVERS_SCENIC_AREA",
day_key,
area_key,
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
)
for product_key in rel_sources(builder, "HAS_DAY", day_key):
builder.add_rel(
"PRODUCT_COVERS_SCENIC_AREA",
product_key,
area_key,
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
)
def add_slot_candidate_layers(builder: KGBuilder) -> None:
hotel_keys = [key for key, node in builder.nodes.items() if node.get("label") == "HotelResource"]
restaurant_keys = [key for key, node in builder.nodes.items() if node.get("label") == "RestaurantResource"]
vehicle_keys = [key for key, node in builder.nodes.items() if node.get("label") == "VehicleResource"]
for slot_key, slot in list(builder.nodes.items()):
if slot.get("label") != "ResourceSlot":
continue
slot_type = slot.get("slot_type")
day_keys = rel_sources(builder, "DAY_HAS_SLOT", slot_key)
if slot_type in {"lodging", "meal"} and day_keys:
targets = hotel_keys if slot_type == "lodging" else restaurant_keys
candidate_keys: list[str] = []
for day_key in day_keys:
stop_keys = rel_targets(builder, day_key, "DAY_HAS_STOP")
anchor_keys: list[str] = []
for stop_key in stop_keys:
anchor_keys.extend(rel_targets(builder, stop_key, "STOP_VISITS_SCENIC_AREA"))
for attr_key in rel_targets(builder, stop_key, "STOP_VISITS_ATTRACTION"):
attr = builder.nodes.get(attr_key, {})
if attr.get("is_independent_destination") is not False:
anchor_keys.append(attr_key)
anchor_keys = list(dict.fromkeys(anchor_keys))
for anchor_key in anchor_keys:
anchor = builder.nodes.get(anchor_key, {})
for rel in builder.relations:
if rel["relation_type"] == "NEARBY_LOCATION_RESOURCE" and rel["source"] == anchor_key and rel["target"] in targets:
candidate_keys.append(rel["target"])
builder.add_rel(
"SLOT_CAN_USE_LOCATION_RESOURCE",
slot_key,
rel["target"],
candidate_basis="visited_scenic_area_nearby" if anchor.get("label") == "ScenicArea" else "visited_independent_attraction_nearby",
via_attraction=anchor.get("name"),
via_anchor_type=anchor.get("label"),
distance_km=(rel.get("properties") or {}).get("distance_km"),
drive_distance_km=(rel.get("properties") or {}).get("drive_distance_km"),
drive_duration_min=(rel.get("properties") or {}).get("drive_duration_min"),
region_match_level=(rel.get("properties") or {}).get("region_match_level"),
)
if candidate_keys:
continue
day_region_keys = rel_targets(builder, day_key, "DAY_COVERS_REGION")
for resource_key in targets:
if set(rel_targets(builder, resource_key, "LOCATED_IN_REGION")) & set(day_region_keys):
candidate_keys.append(resource_key)
builder.add_rel(
"SLOT_CAN_USE_LOCATION_RESOURCE",
slot_key,
resource_key,
candidate_basis="same_day_region",
)
if slot_type == "vehicle":
product_keys = rel_sources(builder, "PRODUCT_HAS_SLOT", slot_key)
for vehicle_key in vehicle_keys:
builder.add_rel(
"SLOT_CAN_USE_SERVICE_RESOURCE",
slot_key,
vehicle_key,
candidate_basis="vehicle_service_capacity",
)
for product_key in product_keys:
builder.add_rel(
"VEHICLE_SUITABLE_FOR_PRODUCT",
vehicle_key,
product_key,
candidate_basis="product_vehicle_slot",
)
def extract_frontmatter(markdown: str) -> tuple[dict[str, Any], str]:
if markdown.startswith("---"):
end = markdown.find("---", 3)
if end > 0:
block = markdown[3:end].strip()
try:
return json.loads(block), markdown[end + 3:]
except Exception:
return {}, markdown
return {}, markdown
def parse_fixed_route_rows(markdown: str) -> list[dict[str, str]]:
rows: list[dict[str, str]] = []
for line in markdown.splitlines():
line = line.strip()
if not line.startswith("| D") or "---" in line:
continue
cells = [c.strip().replace("\\|", "|") for c in line.strip("|").split("|")]
if len(cells) >= 4 and re.fullmatch(r"D\d+", cells[0]):
rows.append({"day": cells[0], "day_index": re.search(r"\d+", cells[0]).group(), "route": cells[1], "meals": cells[2], "accommodation": cells[3]})
return rows
def raw_text_block(markdown: str) -> str:
m = re.search(r"## 6\. 原文保留.*?```text\n(.*?)\n```", markdown, flags=re.S)
return clean(m.group(1)) if m else ""
def fallback_route_rows(markdown: str, fm: dict[str, Any]) -> list[dict[str, str]]:
"""Build conservative day rows when the source has readable prose but no D table."""
duration = int(fm.get("duration_days") or 0)
attractions = [compact(x) for x in (fm.get("core_attractions") or []) if compact(x)]
raw = raw_text_block(markdown)
if duration != 1 or not raw or not attractions:
return []
attraction = attractions[0]
start = "贵阳" if "贵阳" in raw or "延安西路" in raw or "旅游集散中心" in raw else "出发地"
end = "贵阳" if "返回贵阳" in raw or "贵阳统一散团" in raw else start
meal = ""
meal_match = re.search(r"用餐[::,,]?\s*([^。\n]{0,80})", raw)
if not meal_match:
meal_match = re.search(r"中餐[::,,]\s*([^。\n]{0,80})", raw)
if meal_match:
meal = "用餐:" + compact(meal_match.group(1))
return [{"day": "D1", "day_index": "1", "route": f"{start}→{attraction}→{end}", "meals": meal, "accommodation": "/"}]
def route_points(route: str) -> list[str]:
parts = re.split(r"→|->|>>|>|—|-|~|~", compact(route))
out: list[str] = []
for part in parts:
part = re.sub(r"^(D\d+|第[一二三四五六七八九十]+天)", "", part).strip()
if part and part not in out:
out.append(part)
return out
def find_attraction(name: str, alias_to_key: dict[str, str]) -> str:
n = norm(name)
for alias, key in alias_to_key.items():
a = norm(alias)
if a and (a in n or n in a):
return key
return ""
def ensure_series(builder: KGBuilder, name: str, family: str, attractions: list[str]) -> str:
series = family or "未分组"
if "游黔程" in name:
series = "游黔程"
elif "游黔途" in name or "1+1" in name:
series = "1+1游黔途"
elif "轻奢" in family or "头等舱" in name:
series = "轻奢纯玩"
elif "经典" in family:
series = "经典纯玩"
key = builder.add_node(
"ProductSeries",
f"series:{slug(series)}",
series,
series_id=f"SER-{digest(series, length=8)}",
series_type=family,
main_destinations=attractions,
notes="按产品名称/车型/酒店等级归并的已有路线系列",
)
return key
def product_key_for_name(name: str) -> str:
return f"product:{slug(name)}"
def find_product_key(builder: KGBuilder, product_name: str) -> str:
target = norm(product_name)
best = ""
best_score = 0
for key, node in builder.nodes.items():
if node.get("label") != "TourProduct":
continue
n = norm(node.get("name"))
if not n or not target:
continue
score = 0
if n == target:
score = 100
elif n in target or target in n:
score = min(len(n), len(target))
else:
common = len(set(re.findall(r"[\u4e00-\u9fff]+", n)) & set(re.findall(r"[\u4e00-\u9fff]+", target)))
score = common
if score > best_score:
best_score = score
best = key
return best if best_score >= 4 else ""
HOTEL_REGION_KEYWORDS = {
"贵阳区域": ["贵阳", "龙洞堡", "双龙", "甲秀楼", "青岩", "花溪", "黔灵山"],
"黄果树区域": ["安顺", "黄果树", "天星桥", "陡坡塘", "龙宫", "平坝"],
"西江千户苗寨区域": ["西江", "苗寨", "雷山", "千户苗寨"],
"镇远古镇区域": ["镇远", "镇远古镇", "镇远古城"],
"梵净山区域": ["梵净山", "江口", "铜仁", "中南门"],
"织金/荔波区域": ["织金", "织金洞", "荔波", "小七孔", "大七孔"],
"毕节区域": ["毕节", "百里杜鹃", "大方", "黔西"],
"开阳/猴耳天坑区域": ["开阳", "猴耳天坑", "南江大峡谷"],
"遵义区域": ["遵义", "乌江寨", "茅台", "遵义会议"],
}
RESTAURANT_REGION_KEYWORDS = HOTEL_REGION_KEYWORDS
RESOURCE_GROUP_ATTRACTION_MAP = {
"贵阳区域": ["青岩古镇", "甲秀楼", "黔灵山公园", "天河潭"],
"黄果树区域": ["黄果树", "天星桥", "陡坡塘瀑布", "龙宫", "平坝樱花"],
"西江千户苗寨区域": ["西江千户苗寨"],
"镇远古镇区域": ["镇远古城"],
"梵净山区域": ["梵净山", "中南门古城"],
"织金/荔波区域": ["荔波小七孔", "织金洞", "中国天眼"],
"毕节区域": ["百里杜鹃"],
"遵义区域": ["乌江寨", "茅台镇", "遵义会议会址"],
}
def option_groups_for_text(groups: dict[str, str], text: str, keyword_map: dict[str, list[str]]) -> list[str]:
n = norm(text)
out: list[str] = []
for region, key in groups.items():
region_name = clean_region_label(region).replace("区域", "")
canonical = clean_region_label(region)
candidates = [region_name, canonical, *split_items(region_name, limit=8), *keyword_map.get(canonical, [])]
if any(norm(part) and (norm(part) in n or n in norm(part)) for part in candidates):
out.append(key)
if out:
return list(dict.fromkeys(out))
for region, key in groups.items():
region_name = clean_region_label(region).replace("区域", "")
if region_name[:2] and region_name[:2] in text:
return [key]
return []
def hotel_groups_for_text(hotel_groups: dict[str, str], text: str) -> list[str]:
return option_groups_for_text(hotel_groups, text, HOTEL_REGION_KEYWORDS)
def hotel_group_for_text(hotel_groups: dict[str, str], text: str) -> str:
keys = hotel_groups_for_text(hotel_groups, text)
return keys[0] if keys else ""
def restaurant_group_for_text(restaurant_groups: dict[str, str], text: str) -> str:
keys = restaurant_groups_for_text(restaurant_groups, text)
if keys:
return keys[0]
return next(iter(restaurant_groups.values()), "")
def restaurant_groups_for_text(restaurant_groups: dict[str, str], text: str) -> list[str]:
keys = option_groups_for_text(restaurant_groups, text, RESTAURANT_REGION_KEYWORDS)
return keys or ([next(iter(restaurant_groups.values()))] if restaurant_groups else [])
def add_resource_slot(
builder: KGBuilder,
product_key: str,
day_key: str | None,
slot_type: str,
slot_name: str,
scope: str,
day_index: int | None,
default_text: str,
default_level: str,
change_mode: str,
pricing_mode: str,
source_file: str,
option_group_key: str | list[str] = "",
) -> str:
slot_key = builder.add_node(
"ResourceSlot",
f"slot:{product_key}:{day_index or 0}:{slot_type}:{slug(default_text or slot_name)}",
slot_name,
slot_id=f"SLOT-{digest(product_key, day_index, slot_type, default_text, length=10)}",
slot_name=slot_name,
slot_type=slot_type,
scope=scope,
day_index=day_index,
default_text=default_text,
default_level=default_level,
required=slot_type not in {"transfer", "gift_service"},
change_mode=change_mode,
changeable=change_mode != "fixed",
customer_visible=True,
pricing_mode=pricing_mode,
evidence_text=default_text,
source_file=source_file,
)
builder.add_rel("PRODUCT_HAS_SLOT", product_key, slot_key)
if day_key:
builder.add_rel("DAY_HAS_SLOT", day_key, slot_key)
option_group_keys = [option_group_key] if isinstance(option_group_key, str) else list(option_group_key or [])
option_group_keys = [key for key in dict.fromkeys(option_group_keys) if key]
if option_group_keys:
builder.add_rel("SLOT_DEFAULT_GROUP", slot_key, option_group_keys[0])
for key in option_group_keys:
builder.add_rel("SLOT_ALLOWED_GROUP", slot_key, key)
return slot_key
def fee_type_from_context(text: str) -> str:
if "保险" in text:
return "景区保险"
if any(x in text for x in ("观光车", "环保车", "电瓶车", "小交通", "景交")):
return "景区小交通"
if any(x in text for x in ("扶梯", "索道", "游船", "漂流")):
return "自愿项目"
if any(x in text for x in ("门票", "票")):
return "门票"
if any(x in text for x in ("餐标", "正餐", "人均")):
return "餐标"
if "单房差" in text:
return "单房差"
if "儿童" in text:
return "儿童价"
return "费用说明"
FEE_TARGET_TERMS = [
("黄果树旅游景区", ["黄果树景区", "黄果树大瀑布风景名胜区", "黄果树大瀑布", "黄果树"]),
("荔波小七孔", ["小七孔", "荔波小七孔", "鸳鸯湖"]),
("西江千户苗寨", ["西江千户", "西江苗寨", "西江"]),
("青岩古镇", ["青岩古镇", "青岩"]),
("梵净山", ["梵净山"]),
("镇远古城", ["镇远古城", "镇远古镇", "镇远"]),
("百里杜鹃", ["百里杜鹃"]),
("织金洞", ["织金洞"]),
("中国天眼", ["中国天眼", "天眼"]),
("天星桥", ["天星桥"]),
("陡坡塘瀑布", ["陡坡塘"]),
("龙宫", ["龙宫"]),
("西江千户苗寨", ["苗寨"]),
]
def find_named_node(builder: KGBuilder, label: str, name: str) -> str:
target = norm(name)
for key, node in builder.nodes.items():
if node.get("label") == label and norm(node.get("name")) == target:
return key
return ""
def find_named_or_alias_node(builder: KGBuilder, label: str, names: list[str]) -> str:
normalized = [norm(name) for name in names if norm(name)]
for key, node in builder.nodes.items():
if node.get("label") != label:
continue
node_name = norm(node.get("name"))
aliases = [norm(alias) for alias in (node.get("aliases") or []) if norm(alias)]
if any(term and (term == node_name or term in aliases or node_name in term or term in node_name) for term in normalized):
return key
return ""
def fee_target_from_text(builder: KGBuilder, text: str, fallback_key: str = "") -> tuple[str, str, str]:
for canonical, aliases in FEE_TARGET_TERMS:
if any(alias and alias in text for alias in aliases):
area_key = find_named_or_alias_node(builder, "ScenicArea", [canonical, *aliases])
if area_key:
return area_key, "ScenicArea", builder.nodes[area_key]["name"]
attr_key = find_named_or_alias_node(builder, "ScenicAttraction", [canonical, *aliases])
if attr_key:
return attr_key, "ScenicAttraction", builder.nodes[attr_key]["name"]
if fallback_key and fallback_key in builder.nodes:
node = builder.nodes[fallback_key]
return fallback_key, node.get("label", ""), node.get("name", "")
return "", "", ""
def service_from_fee_part(part: str, previous_service: str = "", joined_with_previous_without_amount: bool = False) -> tuple[str, str]:
text = compact(part)
if "保险" in text and previous_service.endswith("+保险"):
return "景区小交通+保险", previous_service
vehicle_word = next((word for word in ["环保车", "观光车", "电瓶车", "摆渡车", "景交", "小交通"] if word in text), "")
if "保险" in text and vehicle_word:
service = "景区小交通+保险"
name = "景区交通+保险" if vehicle_word in {"景交", "小交通"} else f"{vehicle_word}+保险"
return service, name
if "保险" in text and joined_with_previous_without_amount and previous_service in {"环保车", "观光车", "电瓶车", "摆渡车", "景区交通"}:
return "景区小交通+保险", f"{previous_service}+保险"
if "保险" in text:
return "景区保险", "保险"
if vehicle_word:
service = "景区小交通"
return service, "景区交通" if vehicle_word in {"景交", "小交通"} else vehicle_word
if "扶梯" in text or (("单程" in text or "双程" in text or "往返" in text) and previous_service.startswith("扶梯")):
direction = "双程" if any(x in text for x in ["双程", "往返"]) else ("单程" if "单程" in text else "")
return "自愿项目", f"扶梯{direction}".strip()
if "索道" in text:
direction = "往返" if "往返" in text else ("单程" if "单程" in text else "")
return "自愿项目", f"索道{direction}".strip()
if any(word in text for word in ["游船", "划船", "船票"]):
return "自愿项目", "游船"
if "演出" in text or "红飘带" in text:
return "自愿项目", "演出"
if "门票" in text or "大门票" in text or "首道" in text:
return "门票", "门票"
if "餐标" in text or "正餐" in text:
return "餐标", "餐标"
if previous_service:
return fee_type_from_context(previous_service), previous_service
return fee_type_from_context(text), ""
def inclusion_from_fee_line(line: str, part: str, fee_type: str) -> tuple[str, str, bool]:
text = f"{line} {part}"
if any(x in text for x in ["免费", "赠送"]):
return "免费/赠送", "included_free", False
if any(x in text for x in ["自愿", "自理自愿", "不必须", "可选", "酌情自愿"]) or fee_type == "自愿项目":
return "不含/自愿自理", "optional_self_pay", True
if any(x in text for x in ["必须", "需自费", "需自理", "客人需自理", "自费", "不含", "另付", "必消", "必销"]):
return "不含/自理", "mandatory_self_pay", False
if "含" in text and "不含" not in text:
return "已含", "included", False
return "需核实", "policy_dependent", False
def clean_fee_prefix(part: str, amount_start: int) -> str:
prefix = part[:amount_start]
prefix = re.sub(r"^[\s::,,、++()()【】《》\\-—]*", "", prefix)
prefix = re.sub(r".*(费用不含|必须自理费用|自愿自费项目|自愿消费项目|不含自愿消费|不含|含|自理|自费|需自费|:|:)", "", prefix)
prefix = re.sub(r"(客人|游客|费用|共计|景区内|换乘|乘坐|进入|需要|需|可)?$", "", prefix).strip()
return prefix[-18:]
def is_total_amount_match(part: str, amount_start: int) -> bool:
prefix = part[:amount_start]
return bool(re.search(r"(共计|合计|总计|小计|共|共自理|请自理|客人请自理|客人共自理)\s*[::]?\s*$", prefix))
def add_structured_fee_node(
builder: KGBuilder,
owner_key: str,
product_name: str,
source_file: str,
line: str,
target_key: str,
target_label: str,
target_name: str,
service_name: str,
fee_type: str,
amount: float,
unit: str,
inclusion_status: str,
mandatory_level: str,
is_optional: bool,
consumer_group: str,
extraction_rule: str,
) -> str:
owner_label = builder.nodes.get(owner_key, {}).get("label")
fee_rel_type = "PRICE_PACKAGE_HAS_FEE" if owner_label == "ProductPricePackage" else "SLOT_HAS_FEE"
item_name = " ".join(x for x in [target_name, service_name or fee_type] if x).strip() or service_name or fee_type
identity = f"{product_name}|{target_key}|{item_name}|{amount}|{consumer_group}|{line[:80]}"
label = "TicketFee" if fee_type in {"景区保险", "景区小交通", "景区小交通+保险", "自愿项目", "门票"} else "FeeItem"
common = dict(
amount_text=f"{amount:g}元/{unit or '人'}",
amount_value=amount,
price_text=f"{amount:g}元/{unit or '人'}",
adult_price=amount if consumer_group != "child" else None,
child_price=amount if consumer_group == "child" else None,
currency="CNY",
unit=unit or "人",
inclusion_status=inclusion_status,
mandatory_level=mandatory_level,
applies_to=product_name,
source_product_name=product_name,
scenic_target_name=target_name,
scenic_target_type=target_label,
consumer_group=consumer_group,
is_free=False,
is_optional=is_optional,
price_status="source_explicit",
child_price_status="同成人价/原文未单列儿童价" if consumer_group == "all" else ("儿童价原文明确" if consumer_group == "child" else ""),
rule_text=line,
evidence_excerpt=line,
source_file=source_file,
data_quality_status="structured_from_explicit_price",
extraction_rule=extraction_rule,
)
if label == "TicketFee":
fee_key = builder.add_node(
"TicketFee",
f"ticket_fee:{digest(source_file, owner_key, identity)}",
item_name,
ticket_fee_id=f"TF-{digest(source_file, owner_key, identity, length=10)}",
fee_name=item_name,
fee_type=fee_type,
**common,
)
else:
fee_key = builder.add_node(
"FeeItem",
f"fee:{digest(source_file, owner_key, identity)}",
item_name,
fee_item_id=f"FEE-{digest(source_file, owner_key, identity, length=10)}",
fee_type=fee_type,
item_name=item_name,
**common,
)
builder.add_rel(fee_rel_type, owner_key, fee_key)
if target_key:
if target_label == "ScenicArea":
builder.add_rel("SCENIC_AREA_HAS_FEE", target_key, fee_key, source_product_name=product_name)
elif target_label == "ScenicAttraction":
builder.add_rel("ATTRACTION_HAS_FEE", target_key, fee_key, source_product_name=product_name)
product_key = find_product_key(builder, product_name)
if product_key:
builder.add_rel("PRODUCT_HAS_FEE", product_key, fee_key)
return fee_key
def add_free_project_fee(
builder: KGBuilder,
owner_key: str,
product_name: str,
source_file: str,
line: str,
target_key: str,
target_label: str,
target_name: str,
project_name: str,
) -> None:
if not project_name:
return
item_name = " ".join(x for x in [target_name, project_name] if x).strip()
identity = f"free|{product_name}|{target_key}|{item_name}|{line[:80]}"
fee_key = builder.add_node(
"TicketFee",
f"ticket_fee:{digest(source_file, owner_key, identity)}",
item_name,
ticket_fee_id=f"TF-{digest(source_file, owner_key, identity, length=10)}",
fee_name=item_name,
fee_type="免费/赠送项目",
amount_text="0元",
amount_value=0,
price_text="免费/赠送",
adult_price=0,
child_price=0,
currency="CNY",
unit="人",
inclusion_status="免费/赠送",
mandatory_level="included_free",
applies_to=product_name,
source_product_name=product_name,
scenic_target_name=target_name,
scenic_target_type=target_label,
consumer_group="all",
is_free=True,
is_optional=False,
price_status="source_free",
child_price_status="同成人免费",
rule_text=line,
evidence_excerpt=line,
source_file=source_file,
data_quality_status="structured_from_free_text",
extraction_rule="free_or_gift_project_line",
)
owner_label = builder.nodes.get(owner_key, {}).get("label")
builder.add_rel("PRICE_PACKAGE_HAS_FEE" if owner_label == "ProductPricePackage" else "SLOT_HAS_FEE", owner_key, fee_key)
if target_key:
builder.add_rel("SCENIC_AREA_HAS_FEE" if target_label == "ScenicArea" else "ATTRACTION_HAS_FEE", target_key, fee_key, source_product_name=product_name)
product_key = find_product_key(builder, product_name)
if product_key:
builder.add_rel("PRODUCT_HAS_FEE", product_key, fee_key)
def extract_fee_nodes(builder: KGBuilder, owner_key: str, product_name: str, text: str, source_file: str) -> None:
seen: set[str] = set()
for line in re.split(r"[。;;\n]+", text):
line = compact(line)
if len(line) < 4:
continue
if ("免费" in line or "赠送" in line) and not any(x in line for x in ["并非", "不是免费", "不免费", "非免费"]):
target_key, target_label, target_name = fee_target_from_text(builder, line)
free_names = []
for m in re.finditer(r"(?:赠送|免费)([^,,。;;]{2,24})(?:体验|券|项目|服务)?", line):
project = compact(m.group(1))
if project and not any(bad in project for bad in ["门票退", "无退费", "早餐"]):
free_names.append(project if project.endswith(("体验", "券", "服务")) else f"{project}体验")
for project in free_names[:3]:
identity = f"free|{target_key}|{project}|{line[:80]}"
if identity in seen:
continue
seen.add(identity)
add_free_project_fee(builder, owner_key, product_name, source_file, line, target_key, target_label, target_name, project)
if not re.search(r"\d+(?:\.\d+)?\s*元", line):
continue
if not any(x in line for x in ("元", "票", "保险", "车", "扶梯", "索道", "餐", "儿童", "游船", "演出")):
continue
current_target_key, current_target_label, current_target_name = fee_target_from_text(builder, line)
current_service = ""
split_tokens = re.split(r"([,,、++])", line)
part_rows: list[tuple[str, str]] = []
pending_sep = ""
for token in split_tokens:
token = compact(token)
if not token:
continue
if token in {",", ",", "、", "+", "+"}:
pending_sep = token
continue
part_rows.append((pending_sep, token))
pending_sep = ""
previous_part_had_amount = False
for previous_sep, part in part_rows:
part = compact(part)
detected_key, detected_label, detected_name = fee_target_from_text(builder, part, current_target_key)
if detected_key:
current_target_key, current_target_label, current_target_name = detected_key, detected_label, detected_name
fee_type, service_name = service_from_fee_part(
part,
current_service,
joined_with_previous_without_amount=previous_sep in {"+", "+"} and not previous_part_had_amount,
)
if service_name:
current_service = service_name
if fee_type == "费用说明" and not current_service:
previous_part_had_amount = bool(re.search(r"\d+(?:\.\d+)?\s*元", part))
continue
for match in re.finditer(r"(\d+(?:\.\d+)?)\s*元\s*(?:/|/)?\s*(人|趟|间|份|餐)?", part):
if is_total_amount_match(part, match.start()):
continue
prefix = clean_fee_prefix(part, match.start())
amount = float(match.group(1))
unit = match.group(2) or ("人" if "/人" in part or "每人" in part else "")
local_fee_type, local_service = service_from_fee_part(
prefix + part[match.start():],
current_service,
joined_with_previous_without_amount=previous_sep in {"+", "+"} and not previous_part_had_amount,
)
if local_service:
fee_type, service_name = local_fee_type, local_service
current_service = service_name
if not service_name and prefix:
service_name = prefix
if not service_name or service_name in {"费用说明"}:
continue
consumer_group = "child" if "儿童" in part or "儿童" in line[: max(0, line.find(part)) + len(part)] and "成人" not in part else "all"
inclusion_status, mandatory_level, is_optional = inclusion_from_fee_line(line, part, fee_type)
identity = f"{current_target_key}|{service_name}|{amount}|{consumer_group}|{line[:80]}"
if identity in seen:
continue
seen.add(identity)
add_structured_fee_node(
builder, owner_key, product_name, source_file, line,
current_target_key, current_target_label, current_target_name,
service_name, fee_type, amount, unit, inclusion_status, mandatory_level,
is_optional, consumer_group, "explicit_project_price_pattern",
)
previous_part_had_amount = bool(re.search(r"\d+(?:\.\d+)?\s*元", part))
if "半票" in line or "免票" in line:
rule_key = builder.add_node(
"BusinessRule",
f"rule:ticket_discount:{digest(source_file, owner_key, line)}",
f"{product_name}门票优惠/退费规则",
rule_id=f"RULE-{digest(source_file, owner_key, line, length=10)}",
rule_type="ticket_discount",
applies_to_type="TicketFee",
applies_to_text=product_name,
rule_text=line,
severity="提示",
source_file=source_file,
)
product_key = find_product_key(builder, product_name)
if product_key:
builder.add_rel("PRODUCT_HAS_RULE", product_key, rule_key)
def parse_existing_route_markdown(
builder: KGBuilder,
alias_to_key: dict[str, str],
hotel_groups: dict[str, str],
restaurant_groups: dict[str, str],
vehicle_group_key: str,
) -> list[str]:
index_path = ROUTE_MD_DIR / "产品索引.json"
items = json.loads(index_path.read_text(encoding="utf-8"))
product_keys: list[str] = []
for item in items:
md_path = ROUTE_MD_DIR / item["markdown_filename"]
markdown = md_path.read_text(encoding="utf-8")
fm, _ = extract_frontmatter(markdown)
name = fm.get("product_name") or item.get("product_name")
if not name:
continue
source_file = fm.get("source_file") or item.get("source_file") or str(md_path)
attractions = fm.get("core_attractions") or []
series_key = ensure_series(builder, name, fm.get("product_family") or "", attractions)
product_key = builder.add_node(
"TourProduct",
product_key_for_name(name + source_file),
name,
product_id=f"TP-{digest(source_file, name, length=10)}",
product_series=builder.nodes[series_key]["name"],
product_family=fm.get("product_family"),
product_type="已有路线产品",
duration_days=fm.get("duration_days"),
duration_nights=max(int(fm.get("duration_days") or 0) - 1, 0) if fm.get("duration_days") else None,
route_immutable=True,
default_group_mode="固定产品/按产品说明",
default_vehicle_type=fm.get("default_vehicle_type"),
default_hotel_grade=fm.get("default_hotel_grade"),
default_meal_standard="按产品每日用餐/接待标准",
service_promise="按产品原文承诺",
selling_points=attractions,
included_summary="见产品Markdown费用与接待标准",
excluded_summary="见产品Markdown费用与规则候选",
booking_notes="路线骨架固定;仅允许在资源槽位范围内替换/升级/二次核价。",
source_files=[source_file, str(md_path)],
evidence_excerpt=markdown[:1200],
)
builder.add_rel("BELONGS_TO_SERIES", product_key, series_key)
product_keys.append(product_key)
rows = parse_fixed_route_rows(markdown)
if not rows:
rows = fallback_route_rows(markdown, fm)
product_stop_keys: list[str] = []
product_stop_names: list[str] = []
product_day_keys: list[str] = []
product_day_summaries: list[str] = []
for row in rows:
day_index = int(row["day_index"])
day_key = builder.add_node(
"ProductDay",
f"product_day:{product_key}:{day_index}",
f"{name} {row['day']}",
day_id=f"DAY-{digest(product_key, day_index, length=10)}",
day_index=day_index,
title=row["route"],
route_path=row["route"],
route_line_name=name,
route_line_id=f"TP-{digest(source_file, name, length=10)}",
route_sequence_label=row["day"],
route_display_name=f"{name} {row['day']}",
meal_text=row["meals"],
accommodation_text=row["accommodation"],
transport_summary="按固定路线骨架执行",
source_file=source_file,
)
builder.add_rel("HAS_DAY", product_key, day_key, day_index=day_index)
product_day_keys.append(day_key)
product_day_summaries.append(f"{row['day']} {row['route']}")
points = route_points(row["route"])
stop_keys: list[str] = []
for order, point in enumerate(points, start=1):
attr_key = find_attraction(point, alias_to_key)
stop_type = "scenic" if attr_key else ("city_or_area" if point else "unknown")
global_order = len(product_stop_keys) + 1
route_sequence_label = f"{row['day']}-{order:02d}"
station_like_name = f"{route_sequence_label} {point}"
stop_key = builder.add_node(
"RouteStop",
f"route_stop:{product_key}:{day_index}:{order}:{slug(point)}",
station_like_name,
stop_id=f"STOP-{digest(product_key, day_index, order, point, length=10)}",
stop_order=order,
day_stop_order=order,
global_stop_order=global_order,
route_sequence_label=route_sequence_label,
station_like_name=station_like_name,
route_line_name=name,
route_line_id=f"TP-{digest(source_file, name, length=10)}",
route_display_name=f"{name} {route_sequence_label} {point}",
stop_type=stop_type,
city_or_area=point,
is_core_visit=bool(attr_key),
evidence_text=row["route"],
meal_text=row["meals"],
accommodation_text=row["accommodation"],
source_file=source_file,
)
builder.add_rel("DAY_HAS_STOP", day_key, stop_key, stop_order=order)
builder.add_rel(
"PRODUCT_HAS_ORDERED_STOP",
product_key,
stop_key,
global_stop_order=global_order,
day_index=day_index,
day_stop_order=order,
route_sequence_label=route_sequence_label,
)
if attr_key:
builder.add_rel("STOP_VISITS_ATTRACTION", stop_key, attr_key)
stop_keys.append(stop_key)
product_stop_keys.append(stop_key)
product_stop_names.append(point)
if stop_keys:
builder.nodes[day_key].update({
"route_stop_sequence": " -> ".join(points),
"route_display_text": f"{name} {row['day']}:{' -> '.join(points)}",
"total_stop_count": len(stop_keys),
})
for idx in range(len(stop_keys) - 1):
origin = builder.nodes[stop_keys[idx]]["name"]
dest = builder.nodes[stop_keys[idx + 1]]["name"]
seg_key = builder.add_node(
"RouteSegment",
f"route_segment:{product_key}:{day_index}:{idx+1}",
f"{row['day']} {origin}->{dest}",
segment_id=f"SEG-{digest(product_key, day_index, idx, origin, dest, length=10)}",
day_index=day_index,
origin_text=origin,
destination_text=dest,
transport_mode=fm.get("default_vehicle_type") or "按产品安排",
source_file=source_file,
)
builder.add_rel("DAY_HAS_SEGMENT", day_key, seg_key)
builder.add_rel("SEGMENT_FROM_STOP", seg_key, stop_keys[idx])
builder.add_rel("SEGMENT_TO_STOP", seg_key, stop_keys[idx + 1])
if row["accommodation"] and row["accommodation"] != "/":
group_key = hotel_groups_for_text(hotel_groups, row["accommodation"])
add_resource_slot(
builder, product_key, day_key, "lodging", f"{name} {row['day']}住宿槽位", "day",
day_index, row["accommodation"], fm.get("default_hotel_grade") or "", "same_level_replace",
"same_level_or_second_quote", source_file, group_key,
)
if row["meals"]:
group_key = restaurant_groups_for_text(restaurant_groups, row["route"] + row["accommodation"])
add_resource_slot(
builder, product_key, day_key, "meal", f"{name} {row['day']}餐饮槽位", "day",
day_index, row["meals"], "按产品餐标", "optional_addon", "included_or_second_quote",
source_file, group_key,
)
for idx in range(len(product_day_keys) - 1):
builder.add_rel(
"DAY_NEXT_DAY",
product_day_keys[idx],
product_day_keys[idx + 1],
sequence_order=idx + 1,
)
for idx in range(len(product_stop_keys) - 1):
current_key = product_stop_keys[idx]
next_key = product_stop_keys[idx + 1]
current = builder.nodes[current_key]
next_stop = builder.nodes[next_key]
current["next_stop_name"] = next_stop.get("city_or_area") or next_stop.get("name")
next_stop["previous_stop_name"] = current.get("city_or_area") or current.get("name")
builder.add_rel(
"ROUTE_STOP_NEXT",
current_key,
next_key,
route_line_name=name,
sequence_order=idx + 1,
from_stop_order=current.get("global_stop_order"),
to_stop_order=next_stop.get("global_stop_order"),
)
if product_stop_keys:
stop_sequence = " -> ".join([compact(x) for x in product_stop_names if compact(x)])
for stop_key in product_stop_keys:
builder.nodes[stop_key]["total_stop_count"] = len(product_stop_keys)
builder.nodes[stop_key]["route_stop_sequence"] = stop_sequence
builder.nodes[product_key].update({
"route_line_name": name,
"route_line_id": f"TP-{digest(source_file, name, length=10)}",
"route_stop_sequence": stop_sequence,
"route_day_summary": ";".join(product_day_summaries),
"route_display_text": f"{name}:{stop_sequence}",
"total_stop_count": len(product_stop_keys),
})
if fm.get("default_vehicle_type"):
add_resource_slot(
builder, product_key, None, "vehicle", f"{name}默认交通槽位", "product",
None, fm.get("default_vehicle_type"), fm.get("default_vehicle_type"), "upgradeable",
"second_quote", source_file, vehicle_group_key,
)
fee_slot = add_resource_slot(
builder, product_key, None, "ticket_transport_fee", f"{name}门票小交通费用槽位", "product",
None, "按产品费用说明", "", "mandatory_self_pay", "policy_dependent", source_file,
)
fee_section = markdown[markdown.find("## 5. 费用与规则候选"):] if "## 5. 费用与规则候选" in markdown else markdown
extract_fee_nodes(builder, fee_slot, name, fee_section, source_file)
for line in re.findall(r"^- (.+)$", fee_section, flags=re.M)[:80]:
if any(k in line for k in ["老人", "儿童", "学生", "退", "预约", "不接待", "投诉", "满房", "同级", "孕妇", "不可抗力", "路滑", "少走路"]):
rule_type = "discount" if any(k in line for k in ["老人", "儿童", "学生", "军人", "免票", "半票"]) else ("refund" if "退" in line else ("risk" if any(k in line for k in ["预约", "路滑", "孕妇", "不接待", "不可抗力"]) else "booking"))
rule_key = builder.add_node(
"BusinessRule",
f"rule:{digest(source_file, name, line)}",
f"{name} {rule_type}规则",
rule_id=f"RULE-{digest(source_file, name, line, length=10)}",
rule_type=rule_type,
applies_to_type="TourProduct",
applies_to_text=name,
rule_text=line,
severity="关键限制" if rule_type in {"refund", "risk"} else "提示",
source_file=source_file,
)
builder.add_rel("PRODUCT_HAS_RULE", product_key, rule_key)
return product_keys
def ensure_pricing_subject(
builder: KGBuilder,
product_name: str,
source_file: str,
catalog_type: str,
product_direction: str = "",
route_preview: str = "",
) -> tuple[str, str, str]:
existing = find_product_key(builder, product_name)
if existing:
return existing, "PRODUCT_HAS_PRICE_PACKAGE", "PRODUCT_HAS_RULE"
catalog_key = builder.add_node(
"PriceCatalogItem",
f"price_catalog:{slug(product_name + source_file)}",
product_name,
catalog_item_id=f"PCI-{digest(source_file, product_name, catalog_type, length=10)}",
catalog_type=catalog_type,
duration_days=duration_from_text(product_name),
product_direction=product_direction,
route_preview=route_preview,
price_validity="见价格表",
source_file=source_file,
booking_notes="来自价格表;未在路线文档中形成完整每日行程骨架时,不作为完整路线推荐。",
)
return catalog_key, "CATALOG_ITEM_HAS_PRICE_PACKAGE", "CATALOG_ITEM_HAS_RULE"
def parse_small_group_prices(builder: KGBuilder) -> None:
path = SOURCE_ROOT / "滨海国旅2-8人拼小团计划 ( 26年4月1号-4月28号。。26年5月4号--6月30号 ~)(五一节除外).xlsx"
df = pd.read_excel(path, header=None)
notes = compact(df.iloc[1, 0]) if len(df) > 1 else ""
current_vehicle = ""
current_product = ""
current_collection = ""
current_schedule = ""
current_inner_fee = ""
current_refund = ""
for idx in range(3, len(df)):
row = [compact(x) for x in df.iloc[idx].tolist()[:10]]
if row[0]:
current_vehicle = row[0]
if row[1]:
current_product = re.split(r"\n|(镇远|注:", row[1])[0].strip()
if row[2]:
current_collection = row[2]
if row[3]:
current_schedule = row[3]
if row[8]:
current_inner_fee = row[8]
if row[9]:
current_refund = row[9]
if not current_product or not row[4] or money(row[5]) is None:
continue
subject_key, price_rel_type, rule_rel_type = ensure_pricing_subject(
builder, current_product, str(path), "1-8人拼小团报价目录"
)
package_name = f"{current_product} {row[4]}"
price_key = builder.add_node(
"ProductPricePackage",
f"price:small:{idx}:{slug(package_name)}",
package_name,
price_package_id=f"PRICE-SG-{idx:03d}",
package_name=package_name,
season="2026年4月/5-6月平季(五一除外)",
date_range="2026-04-01~2026-04-28;2026-05-04~2026-06-30",
group_size_band=current_vehicle or "1-8人拼小团",
room_type=row[4],
hotel_grade=row[4],
vehicle_type="按人数派5/7/9座车",
adult_price=money(row[5]),
child_price=money(row[6]),
single_room_supplement=money(row[7]),
inner_transport_fee_text=current_inner_fee,
refund_policy_text=current_refund,
source_file=str(path),
collection_method=current_collection,
schedule_rule=current_schedule,
notes=notes,
)
builder.add_rel(price_rel_type, subject_key, price_key)
if current_inner_fee:
extract_fee_nodes(builder, price_key, current_product, current_inner_fee, str(path))
if current_refund:
rule_key = builder.add_node(
"BusinessRule",
f"rule:small_refund:{slug(current_product + current_refund)}",
f"{current_product}证件退费规则",
rule_id=f"RULE-{digest(current_product, current_refund, length=10)}",
rule_type="refund",
applies_to_type="ProductPricePackage",
applies_to_text=current_product,
rule_text=current_refund,
severity="报价必看",
source_file=str(path),
)
builder.add_rel(rule_rel_type, subject_key, rule_key)
if notes:
for i, line in enumerate(re.split(r"\s*(?=\d+:|关于)", notes)[:20], start=1):
line = compact(line)
if len(line) < 8:
continue
if any(k in line for k in ["用车", "行李", "酒店", "用餐", "不接待", "老人", "孕妇", "儿童"]):
rule_key = builder.add_node(
"BusinessRule",
f"rule:small_note:{i}",
f"拼小团注意事项{i}",
rule_id=f"RULE-SMALL-{i:02d}",
rule_type="eligibility" if any(k in line for k in ["不接待", "老人", "孕妇", "儿童"]) else "booking",
applies_to_type="TourProduct",
applies_to_text="1-8人拼小团",
rule_text=line,
severity="关键限制" if "不接待" in line else "提示",
source_file=str(path),
)
for key, node in list(builder.nodes.items()):
if node.get("label") == "TourProduct" and "拼小团" in compact(node.get("default_group_mode") or node.get("product_type")):
builder.add_rel("PRODUCT_HAS_RULE", key, rule_key)
if node.get("label") == "PriceCatalogItem" and "拼小团" in compact(node.get("catalog_type")):
builder.add_rel("CATALOG_ITEM_HAS_RULE", key, rule_key)
def parse_independent_prices(builder: KGBuilder) -> None:
path = SOURCE_ROOT / "20-25人独立成团.xlsx"
xl = pd.ExcelFile(path)
for sheet in xl.sheet_names:
df = pd.read_excel(path, sheet_name=sheet, header=None)
current_product = ""
current_direction = ""
for idx in range(4, len(df)):
row = [compact(x) for x in df.iloc[idx].tolist()[:10]]
if not any(row):
continue
if row[0] and row[0] not in {"产品方向", "报名建议"}:
current_direction = row[0]
if row[1] and row[1] != "参考酒店":
current_product = row[1]
if not current_product:
continue
hotel_text = row[5] or row[3] or row[4]
price_cols = [(6, "20人"), (7, "25人")]
for col, group_size in price_cols:
if col >= len(row) or money(row[col]) is None:
continue
subject_key, price_rel_type, _rule_rel_type = ensure_pricing_subject(
builder, current_product, str(path), "20-25人独立成团报价目录", current_direction, row[2]
)
package_name = f"{current_product} {sheet} {group_size} {hotel_text or ''}"
price_key = builder.add_node(
"ProductPricePackage",
f"price:ind:{sheet}:{idx}:{group_size}:{slug(package_name)}",
package_name,
price_package_id=f"PRICE-IND-{digest(sheet, idx, group_size, length=10)}",
package_name=package_name,
season=sheet,
group_size_band=group_size,
room_type=hotel_text,
hotel_grade=hotel_text,
vehicle_type="32-38座2+1座大巴",
adult_price=money(row[col]),
single_room_supplement=money(row[8]),
inner_transport_fee_text=row[3] if "门票" in row[3] or "观光车" in row[3] else "",
source_file=str(path),
product_direction=current_direction,
route_preview=row[2],
meal_standard=row[4] if "餐" in row[4] else row[5],
)
builder.add_rel(price_rel_type, subject_key, price_key)
if idx == 1 or "报名建议" in row[0]:
continue
note = compact(df.iloc[1, 0]) if len(df) > 1 else ""
if note:
for product_key, node in list(builder.nodes.items()):
if (
(node.get("label") == "TourProduct" and "独立成团" in compact(node.get("product_type")))
or (node.get("label") == "PriceCatalogItem" and "独立成团" in compact(node.get("catalog_type")))
):
rule_key = builder.add_node(
"BusinessRule",
f"rule:ind:{slug(sheet + note[:80])}",
f"{sheet}独立成团报名限制",
rule_id=f"RULE-IND-{digest(sheet, note[:120], length=10)}",
rule_type="eligibility",
applies_to_type="TourProduct",
applies_to_text="20-25人独立成团",
rule_text=note[:1200],
severity="关键限制",
source_file=str(path),
)
rel_type = "CATALOG_ITEM_HAS_RULE" if node.get("label") == "PriceCatalogItem" else "PRODUCT_HAS_RULE"
builder.add_rel(rel_type, product_key, rule_key)
def parse_transfer_quotes(builder: KGBuilder) -> None:
path = SOURCE_ROOT / "黔玩转接送组报价.docx"
text = read_office_text(path)
current_vehicle = ""
for raw in text.splitlines():
line = compact(raw)
if not line:
continue
if re.match(r"^\d+座|^7座", line) and "趟" not in line:
current_vehicle = line
continue
m = re.search(r"(.+?)[-—–]+(.+?)(\d+)\s*/?\s*趟", line)
if not m or not current_vehicle:
continue
origin, destination, price = compact(m.group(1)), compact(m.group(2)), float(m.group(3))
key = builder.add_node(
"TransferQuote",
f"transfer:{slug(current_vehicle + origin + destination + str(price))}",
f"{current_vehicle} {origin}->{destination}",
transfer_quote_id=f"TQ-{digest(current_vehicle, origin, destination, price, length=10)}",
origin_text=origin,
destination_text=destination,
vehicle_type=current_vehicle,
price_per_trip=price,
quote_unit="趟",
quote_notes=line,
source_file=str(path),
)
for area_text, rel_type in [(origin, "LOCATED_IN"), (destination, "LOCATED_IN")]:
for part in split_items(area_text.replace("、", ","), limit=6):
area_key = builder.add_node("Area", f"area:{slug(part)}", part, area_id=f"AREA-{digest(part, length=8)}", area_type="接送区域")
# LOCATED_IN start type does not include TransferQuote in schema; keep area text in quote properties.
def parse_sales_scripts(builder: KGBuilder) -> None:
path = SOURCE_ROOT / "线上客资回复话术.docx"
text = read_office_text(path)
chunks = re.split(r"(?=Step\d+\.|STEP\d+|步骤[一二三四五六七八九十])", text)
for idx, chunk in enumerate(chunks):
msg = compact(chunk)
if len(msg) < 30:
continue
channel = "微信" if "微信" in msg or idx > 1 else "小红书"
stage = "产品推荐" if "产品" in msg or "路线" in msg else ("留资引导" if "VX" in msg or "加V" in msg else "首次沟通")
key = builder.add_node(
"SalesScript",
f"script:{idx}:{slug(msg[:40])}",
f"{channel}-{stage}-{idx}",
script_id=f"SCRIPT-{idx:03d}",
channel=channel,
funnel_stage=stage,
trigger_scenario=msg[:120],
message_template=msg[:1500],
intent_tags=[tag for tag in ["留资", "报价", "费用包含", "纯玩", "房间数", "老人小孩", "产品推荐"] if tag in msg],
required_customer_fields=[field for field in ["月份", "人数", "天数", "房间数", "老人", "小孩", "酒店", "预算"] if field in msg],
source_file=str(path),
)
# Link to explain broad objects after product creation is complete in a lightweight way.
for product_key, product in list(builder.nodes.items())[:200]:
if product.get("label") == "TourProduct" and any(token and token in msg for token in split_items(product.get("name"), limit=3)):
builder.add_rel("SCRIPT_EXPLAINS", key, product_key)
def add_resource_library_layer(builder: KGBuilder) -> None:
root_key = builder.add_node(
"ResourceLibrary",
"library:root:travel_graph",
"旅行社线路制定业务资料库",
library_id="LIB-TRAVEL-GRAPH",
library_type="root",
description="统一组织已有路线产品、基础资源、价格报价、规则和图片素材,方便业务视角浏览图谱。",
scope="旅行社线路制定",
)
sections = {
"route_products": ("已有路线产品库", "product_catalog", "已成型路线产品,路线骨架固定。"),
"base_resources": ("基础资源库", "base_resource", "景点、酒店、餐饮、车辆、接送等可复用资料。"),
"price_catalog": ("报价目录库", "price_catalog", "价格表中不等同于完整路线的报价对象。"),
"rule_library": ("业务规则库", "rule", "退费、限制、优惠、风险、预订等规则。"),
"media_library": ("图片素材库", "media", "可靠图片 URL 与素材组。"),
}
section_keys: dict[str, str] = {}
for code, (name, library_type, desc) in sections.items():
key = builder.add_node(
"ResourceLibrary",
f"library:{code}",
name,
library_id=f"LIB-{code.upper()}",
library_type=library_type,
description=desc,
scope="旅行社线路制定",
)
builder.add_rel("LIBRARY_HAS_SECTION", root_key, key)
section_keys[code] = key
resource_rel_by_label = {
"ScenicArea": "LIBRARY_CONTAINS_SCENIC_AREA",
"ScenicAttraction": "LIBRARY_CONTAINS_ATTRACTION",
"HotelResource": "LIBRARY_CONTAINS_HOTEL",
"RestaurantResource": "LIBRARY_CONTAINS_RESTAURANT",
"VehicleResource": "LIBRARY_CONTAINS_VEHICLE",
"TransferQuote": "LIBRARY_CONTAINS_TRANSFER",
"ResourceOptionGroup": "LIBRARY_CONTAINS_OPTION_GROUP",
}
for key, node in list(builder.nodes.items()):
label = node.get("label")
if label == "TourProduct":
builder.add_rel("LIBRARY_CONTAINS_PRODUCT", section_keys["route_products"], key)
elif label == "PriceCatalogItem":
builder.add_rel("LIBRARY_CONTAINS_CATALOG_ITEM", section_keys["price_catalog"], key)
elif label in resource_rel_by_label:
builder.add_rel(resource_rel_by_label[label], section_keys["base_resources"], key)
elif label == "BusinessRule":
builder.add_rel("LIBRARY_CONTAINS_RULE", section_keys["rule_library"], key)
elif label == "MediaResource":
builder.add_rel("LIBRARY_CONTAINS_MEDIA", section_keys["media_library"], key)
def write_outputs(builder: KGBuilder, schema: dict[str, Any]) -> None:
OUT_DIR.mkdir(parents=True, exist_ok=True)
SCHEMA_OUT_DIR.mkdir(parents=True, exist_ok=True)
schema_json = SCHEMA_OUT_DIR / "travel_graph_existing_product_schema.v1.4.json"
schema_dsl = SCHEMA_OUT_DIR / "travel_graph_existing_product_schema.v1.4.dsl.md"
schema_json.write_text(json.dumps(schema, ensure_ascii=False, indent=2), encoding="utf-8")
schema_dsl.write_text(schema_to_dsl(schema), encoding="utf-8")
(OUT_DIR / schema_json.name).write_text(schema_json.read_text(encoding="utf-8"), encoding="utf-8")
(OUT_DIR / schema_dsl.name).write_text(schema_dsl.read_text(encoding="utf-8"), encoding="utf-8")
nodes = list(builder.nodes.values())
rels = builder.relations
(OUT_DIR / "抽取结果_nodes.json").write_text(json.dumps(nodes, ensure_ascii=False, indent=2), encoding="utf-8")
(OUT_DIR / "抽取结果_relations.json").write_text(json.dumps(rels, ensure_ascii=False, indent=2), encoding="utf-8")
with (OUT_DIR / "抽取结果_nodes.csv").open("w", newline="", encoding="utf-8-sig") as fh:
writer = csv.DictWriter(fh, fieldnames=["label", "natural_key", "name", "summary"])
writer.writeheader()
for node in nodes:
writer.writerow({
"label": node["label"],
"natural_key": node["natural_key"],
"name": node["name"],
"summary": node.get("route_path") or node.get("package_name") or node.get("rule_text") or node.get("primary_image_url") or "",
})
with (OUT_DIR / "抽取结果_relations.csv").open("w", newline="", encoding="utf-8-sig") as fh:
writer = csv.DictWriter(fh, fieldnames=["relation_type", "source", "target", "properties"])
writer.writeheader()
for rel in rels:
writer.writerow({**rel, "properties": json.dumps(rel.get("properties") or {}, ensure_ascii=False)})
node_counts = Counter(node["label"] for node in nodes)
rel_counts = Counter(rel["relation_type"] for rel in rels)
report = [
"# travel_graph 旅行社线路制定图谱入库说明",
"",
f"生成时间:{datetime.now().strftime('%Y-%m-%d %H:%M:%S')}",
"",
"## 项目",
f"- Project ID: `{PROJECT_ID}`",
f"- Tenant ID: `{TENANT_ID}`",
f"- FalkorDB Graph Name: `{GRAPH_NAME}`",
f"- 项目名称:{PROJECT_NAME}",
"",
"## 本期业务边界",
"- 只做已有路线产品,不做从零自由定制路线。",
"- 固定路线骨架不可变,住宿/餐饮/车辆/接送/房型/门票小交通以资源槽位方式支持客户微调。",
"- 景点到景点、景区片区到合作酒店/餐厅采用高德驾车距离与耗时;不把纯直线距离作为最终候选关系。",
"- 价格进入 ProductPricePackage、HotelResource、RestaurantResource、TransferQuote、TicketFee/FeeItem。",
"- 图片 URL 进入 MediaResource,并同步写入匹配到的景点/酒店/车辆/资源组选项实体。",
"- 车辆资源只采用图片资源库中 `标签=车辆` 且高/中可靠的资源;待确认、低可靠、无图或禁止冒充资源不作为车辆实体。",
"",
"## 节点统计",
*[f"- {k}: {v}" for k, v in node_counts.most_common()],
"",
"## 关系统计",
*[f"- {k}: {v}" for k, v in rel_counts.most_common()],
]
(OUT_DIR / "入库说明.md").write_text("\n".join(report), encoding="utf-8")
def upsert_postgres(builder: KGBuilder, schema: dict[str, Any]) -> dict[str, int]:
with psycopg.connect(DB_URL, row_factory=dict_row) as conn:
with conn.cursor() as cur:
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.projects (
tenant_id, project_id, display_name, description, status,
default_namespace, metadata_jsonb, created_by, updated_at
)
VALUES (%s,%s,%s,%s,'active',%s,%s,'codex-import',now())
ON CONFLICT (tenant_id, project_id) DO UPDATE
SET display_name=EXCLUDED.display_name,
description=EXCLUDED.description,
status='active',
default_namespace=EXCLUDED.default_namespace,
metadata_jsonb=EXCLUDED.metadata_jsonb,
updated_at=now()
""",
(
TENANT_ID,
PROJECT_ID,
PROJECT_NAME,
"已有路线产品知识图谱:固定路线骨架、可配置资源槽位、价格、规则和图片素材。",
SCHEMA_NAMESPACE,
Jsonb({"business": "travel_agency_existing_product", "graph_name": GRAPH_NAME}),
),
)
cur.execute(
f"""
UPDATE {DB_SCHEMA}.ontology_schemas
SET status='archived', updated_at=now()
WHERE tenant_id=%s AND project_id=%s AND namespace=%s AND version <> %s
""",
(TENANT_ID, PROJECT_ID, SCHEMA_NAMESPACE, SCHEMA_VERSION),
)
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.ontology_schemas (
tenant_id, project_id, namespace, version, display_name, description,
status, schema_jsonb, created_by, published_by, published_at, updated_at
)
VALUES (%s,%s,%s,%s,%s,%s,'active',%s,'codex-import','codex-import',now(),now())
ON CONFLICT (tenant_id, project_id, namespace, version) DO UPDATE
SET display_name=EXCLUDED.display_name,
description=EXCLUDED.description,
status='active',
schema_jsonb=EXCLUDED.schema_jsonb,
published_by='codex-import',
published_at=now(),
updated_at=now()
RETURNING id
""",
(TENANT_ID, PROJECT_ID, SCHEMA_NAMESPACE, SCHEMA_VERSION, schema["display_name"], schema["purpose"], Jsonb(schema)),
)
schema_id = cur.fetchone()["id"]
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.graph_releases (
tenant_id, project_id, graph_release_id, graph_name, alias, status,
schema_id, source_dataset_version, metadata_jsonb, created_by,
published_at, activated_at, updated_at
)
VALUES (%s,%s,%s,%s,'active','active',%s,%s,%s,'codex-import',now(),now(),now())
ON CONFLICT (tenant_id, project_id, alias) DO UPDATE
SET graph_release_id=EXCLUDED.graph_release_id,
graph_name=EXCLUDED.graph_name,
status='active',
schema_id=EXCLUDED.schema_id,
source_dataset_version=EXCLUDED.source_dataset_version,
metadata_jsonb=EXCLUDED.metadata_jsonb,
activated_at=now(),
updated_at=now()
""",
(
TENANT_ID, PROJECT_ID, "travel_graph_v1", GRAPH_NAME, schema_id,
"existing-route-products-md-resource-workbooks-amap-driving-2026",
Jsonb({"node_count": len(builder.nodes), "relation_count": len(builder.relations), "output_dir": str(OUT_DIR)}),
),
)
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.import_templates (
template_id, version, display_name, primary_entity, template_jsonb, status, updated_at
)
VALUES (%s,1,%s,'TourProduct',%s,'active',now())
ON CONFLICT (template_id, version) DO UPDATE
SET display_name=EXCLUDED.display_name,
template_jsonb=EXCLUDED.template_jsonb,
status='active',
updated_at=now()
""",
(TEMPLATE_ID, "旅行社线路制定已有产品导入模板", Jsonb(schema)),
)
cur.execute(f"DELETE FROM {DB_SCHEMA}.candidate_relations WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
cur.execute(f"DELETE FROM {DB_SCHEMA}.candidate_entities WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
cur.execute(
f"""
DELETE FROM {DB_SCHEMA}.raw_records rr
USING {DB_SCHEMA}.import_batches ib
WHERE rr.batch_id=ib.id AND ib.tenant_id=%s AND ib.project_id=%s
""",
(TENANT_ID, PROJECT_ID),
)
cur.execute(f"DELETE FROM {DB_SCHEMA}.import_batches WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
file_hash = hashlib.md5(json.dumps({"nodes": list(builder.nodes), "rels": builder.relations}, ensure_ascii=False).encode()).hexdigest()
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.import_batches (
tenant_id, project_id, graph_name, template_id, source_name, file_name,
file_hash, status, total_rows, success_rows, failed_rows, created_by, updated_at
)
VALUES (%s,%s,%s,%s,%s,%s,%s,'published',%s,%s,0,'codex-import',now())
RETURNING id
""",
(
TENANT_ID, PROJECT_ID, GRAPH_NAME, TEMPLATE_ID, "旅行社已有路线产品+资源库",
str(SOURCE_ROOT), file_hash, len(builder.nodes) + len(builder.relations), len(builder.nodes) + len(builder.relations),
),
)
batch_id = cur.fetchone()["id"]
id_by_key: dict[str, int] = {}
for row_number, (key, node) in enumerate(builder.nodes.items(), start=1):
payload = {k: v for k, v in node.items() if k not in {"label", "natural_key", "name"}}
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.candidate_entities (
tenant_id, project_id, batch_id, template_id, entity_type, natural_key,
display_name, payload_jsonb, confidence, status, reviewed_by, reviewed_at, updated_at
)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,0.94,'published','codex-import',now(),now())
RETURNING id
""",
(TENANT_ID, PROJECT_ID, batch_id, TEMPLATE_ID, node["label"], key, node.get("name") or key, Jsonb(payload)),
)
id_by_key[key] = cur.fetchone()["id"]
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.raw_records (batch_id, row_number, raw_jsonb, row_hash, parse_status)
VALUES (%s,%s,%s,%s,'parsed')
ON CONFLICT (batch_id, row_number) DO NOTHING
""",
(batch_id, row_number, Jsonb(node), hashlib.md5(json.dumps(node, ensure_ascii=False, sort_keys=True).encode()).hexdigest()),
)
for rel in builder.relations:
src_id = id_by_key.get(rel["source"])
dst_id = id_by_key.get(rel["target"])
if not src_id or not dst_id:
continue
cur.execute(
f"""
INSERT INTO {DB_SCHEMA}.candidate_relations (
tenant_id, project_id, batch_id, source_candidate_id, relation_type,
target_candidate_id, target_ref_jsonb, payload_jsonb, status
)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,'published')
""",
(TENANT_ID, PROJECT_ID, batch_id, src_id, rel["relation_type"], dst_id, Jsonb({"natural_key": rel["target"]}), Jsonb(rel.get("properties") or {})),
)
conn.commit()
return {"schema_id": schema_id, "batch_id": batch_id}
def write_falkor(builder: KGBuilder) -> dict[str, int]:
db = FalkorDB(host="localhost", port=6380)
if GRAPH_NAME in db.list_graphs():
db.select_graph(GRAPH_NAME).delete()
graph = db.select_graph(GRAPH_NAME)
for node in builder.nodes.values():
label = re.sub(r"[^A-Za-z0-9_]", "", node["label"]) or "Entity"
props = graph_safe_props(node)
graph.query(f"MERGE (n:{label} {{natural_key:$natural_key}}) SET n += $props", {"natural_key": node["natural_key"], "props": props})
for rel in builder.relations:
rel_type = re.sub(r"[^A-Z0-9_]", "", rel["relation_type"].upper()) or "RELATED_TO"
props = graph_safe_props({"natural_key": f"{rel['source']}->{rel_type}->{rel['target']}", **(rel.get("properties") or {})})
graph.query(
"""
MATCH (a {natural_key:$source}), (b {natural_key:$target})
MERGE (a)-[r:%s]->(b)
SET r += $props
""" % rel_type,
{"source": rel["source"], "target": rel["target"], "props": props},
)
return {
"graph_nodes": graph.query("MATCH (n) RETURN count(n)").result_set[0][0],
"graph_relations": graph.query("MATCH ()-[r]->() RETURN count(r)").result_set[0][0],
}
def build() -> dict[str, Any]:
builder = KGBuilder()
schema = load_schema()
amap_cache = load_amap_enrichment_cache()
driving_metric_cache = load_amap_driving_metric_cache()
media_index = load_media_resources(builder)
alias_to_key = seed_attractions(builder, media_index)
_vehicle_keys, vehicle_group_key = seed_vehicles_from_media(builder, media_index)
hotel_groups, restaurant_groups = parse_resource_workbooks(builder, media_index)
parse_existing_route_markdown(builder, alias_to_key, hotel_groups, restaurant_groups, vehicle_group_key)
parse_small_group_prices(builder)
parse_independent_prices(builder)
parse_transfer_quotes(builder)
parse_sales_scripts(builder)
apply_amap_enrichment_layer(builder, amap_cache)
derive_scenic_area_coordinates(builder)
add_administrative_region_layer(builder)
apply_driving_metric_layer(builder, driving_metric_cache)
add_route_region_layer(builder)
add_route_scenic_area_layer(builder)
add_slot_candidate_layers(builder)
add_resource_library_layer(builder)
write_outputs(builder, schema)
pg_info = upsert_postgres(builder, schema)
graph_info = write_falkor(builder)
summary = {
"tenant_id": TENANT_ID,
"project_id": PROJECT_ID,
"project_name": PROJECT_NAME,
"graph_name": GRAPH_NAME,
"nodes": len(builder.nodes),
"relations": len(builder.relations),
**pg_info,
**graph_info,
"output_dir": str(OUT_DIR),
}
OUT_DIR.mkdir(parents=True, exist_ok=True)
(OUT_DIR / "入库执行摘要.json").write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
return summary
if __name__ == "__main__":
print(json.dumps(build(), ensure_ascii=False, indent=2))