2898 lines
140 KiB
Python
2898 lines
140 KiB
Python
from __future__ import annotations
|
||
|
||
import csv
|
||
import hashlib
|
||
import json
|
||
import math
|
||
import re
|
||
import subprocess
|
||
from collections import Counter, defaultdict
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
import pandas as pd
|
||
import psycopg
|
||
from falkordb import FalkorDB
|
||
from psycopg.rows import dict_row
|
||
from psycopg.types.json import Jsonb
|
||
|
||
from common_paths import PROJECT_ROOT, TRAVEL_AGENCY_SOURCE_ROOT, TRAVEL_KG_EXPORT_ROOT
|
||
|
||
SOURCE_ROOT = TRAVEL_AGENCY_SOURCE_ROOT
|
||
ROUTE_MD_DIR = SOURCE_ROOT / "2026年新行程打包_md整理"
|
||
ROUTE_MD_PRODUCTS = ROUTE_MD_DIR / "products"
|
||
SCHEMA_SRC = PROJECT_ROOT / "schema搭建/travel_agency_business/travel_agency_existing_product_schema.v1.json"
|
||
SCHEMA_OUT_DIR = PROJECT_ROOT / "schema搭建/travel_graph_existing_product"
|
||
OUT_DIR = TRAVEL_KG_EXPORT_ROOT / "travel_graph_旅行社线路制定"
|
||
AMAP_CACHE_PATH = OUT_DIR / "amap_poi_enrichment_cache.json"
|
||
AMAP_DRIVING_CACHE_PATH = OUT_DIR / "amap_driving_distance_cache.json"
|
||
|
||
DB_URL = "postgresql://admin:password@localhost:5433/kg_admin"
|
||
DB_SCHEMA = "kg_admin_new2"
|
||
TENANT_ID = "travel_agency"
|
||
PROJECT_ID = "travel_graph"
|
||
PROJECT_NAME = "旅行社线路制定"
|
||
GRAPH_NAME = "travel_graph"
|
||
SCHEMA_NAMESPACE = "travel_agency_existing_product"
|
||
SCHEMA_VERSION = 4
|
||
TEMPLATE_ID = "travel_graph_existing_product_v4"
|
||
|
||
|
||
ATTRACTION_SEEDS = [
|
||
("黄果树", ["黄果树", "黄果树瀑布", "黄果树大瀑布", "黄果树风景名胜区"], "安顺", "瀑布/5A", "贵州龙头景区,瀑布群核心卖点。"),
|
||
("天星桥", ["天星桥", "天星桥景区"], "安顺", "喀斯特/黄果树景区", "水上石林、天然盆景。"),
|
||
("陡坡塘瀑布", ["陡坡塘", "陡坡塘瀑布"], "安顺", "瀑布/黄果树景区", "瀑面宽,西游记取景。"),
|
||
("荔波小七孔", ["小七孔", "荔波小七孔", "小七孔景区"], "黔南", "山水/5A", "世界自然遗产,水上森林、卧龙潭等。"),
|
||
("西江千户苗寨", ["西江", "西江苗寨", "西江千户苗寨"], "黔东南", "民族村寨/4A", "苗寨夜景、长桌宴、吊脚楼。"),
|
||
("镇远古城", ["镇远", "镇远古镇", "镇远古城"], "黔东南", "古城/5A", "古城夜景、舞阳河沿岸住宿。"),
|
||
("梵净山", ["梵净山"], "铜仁", "山岳/5A", "弥勒道场、蘑菇石、金顶。"),
|
||
("青岩古镇", ["青岩", "青岩古镇"], "贵阳", "古镇/5A", "卤猪脚、小吃、送机前半日游。"),
|
||
("百里杜鹃", ["百里杜鹃"], "毕节", "赏花", "3-4月花期主题。"),
|
||
("平坝樱花", ["平坝樱花", "平坝农场"], "安顺", "赏花", "春季樱花主题。"),
|
||
("织金洞", ["织金洞"], "毕节", "溶洞/5A", "大型喀斯特溶洞。"),
|
||
("中国天眼", ["天眼", "中国天眼", "FAST"], "黔南", "科技研学", "天文研学卖点。"),
|
||
("茅台镇", ["茅台", "茅台镇"], "遵义", "酒文化", "酱酒文化体验。"),
|
||
("遵义会议会址", ["遵义会址", "遵义会议会址"], "遵义", "红色文化", "红色研学路线核心。"),
|
||
("兴义万峰林", ["万峰林", "兴义万峰林"], "黔西南", "峰林", "黔西南山水。"),
|
||
("万峰湖", ["万峰湖"], "黔西南", "湖泊", "兴义水上体验。"),
|
||
("马岭河峡谷", ["马岭河", "马岭河峡谷"], "黔西南", "峡谷", "兴义峡谷景观。"),
|
||
("花江大桥", ["花江大桥"], "安顺/黔西南", "桥梁景观", "桥见贵州特色线路。"),
|
||
("龙宫", ["龙宫"], "安顺", "溶洞/5A", "安顺秘境类产品。"),
|
||
("天河潭", ["天河潭"], "贵阳", "山水", "贵阳近郊半日/首日。"),
|
||
("甲秀楼", ["甲秀楼"], "贵阳", "城市地标", "贵阳市区地标。"),
|
||
("黔灵山公园", ["黔灵公园", "黔灵山"], "贵阳", "城市公园", "贵阳市区轻量游。"),
|
||
("乌江寨", ["乌江寨"], "遵义", "度假街区", "夜游/住宿度假。"),
|
||
("野洞河", ["野洞河"], "黔东南", "漂流", "漂流体验。"),
|
||
("安顺古城", ["安顺古城"], "安顺", "城市/夜游", "安顺中转游览。"),
|
||
("中南门古城", ["中南门", "中南门古城"], "铜仁", "古城/夜游", "铜仁夜游街区。"),
|
||
]
|
||
|
||
|
||
LOCATION_RESOURCE_LABELS = {"ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"}
|
||
|
||
|
||
SCENIC_AREA_DEFINITIONS = {
|
||
"黄果树旅游景区": {
|
||
"aliases": ["黄果树景区", "黄果树风景名胜区", "黄果树瀑布景区"],
|
||
"city": "安顺",
|
||
"admin_hint": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"area_type": "国家级风景名胜区/5A景区",
|
||
"route_anchor": True,
|
||
"members": {
|
||
"黄果树": "核心瀑布游览点",
|
||
"天星桥": "黄果树景区子景点/游览点",
|
||
"陡坡塘瀑布": "黄果树景区子景点/游览点",
|
||
},
|
||
"note": "黄果树片区作为资源池锚点;天星桥、陡坡塘、大瀑布不作为平级主目的地推荐。",
|
||
},
|
||
}
|
||
|
||
SCENIC_MEMBER_PARENT = {
|
||
member: (area_name, role)
|
||
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
|
||
for member, role in spec["members"].items()
|
||
}
|
||
|
||
|
||
ATTRACTION_ADMIN_HINTS = {
|
||
"黄果树": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"天星桥": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"陡坡塘瀑布": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"荔波小七孔": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
|
||
"西江千户苗寨": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
|
||
"镇远古城": ("贵州省", "黔东南苗族侗族自治州", "镇远县"),
|
||
"梵净山": ("贵州省", "铜仁市", "江口县"),
|
||
"青岩古镇": ("贵州省", "贵阳市", "花溪区"),
|
||
"百里杜鹃": ("贵州省", "毕节市", "百里杜鹃管理区"),
|
||
"平坝樱花": ("贵州省", "安顺市", "平坝区"),
|
||
"织金洞": ("贵州省", "毕节市", "织金县"),
|
||
"中国天眼": ("贵州省", "黔南布依族苗族自治州", "平塘县"),
|
||
"茅台镇": ("贵州省", "遵义市", "仁怀市"),
|
||
"遵义会议会址": ("贵州省", "遵义市", "红花岗区"),
|
||
"兴义万峰林": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"万峰湖": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"马岭河峡谷": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"花江大桥": ("贵州省", "安顺市", "关岭布依族苗族自治县"),
|
||
"龙宫": ("贵州省", "安顺市", "西秀区"),
|
||
"天河潭": ("贵州省", "贵阳市", "花溪区"),
|
||
"甲秀楼": ("贵州省", "贵阳市", "南明区"),
|
||
"黔灵山公园": ("贵州省", "贵阳市", "云岩区"),
|
||
"乌江寨": ("贵州省", "遵义市", "播州区"),
|
||
"野洞河": ("贵州省", "黔东南苗族侗族自治州", ""),
|
||
"安顺古城": ("贵州省", "安顺市", "西秀区"),
|
||
"中南门古城": ("贵州省", "铜仁市", "碧江区"),
|
||
}
|
||
|
||
|
||
def scenic_area_name_for_attraction(name: str) -> str:
|
||
if name in {"茅台镇"}:
|
||
return f"{name}酒文化片区"
|
||
if name in {"花江大桥"}:
|
||
return f"{name}观景片区"
|
||
if name in {"甲秀楼", "中南门古城", "安顺古城"}:
|
||
return f"{name}游览片区"
|
||
return f"{name}景区"
|
||
|
||
|
||
def expand_scenic_area_definitions() -> None:
|
||
"""Every route destination needs a scenic-area/resource-pool anchor.
|
||
|
||
黄果树 is a compound scenic area with child visit points. Other core route
|
||
destinations currently have one representative ScenicAttraction, but still
|
||
need a ScenicArea anchor so the browser can show:
|
||
行政区 -> 景区/片区 -> 酒店/餐饮/路线产品。
|
||
"""
|
||
existing_members = {
|
||
member
|
||
for spec in SCENIC_AREA_DEFINITIONS.values()
|
||
for member in (spec.get("members") or {})
|
||
}
|
||
for name, aliases, city, attraction_type, point in ATTRACTION_SEEDS:
|
||
if name in existing_members:
|
||
continue
|
||
area_name = scenic_area_name_for_attraction(name)
|
||
if area_name in SCENIC_AREA_DEFINITIONS:
|
||
continue
|
||
admin_hint = ATTRACTION_ADMIN_HINTS.get(name, ("贵州省", city, ""))
|
||
SCENIC_AREA_DEFINITIONS[area_name] = {
|
||
"aliases": sorted(set([area_name, name, *aliases])),
|
||
"city": city,
|
||
"admin_hint": admin_hint,
|
||
"area_type": attraction_type,
|
||
"route_anchor": True,
|
||
"standalone_area": True,
|
||
"representative_member": name,
|
||
"members": {name: "核心游览点/主目的地"},
|
||
"note": f"{area_name}作为{city}线路资源池锚点;酒店/餐饮候选以景区锚点高德车程计算。",
|
||
}
|
||
|
||
|
||
expand_scenic_area_definitions()
|
||
SCENIC_MEMBER_PARENT = {
|
||
member: (area_name, role)
|
||
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
|
||
for member, role in (spec.get("members") or {}).items()
|
||
}
|
||
SCENIC_MEMBER_IS_CHILD = {
|
||
member: not bool(spec.get("standalone_area"))
|
||
for area_name, spec in SCENIC_AREA_DEFINITIONS.items()
|
||
for member in (spec.get("members") or {})
|
||
}
|
||
|
||
|
||
REGION_TEXT_HINTS = {
|
||
"贵阳": ("贵州省", "贵阳市", ""),
|
||
"龙洞堡": ("贵州省", "贵阳市", "南明区"),
|
||
"双龙": ("贵州省", "贵阳市", ""),
|
||
"青岩": ("贵州省", "贵阳市", "花溪区"),
|
||
"花溪": ("贵州省", "贵阳市", "花溪区"),
|
||
"甲秀": ("贵州省", "贵阳市", "南明区"),
|
||
"黔灵": ("贵州省", "贵阳市", "云岩区"),
|
||
"安顺": ("贵州省", "安顺市", ""),
|
||
"黄果树": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"天星桥": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"陡坡塘": ("贵州省", "安顺市", "镇宁布依族苗族自治县"),
|
||
"龙宫": ("贵州省", "安顺市", "西秀区"),
|
||
"西江": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
|
||
"苗寨": ("贵州省", "黔东南苗族侗族自治州", "雷山县"),
|
||
"镇远": ("贵州省", "黔东南苗族侗族自治州", "镇远县"),
|
||
"梵净山": ("贵州省", "铜仁市", "江口县"),
|
||
"江口": ("贵州省", "铜仁市", "江口县"),
|
||
"铜仁": ("贵州省", "铜仁市", ""),
|
||
"荔波": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
|
||
"小七孔": ("贵州省", "黔南布依族苗族自治州", "荔波县"),
|
||
"织金": ("贵州省", "毕节市", "织金县"),
|
||
"毕节": ("贵州省", "毕节市", ""),
|
||
"百里杜鹃": ("贵州省", "毕节市", "百里杜鹃管理区"),
|
||
"开阳": ("贵州省", "贵阳市", "开阳县"),
|
||
"猴耳天坑": ("贵州省", "贵阳市", "开阳县"),
|
||
"遵义": ("贵州省", "遵义市", ""),
|
||
"乌江寨": ("贵州省", "遵义市", "播州区"),
|
||
"茅台": ("贵州省", "遵义市", "仁怀市"),
|
||
"遵义会址": ("贵州省", "遵义市", "红花岗区"),
|
||
"习水": ("贵州省", "遵义市", "习水县"),
|
||
"四渡赤水": ("贵州省", "遵义市", "习水县"),
|
||
"独山": ("贵州省", "黔南布依族苗族自治州", "独山县"),
|
||
"都匀": ("贵州省", "黔南布依族苗族自治州", "都匀市"),
|
||
"龙里": ("贵州省", "黔南布依族苗族自治州", "龙里县"),
|
||
"肇兴": ("贵州省", "黔东南苗族侗族自治州", "黎平县"),
|
||
"从江": ("贵州省", "黔东南苗族侗族自治州", "从江县"),
|
||
"丹寨": ("贵州省", "黔东南苗族侗族自治州", "丹寨县"),
|
||
"兴义": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"万峰林": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"马岭河": ("贵州省", "黔西南布依族苗族自治州", "兴义市"),
|
||
"晴隆": ("贵州省", "黔西南布依族苗族自治州", "晴隆县"),
|
||
}
|
||
|
||
|
||
RESTAURANT_SECTION_REGION_ALIASES = {
|
||
"黔东南": "黔东南区域",
|
||
"黔北": "遵义区域",
|
||
"黔西": "黔西南区域",
|
||
"黔西南": "黔西南区域",
|
||
}
|
||
|
||
|
||
def clean(value: Any) -> str:
|
||
if value is None:
|
||
return ""
|
||
if isinstance(value, float) and pd.isna(value):
|
||
return ""
|
||
text = str(value).replace("\x00", "").replace("\u200b", "").replace("\u200f", "")
|
||
text = re.sub(r"[ \t]+", " ", text)
|
||
return re.sub(r"\n{3,}", "\n\n", text).strip()
|
||
|
||
|
||
def compact(value: Any) -> str:
|
||
return re.sub(r"\s+", " ", clean(value)).strip()
|
||
|
||
|
||
def norm(value: Any) -> str:
|
||
return re.sub(r"[\s()()《》【】、,。::/\\\\·++\-—_]+", "", compact(value)).lower()
|
||
|
||
|
||
def slug(text: str, prefix: str = "") -> str:
|
||
base = re.sub(r"[\s()()《》【】、,。::/\\\\]+", "_", compact(text))
|
||
base = re.sub(r"_+", "_", base).strip("_")
|
||
digest = hashlib.md5(compact(text).encode("utf-8")).hexdigest()[:8]
|
||
return f"{prefix}{base[:50]}_{digest}"
|
||
|
||
|
||
def digest(*parts: Any, length: int = 10) -> str:
|
||
raw = "||".join(compact(p) for p in parts)
|
||
return hashlib.md5(raw.encode("utf-8")).hexdigest()[:length].upper()
|
||
|
||
|
||
def money(value: Any) -> float | None:
|
||
m = re.search(r"-?\d+(?:\.\d+)?", compact(value))
|
||
return float(m.group()) if m else None
|
||
|
||
|
||
def number(value: Any) -> float | None:
|
||
try:
|
||
if value in (None, ""):
|
||
return None
|
||
return float(value)
|
||
except Exception:
|
||
m = re.search(r"-?\d+(?:\.\d+)?", compact(value))
|
||
return float(m.group()) if m else None
|
||
|
||
|
||
def haversine_km(lng1: float, lat1: float, lng2: float, lat2: float) -> float:
|
||
radius = 6371.0088
|
||
phi1 = math.radians(lat1)
|
||
phi2 = math.radians(lat2)
|
||
d_phi = math.radians(lat2 - lat1)
|
||
d_lam = math.radians(lng2 - lng1)
|
||
a = math.sin(d_phi / 2) ** 2 + math.cos(phi1) * math.cos(phi2) * math.sin(d_lam / 2) ** 2
|
||
return radius * (2 * math.atan2(math.sqrt(a), math.sqrt(1 - a)))
|
||
|
||
|
||
def split_items(text: str, limit: int = 40) -> list[str]:
|
||
parts = re.split(r"[、,,;/;\n]+", compact(text))
|
||
out: list[str] = []
|
||
seen: set[str] = set()
|
||
for part in parts:
|
||
part = part.strip()
|
||
if not part or part.lower() == "nan" or part in seen:
|
||
continue
|
||
seen.add(part)
|
||
out.append(part)
|
||
if len(out) >= limit:
|
||
break
|
||
return out
|
||
|
||
|
||
def clean_region_label(text: str) -> str:
|
||
value = compact(text)
|
||
value = re.sub(r"^\s*\d+(?:\.\d+)?\s*", "", value)
|
||
return value.strip()
|
||
|
||
|
||
def urls_from_text(text: Any) -> list[str]:
|
||
return re.findall(r"https?://[^\s,,;;]+", clean(text))
|
||
|
||
|
||
def read_office_text(path: Path) -> str:
|
||
proc = subprocess.run(["textutil", "-convert", "txt", "-stdout", str(path)], capture_output=True, text=True, check=False)
|
||
return proc.stdout if proc.returncode == 0 else ""
|
||
|
||
|
||
def duration_from_text(text: str) -> int | None:
|
||
m = re.search(r"(\d+)\s*(?:日游|天|日)", text)
|
||
if m:
|
||
return int(m.group(1))
|
||
cn = {"一": 1, "二": 2, "两": 2, "三": 3, "四": 4, "五": 5, "六": 6, "七": 7, "八": 8}
|
||
m = re.search(r"([一二两三四五六七八])日游", text)
|
||
return cn.get(m.group(1)) if m else None
|
||
|
||
|
||
def graph_safe_props(node: dict[str, Any]) -> dict[str, Any]:
|
||
props: dict[str, Any] = {}
|
||
for key, value in node.items():
|
||
if key == "label" or value is None:
|
||
continue
|
||
if isinstance(value, (dict, list)):
|
||
props[key] = json.dumps(value, ensure_ascii=False)
|
||
elif isinstance(value, (int, float, bool, str)):
|
||
props[key] = value
|
||
else:
|
||
props[key] = str(value)
|
||
return props
|
||
|
||
|
||
class KGBuilder:
|
||
def __init__(self) -> None:
|
||
self.nodes: dict[str, dict[str, Any]] = {}
|
||
self.relations: list[dict[str, Any]] = []
|
||
self._rel_seen: set[tuple[str, str, str, str]] = set()
|
||
|
||
def add_node(self, label: str, key: str, name: str, **props: Any) -> str:
|
||
if not key:
|
||
key = f"{label.lower()}:{slug(name)}"
|
||
payload = {
|
||
"label": label,
|
||
"natural_key": key,
|
||
"name": compact(name) or key,
|
||
**{k: v for k, v in props.items() if v not in (None, "", [], {})},
|
||
}
|
||
existing = self.nodes.get(key)
|
||
if existing:
|
||
merged = {**existing, **payload}
|
||
for field in (
|
||
"aliases", "source_files", "selling_points", "features", "applicable_products",
|
||
"signature_dishes", "meal_scene", "image_urls", "image_ids", "media_resource_ids",
|
||
"main_destinations", "service_scope",
|
||
):
|
||
vals: list[str] = []
|
||
for src in (existing.get(field), payload.get(field)):
|
||
if isinstance(src, list):
|
||
vals.extend(compact(x) for x in src if compact(x))
|
||
elif src:
|
||
vals.append(compact(src))
|
||
if vals:
|
||
merged[field] = sorted(set(vals))
|
||
self.nodes[key] = merged
|
||
else:
|
||
self.nodes[key] = payload
|
||
return key
|
||
|
||
def add_rel(self, rel_type: str, source: str, target: str, **props: Any) -> None:
|
||
if not source or not target or source == target:
|
||
return
|
||
payload = {k: v for k, v in props.items() if v not in (None, "", [], {})}
|
||
identity = (rel_type, source, target, json.dumps(payload, ensure_ascii=False, sort_keys=True))
|
||
if identity in self._rel_seen:
|
||
return
|
||
self._rel_seen.add(identity)
|
||
self.relations.append({"relation_type": rel_type, "source": source, "target": target, "properties": payload})
|
||
|
||
|
||
def load_schema() -> dict[str, Any]:
|
||
schema = json.loads(SCHEMA_SRC.read_text(encoding="utf-8"))
|
||
schema["namespace"] = SCHEMA_NAMESPACE
|
||
schema["version"] = "1.4"
|
||
schema["display_name"] = "旅行社线路制定知识图谱 Schema"
|
||
schema["purpose"] = "面向已有路线产品的客服问答和产品微调:固定路线骨架、可配置资源槽位、价格、规则、图片素材统一建模。"
|
||
schema["entity_types"]["PriceCatalogItem"] = {
|
||
"cn": "报价目录项",
|
||
"definition": "价格表中出现但未在路线文档中形成完整每日行程骨架的报价对象;用于保留报价,不作为完整路线推荐。",
|
||
"fields": [
|
||
"catalog_item_id", "name", "catalog_type", "duration_days", "product_direction",
|
||
"route_preview", "price_validity", "booking_notes", "source_file", "evidence_excerpt",
|
||
],
|
||
}
|
||
schema["entity_types"]["ResourceLibrary"] = {
|
||
"cn": "业务资料库",
|
||
"definition": "用于表达业务资料库/基础资源库/已有路线产品库等聚合目录,让图谱在业务视角下可浏览、可解释。",
|
||
"fields": ["library_id", "library_type", "name", "description", "scope", "source_file"],
|
||
}
|
||
schema["entity_types"]["ScenicArea"] = {
|
||
"cn": "景区/景点片区",
|
||
"definition": "面向路线推荐的景区级资源池锚点,例如黄果树旅游景区;用于把天星桥、陡坡塘等子游览点归到同一景区片区。",
|
||
"fields": [
|
||
"scenic_area_id", "name", "aliases", "city", "area_type", "route_anchor",
|
||
"admin_region_name", "admin_region_level", "resource_spatial_class",
|
||
"management_note", "source_files",
|
||
],
|
||
}
|
||
for entity in ["TicketFee", "FeeItem"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in [
|
||
"price_text", "adult_price", "child_price", "currency", "consumer_group",
|
||
"is_free", "is_optional", "price_status", "child_price_status",
|
||
"scenic_target_name", "scenic_target_type", "source_product_name",
|
||
"evidence_excerpt", "data_quality_status", "extraction_rule",
|
||
]:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
schema["entity_types"]["AdministrativeRegion"] = {
|
||
"cn": "行政区",
|
||
"definition": "贵州旅游资源的行政区底座,只为当前已入图景点及其相关资源建立省/市州/区县层级。",
|
||
"fields": [
|
||
"region_id", "name", "level", "province", "city", "county", "town",
|
||
"parent_id", "adcode", "center_lng", "center_lat", "geo_polygon",
|
||
"source_system", "source_file", "scope_note",
|
||
],
|
||
}
|
||
schema["entity_types"]["GeoPoint"] = {
|
||
"cn": "地理坐标点",
|
||
"definition": "资源的经纬度坐标,用于附近酒店/餐厅和距离推荐;车辆等服务资源不按附近关系建模。",
|
||
"fields": [
|
||
"geo_point_id", "lng", "lat", "address_text", "provider",
|
||
"precision_level", "source_system", "source_file",
|
||
],
|
||
}
|
||
image_fields = ["primary_image_url", "image_urls", "media_resource_ids", "media_reliability", "media_usage_note"]
|
||
for entity in ["ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource", "VehicleResource", "ResourceOptionGroup", "TourProduct"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in image_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
amap_fields = [
|
||
"amap_poi_id", "amap_name", "amap_type", "amap_typecode", "amap_address",
|
||
"amap_location", "amap_lng", "amap_lat", "amap_pname", "amap_cityname",
|
||
"amap_adname", "amap_adcode", "amap_tel", "amap_rating", "amap_cost", "amap_open_time",
|
||
"amap_photo_urls", "amap_match_status", "amap_match_confidence",
|
||
"amap_match_reason", "data_completeness_note",
|
||
]
|
||
for entity in ["ScenicAttraction", "HotelResource", "RestaurantResource"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in amap_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
location_fields = [
|
||
"resource_spatial_class", "location_lng", "location_lat", "admin_region_name",
|
||
"admin_region_level", "admin_region_source", "geo_match_status",
|
||
]
|
||
for entity in ["ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in location_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
attraction_hierarchy_fields = [
|
||
"attraction_level", "parent_scenic_area_name", "is_independent_destination",
|
||
"route_role", "walk_intensity_hint", "recommendation_note",
|
||
]
|
||
fields = schema["entity_types"]["ScenicAttraction"]["fields"]
|
||
for field in attraction_hierarchy_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
route_region_fields = ["admin_region_name", "admin_region_level", "route_region_path", "origin_region_name", "destination_region_name"]
|
||
for entity in ["ProductDay", "RouteStop", "RouteSegment"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in route_region_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
route_line_fields = [
|
||
"route_line_name", "route_line_id", "route_sequence_label", "route_display_name",
|
||
"route_stop_sequence", "route_day_summary", "route_display_text",
|
||
"global_stop_order", "day_stop_order", "station_like_name", "total_stop_count",
|
||
"previous_stop_name", "next_stop_name",
|
||
]
|
||
for entity in ["TourProduct", "ProductDay", "RouteStop"]:
|
||
fields = schema["entity_types"][entity]["fields"]
|
||
for field in route_line_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
vehicle_fields = ["resource_service_class", "service_region_mode", "suitable_group_size", "dispatch_basis"]
|
||
fields = schema["entity_types"]["VehicleResource"]["fields"]
|
||
for field in vehicle_fields:
|
||
if field not in fields:
|
||
fields.append(field)
|
||
schema["relation_types"]["PRICE_PACKAGE_HAS_FEE"] = ["ProductPricePackage", "FeeItem|TicketFee", "价格包包含费用项/门票小交通费用"]
|
||
schema["relation_types"]["CATALOG_ITEM_HAS_PRICE_PACKAGE"] = ["PriceCatalogItem", "ProductPricePackage", "报价目录项拥有价格包"]
|
||
schema["relation_types"]["CATALOG_ITEM_HAS_RULE"] = ["PriceCatalogItem", "BusinessRule", "报价目录项适用规则"]
|
||
schema["relation_types"]["CATALOG_ITEM_MATCHES_PRODUCT"] = ["PriceCatalogItem", "TourProduct", "报价目录项匹配已有路线产品"]
|
||
schema["relation_types"]["LIBRARY_HAS_SECTION"] = ["ResourceLibrary", "ResourceLibrary", "资料库包含子库"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_PRODUCT"] = ["ResourceLibrary", "TourProduct", "路线产品库包含已有路线产品"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_CATALOG_ITEM"] = ["ResourceLibrary", "PriceCatalogItem", "报价目录库包含报价项"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_ATTRACTION"] = ["ResourceLibrary", "ScenicAttraction", "景点资源库包含景点"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_SCENIC_AREA"] = ["ResourceLibrary", "ScenicArea", "资料库包含景区/片区锚点"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_HOTEL"] = ["ResourceLibrary", "HotelResource", "酒店资源库包含酒店"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_RESTAURANT"] = ["ResourceLibrary", "RestaurantResource", "餐饮资源库包含餐厅"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_VEHICLE"] = ["ResourceLibrary", "VehicleResource", "车辆资源库包含车辆"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_TRANSFER"] = ["ResourceLibrary", "TransferQuote", "接送报价库包含接送报价"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_OPTION_GROUP"] = ["ResourceLibrary", "ResourceOptionGroup", "基础资源库包含可选资源组"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_RULE"] = ["ResourceLibrary", "BusinessRule", "规则库包含业务规则"]
|
||
schema["relation_types"]["LIBRARY_CONTAINS_MEDIA"] = ["ResourceLibrary", "MediaResource", "素材库包含图片资源"]
|
||
schema["relation_types"]["GROUP_SERVES_ATTRACTION"] = ["ResourceOptionGroup", "ScenicAttraction", "资源组选项服务于景区/景点"]
|
||
schema["relation_types"]["HOTEL_SERVES_ATTRACTION"] = ["HotelResource", "ScenicAttraction", "酒店资源可服务于景区/景点住宿"]
|
||
schema["relation_types"]["RESTAURANT_SERVES_ATTRACTION"] = ["RestaurantResource", "ScenicAttraction", "餐厅资源可服务于景区/景点用餐"]
|
||
schema["relation_types"]["REGION_PARENT_OF"] = ["AdministrativeRegion", "AdministrativeRegion", "上级行政区包含下级行政区"]
|
||
schema["relation_types"]["SCENIC_AREA_HAS_ATTRACTION"] = ["ScenicArea", "ScenicAttraction", "景区包含具体游览点/子景点"]
|
||
schema["relation_types"]["ATTRACTION_PART_OF_SCENIC_AREA"] = ["ScenicAttraction", "ScenicArea", "具体游览点归属于景区片区"]
|
||
schema["relation_types"]["SCENIC_AREA_HAS_FEE"] = ["ScenicArea", "TicketFee|FeeItem", "景区片区拥有门票、小交通、保险或二次消费费用"]
|
||
schema["relation_types"]["ATTRACTION_HAS_FEE"] = ["ScenicAttraction", "TicketFee|FeeItem", "具体景点/游览点拥有门票、小交通、保险或二次消费费用"]
|
||
schema["relation_types"]["PRODUCT_HAS_FEE"] = ["TourProduct", "TicketFee|FeeItem", "产品涉及费用项目"]
|
||
schema["relation_types"]["LOCATED_IN_REGION"] = ["ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "AdministrativeRegion", "位置资源位于行政区"]
|
||
schema["relation_types"]["RESOURCE_HAS_GEOPOINT"] = ["ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "GeoPoint", "位置资源拥有经纬度坐标"]
|
||
schema["relation_types"]["NEARBY_LOCATION_RESOURCE"] = ["ScenicArea|ScenicAttraction", "HotelResource|RestaurantResource", "景区片区或独立景点附近位置资源,最终关系必须来自高德驾车距离/耗时;子景点不直接连酒店餐饮"]
|
||
schema["relation_types"]["DRIVING_ROUTE_METRIC"] = ["ScenicArea|ScenicAttraction", "ScenicArea|ScenicAttraction|HotelResource|RestaurantResource", "高德驾车距离与耗时指标,用于景区顺序、酒店/餐饮候选排序和行程可行性判断"]
|
||
schema["relation_types"]["VEHICLE_SUITABLE_FOR_PRODUCT"] = ["VehicleResource", "TourProduct", "车辆按人数/车型/服务能力适配产品,不按景点附近推荐"]
|
||
schema["relation_types"]["STOP_LOCATED_IN_REGION"] = ["RouteStop", "AdministrativeRegion", "停靠点落入行政区"]
|
||
schema["relation_types"]["STOP_VISITS_SCENIC_AREA"] = ["RouteStop", "ScenicArea", "路线停靠点访问景区片区"]
|
||
schema["relation_types"]["PRODUCT_HAS_ORDERED_STOP"] = ["TourProduct", "RouteStop", "产品线路按全程顺序包含停靠点,等价公交线路-站点序列"]
|
||
schema["relation_types"]["ROUTE_STOP_NEXT"] = ["RouteStop", "RouteStop", "同一产品线路中相邻停靠点的先后关系"]
|
||
schema["relation_types"]["DAY_NEXT_DAY"] = ["ProductDay", "ProductDay", "同一产品线路中每日行程的先后关系"]
|
||
schema["relation_types"]["DAY_COVERS_SCENIC_AREA"] = ["ProductDay", "ScenicArea", "每日行程覆盖景区片区"]
|
||
schema["relation_types"]["PRODUCT_COVERS_SCENIC_AREA"] = ["TourProduct", "ScenicArea", "产品覆盖景区片区"]
|
||
schema["relation_types"]["DAY_COVERS_REGION"] = ["ProductDay", "AdministrativeRegion", "每日行程覆盖行政区"]
|
||
schema["relation_types"]["SEGMENT_FROM_REGION"] = ["RouteSegment", "AdministrativeRegion", "移动段起点行政区"]
|
||
schema["relation_types"]["SEGMENT_TO_REGION"] = ["RouteSegment", "AdministrativeRegion", "移动段终点行政区"]
|
||
schema["relation_types"]["SLOT_CAN_USE_LOCATION_RESOURCE"] = ["ResourceSlot", "HotelResource|RestaurantResource", "住宿/餐饮槽位可用合作位置资源"]
|
||
schema["relation_types"]["SLOT_CAN_USE_SERVICE_RESOURCE"] = ["ResourceSlot", "VehicleResource", "车辆等服务槽位可用服务资源"]
|
||
return schema
|
||
|
||
|
||
def schema_to_dsl(schema: dict[str, Any]) -> str:
|
||
list_fields = {
|
||
"aliases", "source_files", "selling_points", "features", "applicable_products", "signature_dishes",
|
||
"meal_scene", "image_urls", "image_ids", "image_urls", "media_resource_ids", "main_destinations",
|
||
"image_urls", "image_ids", "image_urls", "service_scope", "amap_photo_urls",
|
||
"suitable_group_size",
|
||
}
|
||
lines = ["```text", f"namespace {schema['namespace']}", f"version {schema['version']}", ""]
|
||
for name, spec in schema["entity_types"].items():
|
||
lines.append(f"{name}({spec['cn']}): EntityType")
|
||
lines.append(" properties:")
|
||
for field in spec["fields"]:
|
||
typ = "TextList" if field in list_fields else "Text"
|
||
if any(token in field for token in ("count", "days", "nights", "price", "value", "order", "index", "min", "max", "score", "lng", "lat", "distance")):
|
||
typ = "Number"
|
||
if field in {"route_immutable", "required", "changeable", "customer_visible", "default_option", "confirmed", "confirmation_required", "refundable", "is_core_visit"}:
|
||
typ = "Boolean"
|
||
lines.append(f" {field}: {typ}")
|
||
lines.append("")
|
||
for rel, (start, end, desc) in schema["relation_types"].items():
|
||
lines.append(f"{rel}({desc}): RelationType")
|
||
lines.append(f" startNode: {start}")
|
||
lines.append(f" endNode: {end}")
|
||
lines.append("")
|
||
lines.append("```")
|
||
return "\n".join(lines)
|
||
|
||
|
||
class MediaIndex:
|
||
def __init__(self) -> None:
|
||
self.by_alias: dict[str, list[str]] = defaultdict(list)
|
||
self.by_id: dict[str, str] = {}
|
||
self.label_by_key: dict[str, str] = {}
|
||
self.accepted_vehicle_names: set[str] = set()
|
||
|
||
def add_alias(self, alias: str, media_key: str) -> None:
|
||
n = norm(alias)
|
||
if len(n) >= 2 and media_key not in self.by_alias[n]:
|
||
self.by_alias[n].append(media_key)
|
||
|
||
def find_for_name(self, name: str, aliases: list[str] | None = None, allowed_labels: set[str] | None = None) -> list[str]:
|
||
candidates: list[str] = []
|
||
names = [name] + (aliases or [])
|
||
normalized = [norm(x) for x in names if len(norm(x)) >= 2]
|
||
for n in normalized:
|
||
candidates.extend(self.by_alias.get(n, []))
|
||
if not candidates:
|
||
for alias_norm, media_keys in self.by_alias.items():
|
||
if any((n and len(n) >= 3 and (n in alias_norm or alias_norm in n)) for n in normalized):
|
||
candidates.extend(media_keys)
|
||
out: list[str] = []
|
||
for key in candidates:
|
||
if allowed_labels and self.label_by_key.get(key) not in allowed_labels:
|
||
continue
|
||
if key not in out:
|
||
out.append(key)
|
||
return out[:3]
|
||
|
||
|
||
def apply_media(
|
||
builder: KGBuilder,
|
||
media_index: MediaIndex,
|
||
node_key: str,
|
||
name: str,
|
||
aliases: list[str] | None = None,
|
||
allowed_labels: set[str] | None = None,
|
||
) -> None:
|
||
media_keys = media_index.find_for_name(name, aliases, allowed_labels)
|
||
if not media_keys:
|
||
return
|
||
node = builder.nodes[node_key]
|
||
image_urls: list[str] = []
|
||
image_ids: list[str] = []
|
||
reliabilities: list[str] = []
|
||
usage_notes: list[str] = []
|
||
for media_key in media_keys:
|
||
media = builder.nodes.get(media_key, {})
|
||
image_urls.extend(media.get("image_urls") or [])
|
||
image_ids.extend(media.get("image_ids") or [])
|
||
if media.get("reliability_level"):
|
||
reliabilities.append(media["reliability_level"])
|
||
if media.get("usage_note"):
|
||
usage_notes.append(media["usage_note"])
|
||
builder.add_rel("RESOURCE_HAS_MEDIA", node_key, media_key, match_rule="名称/别名匹配图片资源库")
|
||
if image_urls:
|
||
node["primary_image_url"] = image_urls[0]
|
||
node["image_urls"] = sorted(set(image_urls))
|
||
node["media_resource_ids"] = [builder.nodes[m].get("media_id") for m in media_keys if builder.nodes.get(m)]
|
||
node["media_reliability"] = ";".join(sorted(set(reliabilities)))
|
||
node["media_usage_note"] = ";".join(sorted(set(usage_notes)))[:500]
|
||
|
||
|
||
def load_amap_enrichment_cache() -> dict[str, dict[str, Any]]:
|
||
if not AMAP_CACHE_PATH.exists():
|
||
return {}
|
||
try:
|
||
payload = json.loads(AMAP_CACHE_PATH.read_text(encoding="utf-8"))
|
||
except Exception:
|
||
return {}
|
||
if isinstance(payload, dict) and isinstance(payload.get("items"), dict):
|
||
return payload["items"]
|
||
if isinstance(payload, dict):
|
||
return payload
|
||
return {}
|
||
|
||
|
||
def load_amap_driving_metric_cache() -> dict[str, dict[str, Any]]:
|
||
if not AMAP_DRIVING_CACHE_PATH.exists():
|
||
return {}
|
||
try:
|
||
payload = json.loads(AMAP_DRIVING_CACHE_PATH.read_text(encoding="utf-8"))
|
||
except Exception:
|
||
return {}
|
||
if isinstance(payload, dict) and isinstance(payload.get("items"), dict):
|
||
return payload["items"]
|
||
if isinstance(payload, dict):
|
||
return payload
|
||
return {}
|
||
|
||
|
||
def apply_amap_enrichment_layer(builder: KGBuilder, cache: dict[str, dict[str, Any]]) -> None:
|
||
if not cache:
|
||
return
|
||
target_labels = {"ScenicAttraction", "HotelResource", "RestaurantResource"}
|
||
for key, node in builder.nodes.items():
|
||
if node.get("label") not in target_labels:
|
||
continue
|
||
item = cache.get(key) or cache.get(node.get("natural_key", ""))
|
||
if not isinstance(item, dict):
|
||
continue
|
||
fields = item.get("fields") if isinstance(item.get("fields"), dict) else item
|
||
status = compact(fields.get("amap_match_status") or item.get("status"))
|
||
if status:
|
||
node["amap_match_status"] = status
|
||
for field in [
|
||
"amap_poi_id", "amap_name", "amap_type", "amap_typecode", "amap_address",
|
||
"amap_location", "amap_lng", "amap_lat", "amap_pname", "amap_cityname",
|
||
"amap_adname", "amap_adcode", "amap_tel", "amap_rating", "amap_cost", "amap_open_time",
|
||
"amap_photo_urls", "amap_match_confidence", "amap_match_reason",
|
||
"data_completeness_note",
|
||
]:
|
||
value = fields.get(field)
|
||
if value not in (None, "", [], {}):
|
||
node[field] = value
|
||
|
||
|
||
def upsert_relation_props(builder: KGBuilder, rel_type: str, source: str, target: str, **props: Any) -> None:
|
||
payload = {k: v for k, v in props.items() if v not in (None, "", [], {})}
|
||
if not payload:
|
||
return
|
||
for rel in builder.relations:
|
||
if rel["relation_type"] == rel_type and rel["source"] == source and rel["target"] == target:
|
||
rel.setdefault("properties", {}).update(payload)
|
||
return
|
||
builder.add_rel(rel_type, source, target, **payload)
|
||
|
||
|
||
def apply_driving_metric_layer(builder: KGBuilder, cache: dict[str, dict[str, Any]]) -> None:
|
||
if not cache:
|
||
return
|
||
for item in cache.values():
|
||
if not isinstance(item, dict):
|
||
continue
|
||
if item.get("status") != "matched":
|
||
continue
|
||
source_key = compact(item.get("source_key"))
|
||
target_key = compact(item.get("target_key"))
|
||
if source_key not in builder.nodes or target_key not in builder.nodes:
|
||
continue
|
||
source = builder.nodes[source_key]
|
||
target = builder.nodes[target_key]
|
||
source_label = source.get("label")
|
||
if source_label not in {"ScenicArea", "ScenicAttraction"}:
|
||
continue
|
||
if source_label == "ScenicAttraction" and source.get("is_independent_destination") is False:
|
||
continue
|
||
target_label = target.get("label")
|
||
if target_label not in {"ScenicArea", "ScenicAttraction", "HotelResource", "RestaurantResource"}:
|
||
continue
|
||
if target_label == "ScenicAttraction" and target.get("is_independent_destination") is False:
|
||
continue
|
||
distance_km = number(item.get("drive_distance_km"))
|
||
duration_min = number(item.get("drive_duration_min"))
|
||
metric_props = {
|
||
"metric_scope": item.get("metric_scope"),
|
||
"resource_type": item.get("target_resource_type"),
|
||
"drive_distance_km": distance_km,
|
||
"drive_duration_min": duration_min,
|
||
"amap_distance_m": item.get("amap_distance_m"),
|
||
"amap_duration_s": item.get("amap_duration_s"),
|
||
"provider": item.get("provider") or "amap",
|
||
"api": item.get("api") or "amap_distance",
|
||
"route_type": item.get("route_type") or "driving",
|
||
"region_match_level": item.get("region_match_level"),
|
||
"same_admin_region": item.get("same_admin_region"),
|
||
"source_region": item.get("source_region"),
|
||
"target_region": item.get("target_region"),
|
||
"origin_location": item.get("origin_location"),
|
||
"destination_location": item.get("destination_location"),
|
||
"updated_at": item.get("updated_at"),
|
||
"rule": "高德驾车距离/耗时;用于候选资源排序与行程可行性判断,价格和房态仍需二次确认",
|
||
}
|
||
builder.add_rel("DRIVING_ROUTE_METRIC", source_key, target_key, **metric_props)
|
||
if target_label in {"HotelResource", "RestaurantResource"}:
|
||
upsert_relation_props(
|
||
builder,
|
||
"NEARBY_LOCATION_RESOURCE",
|
||
source_key,
|
||
target_key,
|
||
resource_type="hotel" if target_label == "HotelResource" else "restaurant",
|
||
distance_km=item.get("straight_distance_km"),
|
||
drive_distance_km=distance_km,
|
||
drive_duration_min=duration_min,
|
||
metric_scope=item.get("metric_scope"),
|
||
region_match_level=item.get("region_match_level"),
|
||
candidate_basis="same_region_with_amap_driving_metric",
|
||
rule="同区县/同城可服务资源,并已补充高德驾车距离与耗时;推荐时优先按车程、房态、餐标过滤",
|
||
)
|
||
|
||
|
||
def load_media_resources(builder: KGBuilder) -> MediaIndex:
|
||
media_index = MediaIndex()
|
||
path = SOURCE_ROOT / "图片资源库_全品类别名索引.xlsx"
|
||
alias_df = pd.read_excel(path, sheet_name="别名索引")
|
||
group_df = pd.read_excel(path, sheet_name="资源分组")
|
||
alias_map: dict[str, list[str]] = defaultdict(list)
|
||
no_match_notes: dict[str, str] = {}
|
||
for _, row in alias_df.iterrows():
|
||
resource_id = compact(row.get("资源ID"))
|
||
alias = compact(row.get("别名"))
|
||
if resource_id and alias:
|
||
alias_map[resource_id].append(alias)
|
||
if compact(row.get("禁止误匹配")):
|
||
no_match_notes[resource_id] = compact(row.get("禁止误匹配"))
|
||
for _, row in group_df.iterrows():
|
||
resource_id = compact(row.get("资源ID"))
|
||
label = compact(row.get("标签"))
|
||
title = compact(row.get("主标题"))
|
||
reliability = compact(row.get("匹配级别"))
|
||
urls = urls_from_text(row.get("图片链接列表"))
|
||
if not resource_id or not title or not urls:
|
||
continue
|
||
if label == "待确认图片" or reliability in {"低", "无图/禁止冒充"}:
|
||
continue
|
||
aliases = split_items(row.get("别名列表"), limit=80) + alias_map.get(resource_id, [])
|
||
label_for_match = label
|
||
if label == "景点" and any(token in title for token in ["酒店", "客栈", "民宿", "宾馆", "维也纳", "住宿"]):
|
||
label_for_match = "酒店"
|
||
key = builder.add_node(
|
||
"MediaResource",
|
||
f"media:{resource_id}",
|
||
title,
|
||
media_id=resource_id,
|
||
title=title,
|
||
media_type="image_group",
|
||
resource_label=label_for_match,
|
||
original_resource_label=label,
|
||
reliability_level=reliability,
|
||
image_ids=split_items(row.get("图片编号列表"), limit=80),
|
||
image_urls=urls,
|
||
primary_image_url=urls[0],
|
||
aliases=sorted(set(a for a in aliases if a)),
|
||
usage_note=compact(row.get("备注")) or compact(row.get("说明")) or ("禁止误匹配:" + no_match_notes.get(resource_id, "") if no_match_notes.get(resource_id) else ""),
|
||
source_file=str(path),
|
||
)
|
||
media_index.by_id[resource_id] = key
|
||
media_index.label_by_key[key] = label_for_match
|
||
media_index.add_alias(title, key)
|
||
for alias in aliases:
|
||
media_index.add_alias(alias, key)
|
||
if label_for_match == "车辆" and reliability in {"高", "中"} and "待确认" not in title and "缺少可靠" not in title:
|
||
media_index.accepted_vehicle_names.add(title)
|
||
return media_index
|
||
|
||
|
||
def seed_attractions(builder: KGBuilder, media_index: MediaIndex) -> dict[str, str]:
|
||
alias_to_key: dict[str, str] = {}
|
||
scenic_area_keys: dict[str, str] = {}
|
||
for area_name, spec in SCENIC_AREA_DEFINITIONS.items():
|
||
key = builder.add_node(
|
||
"ScenicArea",
|
||
f"scenic_area:{slug(area_name)}",
|
||
area_name,
|
||
scenic_area_id=f"SA-{digest(area_name, length=8)}",
|
||
aliases=spec.get("aliases") or [],
|
||
city=spec.get("city"),
|
||
area_type=spec.get("area_type"),
|
||
route_anchor=bool(spec.get("route_anchor")),
|
||
resource_spatial_class="location_resource",
|
||
management_note=spec.get("note"),
|
||
source_files=["种子景区层级+产品路线+高德POI"],
|
||
)
|
||
scenic_area_keys[area_name] = key
|
||
for name, aliases, city, attraction_type, point in ATTRACTION_SEEDS:
|
||
area_key = builder.add_node("Area", f"area:{slug(city)}", city, area_id=f"AREA-{digest(city, length=8)}", area_type="目的地区域")
|
||
parent_area_name, member_role = SCENIC_MEMBER_PARENT.get(name, ("", ""))
|
||
is_sub_spot = bool(parent_area_name) and SCENIC_MEMBER_IS_CHILD.get(name, False)
|
||
key = builder.add_node(
|
||
"ScenicAttraction",
|
||
f"attraction:{slug(name)}",
|
||
name,
|
||
attraction_id=f"ATTR-{digest(name, length=8)}",
|
||
aliases=aliases,
|
||
city=city,
|
||
attraction_type=attraction_type,
|
||
attraction_level="scenic_spot" if is_sub_spot else ("representative_attraction" if parent_area_name else "standalone_attraction"),
|
||
parent_scenic_area_name=parent_area_name,
|
||
is_independent_destination=not is_sub_spot,
|
||
route_role=member_role or "独立线路目的地",
|
||
walk_intensity_hint="子景点步行强度需结合游客体力与景区交通确认" if is_sub_spot else "按产品行程安排确认",
|
||
recommendation_note=(
|
||
f"{name}是{parent_area_name}下的具体游览点,不作为独立主目的地;用于计算车程、游览顺序和客户偏好。"
|
||
if is_sub_spot else "可作为产品线路中的主目的地或独立景点资源。"
|
||
),
|
||
resource_spatial_class="location_resource",
|
||
selling_points=[point],
|
||
source_files=["种子景点+产品路线+图片资源库"],
|
||
)
|
||
builder.add_rel("LOCATED_IN", key, area_key)
|
||
if parent_area_name and parent_area_name in scenic_area_keys:
|
||
scenic_area_key = scenic_area_keys[parent_area_name]
|
||
builder.add_rel(
|
||
"SCENIC_AREA_HAS_ATTRACTION",
|
||
scenic_area_key,
|
||
key,
|
||
member_role=member_role,
|
||
independent_destination=not is_sub_spot,
|
||
relation_note=(
|
||
"景区片区包含具体子游览点,避免子景点在推荐中被误认为平级主目的地"
|
||
if is_sub_spot
|
||
else "景区片区锚点连接代表游览点,用于行政区资源池、路线覆盖和车程候选"
|
||
),
|
||
)
|
||
builder.add_rel(
|
||
"ATTRACTION_PART_OF_SCENIC_AREA",
|
||
key,
|
||
scenic_area_key,
|
||
member_role=member_role,
|
||
independent_destination=not is_sub_spot,
|
||
)
|
||
apply_media(builder, media_index, key, name, aliases, {"景点"})
|
||
for alias in aliases:
|
||
alias_to_key[alias] = key
|
||
return alias_to_key
|
||
|
||
|
||
def seed_vehicles_from_media(builder: KGBuilder, media_index: MediaIndex) -> tuple[dict[str, str], str]:
|
||
out: dict[str, str] = {}
|
||
group_key = builder.add_node(
|
||
"ResourceOptionGroup",
|
||
"option_group:vehicle:reliable_reference",
|
||
"可靠车辆资源组",
|
||
option_group_id="OG-VEHICLE-RELIABLE",
|
||
resource_type="vehicle",
|
||
city_or_area="贵州/全省",
|
||
grade_or_level="图片资源库高/中可靠车辆",
|
||
option_policy="车辆名称仅采用图片资源库中可靠车辆分组;实际派车按人数、行李、档期二次确认。",
|
||
default_option=False,
|
||
source_file=str(SOURCE_ROOT / "图片资源库_全品类别名索引.xlsx"),
|
||
)
|
||
for media_key, media in list(builder.nodes.items()):
|
||
if media.get("label") != "MediaResource" or media.get("resource_label") != "车辆":
|
||
continue
|
||
title = compact(media.get("title"))
|
||
if title not in media_index.accepted_vehicle_names:
|
||
continue
|
||
aliases = media.get("aliases") or []
|
||
seat_count = None
|
||
joined = f"{title} {' '.join(aliases)}"
|
||
m = re.search(r"(\d+)\s*座", joined)
|
||
if m:
|
||
seat_count = int(m.group(1))
|
||
elif "七座" in joined:
|
||
seat_count = 7
|
||
key = builder.add_node(
|
||
"VehicleResource",
|
||
f"vehicle:{slug(title)}",
|
||
title,
|
||
vehicle_id=f"VEH-{digest(title, length=8)}",
|
||
vehicle_type=title,
|
||
seat_count=seat_count,
|
||
seat_layout="参考图",
|
||
comfort_level="按实际派车确认",
|
||
capacity_min=None,
|
||
capacity_max=seat_count,
|
||
service_scope=["产品用车参考图", "接送/小团车型参考"],
|
||
aliases=aliases,
|
||
resource_service_class="service_resource",
|
||
service_region_mode="贵州全省调度/按产品确认",
|
||
suitable_group_size=f"建议不超过{seat_count}人" if seat_count else "按车型与行李二次确认",
|
||
dispatch_basis="按人数、行李、预算、车型等级和司机档期推荐;不按景点附近推荐。",
|
||
primary_image_url=media.get("primary_image_url"),
|
||
image_urls=media.get("image_urls"),
|
||
media_resource_ids=[media.get("media_id")],
|
||
media_reliability=media.get("reliability_level"),
|
||
media_usage_note=media.get("usage_note"),
|
||
)
|
||
builder.add_rel("RESOURCE_HAS_MEDIA", key, media_key, match_rule="车辆资源来自图片资源库资源分组")
|
||
builder.add_rel("GROUP_CONTAINS_VEHICLE", group_key, key)
|
||
out[title] = key
|
||
return out, group_key
|
||
|
||
|
||
def parse_resource_workbooks(builder: KGBuilder, media_index: MediaIndex) -> tuple[dict[str, str], dict[str, str]]:
|
||
hotel_groups: dict[str, str] = {}
|
||
restaurant_groups: dict[str, str] = {}
|
||
hotel_path = SOURCE_ROOT / "住宿资源库(四钻及以上).xlsx"
|
||
df = pd.read_excel(hotel_path, header=None)
|
||
region = ""
|
||
for _, row in df.iterrows():
|
||
values = [compact(x) for x in row.tolist()]
|
||
if values[0] and "区域" in values[0] and not values[1]:
|
||
region = clean_region_label(values[0])
|
||
group_key = builder.add_node(
|
||
"ResourceOptionGroup",
|
||
f"option_group:hotel:{slug(region)}",
|
||
f"{region}酒店资源组",
|
||
option_group_id=f"OG-HOTEL-{digest(region, length=8)}",
|
||
resource_type="hotel",
|
||
city_or_area=region,
|
||
grade_or_level="四钻及以上",
|
||
option_policy="可作为产品住宿槽位的同级/升级参考,具体房态和差价需二次确认。",
|
||
default_option=False,
|
||
source_file=str(hotel_path),
|
||
)
|
||
apply_media(builder, media_index, group_key, f"{region}参考酒店", [region, f"{region}4钻参考酒店"], {"酒店"})
|
||
hotel_groups[region] = group_key
|
||
continue
|
||
if values[0] in {"酒店名称", ""} or not values[0]:
|
||
continue
|
||
name = values[0]
|
||
key = builder.add_node(
|
||
"HotelResource",
|
||
f"hotel:{slug(name)}",
|
||
name,
|
||
hotel_id=f"HOTEL-{digest(name, length=8)}",
|
||
hotel_grade=values[1],
|
||
city_or_area=region,
|
||
address=values[2],
|
||
resource_spatial_class="location_resource",
|
||
contact_name=values[3],
|
||
features=split_items(values[4]),
|
||
listed_price_text=values[5],
|
||
off_season_price_text=values[6],
|
||
peak_season_price_text=values[7],
|
||
applicable_products=split_items(values[8]),
|
||
source_file=str(hotel_path),
|
||
)
|
||
apply_media(builder, media_index, key, name, [region, f"{region}4钻参考酒店"], {"酒店"})
|
||
if region:
|
||
area_key = builder.add_node("Area", f"area:{slug(region)}", region, area_id=f"AREA-{digest(region, length=8)}", area_type="酒店区域")
|
||
builder.add_rel("LOCATED_IN", key, area_key)
|
||
group_key = hotel_groups.get(region)
|
||
if group_key:
|
||
builder.add_rel("GROUP_CONTAINS_HOTEL", group_key, key)
|
||
|
||
rest_path = SOURCE_ROOT / "餐厅资源库.xlsx"
|
||
df = pd.read_excel(rest_path, header=None)
|
||
region = ""
|
||
for _, row in df.iterrows():
|
||
values = [compact(x) for x in row.tolist()]
|
||
section_region = RESTAURANT_SECTION_REGION_ALIASES.get(values[0]) if values[0] and not any(values[1:]) else ""
|
||
if values[0] and ("区域" in values[0] and not values[1] or section_region):
|
||
region = clean_region_label(section_region or values[0])
|
||
group_key = builder.add_node(
|
||
"ResourceOptionGroup",
|
||
f"option_group:restaurant:{slug(region)}",
|
||
f"{region}餐饮资源组",
|
||
option_group_id=f"OG-REST-{digest(region, length=8)}",
|
||
resource_type="restaurant",
|
||
city_or_area=region,
|
||
grade_or_level="按人均/特色菜选择",
|
||
option_policy="可作为产品餐饮槽位的团队餐或特色餐参考,需按人数、餐标、桌数确认。",
|
||
default_option=False,
|
||
source_file=str(rest_path),
|
||
)
|
||
restaurant_groups[region] = group_key
|
||
continue
|
||
if values[0] in {"餐厅名称", ""} or not values[0]:
|
||
continue
|
||
name = values[0]
|
||
key = builder.add_node(
|
||
"RestaurantResource",
|
||
f"restaurant:{slug(name)}",
|
||
name,
|
||
restaurant_id=f"REST-{digest(name, length=8)}",
|
||
city_or_area=region,
|
||
address=values[1] or values[5],
|
||
resource_spatial_class="location_resource",
|
||
per_capita_price_text=values[2],
|
||
signature_dishes=split_items(values[3]),
|
||
contact_name=values[4],
|
||
meal_scene=split_items(values[6]),
|
||
source_file=str(rest_path),
|
||
)
|
||
apply_media(builder, media_index, key, name, [region], {"餐饮"})
|
||
if region:
|
||
area_key = builder.add_node("Area", f"area:{slug(region)}", region, area_id=f"AREA-{digest(region, length=8)}", area_type="餐饮区域")
|
||
builder.add_rel("LOCATED_IN", key, area_key)
|
||
group_key = restaurant_groups.get(region)
|
||
if group_key:
|
||
builder.add_rel("GROUP_CONTAINS_RESTAURANT", group_key, key)
|
||
return hotel_groups, restaurant_groups
|
||
|
||
|
||
def group_member_keys(builder: KGBuilder, group_key: str, rel_type: str) -> list[str]:
|
||
return [rel["target"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["source"] == group_key]
|
||
|
||
|
||
def rel_targets(builder: KGBuilder, source_key: str, rel_type: str) -> list[str]:
|
||
return [rel["target"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["source"] == source_key]
|
||
|
||
|
||
def rel_sources(builder: KGBuilder, rel_type: str, target_key: str) -> list[str]:
|
||
return [rel["source"] for rel in builder.relations if rel["relation_type"] == rel_type and rel["target"] == target_key]
|
||
|
||
|
||
def link_resource_groups_to_attractions(
|
||
builder: KGBuilder,
|
||
alias_to_key: dict[str, str],
|
||
hotel_groups: dict[str, str],
|
||
restaurant_groups: dict[str, str],
|
||
) -> None:
|
||
for region, group_key in hotel_groups.items():
|
||
attraction_names = RESOURCE_GROUP_ATTRACTION_MAP.get(clean_region_label(region), [])
|
||
for attraction_name in attraction_names:
|
||
attraction_key = find_attraction(attraction_name, alias_to_key)
|
||
if not attraction_key:
|
||
continue
|
||
builder.add_rel("GROUP_SERVES_ATTRACTION", group_key, attraction_key, basis="酒店区域与景区服务半径匹配")
|
||
for hotel_key in group_member_keys(builder, group_key, "GROUP_CONTAINS_HOTEL"):
|
||
builder.add_rel("HOTEL_SERVES_ATTRACTION", hotel_key, attraction_key, basis="继承酒店资源组服务景区")
|
||
for region, group_key in restaurant_groups.items():
|
||
attraction_names = RESOURCE_GROUP_ATTRACTION_MAP.get(clean_region_label(region), [])
|
||
for attraction_name in attraction_names:
|
||
attraction_key = find_attraction(attraction_name, alias_to_key)
|
||
if not attraction_key:
|
||
continue
|
||
builder.add_rel("GROUP_SERVES_ATTRACTION", group_key, attraction_key, basis="餐饮区域与景区服务半径匹配")
|
||
for restaurant_key in group_member_keys(builder, group_key, "GROUP_CONTAINS_RESTAURANT"):
|
||
builder.add_rel("RESTAURANT_SERVES_ATTRACTION", restaurant_key, attraction_key, basis="继承餐饮资源组服务景区")
|
||
|
||
|
||
def region_node_key(name: str, level: str, parent_key: str = "") -> str:
|
||
return f"admin_region:{level}:{slug(parent_key + ':' + name if parent_key else name)}"
|
||
|
||
|
||
def add_admin_region(
|
||
builder: KGBuilder,
|
||
name: str,
|
||
level: str,
|
||
province: str,
|
||
city: str = "",
|
||
county: str = "",
|
||
town: str = "",
|
||
parent_key: str = "",
|
||
adcode: str = "",
|
||
source_system: str = "",
|
||
scope_note: str = "",
|
||
) -> str:
|
||
key = region_node_key(name, level, parent_key)
|
||
builder.add_node(
|
||
"AdministrativeRegion",
|
||
key,
|
||
name,
|
||
region_id=f"REG-{digest(level, name, parent_key, length=10)}",
|
||
level=level,
|
||
province=province,
|
||
city=city,
|
||
county=county,
|
||
town=town,
|
||
parent_id=parent_key,
|
||
adcode=adcode,
|
||
source_system=source_system,
|
||
scope_note=scope_note or "仅为当前旅行社路线涉及资源建立行政区节点",
|
||
)
|
||
if parent_key:
|
||
builder.add_rel("REGION_PARENT_OF", parent_key, key)
|
||
return key
|
||
|
||
|
||
def add_region_hierarchy(
|
||
builder: KGBuilder,
|
||
province: str,
|
||
city: str = "",
|
||
county: str = "",
|
||
town: str = "",
|
||
adcode: str = "",
|
||
source_system: str = "",
|
||
) -> str:
|
||
province = province or "贵州省"
|
||
province_key = add_admin_region(builder, province, "province", province, source_system=source_system)
|
||
parent = province_key
|
||
most_specific = province_key
|
||
if city:
|
||
city_key = add_admin_region(
|
||
builder, city, "city", province, city=city, parent_key=parent,
|
||
source_system=source_system,
|
||
)
|
||
parent = city_key
|
||
most_specific = city_key
|
||
if county and county != city:
|
||
county_key = add_admin_region(
|
||
builder, county, "county", province, city=city, county=county, parent_key=parent,
|
||
adcode=adcode, source_system=source_system,
|
||
)
|
||
parent = county_key
|
||
most_specific = county_key
|
||
if town:
|
||
most_specific = add_admin_region(
|
||
builder, town, "town", province, city=city, county=county, town=town, parent_key=parent,
|
||
source_system=source_system,
|
||
)
|
||
return most_specific
|
||
|
||
|
||
def region_hint_for_text(text: str) -> tuple[str, str, str] | None:
|
||
n = compact(text)
|
||
for keyword, region in REGION_TEXT_HINTS.items():
|
||
if keyword in n:
|
||
return region
|
||
return None
|
||
|
||
|
||
def region_for_location_node(node: dict[str, Any]) -> tuple[str, str, str, str, str]:
|
||
province = compact(node.get("amap_pname")) or "贵州省"
|
||
city = compact(node.get("amap_cityname"))
|
||
county = compact(node.get("amap_adname"))
|
||
adcode = compact(node.get("amap_adcode"))
|
||
if city or county:
|
||
return province, city, county, adcode, "amap"
|
||
if node.get("label") == "ScenicArea":
|
||
area_spec = SCENIC_AREA_DEFINITIONS.get(compact(node.get("name"))) or {}
|
||
hint = area_spec.get("admin_hint")
|
||
if hint:
|
||
return hint[0], hint[1], hint[2], "", "scenic_area_admin_hint"
|
||
if node.get("label") == "ScenicAttraction":
|
||
hint = ATTRACTION_ADMIN_HINTS.get(compact(node.get("name")))
|
||
if hint:
|
||
return hint[0], hint[1], hint[2], "", "seed_admin_hint"
|
||
hint = region_hint_for_text(" ".join([
|
||
compact(node.get("name")),
|
||
compact(node.get("city_or_area")),
|
||
compact(node.get("address")),
|
||
compact(node.get("city")),
|
||
]))
|
||
if hint:
|
||
return hint[0], hint[1], hint[2], "", "business_region_hint"
|
||
return "贵州省", "", "", "", "province_fallback"
|
||
|
||
|
||
def add_geopoint_for_node(builder: KGBuilder, node_key: str, node: dict[str, Any]) -> None:
|
||
lng = number(node.get("amap_lng") or node.get("location_lng"))
|
||
lat = number(node.get("amap_lat") or node.get("location_lat"))
|
||
if lng is None or lat is None:
|
||
node["geo_match_status"] = node.get("geo_match_status") or "no_coordinate"
|
||
return
|
||
node["location_lng"] = lng
|
||
node["location_lat"] = lat
|
||
node["geo_match_status"] = node.get("amap_match_status") or "matched"
|
||
gp_key = f"geopoint:{node_key}"
|
||
builder.add_node(
|
||
"GeoPoint",
|
||
gp_key,
|
||
f"{node.get('name')}坐标",
|
||
geo_point_id=f"GEO-{digest(node_key, lng, lat, length=10)}",
|
||
lng=lng,
|
||
lat=lat,
|
||
address_text=node.get("amap_address") or node.get("address"),
|
||
provider="amap" if node.get("amap_poi_id") else "business_hint",
|
||
precision_level="poi" if node.get("amap_poi_id") else "region_hint",
|
||
source_system="amap" if node.get("amap_poi_id") else "business_source",
|
||
)
|
||
builder.add_rel("RESOURCE_HAS_GEOPOINT", node_key, gp_key)
|
||
|
||
|
||
def derive_scenic_area_coordinates(builder: KGBuilder) -> None:
|
||
"""Use the representative member POI as the scenic-area anchor coordinate.
|
||
|
||
Resource recommendation distances should start from scenic areas, not every
|
||
child spot. For 黄果树 this means the resource pool is anchored at 黄果树旅游景区,
|
||
while 天星桥/陡坡塘 remain child visit points.
|
||
"""
|
||
for area_key, area in list(builder.nodes.items()):
|
||
if area.get("label") != "ScenicArea":
|
||
continue
|
||
member_keys = rel_targets(builder, area_key, "SCENIC_AREA_HAS_ATTRACTION")
|
||
if not member_keys:
|
||
continue
|
||
preferred = ""
|
||
for member_key in member_keys:
|
||
member = builder.nodes.get(member_key, {})
|
||
if compact(member.get("name")) in {"黄果树", "黄果树瀑布"}:
|
||
preferred = member_key
|
||
break
|
||
preferred = preferred or member_keys[0]
|
||
member = builder.nodes.get(preferred, {})
|
||
lng = number(member.get("amap_lng") or member.get("location_lng"))
|
||
lat = number(member.get("amap_lat") or member.get("location_lat"))
|
||
if lng is None or lat is None:
|
||
continue
|
||
area["location_lng"] = lng
|
||
area["location_lat"] = lat
|
||
area["geo_match_status"] = "derived_from_representative_attraction"
|
||
area["data_completeness_note"] = f"景区片区坐标派生自代表游览点:{member.get('name')}"
|
||
|
||
|
||
def add_administrative_region_layer(builder: KGBuilder) -> None:
|
||
for node_key, node in list(builder.nodes.items()):
|
||
if node.get("label") not in LOCATION_RESOURCE_LABELS:
|
||
continue
|
||
province, city, county, adcode, source = region_for_location_node(node)
|
||
region_key = add_region_hierarchy(builder, province, city, county, adcode=adcode, source_system=source)
|
||
region_node = builder.nodes.get(region_key, {})
|
||
node["admin_region_name"] = region_node.get("name")
|
||
node["admin_region_level"] = region_node.get("level")
|
||
node["admin_region_source"] = source
|
||
node["resource_spatial_class"] = "location_resource"
|
||
builder.add_rel("LOCATED_IN_REGION", node_key, region_key, region_source=source)
|
||
add_geopoint_for_node(builder, node_key, node)
|
||
|
||
|
||
def same_region_enough(a: dict[str, Any], b: dict[str, Any]) -> bool:
|
||
ar = compact(a.get("admin_region_name"))
|
||
br = compact(b.get("admin_region_name"))
|
||
if ar and br and ar == br:
|
||
return True
|
||
ac = compact(a.get("amap_cityname") or a.get("city"))
|
||
bc = compact(b.get("amap_cityname") or b.get("city_or_area"))
|
||
return bool(ac and bc and ac in bc)
|
||
|
||
|
||
def add_nearby_location_relations(builder: KGBuilder) -> None:
|
||
sources = [
|
||
(k, n)
|
||
for k, n in builder.nodes.items()
|
||
if n.get("label") == "ScenicArea"
|
||
or (n.get("label") == "ScenicAttraction" and n.get("is_independent_destination") is not False)
|
||
]
|
||
hotels = [(k, n) for k, n in builder.nodes.items() if n.get("label") == "HotelResource"]
|
||
restaurants = [(k, n) for k, n in builder.nodes.items() if n.get("label") == "RestaurantResource"]
|
||
for source_key, source in sources:
|
||
a_lng = number(source.get("location_lng"))
|
||
a_lat = number(source.get("location_lat"))
|
||
if a_lng is None or a_lat is None:
|
||
continue
|
||
for target_key, target in hotels + restaurants:
|
||
t_lng = number(target.get("location_lng"))
|
||
t_lat = number(target.get("location_lat"))
|
||
if t_lng is None or t_lat is None:
|
||
continue
|
||
distance = haversine_km(a_lng, a_lat, t_lng, t_lat)
|
||
threshold = 35.0 if target.get("label") == "HotelResource" else 15.0
|
||
if distance <= threshold or (distance <= threshold * 1.6 and same_region_enough(source, target)):
|
||
builder.add_rel(
|
||
"NEARBY_LOCATION_RESOURCE",
|
||
source_key,
|
||
target_key,
|
||
resource_type="hotel" if target.get("label") == "HotelResource" else "restaurant",
|
||
distance_km=round(distance, 2),
|
||
source_scope="scenic_area_anchor" if source.get("label") == "ScenicArea" else "independent_attraction",
|
||
rule=f"以景区片区或独立景点为起点,直线距离不超过{threshold:g}km;子景点不直接连接酒店餐饮,实际推荐仍需结合车程、房态和餐标二次确认",
|
||
)
|
||
|
||
|
||
def route_stop_region_keys(builder: KGBuilder, stop_key: str) -> list[str]:
|
||
region_keys = rel_targets(builder, stop_key, "STOP_LOCATED_IN_REGION")
|
||
if region_keys:
|
||
return region_keys
|
||
attr_keys = rel_targets(builder, stop_key, "STOP_VISITS_ATTRACTION")
|
||
for attr_key in attr_keys:
|
||
region_keys.extend(rel_targets(builder, attr_key, "LOCATED_IN_REGION"))
|
||
if region_keys:
|
||
return list(dict.fromkeys(region_keys))
|
||
stop = builder.nodes.get(stop_key, {})
|
||
hint = region_hint_for_text(" ".join([compact(stop.get("name")), compact(stop.get("city_or_area"))]))
|
||
if not hint:
|
||
return []
|
||
return [add_region_hierarchy(builder, hint[0], hint[1], hint[2], source_system="route_stop_hint")]
|
||
|
||
|
||
def add_route_region_layer(builder: KGBuilder) -> None:
|
||
for stop_key, stop in list(builder.nodes.items()):
|
||
if stop.get("label") != "RouteStop":
|
||
continue
|
||
region_keys = route_stop_region_keys(builder, stop_key)
|
||
if not region_keys:
|
||
continue
|
||
first_region = builder.nodes.get(region_keys[0], {})
|
||
stop["admin_region_name"] = first_region.get("name")
|
||
stop["admin_region_level"] = first_region.get("level")
|
||
for region_key in region_keys:
|
||
builder.add_rel("STOP_LOCATED_IN_REGION", stop_key, region_key)
|
||
|
||
for day_key, day in list(builder.nodes.items()):
|
||
if day.get("label") != "ProductDay":
|
||
continue
|
||
stop_keys = rel_targets(builder, day_key, "DAY_HAS_STOP")
|
||
region_path: list[str] = []
|
||
for stop_key in stop_keys:
|
||
for region_key in rel_targets(builder, stop_key, "STOP_LOCATED_IN_REGION"):
|
||
region = builder.nodes.get(region_key, {})
|
||
region_name = compact(region.get("name"))
|
||
if region_name and region_name not in region_path:
|
||
region_path.append(region_name)
|
||
builder.add_rel("DAY_COVERS_REGION", day_key, region_key)
|
||
if region_path:
|
||
day["route_region_path"] = " -> ".join(region_path)
|
||
day["admin_region_name"] = region_path[-1]
|
||
|
||
for segment_key, segment in list(builder.nodes.items()):
|
||
if segment.get("label") != "RouteSegment":
|
||
continue
|
||
from_stops = rel_targets(builder, segment_key, "SEGMENT_FROM_STOP")
|
||
to_stops = rel_targets(builder, segment_key, "SEGMENT_TO_STOP")
|
||
from_regions = route_stop_region_keys(builder, from_stops[0]) if from_stops else []
|
||
to_regions = route_stop_region_keys(builder, to_stops[0]) if to_stops else []
|
||
if from_regions:
|
||
builder.add_rel("SEGMENT_FROM_REGION", segment_key, from_regions[0])
|
||
segment["origin_region_name"] = builder.nodes.get(from_regions[0], {}).get("name")
|
||
if to_regions:
|
||
builder.add_rel("SEGMENT_TO_REGION", segment_key, to_regions[0])
|
||
segment["destination_region_name"] = builder.nodes.get(to_regions[0], {}).get("name")
|
||
|
||
|
||
def add_route_scenic_area_layer(builder: KGBuilder) -> None:
|
||
for rel in list(builder.relations):
|
||
if rel["relation_type"] != "STOP_VISITS_ATTRACTION":
|
||
continue
|
||
stop_key = rel["source"]
|
||
attr_key = rel["target"]
|
||
area_keys = rel_targets(builder, attr_key, "ATTRACTION_PART_OF_SCENIC_AREA")
|
||
if not area_keys:
|
||
continue
|
||
for area_key in area_keys:
|
||
builder.add_rel(
|
||
"STOP_VISITS_SCENIC_AREA",
|
||
stop_key,
|
||
area_key,
|
||
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
|
||
relation_note="停靠点访问景区片区;具体游览点仍保留用于车程和游览强度判断",
|
||
)
|
||
for day_key in rel_sources(builder, "DAY_HAS_STOP", stop_key):
|
||
builder.add_rel(
|
||
"DAY_COVERS_SCENIC_AREA",
|
||
day_key,
|
||
area_key,
|
||
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
|
||
)
|
||
for product_key in rel_sources(builder, "HAS_DAY", day_key):
|
||
builder.add_rel(
|
||
"PRODUCT_COVERS_SCENIC_AREA",
|
||
product_key,
|
||
area_key,
|
||
via_attraction=builder.nodes.get(attr_key, {}).get("name"),
|
||
)
|
||
|
||
|
||
def add_slot_candidate_layers(builder: KGBuilder) -> None:
|
||
hotel_keys = [key for key, node in builder.nodes.items() if node.get("label") == "HotelResource"]
|
||
restaurant_keys = [key for key, node in builder.nodes.items() if node.get("label") == "RestaurantResource"]
|
||
vehicle_keys = [key for key, node in builder.nodes.items() if node.get("label") == "VehicleResource"]
|
||
for slot_key, slot in list(builder.nodes.items()):
|
||
if slot.get("label") != "ResourceSlot":
|
||
continue
|
||
slot_type = slot.get("slot_type")
|
||
day_keys = rel_sources(builder, "DAY_HAS_SLOT", slot_key)
|
||
if slot_type in {"lodging", "meal"} and day_keys:
|
||
targets = hotel_keys if slot_type == "lodging" else restaurant_keys
|
||
candidate_keys: list[str] = []
|
||
for day_key in day_keys:
|
||
stop_keys = rel_targets(builder, day_key, "DAY_HAS_STOP")
|
||
anchor_keys: list[str] = []
|
||
for stop_key in stop_keys:
|
||
anchor_keys.extend(rel_targets(builder, stop_key, "STOP_VISITS_SCENIC_AREA"))
|
||
for attr_key in rel_targets(builder, stop_key, "STOP_VISITS_ATTRACTION"):
|
||
attr = builder.nodes.get(attr_key, {})
|
||
if attr.get("is_independent_destination") is not False:
|
||
anchor_keys.append(attr_key)
|
||
anchor_keys = list(dict.fromkeys(anchor_keys))
|
||
for anchor_key in anchor_keys:
|
||
anchor = builder.nodes.get(anchor_key, {})
|
||
for rel in builder.relations:
|
||
if rel["relation_type"] == "NEARBY_LOCATION_RESOURCE" and rel["source"] == anchor_key and rel["target"] in targets:
|
||
candidate_keys.append(rel["target"])
|
||
builder.add_rel(
|
||
"SLOT_CAN_USE_LOCATION_RESOURCE",
|
||
slot_key,
|
||
rel["target"],
|
||
candidate_basis="visited_scenic_area_nearby" if anchor.get("label") == "ScenicArea" else "visited_independent_attraction_nearby",
|
||
via_attraction=anchor.get("name"),
|
||
via_anchor_type=anchor.get("label"),
|
||
distance_km=(rel.get("properties") or {}).get("distance_km"),
|
||
drive_distance_km=(rel.get("properties") or {}).get("drive_distance_km"),
|
||
drive_duration_min=(rel.get("properties") or {}).get("drive_duration_min"),
|
||
region_match_level=(rel.get("properties") or {}).get("region_match_level"),
|
||
)
|
||
if candidate_keys:
|
||
continue
|
||
day_region_keys = rel_targets(builder, day_key, "DAY_COVERS_REGION")
|
||
for resource_key in targets:
|
||
if set(rel_targets(builder, resource_key, "LOCATED_IN_REGION")) & set(day_region_keys):
|
||
candidate_keys.append(resource_key)
|
||
builder.add_rel(
|
||
"SLOT_CAN_USE_LOCATION_RESOURCE",
|
||
slot_key,
|
||
resource_key,
|
||
candidate_basis="same_day_region",
|
||
)
|
||
if slot_type == "vehicle":
|
||
product_keys = rel_sources(builder, "PRODUCT_HAS_SLOT", slot_key)
|
||
for vehicle_key in vehicle_keys:
|
||
builder.add_rel(
|
||
"SLOT_CAN_USE_SERVICE_RESOURCE",
|
||
slot_key,
|
||
vehicle_key,
|
||
candidate_basis="vehicle_service_capacity",
|
||
)
|
||
for product_key in product_keys:
|
||
builder.add_rel(
|
||
"VEHICLE_SUITABLE_FOR_PRODUCT",
|
||
vehicle_key,
|
||
product_key,
|
||
candidate_basis="product_vehicle_slot",
|
||
)
|
||
|
||
|
||
def extract_frontmatter(markdown: str) -> tuple[dict[str, Any], str]:
|
||
if markdown.startswith("---"):
|
||
end = markdown.find("---", 3)
|
||
if end > 0:
|
||
block = markdown[3:end].strip()
|
||
try:
|
||
return json.loads(block), markdown[end + 3:]
|
||
except Exception:
|
||
return {}, markdown
|
||
return {}, markdown
|
||
|
||
|
||
def parse_fixed_route_rows(markdown: str) -> list[dict[str, str]]:
|
||
rows: list[dict[str, str]] = []
|
||
for line in markdown.splitlines():
|
||
line = line.strip()
|
||
if not line.startswith("| D") or "---" in line:
|
||
continue
|
||
cells = [c.strip().replace("\\|", "|") for c in line.strip("|").split("|")]
|
||
if len(cells) >= 4 and re.fullmatch(r"D\d+", cells[0]):
|
||
rows.append({"day": cells[0], "day_index": re.search(r"\d+", cells[0]).group(), "route": cells[1], "meals": cells[2], "accommodation": cells[3]})
|
||
return rows
|
||
|
||
|
||
def raw_text_block(markdown: str) -> str:
|
||
m = re.search(r"## 6\. 原文保留.*?```text\n(.*?)\n```", markdown, flags=re.S)
|
||
return clean(m.group(1)) if m else ""
|
||
|
||
|
||
def fallback_route_rows(markdown: str, fm: dict[str, Any]) -> list[dict[str, str]]:
|
||
"""Build conservative day rows when the source has readable prose but no D table."""
|
||
duration = int(fm.get("duration_days") or 0)
|
||
attractions = [compact(x) for x in (fm.get("core_attractions") or []) if compact(x)]
|
||
raw = raw_text_block(markdown)
|
||
if duration != 1 or not raw or not attractions:
|
||
return []
|
||
attraction = attractions[0]
|
||
start = "贵阳" if "贵阳" in raw or "延安西路" in raw or "旅游集散中心" in raw else "出发地"
|
||
end = "贵阳" if "返回贵阳" in raw or "贵阳统一散团" in raw else start
|
||
meal = ""
|
||
meal_match = re.search(r"用餐[::,,]?\s*([^。\n]{0,80})", raw)
|
||
if not meal_match:
|
||
meal_match = re.search(r"中餐[::,,]\s*([^。\n]{0,80})", raw)
|
||
if meal_match:
|
||
meal = "用餐:" + compact(meal_match.group(1))
|
||
return [{"day": "D1", "day_index": "1", "route": f"{start}→{attraction}→{end}", "meals": meal, "accommodation": "/"}]
|
||
|
||
|
||
def route_points(route: str) -> list[str]:
|
||
parts = re.split(r"→|->|>>|>|—|-|~|~", compact(route))
|
||
out: list[str] = []
|
||
for part in parts:
|
||
part = re.sub(r"^(D\d+|第[一二三四五六七八九十]+天)", "", part).strip()
|
||
if part and part not in out:
|
||
out.append(part)
|
||
return out
|
||
|
||
|
||
def find_attraction(name: str, alias_to_key: dict[str, str]) -> str:
|
||
n = norm(name)
|
||
for alias, key in alias_to_key.items():
|
||
a = norm(alias)
|
||
if a and (a in n or n in a):
|
||
return key
|
||
return ""
|
||
|
||
|
||
def ensure_series(builder: KGBuilder, name: str, family: str, attractions: list[str]) -> str:
|
||
series = family or "未分组"
|
||
if "游黔程" in name:
|
||
series = "游黔程"
|
||
elif "游黔途" in name or "1+1" in name:
|
||
series = "1+1游黔途"
|
||
elif "轻奢" in family or "头等舱" in name:
|
||
series = "轻奢纯玩"
|
||
elif "经典" in family:
|
||
series = "经典纯玩"
|
||
key = builder.add_node(
|
||
"ProductSeries",
|
||
f"series:{slug(series)}",
|
||
series,
|
||
series_id=f"SER-{digest(series, length=8)}",
|
||
series_type=family,
|
||
main_destinations=attractions,
|
||
notes="按产品名称/车型/酒店等级归并的已有路线系列",
|
||
)
|
||
return key
|
||
|
||
|
||
def product_key_for_name(name: str) -> str:
|
||
return f"product:{slug(name)}"
|
||
|
||
|
||
def find_product_key(builder: KGBuilder, product_name: str) -> str:
|
||
target = norm(product_name)
|
||
best = ""
|
||
best_score = 0
|
||
for key, node in builder.nodes.items():
|
||
if node.get("label") != "TourProduct":
|
||
continue
|
||
n = norm(node.get("name"))
|
||
if not n or not target:
|
||
continue
|
||
score = 0
|
||
if n == target:
|
||
score = 100
|
||
elif n in target or target in n:
|
||
score = min(len(n), len(target))
|
||
else:
|
||
common = len(set(re.findall(r"[\u4e00-\u9fff]+", n)) & set(re.findall(r"[\u4e00-\u9fff]+", target)))
|
||
score = common
|
||
if score > best_score:
|
||
best_score = score
|
||
best = key
|
||
return best if best_score >= 4 else ""
|
||
|
||
|
||
HOTEL_REGION_KEYWORDS = {
|
||
"贵阳区域": ["贵阳", "龙洞堡", "双龙", "甲秀楼", "青岩", "花溪", "黔灵山"],
|
||
"黄果树区域": ["安顺", "黄果树", "天星桥", "陡坡塘", "龙宫", "平坝"],
|
||
"西江千户苗寨区域": ["西江", "苗寨", "雷山", "千户苗寨"],
|
||
"镇远古镇区域": ["镇远", "镇远古镇", "镇远古城"],
|
||
"梵净山区域": ["梵净山", "江口", "铜仁", "中南门"],
|
||
"织金/荔波区域": ["织金", "织金洞", "荔波", "小七孔", "大七孔"],
|
||
"毕节区域": ["毕节", "百里杜鹃", "大方", "黔西"],
|
||
"开阳/猴耳天坑区域": ["开阳", "猴耳天坑", "南江大峡谷"],
|
||
"遵义区域": ["遵义", "乌江寨", "茅台", "遵义会议"],
|
||
}
|
||
|
||
|
||
RESTAURANT_REGION_KEYWORDS = HOTEL_REGION_KEYWORDS
|
||
|
||
|
||
RESOURCE_GROUP_ATTRACTION_MAP = {
|
||
"贵阳区域": ["青岩古镇", "甲秀楼", "黔灵山公园", "天河潭"],
|
||
"黄果树区域": ["黄果树", "天星桥", "陡坡塘瀑布", "龙宫", "平坝樱花"],
|
||
"西江千户苗寨区域": ["西江千户苗寨"],
|
||
"镇远古镇区域": ["镇远古城"],
|
||
"梵净山区域": ["梵净山", "中南门古城"],
|
||
"织金/荔波区域": ["荔波小七孔", "织金洞", "中国天眼"],
|
||
"毕节区域": ["百里杜鹃"],
|
||
"遵义区域": ["乌江寨", "茅台镇", "遵义会议会址"],
|
||
}
|
||
|
||
|
||
def option_groups_for_text(groups: dict[str, str], text: str, keyword_map: dict[str, list[str]]) -> list[str]:
|
||
n = norm(text)
|
||
out: list[str] = []
|
||
for region, key in groups.items():
|
||
region_name = clean_region_label(region).replace("区域", "")
|
||
canonical = clean_region_label(region)
|
||
candidates = [region_name, canonical, *split_items(region_name, limit=8), *keyword_map.get(canonical, [])]
|
||
if any(norm(part) and (norm(part) in n or n in norm(part)) for part in candidates):
|
||
out.append(key)
|
||
if out:
|
||
return list(dict.fromkeys(out))
|
||
for region, key in groups.items():
|
||
region_name = clean_region_label(region).replace("区域", "")
|
||
if region_name[:2] and region_name[:2] in text:
|
||
return [key]
|
||
return []
|
||
|
||
|
||
def hotel_groups_for_text(hotel_groups: dict[str, str], text: str) -> list[str]:
|
||
return option_groups_for_text(hotel_groups, text, HOTEL_REGION_KEYWORDS)
|
||
|
||
|
||
def hotel_group_for_text(hotel_groups: dict[str, str], text: str) -> str:
|
||
keys = hotel_groups_for_text(hotel_groups, text)
|
||
return keys[0] if keys else ""
|
||
|
||
|
||
def restaurant_group_for_text(restaurant_groups: dict[str, str], text: str) -> str:
|
||
keys = restaurant_groups_for_text(restaurant_groups, text)
|
||
if keys:
|
||
return keys[0]
|
||
return next(iter(restaurant_groups.values()), "")
|
||
|
||
|
||
def restaurant_groups_for_text(restaurant_groups: dict[str, str], text: str) -> list[str]:
|
||
keys = option_groups_for_text(restaurant_groups, text, RESTAURANT_REGION_KEYWORDS)
|
||
return keys or ([next(iter(restaurant_groups.values()))] if restaurant_groups else [])
|
||
|
||
|
||
def add_resource_slot(
|
||
builder: KGBuilder,
|
||
product_key: str,
|
||
day_key: str | None,
|
||
slot_type: str,
|
||
slot_name: str,
|
||
scope: str,
|
||
day_index: int | None,
|
||
default_text: str,
|
||
default_level: str,
|
||
change_mode: str,
|
||
pricing_mode: str,
|
||
source_file: str,
|
||
option_group_key: str | list[str] = "",
|
||
) -> str:
|
||
slot_key = builder.add_node(
|
||
"ResourceSlot",
|
||
f"slot:{product_key}:{day_index or 0}:{slot_type}:{slug(default_text or slot_name)}",
|
||
slot_name,
|
||
slot_id=f"SLOT-{digest(product_key, day_index, slot_type, default_text, length=10)}",
|
||
slot_name=slot_name,
|
||
slot_type=slot_type,
|
||
scope=scope,
|
||
day_index=day_index,
|
||
default_text=default_text,
|
||
default_level=default_level,
|
||
required=slot_type not in {"transfer", "gift_service"},
|
||
change_mode=change_mode,
|
||
changeable=change_mode != "fixed",
|
||
customer_visible=True,
|
||
pricing_mode=pricing_mode,
|
||
evidence_text=default_text,
|
||
source_file=source_file,
|
||
)
|
||
builder.add_rel("PRODUCT_HAS_SLOT", product_key, slot_key)
|
||
if day_key:
|
||
builder.add_rel("DAY_HAS_SLOT", day_key, slot_key)
|
||
option_group_keys = [option_group_key] if isinstance(option_group_key, str) else list(option_group_key or [])
|
||
option_group_keys = [key for key in dict.fromkeys(option_group_keys) if key]
|
||
if option_group_keys:
|
||
builder.add_rel("SLOT_DEFAULT_GROUP", slot_key, option_group_keys[0])
|
||
for key in option_group_keys:
|
||
builder.add_rel("SLOT_ALLOWED_GROUP", slot_key, key)
|
||
return slot_key
|
||
|
||
|
||
def fee_type_from_context(text: str) -> str:
|
||
if "保险" in text:
|
||
return "景区保险"
|
||
if any(x in text for x in ("观光车", "环保车", "电瓶车", "小交通", "景交")):
|
||
return "景区小交通"
|
||
if any(x in text for x in ("扶梯", "索道", "游船", "漂流")):
|
||
return "自愿项目"
|
||
if any(x in text for x in ("门票", "票")):
|
||
return "门票"
|
||
if any(x in text for x in ("餐标", "正餐", "人均")):
|
||
return "餐标"
|
||
if "单房差" in text:
|
||
return "单房差"
|
||
if "儿童" in text:
|
||
return "儿童价"
|
||
return "费用说明"
|
||
|
||
|
||
FEE_TARGET_TERMS = [
|
||
("黄果树旅游景区", ["黄果树景区", "黄果树大瀑布风景名胜区", "黄果树大瀑布", "黄果树"]),
|
||
("荔波小七孔", ["小七孔", "荔波小七孔", "鸳鸯湖"]),
|
||
("西江千户苗寨", ["西江千户", "西江苗寨", "西江"]),
|
||
("青岩古镇", ["青岩古镇", "青岩"]),
|
||
("梵净山", ["梵净山"]),
|
||
("镇远古城", ["镇远古城", "镇远古镇", "镇远"]),
|
||
("百里杜鹃", ["百里杜鹃"]),
|
||
("织金洞", ["织金洞"]),
|
||
("中国天眼", ["中国天眼", "天眼"]),
|
||
("天星桥", ["天星桥"]),
|
||
("陡坡塘瀑布", ["陡坡塘"]),
|
||
("龙宫", ["龙宫"]),
|
||
("西江千户苗寨", ["苗寨"]),
|
||
]
|
||
|
||
|
||
def find_named_node(builder: KGBuilder, label: str, name: str) -> str:
|
||
target = norm(name)
|
||
for key, node in builder.nodes.items():
|
||
if node.get("label") == label and norm(node.get("name")) == target:
|
||
return key
|
||
return ""
|
||
|
||
|
||
def find_named_or_alias_node(builder: KGBuilder, label: str, names: list[str]) -> str:
|
||
normalized = [norm(name) for name in names if norm(name)]
|
||
for key, node in builder.nodes.items():
|
||
if node.get("label") != label:
|
||
continue
|
||
node_name = norm(node.get("name"))
|
||
aliases = [norm(alias) for alias in (node.get("aliases") or []) if norm(alias)]
|
||
if any(term and (term == node_name or term in aliases or node_name in term or term in node_name) for term in normalized):
|
||
return key
|
||
return ""
|
||
|
||
|
||
def fee_target_from_text(builder: KGBuilder, text: str, fallback_key: str = "") -> tuple[str, str, str]:
|
||
for canonical, aliases in FEE_TARGET_TERMS:
|
||
if any(alias and alias in text for alias in aliases):
|
||
area_key = find_named_or_alias_node(builder, "ScenicArea", [canonical, *aliases])
|
||
if area_key:
|
||
return area_key, "ScenicArea", builder.nodes[area_key]["name"]
|
||
attr_key = find_named_or_alias_node(builder, "ScenicAttraction", [canonical, *aliases])
|
||
if attr_key:
|
||
return attr_key, "ScenicAttraction", builder.nodes[attr_key]["name"]
|
||
if fallback_key and fallback_key in builder.nodes:
|
||
node = builder.nodes[fallback_key]
|
||
return fallback_key, node.get("label", ""), node.get("name", "")
|
||
return "", "", ""
|
||
|
||
|
||
def service_from_fee_part(part: str, previous_service: str = "", joined_with_previous_without_amount: bool = False) -> tuple[str, str]:
|
||
text = compact(part)
|
||
if "保险" in text and previous_service.endswith("+保险"):
|
||
return "景区小交通+保险", previous_service
|
||
vehicle_word = next((word for word in ["环保车", "观光车", "电瓶车", "摆渡车", "景交", "小交通"] if word in text), "")
|
||
if "保险" in text and vehicle_word:
|
||
service = "景区小交通+保险"
|
||
name = "景区交通+保险" if vehicle_word in {"景交", "小交通"} else f"{vehicle_word}+保险"
|
||
return service, name
|
||
if "保险" in text and joined_with_previous_without_amount and previous_service in {"环保车", "观光车", "电瓶车", "摆渡车", "景区交通"}:
|
||
return "景区小交通+保险", f"{previous_service}+保险"
|
||
if "保险" in text:
|
||
return "景区保险", "保险"
|
||
if vehicle_word:
|
||
service = "景区小交通"
|
||
return service, "景区交通" if vehicle_word in {"景交", "小交通"} else vehicle_word
|
||
if "扶梯" in text or (("单程" in text or "双程" in text or "往返" in text) and previous_service.startswith("扶梯")):
|
||
direction = "双程" if any(x in text for x in ["双程", "往返"]) else ("单程" if "单程" in text else "")
|
||
return "自愿项目", f"扶梯{direction}".strip()
|
||
if "索道" in text:
|
||
direction = "往返" if "往返" in text else ("单程" if "单程" in text else "")
|
||
return "自愿项目", f"索道{direction}".strip()
|
||
if any(word in text for word in ["游船", "划船", "船票"]):
|
||
return "自愿项目", "游船"
|
||
if "演出" in text or "红飘带" in text:
|
||
return "自愿项目", "演出"
|
||
if "门票" in text or "大门票" in text or "首道" in text:
|
||
return "门票", "门票"
|
||
if "餐标" in text or "正餐" in text:
|
||
return "餐标", "餐标"
|
||
if previous_service:
|
||
return fee_type_from_context(previous_service), previous_service
|
||
return fee_type_from_context(text), ""
|
||
|
||
|
||
def inclusion_from_fee_line(line: str, part: str, fee_type: str) -> tuple[str, str, bool]:
|
||
text = f"{line} {part}"
|
||
if any(x in text for x in ["免费", "赠送"]):
|
||
return "免费/赠送", "included_free", False
|
||
if any(x in text for x in ["自愿", "自理自愿", "不必须", "可选", "酌情自愿"]) or fee_type == "自愿项目":
|
||
return "不含/自愿自理", "optional_self_pay", True
|
||
if any(x in text for x in ["必须", "需自费", "需自理", "客人需自理", "自费", "不含", "另付", "必消", "必销"]):
|
||
return "不含/自理", "mandatory_self_pay", False
|
||
if "含" in text and "不含" not in text:
|
||
return "已含", "included", False
|
||
return "需核实", "policy_dependent", False
|
||
|
||
|
||
def clean_fee_prefix(part: str, amount_start: int) -> str:
|
||
prefix = part[:amount_start]
|
||
prefix = re.sub(r"^[\s::,,、++()()【】《》\\-—]*", "", prefix)
|
||
prefix = re.sub(r".*(费用不含|必须自理费用|自愿自费项目|自愿消费项目|不含自愿消费|不含|含|自理|自费|需自费|:|:)", "", prefix)
|
||
prefix = re.sub(r"(客人|游客|费用|共计|景区内|换乘|乘坐|进入|需要|需|可)?$", "", prefix).strip()
|
||
return prefix[-18:]
|
||
|
||
|
||
def is_total_amount_match(part: str, amount_start: int) -> bool:
|
||
prefix = part[:amount_start]
|
||
return bool(re.search(r"(共计|合计|总计|小计|共|共自理|请自理|客人请自理|客人共自理)\s*[::]?\s*$", prefix))
|
||
|
||
|
||
def add_structured_fee_node(
|
||
builder: KGBuilder,
|
||
owner_key: str,
|
||
product_name: str,
|
||
source_file: str,
|
||
line: str,
|
||
target_key: str,
|
||
target_label: str,
|
||
target_name: str,
|
||
service_name: str,
|
||
fee_type: str,
|
||
amount: float,
|
||
unit: str,
|
||
inclusion_status: str,
|
||
mandatory_level: str,
|
||
is_optional: bool,
|
||
consumer_group: str,
|
||
extraction_rule: str,
|
||
) -> str:
|
||
owner_label = builder.nodes.get(owner_key, {}).get("label")
|
||
fee_rel_type = "PRICE_PACKAGE_HAS_FEE" if owner_label == "ProductPricePackage" else "SLOT_HAS_FEE"
|
||
item_name = " ".join(x for x in [target_name, service_name or fee_type] if x).strip() or service_name or fee_type
|
||
identity = f"{product_name}|{target_key}|{item_name}|{amount}|{consumer_group}|{line[:80]}"
|
||
label = "TicketFee" if fee_type in {"景区保险", "景区小交通", "景区小交通+保险", "自愿项目", "门票"} else "FeeItem"
|
||
common = dict(
|
||
amount_text=f"{amount:g}元/{unit or '人'}",
|
||
amount_value=amount,
|
||
price_text=f"{amount:g}元/{unit or '人'}",
|
||
adult_price=amount if consumer_group != "child" else None,
|
||
child_price=amount if consumer_group == "child" else None,
|
||
currency="CNY",
|
||
unit=unit or "人",
|
||
inclusion_status=inclusion_status,
|
||
mandatory_level=mandatory_level,
|
||
applies_to=product_name,
|
||
source_product_name=product_name,
|
||
scenic_target_name=target_name,
|
||
scenic_target_type=target_label,
|
||
consumer_group=consumer_group,
|
||
is_free=False,
|
||
is_optional=is_optional,
|
||
price_status="source_explicit",
|
||
child_price_status="同成人价/原文未单列儿童价" if consumer_group == "all" else ("儿童价原文明确" if consumer_group == "child" else ""),
|
||
rule_text=line,
|
||
evidence_excerpt=line,
|
||
source_file=source_file,
|
||
data_quality_status="structured_from_explicit_price",
|
||
extraction_rule=extraction_rule,
|
||
)
|
||
if label == "TicketFee":
|
||
fee_key = builder.add_node(
|
||
"TicketFee",
|
||
f"ticket_fee:{digest(source_file, owner_key, identity)}",
|
||
item_name,
|
||
ticket_fee_id=f"TF-{digest(source_file, owner_key, identity, length=10)}",
|
||
fee_name=item_name,
|
||
fee_type=fee_type,
|
||
**common,
|
||
)
|
||
else:
|
||
fee_key = builder.add_node(
|
||
"FeeItem",
|
||
f"fee:{digest(source_file, owner_key, identity)}",
|
||
item_name,
|
||
fee_item_id=f"FEE-{digest(source_file, owner_key, identity, length=10)}",
|
||
fee_type=fee_type,
|
||
item_name=item_name,
|
||
**common,
|
||
)
|
||
builder.add_rel(fee_rel_type, owner_key, fee_key)
|
||
if target_key:
|
||
if target_label == "ScenicArea":
|
||
builder.add_rel("SCENIC_AREA_HAS_FEE", target_key, fee_key, source_product_name=product_name)
|
||
elif target_label == "ScenicAttraction":
|
||
builder.add_rel("ATTRACTION_HAS_FEE", target_key, fee_key, source_product_name=product_name)
|
||
product_key = find_product_key(builder, product_name)
|
||
if product_key:
|
||
builder.add_rel("PRODUCT_HAS_FEE", product_key, fee_key)
|
||
return fee_key
|
||
|
||
|
||
def add_free_project_fee(
|
||
builder: KGBuilder,
|
||
owner_key: str,
|
||
product_name: str,
|
||
source_file: str,
|
||
line: str,
|
||
target_key: str,
|
||
target_label: str,
|
||
target_name: str,
|
||
project_name: str,
|
||
) -> None:
|
||
if not project_name:
|
||
return
|
||
item_name = " ".join(x for x in [target_name, project_name] if x).strip()
|
||
identity = f"free|{product_name}|{target_key}|{item_name}|{line[:80]}"
|
||
fee_key = builder.add_node(
|
||
"TicketFee",
|
||
f"ticket_fee:{digest(source_file, owner_key, identity)}",
|
||
item_name,
|
||
ticket_fee_id=f"TF-{digest(source_file, owner_key, identity, length=10)}",
|
||
fee_name=item_name,
|
||
fee_type="免费/赠送项目",
|
||
amount_text="0元",
|
||
amount_value=0,
|
||
price_text="免费/赠送",
|
||
adult_price=0,
|
||
child_price=0,
|
||
currency="CNY",
|
||
unit="人",
|
||
inclusion_status="免费/赠送",
|
||
mandatory_level="included_free",
|
||
applies_to=product_name,
|
||
source_product_name=product_name,
|
||
scenic_target_name=target_name,
|
||
scenic_target_type=target_label,
|
||
consumer_group="all",
|
||
is_free=True,
|
||
is_optional=False,
|
||
price_status="source_free",
|
||
child_price_status="同成人免费",
|
||
rule_text=line,
|
||
evidence_excerpt=line,
|
||
source_file=source_file,
|
||
data_quality_status="structured_from_free_text",
|
||
extraction_rule="free_or_gift_project_line",
|
||
)
|
||
owner_label = builder.nodes.get(owner_key, {}).get("label")
|
||
builder.add_rel("PRICE_PACKAGE_HAS_FEE" if owner_label == "ProductPricePackage" else "SLOT_HAS_FEE", owner_key, fee_key)
|
||
if target_key:
|
||
builder.add_rel("SCENIC_AREA_HAS_FEE" if target_label == "ScenicArea" else "ATTRACTION_HAS_FEE", target_key, fee_key, source_product_name=product_name)
|
||
product_key = find_product_key(builder, product_name)
|
||
if product_key:
|
||
builder.add_rel("PRODUCT_HAS_FEE", product_key, fee_key)
|
||
|
||
|
||
def extract_fee_nodes(builder: KGBuilder, owner_key: str, product_name: str, text: str, source_file: str) -> None:
|
||
seen: set[str] = set()
|
||
for line in re.split(r"[。;;\n]+", text):
|
||
line = compact(line)
|
||
if len(line) < 4:
|
||
continue
|
||
if ("免费" in line or "赠送" in line) and not any(x in line for x in ["并非", "不是免费", "不免费", "非免费"]):
|
||
target_key, target_label, target_name = fee_target_from_text(builder, line)
|
||
free_names = []
|
||
for m in re.finditer(r"(?:赠送|免费)([^,,。;;]{2,24})(?:体验|券|项目|服务)?", line):
|
||
project = compact(m.group(1))
|
||
if project and not any(bad in project for bad in ["门票退", "无退费", "早餐"]):
|
||
free_names.append(project if project.endswith(("体验", "券", "服务")) else f"{project}体验")
|
||
for project in free_names[:3]:
|
||
identity = f"free|{target_key}|{project}|{line[:80]}"
|
||
if identity in seen:
|
||
continue
|
||
seen.add(identity)
|
||
add_free_project_fee(builder, owner_key, product_name, source_file, line, target_key, target_label, target_name, project)
|
||
|
||
if not re.search(r"\d+(?:\.\d+)?\s*元", line):
|
||
continue
|
||
if not any(x in line for x in ("元", "票", "保险", "车", "扶梯", "索道", "餐", "儿童", "游船", "演出")):
|
||
continue
|
||
current_target_key, current_target_label, current_target_name = fee_target_from_text(builder, line)
|
||
current_service = ""
|
||
split_tokens = re.split(r"([,,、++])", line)
|
||
part_rows: list[tuple[str, str]] = []
|
||
pending_sep = ""
|
||
for token in split_tokens:
|
||
token = compact(token)
|
||
if not token:
|
||
continue
|
||
if token in {",", ",", "、", "+", "+"}:
|
||
pending_sep = token
|
||
continue
|
||
part_rows.append((pending_sep, token))
|
||
pending_sep = ""
|
||
previous_part_had_amount = False
|
||
for previous_sep, part in part_rows:
|
||
part = compact(part)
|
||
detected_key, detected_label, detected_name = fee_target_from_text(builder, part, current_target_key)
|
||
if detected_key:
|
||
current_target_key, current_target_label, current_target_name = detected_key, detected_label, detected_name
|
||
fee_type, service_name = service_from_fee_part(
|
||
part,
|
||
current_service,
|
||
joined_with_previous_without_amount=previous_sep in {"+", "+"} and not previous_part_had_amount,
|
||
)
|
||
if service_name:
|
||
current_service = service_name
|
||
if fee_type == "费用说明" and not current_service:
|
||
previous_part_had_amount = bool(re.search(r"\d+(?:\.\d+)?\s*元", part))
|
||
continue
|
||
for match in re.finditer(r"(\d+(?:\.\d+)?)\s*元\s*(?:/|/)?\s*(人|趟|间|份|餐)?", part):
|
||
if is_total_amount_match(part, match.start()):
|
||
continue
|
||
prefix = clean_fee_prefix(part, match.start())
|
||
amount = float(match.group(1))
|
||
unit = match.group(2) or ("人" if "/人" in part or "每人" in part else "")
|
||
local_fee_type, local_service = service_from_fee_part(
|
||
prefix + part[match.start():],
|
||
current_service,
|
||
joined_with_previous_without_amount=previous_sep in {"+", "+"} and not previous_part_had_amount,
|
||
)
|
||
if local_service:
|
||
fee_type, service_name = local_fee_type, local_service
|
||
current_service = service_name
|
||
if not service_name and prefix:
|
||
service_name = prefix
|
||
if not service_name or service_name in {"费用说明"}:
|
||
continue
|
||
consumer_group = "child" if "儿童" in part or "儿童" in line[: max(0, line.find(part)) + len(part)] and "成人" not in part else "all"
|
||
inclusion_status, mandatory_level, is_optional = inclusion_from_fee_line(line, part, fee_type)
|
||
identity = f"{current_target_key}|{service_name}|{amount}|{consumer_group}|{line[:80]}"
|
||
if identity in seen:
|
||
continue
|
||
seen.add(identity)
|
||
add_structured_fee_node(
|
||
builder, owner_key, product_name, source_file, line,
|
||
current_target_key, current_target_label, current_target_name,
|
||
service_name, fee_type, amount, unit, inclusion_status, mandatory_level,
|
||
is_optional, consumer_group, "explicit_project_price_pattern",
|
||
)
|
||
previous_part_had_amount = bool(re.search(r"\d+(?:\.\d+)?\s*元", part))
|
||
if "半票" in line or "免票" in line:
|
||
rule_key = builder.add_node(
|
||
"BusinessRule",
|
||
f"rule:ticket_discount:{digest(source_file, owner_key, line)}",
|
||
f"{product_name}门票优惠/退费规则",
|
||
rule_id=f"RULE-{digest(source_file, owner_key, line, length=10)}",
|
||
rule_type="ticket_discount",
|
||
applies_to_type="TicketFee",
|
||
applies_to_text=product_name,
|
||
rule_text=line,
|
||
severity="提示",
|
||
source_file=source_file,
|
||
)
|
||
product_key = find_product_key(builder, product_name)
|
||
if product_key:
|
||
builder.add_rel("PRODUCT_HAS_RULE", product_key, rule_key)
|
||
|
||
|
||
def parse_existing_route_markdown(
|
||
builder: KGBuilder,
|
||
alias_to_key: dict[str, str],
|
||
hotel_groups: dict[str, str],
|
||
restaurant_groups: dict[str, str],
|
||
vehicle_group_key: str,
|
||
) -> list[str]:
|
||
index_path = ROUTE_MD_DIR / "产品索引.json"
|
||
items = json.loads(index_path.read_text(encoding="utf-8"))
|
||
product_keys: list[str] = []
|
||
for item in items:
|
||
md_path = ROUTE_MD_DIR / item["markdown_filename"]
|
||
markdown = md_path.read_text(encoding="utf-8")
|
||
fm, _ = extract_frontmatter(markdown)
|
||
name = fm.get("product_name") or item.get("product_name")
|
||
if not name:
|
||
continue
|
||
source_file = fm.get("source_file") or item.get("source_file") or str(md_path)
|
||
attractions = fm.get("core_attractions") or []
|
||
series_key = ensure_series(builder, name, fm.get("product_family") or "", attractions)
|
||
product_key = builder.add_node(
|
||
"TourProduct",
|
||
product_key_for_name(name + source_file),
|
||
name,
|
||
product_id=f"TP-{digest(source_file, name, length=10)}",
|
||
product_series=builder.nodes[series_key]["name"],
|
||
product_family=fm.get("product_family"),
|
||
product_type="已有路线产品",
|
||
duration_days=fm.get("duration_days"),
|
||
duration_nights=max(int(fm.get("duration_days") or 0) - 1, 0) if fm.get("duration_days") else None,
|
||
route_immutable=True,
|
||
default_group_mode="固定产品/按产品说明",
|
||
default_vehicle_type=fm.get("default_vehicle_type"),
|
||
default_hotel_grade=fm.get("default_hotel_grade"),
|
||
default_meal_standard="按产品每日用餐/接待标准",
|
||
service_promise="按产品原文承诺",
|
||
selling_points=attractions,
|
||
included_summary="见产品Markdown费用与接待标准",
|
||
excluded_summary="见产品Markdown费用与规则候选",
|
||
booking_notes="路线骨架固定;仅允许在资源槽位范围内替换/升级/二次核价。",
|
||
source_files=[source_file, str(md_path)],
|
||
evidence_excerpt=markdown[:1200],
|
||
)
|
||
builder.add_rel("BELONGS_TO_SERIES", product_key, series_key)
|
||
product_keys.append(product_key)
|
||
|
||
rows = parse_fixed_route_rows(markdown)
|
||
if not rows:
|
||
rows = fallback_route_rows(markdown, fm)
|
||
product_stop_keys: list[str] = []
|
||
product_stop_names: list[str] = []
|
||
product_day_keys: list[str] = []
|
||
product_day_summaries: list[str] = []
|
||
for row in rows:
|
||
day_index = int(row["day_index"])
|
||
day_key = builder.add_node(
|
||
"ProductDay",
|
||
f"product_day:{product_key}:{day_index}",
|
||
f"{name} {row['day']}",
|
||
day_id=f"DAY-{digest(product_key, day_index, length=10)}",
|
||
day_index=day_index,
|
||
title=row["route"],
|
||
route_path=row["route"],
|
||
route_line_name=name,
|
||
route_line_id=f"TP-{digest(source_file, name, length=10)}",
|
||
route_sequence_label=row["day"],
|
||
route_display_name=f"{name} {row['day']}",
|
||
meal_text=row["meals"],
|
||
accommodation_text=row["accommodation"],
|
||
transport_summary="按固定路线骨架执行",
|
||
source_file=source_file,
|
||
)
|
||
builder.add_rel("HAS_DAY", product_key, day_key, day_index=day_index)
|
||
product_day_keys.append(day_key)
|
||
product_day_summaries.append(f"{row['day']} {row['route']}")
|
||
points = route_points(row["route"])
|
||
stop_keys: list[str] = []
|
||
for order, point in enumerate(points, start=1):
|
||
attr_key = find_attraction(point, alias_to_key)
|
||
stop_type = "scenic" if attr_key else ("city_or_area" if point else "unknown")
|
||
global_order = len(product_stop_keys) + 1
|
||
route_sequence_label = f"{row['day']}-{order:02d}"
|
||
station_like_name = f"{route_sequence_label} {point}"
|
||
stop_key = builder.add_node(
|
||
"RouteStop",
|
||
f"route_stop:{product_key}:{day_index}:{order}:{slug(point)}",
|
||
station_like_name,
|
||
stop_id=f"STOP-{digest(product_key, day_index, order, point, length=10)}",
|
||
stop_order=order,
|
||
day_stop_order=order,
|
||
global_stop_order=global_order,
|
||
route_sequence_label=route_sequence_label,
|
||
station_like_name=station_like_name,
|
||
route_line_name=name,
|
||
route_line_id=f"TP-{digest(source_file, name, length=10)}",
|
||
route_display_name=f"{name} {route_sequence_label} {point}",
|
||
stop_type=stop_type,
|
||
city_or_area=point,
|
||
is_core_visit=bool(attr_key),
|
||
evidence_text=row["route"],
|
||
meal_text=row["meals"],
|
||
accommodation_text=row["accommodation"],
|
||
source_file=source_file,
|
||
)
|
||
builder.add_rel("DAY_HAS_STOP", day_key, stop_key, stop_order=order)
|
||
builder.add_rel(
|
||
"PRODUCT_HAS_ORDERED_STOP",
|
||
product_key,
|
||
stop_key,
|
||
global_stop_order=global_order,
|
||
day_index=day_index,
|
||
day_stop_order=order,
|
||
route_sequence_label=route_sequence_label,
|
||
)
|
||
if attr_key:
|
||
builder.add_rel("STOP_VISITS_ATTRACTION", stop_key, attr_key)
|
||
stop_keys.append(stop_key)
|
||
product_stop_keys.append(stop_key)
|
||
product_stop_names.append(point)
|
||
if stop_keys:
|
||
builder.nodes[day_key].update({
|
||
"route_stop_sequence": " -> ".join(points),
|
||
"route_display_text": f"{name} {row['day']}:{' -> '.join(points)}",
|
||
"total_stop_count": len(stop_keys),
|
||
})
|
||
for idx in range(len(stop_keys) - 1):
|
||
origin = builder.nodes[stop_keys[idx]]["name"]
|
||
dest = builder.nodes[stop_keys[idx + 1]]["name"]
|
||
seg_key = builder.add_node(
|
||
"RouteSegment",
|
||
f"route_segment:{product_key}:{day_index}:{idx+1}",
|
||
f"{row['day']} {origin}->{dest}",
|
||
segment_id=f"SEG-{digest(product_key, day_index, idx, origin, dest, length=10)}",
|
||
day_index=day_index,
|
||
origin_text=origin,
|
||
destination_text=dest,
|
||
transport_mode=fm.get("default_vehicle_type") or "按产品安排",
|
||
source_file=source_file,
|
||
)
|
||
builder.add_rel("DAY_HAS_SEGMENT", day_key, seg_key)
|
||
builder.add_rel("SEGMENT_FROM_STOP", seg_key, stop_keys[idx])
|
||
builder.add_rel("SEGMENT_TO_STOP", seg_key, stop_keys[idx + 1])
|
||
|
||
if row["accommodation"] and row["accommodation"] != "/":
|
||
group_key = hotel_groups_for_text(hotel_groups, row["accommodation"])
|
||
add_resource_slot(
|
||
builder, product_key, day_key, "lodging", f"{name} {row['day']}住宿槽位", "day",
|
||
day_index, row["accommodation"], fm.get("default_hotel_grade") or "", "same_level_replace",
|
||
"same_level_or_second_quote", source_file, group_key,
|
||
)
|
||
if row["meals"]:
|
||
group_key = restaurant_groups_for_text(restaurant_groups, row["route"] + row["accommodation"])
|
||
add_resource_slot(
|
||
builder, product_key, day_key, "meal", f"{name} {row['day']}餐饮槽位", "day",
|
||
day_index, row["meals"], "按产品餐标", "optional_addon", "included_or_second_quote",
|
||
source_file, group_key,
|
||
)
|
||
for idx in range(len(product_day_keys) - 1):
|
||
builder.add_rel(
|
||
"DAY_NEXT_DAY",
|
||
product_day_keys[idx],
|
||
product_day_keys[idx + 1],
|
||
sequence_order=idx + 1,
|
||
)
|
||
for idx in range(len(product_stop_keys) - 1):
|
||
current_key = product_stop_keys[idx]
|
||
next_key = product_stop_keys[idx + 1]
|
||
current = builder.nodes[current_key]
|
||
next_stop = builder.nodes[next_key]
|
||
current["next_stop_name"] = next_stop.get("city_or_area") or next_stop.get("name")
|
||
next_stop["previous_stop_name"] = current.get("city_or_area") or current.get("name")
|
||
builder.add_rel(
|
||
"ROUTE_STOP_NEXT",
|
||
current_key,
|
||
next_key,
|
||
route_line_name=name,
|
||
sequence_order=idx + 1,
|
||
from_stop_order=current.get("global_stop_order"),
|
||
to_stop_order=next_stop.get("global_stop_order"),
|
||
)
|
||
if product_stop_keys:
|
||
stop_sequence = " -> ".join([compact(x) for x in product_stop_names if compact(x)])
|
||
for stop_key in product_stop_keys:
|
||
builder.nodes[stop_key]["total_stop_count"] = len(product_stop_keys)
|
||
builder.nodes[stop_key]["route_stop_sequence"] = stop_sequence
|
||
builder.nodes[product_key].update({
|
||
"route_line_name": name,
|
||
"route_line_id": f"TP-{digest(source_file, name, length=10)}",
|
||
"route_stop_sequence": stop_sequence,
|
||
"route_day_summary": ";".join(product_day_summaries),
|
||
"route_display_text": f"{name}:{stop_sequence}",
|
||
"total_stop_count": len(product_stop_keys),
|
||
})
|
||
if fm.get("default_vehicle_type"):
|
||
add_resource_slot(
|
||
builder, product_key, None, "vehicle", f"{name}默认交通槽位", "product",
|
||
None, fm.get("default_vehicle_type"), fm.get("default_vehicle_type"), "upgradeable",
|
||
"second_quote", source_file, vehicle_group_key,
|
||
)
|
||
fee_slot = add_resource_slot(
|
||
builder, product_key, None, "ticket_transport_fee", f"{name}门票小交通费用槽位", "product",
|
||
None, "按产品费用说明", "", "mandatory_self_pay", "policy_dependent", source_file,
|
||
)
|
||
fee_section = markdown[markdown.find("## 5. 费用与规则候选"):] if "## 5. 费用与规则候选" in markdown else markdown
|
||
extract_fee_nodes(builder, fee_slot, name, fee_section, source_file)
|
||
for line in re.findall(r"^- (.+)$", fee_section, flags=re.M)[:80]:
|
||
if any(k in line for k in ["老人", "儿童", "学生", "退", "预约", "不接待", "投诉", "满房", "同级", "孕妇", "不可抗力", "路滑", "少走路"]):
|
||
rule_type = "discount" if any(k in line for k in ["老人", "儿童", "学生", "军人", "免票", "半票"]) else ("refund" if "退" in line else ("risk" if any(k in line for k in ["预约", "路滑", "孕妇", "不接待", "不可抗力"]) else "booking"))
|
||
rule_key = builder.add_node(
|
||
"BusinessRule",
|
||
f"rule:{digest(source_file, name, line)}",
|
||
f"{name} {rule_type}规则",
|
||
rule_id=f"RULE-{digest(source_file, name, line, length=10)}",
|
||
rule_type=rule_type,
|
||
applies_to_type="TourProduct",
|
||
applies_to_text=name,
|
||
rule_text=line,
|
||
severity="关键限制" if rule_type in {"refund", "risk"} else "提示",
|
||
source_file=source_file,
|
||
)
|
||
builder.add_rel("PRODUCT_HAS_RULE", product_key, rule_key)
|
||
return product_keys
|
||
|
||
|
||
def ensure_pricing_subject(
|
||
builder: KGBuilder,
|
||
product_name: str,
|
||
source_file: str,
|
||
catalog_type: str,
|
||
product_direction: str = "",
|
||
route_preview: str = "",
|
||
) -> tuple[str, str, str]:
|
||
existing = find_product_key(builder, product_name)
|
||
if existing:
|
||
return existing, "PRODUCT_HAS_PRICE_PACKAGE", "PRODUCT_HAS_RULE"
|
||
catalog_key = builder.add_node(
|
||
"PriceCatalogItem",
|
||
f"price_catalog:{slug(product_name + source_file)}",
|
||
product_name,
|
||
catalog_item_id=f"PCI-{digest(source_file, product_name, catalog_type, length=10)}",
|
||
catalog_type=catalog_type,
|
||
duration_days=duration_from_text(product_name),
|
||
product_direction=product_direction,
|
||
route_preview=route_preview,
|
||
price_validity="见价格表",
|
||
source_file=source_file,
|
||
booking_notes="来自价格表;未在路线文档中形成完整每日行程骨架时,不作为完整路线推荐。",
|
||
)
|
||
return catalog_key, "CATALOG_ITEM_HAS_PRICE_PACKAGE", "CATALOG_ITEM_HAS_RULE"
|
||
|
||
|
||
def parse_small_group_prices(builder: KGBuilder) -> None:
|
||
path = SOURCE_ROOT / "滨海国旅2-8人拼小团计划 ( 26年4月1号-4月28号。。26年5月4号--6月30号 ~)(五一节除外).xlsx"
|
||
df = pd.read_excel(path, header=None)
|
||
notes = compact(df.iloc[1, 0]) if len(df) > 1 else ""
|
||
current_vehicle = ""
|
||
current_product = ""
|
||
current_collection = ""
|
||
current_schedule = ""
|
||
current_inner_fee = ""
|
||
current_refund = ""
|
||
for idx in range(3, len(df)):
|
||
row = [compact(x) for x in df.iloc[idx].tolist()[:10]]
|
||
if row[0]:
|
||
current_vehicle = row[0]
|
||
if row[1]:
|
||
current_product = re.split(r"\n|(镇远|注:", row[1])[0].strip()
|
||
if row[2]:
|
||
current_collection = row[2]
|
||
if row[3]:
|
||
current_schedule = row[3]
|
||
if row[8]:
|
||
current_inner_fee = row[8]
|
||
if row[9]:
|
||
current_refund = row[9]
|
||
if not current_product or not row[4] or money(row[5]) is None:
|
||
continue
|
||
subject_key, price_rel_type, rule_rel_type = ensure_pricing_subject(
|
||
builder, current_product, str(path), "1-8人拼小团报价目录"
|
||
)
|
||
package_name = f"{current_product} {row[4]}"
|
||
price_key = builder.add_node(
|
||
"ProductPricePackage",
|
||
f"price:small:{idx}:{slug(package_name)}",
|
||
package_name,
|
||
price_package_id=f"PRICE-SG-{idx:03d}",
|
||
package_name=package_name,
|
||
season="2026年4月/5-6月平季(五一除外)",
|
||
date_range="2026-04-01~2026-04-28;2026-05-04~2026-06-30",
|
||
group_size_band=current_vehicle or "1-8人拼小团",
|
||
room_type=row[4],
|
||
hotel_grade=row[4],
|
||
vehicle_type="按人数派5/7/9座车",
|
||
adult_price=money(row[5]),
|
||
child_price=money(row[6]),
|
||
single_room_supplement=money(row[7]),
|
||
inner_transport_fee_text=current_inner_fee,
|
||
refund_policy_text=current_refund,
|
||
source_file=str(path),
|
||
collection_method=current_collection,
|
||
schedule_rule=current_schedule,
|
||
notes=notes,
|
||
)
|
||
builder.add_rel(price_rel_type, subject_key, price_key)
|
||
if current_inner_fee:
|
||
extract_fee_nodes(builder, price_key, current_product, current_inner_fee, str(path))
|
||
if current_refund:
|
||
rule_key = builder.add_node(
|
||
"BusinessRule",
|
||
f"rule:small_refund:{slug(current_product + current_refund)}",
|
||
f"{current_product}证件退费规则",
|
||
rule_id=f"RULE-{digest(current_product, current_refund, length=10)}",
|
||
rule_type="refund",
|
||
applies_to_type="ProductPricePackage",
|
||
applies_to_text=current_product,
|
||
rule_text=current_refund,
|
||
severity="报价必看",
|
||
source_file=str(path),
|
||
)
|
||
builder.add_rel(rule_rel_type, subject_key, rule_key)
|
||
if notes:
|
||
for i, line in enumerate(re.split(r"\s*(?=\d+:|关于)", notes)[:20], start=1):
|
||
line = compact(line)
|
||
if len(line) < 8:
|
||
continue
|
||
if any(k in line for k in ["用车", "行李", "酒店", "用餐", "不接待", "老人", "孕妇", "儿童"]):
|
||
rule_key = builder.add_node(
|
||
"BusinessRule",
|
||
f"rule:small_note:{i}",
|
||
f"拼小团注意事项{i}",
|
||
rule_id=f"RULE-SMALL-{i:02d}",
|
||
rule_type="eligibility" if any(k in line for k in ["不接待", "老人", "孕妇", "儿童"]) else "booking",
|
||
applies_to_type="TourProduct",
|
||
applies_to_text="1-8人拼小团",
|
||
rule_text=line,
|
||
severity="关键限制" if "不接待" in line else "提示",
|
||
source_file=str(path),
|
||
)
|
||
for key, node in list(builder.nodes.items()):
|
||
if node.get("label") == "TourProduct" and "拼小团" in compact(node.get("default_group_mode") or node.get("product_type")):
|
||
builder.add_rel("PRODUCT_HAS_RULE", key, rule_key)
|
||
if node.get("label") == "PriceCatalogItem" and "拼小团" in compact(node.get("catalog_type")):
|
||
builder.add_rel("CATALOG_ITEM_HAS_RULE", key, rule_key)
|
||
|
||
|
||
def parse_independent_prices(builder: KGBuilder) -> None:
|
||
path = SOURCE_ROOT / "20-25人独立成团.xlsx"
|
||
xl = pd.ExcelFile(path)
|
||
for sheet in xl.sheet_names:
|
||
df = pd.read_excel(path, sheet_name=sheet, header=None)
|
||
current_product = ""
|
||
current_direction = ""
|
||
for idx in range(4, len(df)):
|
||
row = [compact(x) for x in df.iloc[idx].tolist()[:10]]
|
||
if not any(row):
|
||
continue
|
||
if row[0] and row[0] not in {"产品方向", "报名建议"}:
|
||
current_direction = row[0]
|
||
if row[1] and row[1] != "参考酒店":
|
||
current_product = row[1]
|
||
if not current_product:
|
||
continue
|
||
hotel_text = row[5] or row[3] or row[4]
|
||
price_cols = [(6, "20人"), (7, "25人")]
|
||
for col, group_size in price_cols:
|
||
if col >= len(row) or money(row[col]) is None:
|
||
continue
|
||
subject_key, price_rel_type, _rule_rel_type = ensure_pricing_subject(
|
||
builder, current_product, str(path), "20-25人独立成团报价目录", current_direction, row[2]
|
||
)
|
||
package_name = f"{current_product} {sheet} {group_size} {hotel_text or ''}"
|
||
price_key = builder.add_node(
|
||
"ProductPricePackage",
|
||
f"price:ind:{sheet}:{idx}:{group_size}:{slug(package_name)}",
|
||
package_name,
|
||
price_package_id=f"PRICE-IND-{digest(sheet, idx, group_size, length=10)}",
|
||
package_name=package_name,
|
||
season=sheet,
|
||
group_size_band=group_size,
|
||
room_type=hotel_text,
|
||
hotel_grade=hotel_text,
|
||
vehicle_type="32-38座2+1座大巴",
|
||
adult_price=money(row[col]),
|
||
single_room_supplement=money(row[8]),
|
||
inner_transport_fee_text=row[3] if "门票" in row[3] or "观光车" in row[3] else "",
|
||
source_file=str(path),
|
||
product_direction=current_direction,
|
||
route_preview=row[2],
|
||
meal_standard=row[4] if "餐" in row[4] else row[5],
|
||
)
|
||
builder.add_rel(price_rel_type, subject_key, price_key)
|
||
if idx == 1 or "报名建议" in row[0]:
|
||
continue
|
||
note = compact(df.iloc[1, 0]) if len(df) > 1 else ""
|
||
if note:
|
||
for product_key, node in list(builder.nodes.items()):
|
||
if (
|
||
(node.get("label") == "TourProduct" and "独立成团" in compact(node.get("product_type")))
|
||
or (node.get("label") == "PriceCatalogItem" and "独立成团" in compact(node.get("catalog_type")))
|
||
):
|
||
rule_key = builder.add_node(
|
||
"BusinessRule",
|
||
f"rule:ind:{slug(sheet + note[:80])}",
|
||
f"{sheet}独立成团报名限制",
|
||
rule_id=f"RULE-IND-{digest(sheet, note[:120], length=10)}",
|
||
rule_type="eligibility",
|
||
applies_to_type="TourProduct",
|
||
applies_to_text="20-25人独立成团",
|
||
rule_text=note[:1200],
|
||
severity="关键限制",
|
||
source_file=str(path),
|
||
)
|
||
rel_type = "CATALOG_ITEM_HAS_RULE" if node.get("label") == "PriceCatalogItem" else "PRODUCT_HAS_RULE"
|
||
builder.add_rel(rel_type, product_key, rule_key)
|
||
|
||
|
||
def parse_transfer_quotes(builder: KGBuilder) -> None:
|
||
path = SOURCE_ROOT / "黔玩转接送组报价.docx"
|
||
text = read_office_text(path)
|
||
current_vehicle = ""
|
||
for raw in text.splitlines():
|
||
line = compact(raw)
|
||
if not line:
|
||
continue
|
||
if re.match(r"^\d+座|^7座", line) and "趟" not in line:
|
||
current_vehicle = line
|
||
continue
|
||
m = re.search(r"(.+?)[-—–]+(.+?)(\d+)\s*/?\s*趟", line)
|
||
if not m or not current_vehicle:
|
||
continue
|
||
origin, destination, price = compact(m.group(1)), compact(m.group(2)), float(m.group(3))
|
||
key = builder.add_node(
|
||
"TransferQuote",
|
||
f"transfer:{slug(current_vehicle + origin + destination + str(price))}",
|
||
f"{current_vehicle} {origin}->{destination}",
|
||
transfer_quote_id=f"TQ-{digest(current_vehicle, origin, destination, price, length=10)}",
|
||
origin_text=origin,
|
||
destination_text=destination,
|
||
vehicle_type=current_vehicle,
|
||
price_per_trip=price,
|
||
quote_unit="趟",
|
||
quote_notes=line,
|
||
source_file=str(path),
|
||
)
|
||
for area_text, rel_type in [(origin, "LOCATED_IN"), (destination, "LOCATED_IN")]:
|
||
for part in split_items(area_text.replace("、", ","), limit=6):
|
||
area_key = builder.add_node("Area", f"area:{slug(part)}", part, area_id=f"AREA-{digest(part, length=8)}", area_type="接送区域")
|
||
# LOCATED_IN start type does not include TransferQuote in schema; keep area text in quote properties.
|
||
|
||
|
||
def parse_sales_scripts(builder: KGBuilder) -> None:
|
||
path = SOURCE_ROOT / "线上客资回复话术.docx"
|
||
text = read_office_text(path)
|
||
chunks = re.split(r"(?=Step\d+\.|STEP\d+|步骤[一二三四五六七八九十])", text)
|
||
for idx, chunk in enumerate(chunks):
|
||
msg = compact(chunk)
|
||
if len(msg) < 30:
|
||
continue
|
||
channel = "微信" if "微信" in msg or idx > 1 else "小红书"
|
||
stage = "产品推荐" if "产品" in msg or "路线" in msg else ("留资引导" if "VX" in msg or "加V" in msg else "首次沟通")
|
||
key = builder.add_node(
|
||
"SalesScript",
|
||
f"script:{idx}:{slug(msg[:40])}",
|
||
f"{channel}-{stage}-{idx}",
|
||
script_id=f"SCRIPT-{idx:03d}",
|
||
channel=channel,
|
||
funnel_stage=stage,
|
||
trigger_scenario=msg[:120],
|
||
message_template=msg[:1500],
|
||
intent_tags=[tag for tag in ["留资", "报价", "费用包含", "纯玩", "房间数", "老人小孩", "产品推荐"] if tag in msg],
|
||
required_customer_fields=[field for field in ["月份", "人数", "天数", "房间数", "老人", "小孩", "酒店", "预算"] if field in msg],
|
||
source_file=str(path),
|
||
)
|
||
# Link to explain broad objects after product creation is complete in a lightweight way.
|
||
for product_key, product in list(builder.nodes.items())[:200]:
|
||
if product.get("label") == "TourProduct" and any(token and token in msg for token in split_items(product.get("name"), limit=3)):
|
||
builder.add_rel("SCRIPT_EXPLAINS", key, product_key)
|
||
|
||
|
||
def add_resource_library_layer(builder: KGBuilder) -> None:
|
||
root_key = builder.add_node(
|
||
"ResourceLibrary",
|
||
"library:root:travel_graph",
|
||
"旅行社线路制定业务资料库",
|
||
library_id="LIB-TRAVEL-GRAPH",
|
||
library_type="root",
|
||
description="统一组织已有路线产品、基础资源、价格报价、规则和图片素材,方便业务视角浏览图谱。",
|
||
scope="旅行社线路制定",
|
||
)
|
||
sections = {
|
||
"route_products": ("已有路线产品库", "product_catalog", "已成型路线产品,路线骨架固定。"),
|
||
"base_resources": ("基础资源库", "base_resource", "景点、酒店、餐饮、车辆、接送等可复用资料。"),
|
||
"price_catalog": ("报价目录库", "price_catalog", "价格表中不等同于完整路线的报价对象。"),
|
||
"rule_library": ("业务规则库", "rule", "退费、限制、优惠、风险、预订等规则。"),
|
||
"media_library": ("图片素材库", "media", "可靠图片 URL 与素材组。"),
|
||
}
|
||
section_keys: dict[str, str] = {}
|
||
for code, (name, library_type, desc) in sections.items():
|
||
key = builder.add_node(
|
||
"ResourceLibrary",
|
||
f"library:{code}",
|
||
name,
|
||
library_id=f"LIB-{code.upper()}",
|
||
library_type=library_type,
|
||
description=desc,
|
||
scope="旅行社线路制定",
|
||
)
|
||
builder.add_rel("LIBRARY_HAS_SECTION", root_key, key)
|
||
section_keys[code] = key
|
||
|
||
resource_rel_by_label = {
|
||
"ScenicArea": "LIBRARY_CONTAINS_SCENIC_AREA",
|
||
"ScenicAttraction": "LIBRARY_CONTAINS_ATTRACTION",
|
||
"HotelResource": "LIBRARY_CONTAINS_HOTEL",
|
||
"RestaurantResource": "LIBRARY_CONTAINS_RESTAURANT",
|
||
"VehicleResource": "LIBRARY_CONTAINS_VEHICLE",
|
||
"TransferQuote": "LIBRARY_CONTAINS_TRANSFER",
|
||
"ResourceOptionGroup": "LIBRARY_CONTAINS_OPTION_GROUP",
|
||
}
|
||
for key, node in list(builder.nodes.items()):
|
||
label = node.get("label")
|
||
if label == "TourProduct":
|
||
builder.add_rel("LIBRARY_CONTAINS_PRODUCT", section_keys["route_products"], key)
|
||
elif label == "PriceCatalogItem":
|
||
builder.add_rel("LIBRARY_CONTAINS_CATALOG_ITEM", section_keys["price_catalog"], key)
|
||
elif label in resource_rel_by_label:
|
||
builder.add_rel(resource_rel_by_label[label], section_keys["base_resources"], key)
|
||
elif label == "BusinessRule":
|
||
builder.add_rel("LIBRARY_CONTAINS_RULE", section_keys["rule_library"], key)
|
||
elif label == "MediaResource":
|
||
builder.add_rel("LIBRARY_CONTAINS_MEDIA", section_keys["media_library"], key)
|
||
|
||
|
||
def write_outputs(builder: KGBuilder, schema: dict[str, Any]) -> None:
|
||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
SCHEMA_OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
schema_json = SCHEMA_OUT_DIR / "travel_graph_existing_product_schema.v1.4.json"
|
||
schema_dsl = SCHEMA_OUT_DIR / "travel_graph_existing_product_schema.v1.4.dsl.md"
|
||
schema_json.write_text(json.dumps(schema, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
schema_dsl.write_text(schema_to_dsl(schema), encoding="utf-8")
|
||
(OUT_DIR / schema_json.name).write_text(schema_json.read_text(encoding="utf-8"), encoding="utf-8")
|
||
(OUT_DIR / schema_dsl.name).write_text(schema_dsl.read_text(encoding="utf-8"), encoding="utf-8")
|
||
nodes = list(builder.nodes.values())
|
||
rels = builder.relations
|
||
(OUT_DIR / "抽取结果_nodes.json").write_text(json.dumps(nodes, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
(OUT_DIR / "抽取结果_relations.json").write_text(json.dumps(rels, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
with (OUT_DIR / "抽取结果_nodes.csv").open("w", newline="", encoding="utf-8-sig") as fh:
|
||
writer = csv.DictWriter(fh, fieldnames=["label", "natural_key", "name", "summary"])
|
||
writer.writeheader()
|
||
for node in nodes:
|
||
writer.writerow({
|
||
"label": node["label"],
|
||
"natural_key": node["natural_key"],
|
||
"name": node["name"],
|
||
"summary": node.get("route_path") or node.get("package_name") or node.get("rule_text") or node.get("primary_image_url") or "",
|
||
})
|
||
with (OUT_DIR / "抽取结果_relations.csv").open("w", newline="", encoding="utf-8-sig") as fh:
|
||
writer = csv.DictWriter(fh, fieldnames=["relation_type", "source", "target", "properties"])
|
||
writer.writeheader()
|
||
for rel in rels:
|
||
writer.writerow({**rel, "properties": json.dumps(rel.get("properties") or {}, ensure_ascii=False)})
|
||
node_counts = Counter(node["label"] for node in nodes)
|
||
rel_counts = Counter(rel["relation_type"] for rel in rels)
|
||
report = [
|
||
"# travel_graph 旅行社线路制定图谱入库说明",
|
||
"",
|
||
f"生成时间:{datetime.now().strftime('%Y-%m-%d %H:%M:%S')}",
|
||
"",
|
||
"## 项目",
|
||
f"- Project ID: `{PROJECT_ID}`",
|
||
f"- Tenant ID: `{TENANT_ID}`",
|
||
f"- FalkorDB Graph Name: `{GRAPH_NAME}`",
|
||
f"- 项目名称:{PROJECT_NAME}",
|
||
"",
|
||
"## 本期业务边界",
|
||
"- 只做已有路线产品,不做从零自由定制路线。",
|
||
"- 固定路线骨架不可变,住宿/餐饮/车辆/接送/房型/门票小交通以资源槽位方式支持客户微调。",
|
||
"- 景点到景点、景区片区到合作酒店/餐厅采用高德驾车距离与耗时;不把纯直线距离作为最终候选关系。",
|
||
"- 价格进入 ProductPricePackage、HotelResource、RestaurantResource、TransferQuote、TicketFee/FeeItem。",
|
||
"- 图片 URL 进入 MediaResource,并同步写入匹配到的景点/酒店/车辆/资源组选项实体。",
|
||
"- 车辆资源只采用图片资源库中 `标签=车辆` 且高/中可靠的资源;待确认、低可靠、无图或禁止冒充资源不作为车辆实体。",
|
||
"",
|
||
"## 节点统计",
|
||
*[f"- {k}: {v}" for k, v in node_counts.most_common()],
|
||
"",
|
||
"## 关系统计",
|
||
*[f"- {k}: {v}" for k, v in rel_counts.most_common()],
|
||
]
|
||
(OUT_DIR / "入库说明.md").write_text("\n".join(report), encoding="utf-8")
|
||
|
||
|
||
def upsert_postgres(builder: KGBuilder, schema: dict[str, Any]) -> dict[str, int]:
|
||
with psycopg.connect(DB_URL, row_factory=dict_row) as conn:
|
||
with conn.cursor() as cur:
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.projects (
|
||
tenant_id, project_id, display_name, description, status,
|
||
default_namespace, metadata_jsonb, created_by, updated_at
|
||
)
|
||
VALUES (%s,%s,%s,%s,'active',%s,%s,'codex-import',now())
|
||
ON CONFLICT (tenant_id, project_id) DO UPDATE
|
||
SET display_name=EXCLUDED.display_name,
|
||
description=EXCLUDED.description,
|
||
status='active',
|
||
default_namespace=EXCLUDED.default_namespace,
|
||
metadata_jsonb=EXCLUDED.metadata_jsonb,
|
||
updated_at=now()
|
||
""",
|
||
(
|
||
TENANT_ID,
|
||
PROJECT_ID,
|
||
PROJECT_NAME,
|
||
"已有路线产品知识图谱:固定路线骨架、可配置资源槽位、价格、规则和图片素材。",
|
||
SCHEMA_NAMESPACE,
|
||
Jsonb({"business": "travel_agency_existing_product", "graph_name": GRAPH_NAME}),
|
||
),
|
||
)
|
||
cur.execute(
|
||
f"""
|
||
UPDATE {DB_SCHEMA}.ontology_schemas
|
||
SET status='archived', updated_at=now()
|
||
WHERE tenant_id=%s AND project_id=%s AND namespace=%s AND version <> %s
|
||
""",
|
||
(TENANT_ID, PROJECT_ID, SCHEMA_NAMESPACE, SCHEMA_VERSION),
|
||
)
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.ontology_schemas (
|
||
tenant_id, project_id, namespace, version, display_name, description,
|
||
status, schema_jsonb, created_by, published_by, published_at, updated_at
|
||
)
|
||
VALUES (%s,%s,%s,%s,%s,%s,'active',%s,'codex-import','codex-import',now(),now())
|
||
ON CONFLICT (tenant_id, project_id, namespace, version) DO UPDATE
|
||
SET display_name=EXCLUDED.display_name,
|
||
description=EXCLUDED.description,
|
||
status='active',
|
||
schema_jsonb=EXCLUDED.schema_jsonb,
|
||
published_by='codex-import',
|
||
published_at=now(),
|
||
updated_at=now()
|
||
RETURNING id
|
||
""",
|
||
(TENANT_ID, PROJECT_ID, SCHEMA_NAMESPACE, SCHEMA_VERSION, schema["display_name"], schema["purpose"], Jsonb(schema)),
|
||
)
|
||
schema_id = cur.fetchone()["id"]
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.graph_releases (
|
||
tenant_id, project_id, graph_release_id, graph_name, alias, status,
|
||
schema_id, source_dataset_version, metadata_jsonb, created_by,
|
||
published_at, activated_at, updated_at
|
||
)
|
||
VALUES (%s,%s,%s,%s,'active','active',%s,%s,%s,'codex-import',now(),now(),now())
|
||
ON CONFLICT (tenant_id, project_id, alias) DO UPDATE
|
||
SET graph_release_id=EXCLUDED.graph_release_id,
|
||
graph_name=EXCLUDED.graph_name,
|
||
status='active',
|
||
schema_id=EXCLUDED.schema_id,
|
||
source_dataset_version=EXCLUDED.source_dataset_version,
|
||
metadata_jsonb=EXCLUDED.metadata_jsonb,
|
||
activated_at=now(),
|
||
updated_at=now()
|
||
""",
|
||
(
|
||
TENANT_ID, PROJECT_ID, "travel_graph_v1", GRAPH_NAME, schema_id,
|
||
"existing-route-products-md-resource-workbooks-amap-driving-2026",
|
||
Jsonb({"node_count": len(builder.nodes), "relation_count": len(builder.relations), "output_dir": str(OUT_DIR)}),
|
||
),
|
||
)
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.import_templates (
|
||
template_id, version, display_name, primary_entity, template_jsonb, status, updated_at
|
||
)
|
||
VALUES (%s,1,%s,'TourProduct',%s,'active',now())
|
||
ON CONFLICT (template_id, version) DO UPDATE
|
||
SET display_name=EXCLUDED.display_name,
|
||
template_jsonb=EXCLUDED.template_jsonb,
|
||
status='active',
|
||
updated_at=now()
|
||
""",
|
||
(TEMPLATE_ID, "旅行社线路制定已有产品导入模板", Jsonb(schema)),
|
||
)
|
||
cur.execute(f"DELETE FROM {DB_SCHEMA}.candidate_relations WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
|
||
cur.execute(f"DELETE FROM {DB_SCHEMA}.candidate_entities WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
|
||
cur.execute(
|
||
f"""
|
||
DELETE FROM {DB_SCHEMA}.raw_records rr
|
||
USING {DB_SCHEMA}.import_batches ib
|
||
WHERE rr.batch_id=ib.id AND ib.tenant_id=%s AND ib.project_id=%s
|
||
""",
|
||
(TENANT_ID, PROJECT_ID),
|
||
)
|
||
cur.execute(f"DELETE FROM {DB_SCHEMA}.import_batches WHERE tenant_id=%s AND project_id=%s", (TENANT_ID, PROJECT_ID))
|
||
file_hash = hashlib.md5(json.dumps({"nodes": list(builder.nodes), "rels": builder.relations}, ensure_ascii=False).encode()).hexdigest()
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.import_batches (
|
||
tenant_id, project_id, graph_name, template_id, source_name, file_name,
|
||
file_hash, status, total_rows, success_rows, failed_rows, created_by, updated_at
|
||
)
|
||
VALUES (%s,%s,%s,%s,%s,%s,%s,'published',%s,%s,0,'codex-import',now())
|
||
RETURNING id
|
||
""",
|
||
(
|
||
TENANT_ID, PROJECT_ID, GRAPH_NAME, TEMPLATE_ID, "旅行社已有路线产品+资源库",
|
||
str(SOURCE_ROOT), file_hash, len(builder.nodes) + len(builder.relations), len(builder.nodes) + len(builder.relations),
|
||
),
|
||
)
|
||
batch_id = cur.fetchone()["id"]
|
||
id_by_key: dict[str, int] = {}
|
||
for row_number, (key, node) in enumerate(builder.nodes.items(), start=1):
|
||
payload = {k: v for k, v in node.items() if k not in {"label", "natural_key", "name"}}
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.candidate_entities (
|
||
tenant_id, project_id, batch_id, template_id, entity_type, natural_key,
|
||
display_name, payload_jsonb, confidence, status, reviewed_by, reviewed_at, updated_at
|
||
)
|
||
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,0.94,'published','codex-import',now(),now())
|
||
RETURNING id
|
||
""",
|
||
(TENANT_ID, PROJECT_ID, batch_id, TEMPLATE_ID, node["label"], key, node.get("name") or key, Jsonb(payload)),
|
||
)
|
||
id_by_key[key] = cur.fetchone()["id"]
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.raw_records (batch_id, row_number, raw_jsonb, row_hash, parse_status)
|
||
VALUES (%s,%s,%s,%s,'parsed')
|
||
ON CONFLICT (batch_id, row_number) DO NOTHING
|
||
""",
|
||
(batch_id, row_number, Jsonb(node), hashlib.md5(json.dumps(node, ensure_ascii=False, sort_keys=True).encode()).hexdigest()),
|
||
)
|
||
for rel in builder.relations:
|
||
src_id = id_by_key.get(rel["source"])
|
||
dst_id = id_by_key.get(rel["target"])
|
||
if not src_id or not dst_id:
|
||
continue
|
||
cur.execute(
|
||
f"""
|
||
INSERT INTO {DB_SCHEMA}.candidate_relations (
|
||
tenant_id, project_id, batch_id, source_candidate_id, relation_type,
|
||
target_candidate_id, target_ref_jsonb, payload_jsonb, status
|
||
)
|
||
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,'published')
|
||
""",
|
||
(TENANT_ID, PROJECT_ID, batch_id, src_id, rel["relation_type"], dst_id, Jsonb({"natural_key": rel["target"]}), Jsonb(rel.get("properties") or {})),
|
||
)
|
||
conn.commit()
|
||
return {"schema_id": schema_id, "batch_id": batch_id}
|
||
|
||
|
||
def write_falkor(builder: KGBuilder) -> dict[str, int]:
|
||
db = FalkorDB(host="localhost", port=6380)
|
||
if GRAPH_NAME in db.list_graphs():
|
||
db.select_graph(GRAPH_NAME).delete()
|
||
graph = db.select_graph(GRAPH_NAME)
|
||
for node in builder.nodes.values():
|
||
label = re.sub(r"[^A-Za-z0-9_]", "", node["label"]) or "Entity"
|
||
props = graph_safe_props(node)
|
||
graph.query(f"MERGE (n:{label} {{natural_key:$natural_key}}) SET n += $props", {"natural_key": node["natural_key"], "props": props})
|
||
for rel in builder.relations:
|
||
rel_type = re.sub(r"[^A-Z0-9_]", "", rel["relation_type"].upper()) or "RELATED_TO"
|
||
props = graph_safe_props({"natural_key": f"{rel['source']}->{rel_type}->{rel['target']}", **(rel.get("properties") or {})})
|
||
graph.query(
|
||
"""
|
||
MATCH (a {natural_key:$source}), (b {natural_key:$target})
|
||
MERGE (a)-[r:%s]->(b)
|
||
SET r += $props
|
||
""" % rel_type,
|
||
{"source": rel["source"], "target": rel["target"], "props": props},
|
||
)
|
||
return {
|
||
"graph_nodes": graph.query("MATCH (n) RETURN count(n)").result_set[0][0],
|
||
"graph_relations": graph.query("MATCH ()-[r]->() RETURN count(r)").result_set[0][0],
|
||
}
|
||
|
||
|
||
def build() -> dict[str, Any]:
|
||
builder = KGBuilder()
|
||
schema = load_schema()
|
||
amap_cache = load_amap_enrichment_cache()
|
||
driving_metric_cache = load_amap_driving_metric_cache()
|
||
media_index = load_media_resources(builder)
|
||
alias_to_key = seed_attractions(builder, media_index)
|
||
_vehicle_keys, vehicle_group_key = seed_vehicles_from_media(builder, media_index)
|
||
hotel_groups, restaurant_groups = parse_resource_workbooks(builder, media_index)
|
||
parse_existing_route_markdown(builder, alias_to_key, hotel_groups, restaurant_groups, vehicle_group_key)
|
||
parse_small_group_prices(builder)
|
||
parse_independent_prices(builder)
|
||
parse_transfer_quotes(builder)
|
||
parse_sales_scripts(builder)
|
||
apply_amap_enrichment_layer(builder, amap_cache)
|
||
derive_scenic_area_coordinates(builder)
|
||
add_administrative_region_layer(builder)
|
||
apply_driving_metric_layer(builder, driving_metric_cache)
|
||
add_route_region_layer(builder)
|
||
add_route_scenic_area_layer(builder)
|
||
add_slot_candidate_layers(builder)
|
||
add_resource_library_layer(builder)
|
||
write_outputs(builder, schema)
|
||
pg_info = upsert_postgres(builder, schema)
|
||
graph_info = write_falkor(builder)
|
||
summary = {
|
||
"tenant_id": TENANT_ID,
|
||
"project_id": PROJECT_ID,
|
||
"project_name": PROJECT_NAME,
|
||
"graph_name": GRAPH_NAME,
|
||
"nodes": len(builder.nodes),
|
||
"relations": len(builder.relations),
|
||
**pg_info,
|
||
**graph_info,
|
||
"output_dir": str(OUT_DIR),
|
||
}
|
||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
(OUT_DIR / "入库执行摘要.json").write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
return summary
|
||
|
||
|
||
if __name__ == "__main__":
|
||
print(json.dumps(build(), ensure_ascii=False, indent=2))
|