Files
Cloud-Tour-to-Libo/scripts/import_libo_zones.py

538 lines
19 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""Import Libo administrative units and business zones into Postgres.
Two tracks are kept physically separate and are never derived from each other:
* Statutory (法定) — from AMap reverse geocoding + civil-affairs code tables.
Keyed by ``towncode``. Read-only reference data.
* Business (业务) — the玉屏街道 city zone set. Keyed by ``zone_id``. Has no
administrative standing. Zones are irregular polygons; POI membership is
decided by point-in-polygon (NOT bounding boxes — 48% of the 12 zone bboxes
overlap, which would misassign ~37% of POIs).
Active zone set: ``ZS-LB-YUPING-V2`` — 12 surveyed city zones (A01–A12),
replacing the earlier 18-anchor V1 scheme.
Sources (2026-08-06 revised delivery, ``~/Desktop/荔波行政区与片区/``):
* ``1_荔波县_乡镇街道.csv`` -> libo_admin_towns (8)
* ``2_荔波县_村社区.csv`` -> libo_admin_villages (100)
* ``3_玉屏街道_片区.csv`` -> libo_business_zones (12, attrs)
* ``4_玉屏街道_片区边界.geojson`` -> libo_business_zones.geometry + POI 落区
* town-level POI assignment -> amap_spatial_pois.towncode
Usage::
python scripts/import_libo_zones.py --graph-name yunyou_libo
python scripts/import_libo_zones.py --dry-run
"""
from __future__ import annotations
import argparse
import csv
import json
import sys
from pathlib import Path
from typing import Any
import psycopg
from psycopg.rows import dict_row
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from app.config import settings # noqa: E402
HOME = Path.home()
DEFAULT_ZONE_DIR = HOME / "Desktop" / "荔波行政区与片区"
# Statutory town assignment for the full county (all 8 towns), one row per POI.
DEFAULT_TOWN_POI_CSV = (
HOME
/ "Documents"
/ "云游荔波"
/ "outputs"
/ "libo_admin_grid_verified_20260805"
/ "不同业务POI乡镇归属"
/ "荔波县_POI实体_乡镇归属去重_20260805.csv"
)
ZONE_SET_ID = "ZS-LB-YUPING-V2"
# Zone sets superseded by the current import; cleared for the target graph.
LEGACY_ZONE_SET_IDS = ("ZS-LB-YUPING-V1",)
YUPING_TOWNCODE = "522722001000"
SCHEMA_SQL = """
CREATE TABLE IF NOT EXISTS __SCHEMA__.libo_admin_towns (
town_uid TEXT PRIMARY KEY,
civil_adcode TEXT NOT NULL,
name TEXT NOT NULL,
unit_type TEXT NOT NULL,
towncode TEXT NOT NULL UNIQUE,
stat_code TEXT,
community_count INTEGER NOT NULL DEFAULT 0,
village_count INTEGER NOT NULL DEFAULT 0,
unit_total INTEGER NOT NULL DEFAULT 0,
code_note TEXT,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE TABLE IF NOT EXISTS __SCHEMA__.libo_admin_villages (
village_uid TEXT PRIMARY KEY,
parent_civil_adcode TEXT NOT NULL,
town_name TEXT NOT NULL,
parent_type TEXT,
name TEXT NOT NULL,
unit_type TEXT NOT NULL,
alias TEXT,
stat_code TEXT,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE INDEX IF NOT EXISTS idx_libo_admin_villages_parent
ON __SCHEMA__.libo_admin_villages (parent_civil_adcode);
CREATE TABLE IF NOT EXISTS __SCHEMA__.libo_business_zones (
zone_set_id TEXT NOT NULL,
zone_id TEXT NOT NULL,
crs TEXT NOT NULL DEFAULT 'GCJ-02',
zone_name TEXT NOT NULL,
anchor_name TEXT,
anchor_lng DOUBLE PRECISION,
anchor_lat DOUBLE PRECISION,
cap_m DOUBLE PRECISION,
parent_admin_unit_id TEXT,
parent_admin_name TEXT,
poi_count INTEGER NOT NULL DEFAULT 0,
food_count INTEGER NOT NULL DEFAULT 0,
hotel_count INTEGER NOT NULL DEFAULT 0,
transit_count INTEGER NOT NULL DEFAULT 0,
scenic_count INTEGER NOT NULL DEFAULT 0,
area_km2 DOUBLE PRECISION,
dominant_category TEXT,
bank TEXT,
anchor_source TEXT,
geometry JSONB,
geometry_variant TEXT,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
PRIMARY KEY (zone_set_id, zone_id)
);
CREATE TABLE IF NOT EXISTS __SCHEMA__.poi_zone_assignments (
graph_name TEXT NOT NULL,
gaode_poi_id TEXT NOT NULL,
zone_set_id TEXT NOT NULL,
zone_id TEXT NOT NULL,
zone_name TEXT NOT NULL,
method TEXT,
distance_m DOUBLE PRECISION,
review_status TEXT,
computed_at TIMESTAMPTZ,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
PRIMARY KEY (graph_name, gaode_poi_id, zone_set_id)
);
CREATE INDEX IF NOT EXISTS idx_poi_zone_assignments_zone
ON __SCHEMA__.poi_zone_assignments (graph_name, zone_set_id, zone_id);
"""
# Newer schema installs need columns the earlier table version lacked.
ZONE_EXTRA_COLUMNS = (
"scenic_count INTEGER NOT NULL DEFAULT 0",
"area_km2 DOUBLE PRECISION",
"dominant_category TEXT",
"bank TEXT",
)
def read_csv(path: Path) -> list[dict[str, str]]:
with path.open(encoding="utf-8-sig", newline="") as handle:
return list(csv.DictReader(handle))
def as_int(value: str | None) -> int:
try:
return int(float(str(value).strip()))
except (TypeError, ValueError):
return 0
def as_float(value: str | None) -> float | None:
try:
return float(str(value).strip())
except (TypeError, ValueError):
return None
# ── Point-in-polygon (GCJ-02, ray casting) ──────────────────────────────────
def _in_ring(ring: list[list[float]], x: float, y: float) -> bool:
n = len(ring)
inside = False
for i in range(n):
x1, y1 = ring[i]
x2, y2 = ring[(i + 1) % n]
if (y1 > y) != (y2 > y):
xt = (x2 - x1) * (y - y1) / (y2 - y1) + x1
if x < xt:
inside = not inside
return inside
def load_zone_geometry(path: Path) -> dict[str, dict[str, Any]]:
"""zone_id -> {geometry, polys, variant, anchors, area_ha, bank}."""
payload = json.loads(path.read_text(encoding="utf-8"))
result: dict[str, dict[str, Any]] = {}
for feature in payload.get("features", []):
props = feature.get("properties") or {}
zone_id = props.get("zone_id")
if not zone_id:
continue
geom = feature.get("geometry") or {}
polys = (
geom.get("coordinates")
if geom.get("type") == "MultiPolygon"
else [geom.get("coordinates")]
)
result[str(zone_id)] = {
"geometry": geom,
"polys": [p for p in (polys or []) if p],
"variant": props.get("scheme") or props.get("status") or "surveyed",
"anchors": props.get("anchors_gcj02") or [],
"bank": props.get("bank"),
}
return result
def assign_zone(
lng: float,
lat: float,
ordered_zones: list[tuple[str, str, list]],
) -> tuple[str, str]:
for zone_id, zone_name, polys in ordered_zones:
for rings in polys:
if _in_ring(rings[0], lng, lat) and not any(
_in_ring(hole, lng, lat) for hole in rings[1:]
):
return zone_id, zone_name
return "", ""
def import_towns(cur, schema: str, rows: list[dict[str, str]]) -> int:
for row in rows:
cur.execute(
f"""INSERT INTO {schema}.libo_admin_towns
(town_uid, civil_adcode, name, unit_type, towncode, stat_code,
community_count, village_count, unit_total, code_note, updated_at)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s, now())
ON CONFLICT (town_uid) DO UPDATE SET
civil_adcode = EXCLUDED.civil_adcode,
name = EXCLUDED.name,
unit_type = EXCLUDED.unit_type,
towncode = EXCLUDED.towncode,
stat_code = EXCLUDED.stat_code,
community_count = EXCLUDED.community_count,
village_count = EXCLUDED.village_count,
unit_total = EXCLUDED.unit_total,
code_note = EXCLUDED.code_note,
updated_at = now()""",
(
row["内部行政区ID"],
row["民政行政区划代码9位"],
row["名称"],
row["类型"],
row["高德towncode12位"],
row.get("2023统计用乡级代码12位"),
as_int(row.get("社区数")),
as_int(row.get("行政村数")),
as_int(row.get("基层单元合计")),
row.get("代码差异提示"),
),
)
return len(rows)
def import_villages(cur, schema: str, rows: list[dict[str, str]]) -> int:
for row in rows:
cur.execute(
f"""INSERT INTO {schema}.libo_admin_villages
(village_uid, parent_civil_adcode, town_name, parent_type,
name, unit_type, alias, stat_code, updated_at)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s, now())
ON CONFLICT (village_uid) DO UPDATE SET
parent_civil_adcode = EXCLUDED.parent_civil_adcode,
town_name = EXCLUDED.town_name,
parent_type = EXCLUDED.parent_type,
name = EXCLUDED.name,
unit_type = EXCLUDED.unit_type,
alias = EXCLUDED.alias,
stat_code = EXCLUDED.stat_code,
updated_at = now()""",
(
row["基层单元ID"],
row["父级民政代码"],
row["乡镇/街道"],
row.get("父级类型"),
row["现行名称"],
row["单元类型"],
row.get("别名/旧名线索"),
row.get("统计用区划代码12位"),
),
)
return len(rows)
def import_zones(
cur,
schema: str,
rows: list[dict[str, str]],
geometry: dict[str, dict[str, Any]],
) -> tuple[int, list[str]]:
without_face: list[str] = []
for row in rows:
zone_id = row["片区编号"]
face = geometry.get(zone_id)
if not face:
without_face.append(zone_id)
anchors = face["anchors"] if face else []
# 面标签落点:多锚点取均值,否则留空。
if anchors:
anchor_lng = sum(a[0] for a in anchors) / len(anchors)
anchor_lat = sum(a[1] for a in anchors) / len(anchors)
else:
anchor_lng = anchor_lat = None
cur.execute(
f"""INSERT INTO {schema}.libo_business_zones
(zone_set_id, zone_id, crs, zone_name, anchor_name,
anchor_lng, anchor_lat, cap_m, parent_admin_unit_id,
parent_admin_name, poi_count, food_count, hotel_count,
transit_count, scenic_count, area_km2, dominant_category,
bank, anchor_source, geometry, geometry_variant, updated_at)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s, now())
ON CONFLICT (zone_set_id, zone_id) DO UPDATE SET
crs = EXCLUDED.crs,
zone_name = EXCLUDED.zone_name,
anchor_name = EXCLUDED.anchor_name,
anchor_lng = EXCLUDED.anchor_lng,
anchor_lat = EXCLUDED.anchor_lat,
cap_m = EXCLUDED.cap_m,
parent_admin_unit_id = EXCLUDED.parent_admin_unit_id,
parent_admin_name = EXCLUDED.parent_admin_name,
poi_count = EXCLUDED.poi_count,
food_count = EXCLUDED.food_count,
hotel_count = EXCLUDED.hotel_count,
transit_count = EXCLUDED.transit_count,
scenic_count = EXCLUDED.scenic_count,
area_km2 = EXCLUDED.area_km2,
dominant_category = EXCLUDED.dominant_category,
bank = EXCLUDED.bank,
anchor_source = EXCLUDED.anchor_source,
geometry = EXCLUDED.geometry,
geometry_variant = EXCLUDED.geometry_variant,
updated_at = now()""",
(
ZONE_SET_ID,
zone_id,
"GCJ-02",
row["片区名称"],
None,
anchor_lng,
anchor_lat,
None, # cap_m: 面几何为真多边形,不再有半径闸
"LB-522722001",
row.get("所属街道") or "玉屏街道",
as_int(row.get("POI数")),
as_int(row.get("美食")),
as_int(row.get("酒店")),
as_int(row.get("交通")),
as_int(row.get("景区")),
as_float(row.get("面积km²")),
row.get("主导业态"),
face["bank"] if face else None,
"玉屏街道城区实测划分(A方案)",
json.dumps(face["geometry"], ensure_ascii=False) if face else None,
face["variant"] if face else None,
),
)
return len(rows), without_face
def import_assignments(
cur,
schema: str,
graph_name: str,
geometry: dict[str, dict[str, Any]],
zone_names: dict[str, str],
) -> tuple[int, int]:
"""Reassign every project POI to a city zone by point-in-polygon."""
ordered_zones = [
(zone_id, zone_names.get(zone_id, ""), geometry[zone_id]["polys"])
for zone_id in sorted(geometry)
]
cur.execute(
f"""SELECT gaode_poi_id, lng, lat
FROM {schema}.amap_spatial_pois
WHERE graph_name=%s AND lng IS NOT NULL AND lat IS NOT NULL""",
(graph_name,),
)
poi_rows = cur.fetchall()
# Rebuild the active set cleanly: drop old rows for this graph+set first.
cur.execute(
f"""DELETE FROM {schema}.poi_zone_assignments
WHERE graph_name=%s AND zone_set_id=%s""",
(graph_name, ZONE_SET_ID),
)
assigned = 0
for row in poi_rows:
zone_id, zone_name = assign_zone(
float(row["lng"]), float(row["lat"]), ordered_zones
)
if not zone_id:
continue
cur.execute(
f"""INSERT INTO {schema}.poi_zone_assignments
(graph_name, gaode_poi_id, zone_set_id, zone_id, zone_name,
method, review_status, computed_at, updated_at)
VALUES (%s,%s,%s,%s,%s,%s,%s, now(), now())
ON CONFLICT (graph_name, gaode_poi_id, zone_set_id) DO UPDATE SET
zone_id = EXCLUDED.zone_id,
zone_name = EXCLUDED.zone_name,
method = EXCLUDED.method,
review_status = EXCLUDED.review_status,
computed_at = now(),
updated_at = now()""",
(
graph_name,
row["gaode_poi_id"],
ZONE_SET_ID,
zone_id,
zone_name,
"point_in_polygon",
"自动归属",
),
)
assigned += 1
return len(poi_rows), assigned
def backfill_towncode(
cur,
schema: str,
graph_name: str,
rows: list[dict[str, str]],
) -> tuple[int, int]:
"""Write the statutory town assignment onto the POI table."""
cur.execute(
f"ALTER TABLE {schema}.amap_spatial_pois "
"ADD COLUMN IF NOT EXISTS towncode TEXT"
)
cur.execute(
f"ALTER TABLE {schema}.amap_spatial_pois "
"ADD COLUMN IF NOT EXISTS town_name TEXT"
)
cur.execute(
f"CREATE INDEX IF NOT EXISTS idx_amap_spatial_pois_towncode "
f"ON {schema}.amap_spatial_pois (graph_name, towncode)"
)
updated = 0
for row in rows:
towncode = (row.get("高德towncode") or "").strip()
town_name = (row.get("高德乡镇街道") or "").strip()
if not towncode or not town_name:
continue
cur.execute(
f"""UPDATE {schema}.amap_spatial_pois
SET towncode=%s, town_name=%s
WHERE graph_name=%s AND gaode_poi_id=%s""",
(towncode, town_name, graph_name, row["高德POI_ID"]),
)
updated += cur.rowcount
cur.execute(
f"""SELECT count(*) AS n FROM {schema}.amap_spatial_pois
WHERE graph_name=%s AND towncode IS NULL""",
(graph_name,),
)
missing = cur.fetchone()["n"]
return updated, missing
def purge_legacy_zone_sets(cur, schema: str, graph_name: str) -> None:
"""Remove superseded zone sets so no wrong V1 data lingers."""
for legacy in LEGACY_ZONE_SET_IDS:
cur.execute(
f"DELETE FROM {schema}.poi_zone_assignments "
"WHERE graph_name=%s AND zone_set_id=%s",
(graph_name, legacy),
)
cur.execute(
f"DELETE FROM {schema}.libo_business_zones WHERE zone_set_id=%s",
(legacy,),
)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--graph-name", default="yunyou_libo")
parser.add_argument("--zone-dir", type=Path, default=DEFAULT_ZONE_DIR)
parser.add_argument("--town-poi-csv", type=Path, default=DEFAULT_TOWN_POI_CSV)
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
schema = settings.db_schema
towns_csv = args.zone_dir / "1_荔波县_乡镇街道.csv"
villages_csv = args.zone_dir / "2_荔波县_村社区.csv"
zones_csv = args.zone_dir / "3_玉屏街道_片区.csv"
geojson = args.zone_dir / "4_玉屏街道_片区边界.geojson"
for path in (towns_csv, villages_csv, zones_csv, geojson, args.town_poi_csv):
if not path.exists():
print(f"! 源文件不存在: {path}")
return 1
towns = read_csv(towns_csv)
villages = read_csv(villages_csv)
zones = read_csv(zones_csv)
town_pois = read_csv(args.town_poi_csv)
geometry = load_zone_geometry(geojson)
zone_names = {row["片区编号"]: row["片区名称"] for row in zones}
print(
f"读取: 乡级 {len(towns)} · 村社区 {len(villages)} · 片区 {len(zones)} · "
f"面几何 {len(geometry)} · 乡镇归属 {len(town_pois)} · 目标集 {ZONE_SET_ID}"
)
if args.dry_run:
print("dry-run,未写库")
return 0
with psycopg.connect(settings.database_url, row_factory=dict_row) as connection:
with connection.cursor() as cur:
cur.execute(SCHEMA_SQL.replace("__SCHEMA__", schema))
for ddl in ZONE_EXTRA_COLUMNS:
cur.execute(
f"ALTER TABLE {schema}.libo_business_zones "
f"ADD COLUMN IF NOT EXISTS {ddl}"
)
n_towns = import_towns(cur, schema, towns)
n_villages = import_villages(cur, schema, villages)
n_zones, without_face = import_zones(cur, schema, zones, geometry)
n_pois, n_assign = import_assignments(
cur, schema, args.graph_name, geometry, zone_names
)
updated, missing = backfill_towncode(
cur, schema, args.graph_name, town_pois
)
purge_legacy_zone_sets(cur, schema, args.graph_name)
connection.commit()
print(f"✓ libo_admin_towns {n_towns}")
print(f"✓ libo_admin_villages {n_villages}")
print(f"✓ libo_business_zones {n_zones}(无面几何: {', '.join(without_face) or '无'})")
print(f"✓ poi_zone_assignments {n_assign} / {n_pois} 落区(其余为城区外)")
print(f"✓ amap_spatial_pois 乡镇回填 {updated} 行,仍缺 towncode {missing} 行")
print(f"✓ 已清除旧片区集: {', '.join(LEGACY_ZONE_SET_IDS)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())