- 템플릿: Planner.astro, lib/plan.ts·tour.ts, styles/plan.css, planStrings, pages plan·en/plan·404. Base 내비 회복 일정 → /plan, 언어 짝 /plan↔/en/plan, 옛 주소 리다이렉트(vercel.json), 사이트맵 - 워커 planner 단계(recovery 다음): scripts/build_planner_data.mjs 가 업종별 기본 규칙표(scripts/template/planner/procedures.plastic|derm.json)에 병원 시술 페이지 원문(recoveryNotes)을 matchKeywords 로 붙이고, 장소는 briefs/<clinic>/planner.places.json(큐레이션) 또는 범용 기본표(관광공사 기준 좌표)로 만든다 - 브리프: viewclinic·oracle 큐레이션 장소. 빈 템플릿(관광 데이터 없음)도 빌드·검증 통과(plan.test 14건) - 이전 세션의 미커밋 작업(피부과 수집·OCR·게이트·언어 스위치, stay 페이지 제거)도 이 커밋에 함께 들어감. docs/prd 변경은 제외 Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
586 lines
30 KiB
Python
586 lines
30 KiB
Python
#!/usr/bin/env python3
|
|
"""의료관광 연결 데이터 수집 (한국관광공사 TourAPI 다국어). 병원별로 돈다.
|
|
|
|
병원 좌표를 중심으로 숙박·음식점·관광지·문화시설·쇼핑을 모으고, 진행 중인 축제를
|
|
거리와 함께 붙여 프론트가 쓸 JSON 으로 낸다. 영문(EngService2)과 국문(KorService2)을
|
|
한 번에 만들 수 있다.
|
|
|
|
왜 사전 수집인가
|
|
영문 서비스는 도심 커버리지가 얇고 관광지 항목의 대부분이 의료기관이라 필터링이 많이
|
|
필요하다. 그 처리를 수집 시점에 끝내고 프론트는 정제된 결과만 그린다. 미팅 데모 중
|
|
외부 API 상태에 화면이 종속되지도 않는다.
|
|
|
|
무엇을 만들지 않는가
|
|
없는 값을 채우지 않는다. 평점·리뷰·영업시간은 TourAPI 가 주지 않으므로 비워 두고,
|
|
Google Places 승인 후 별도 단계에서 붙인다. 식이 적합성은 병원 입력이며 여기서
|
|
판정하지 않는다. 좌표도 지어내지 않는다. 주소를 좌표로 바꾸지 못하면 종료 코드 1 로 멈춘다.
|
|
|
|
기준 좌표를 정하는 순서
|
|
1. --lat --lng 를 직접 주면 그대로 쓴다 (coordSource: 명령 인자).
|
|
2. --address 와 --name 을 주면 네이버 지역검색(NAVER_CLIENT_ID/SECRET)으로 상호를 찾고,
|
|
결과의 도로명 주소가 준 주소의 도로명·번지와 같을 때만 그 좌표를 쓴다.
|
|
네이버가 못 찾으면 TourAPI 키워드 검색(국문·영문)으로 한 번 더 찾는다.
|
|
3. 아무 인자도 없으면 .env 의 TOUR_API_ORIGIN_LAT/LNG (초기 샘플 사이트용).
|
|
어느 경우든 meta.origin.coordSource 에 출처와 날짜를 적는다.
|
|
|
|
출력은 서포터즈 사이트(Astro)의 데이터다. 화면은 src/pages/stay.astro(국문)와
|
|
src/pages/en/stay.astro(영문)다. 외국인 환자와 보호자가 보는 곳이다.
|
|
|
|
# 워커가 부르는 형태 (팩트 시트 주소 기준, 영문·국문 동시)
|
|
python3 scripts/fetch_medical_tourism.py --clinic oracle --name "오라클피부과" \
|
|
--address "서울 강남구 선릉로 612 한일빌딩" --label "오라클피부과 (선릉로 612)" \
|
|
--out-en ~/supporters-builds/oracle/tourism/medicalTourism.json \
|
|
--out-ko ~/supporters-builds/oracle/tourism/medicalTourismKo.json
|
|
# 좌표를 직접 줄 때
|
|
python3 scripts/fetch_medical_tourism.py --lat 37.5049823 --lng 127.0254182 --label "..." --out-en a.json --out-ko b.json
|
|
# 이전 방식 (.env 좌표, 언어 하나)
|
|
python3 scripts/fetch_medical_tourism.py --lang ko --out supporters/src/data/medicalTourismKo.json
|
|
"""
|
|
import argparse, io, json, math, os, re, ssl, sys, time, urllib.parse, urllib.request
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
# 영문 관광지(76)의 대부분은 관광지가 아니라 의료관광 등록업소다. 반경 5km 100건 중 92건이
|
|
# 분류코드 A02020500 = "Medical Tourism Sites"(categoryCode2 로 확인)이며 병원과 유치 에이전시가
|
|
# 여기 들어 있다. 제목 키워드로는 "하이안과", "닥파인더코리아" 같은 것을 놓치므로 코드로 거른다.
|
|
CAT_MEDICAL_TOURISM = "A02020500"
|
|
# 스파·찜질방은 수술 후 회복 여정에 맞으므로 관광이 아니라 별도 범주로 뺀다.
|
|
CAT_WELLNESS = ("A02020300", "A02020400") # Hot Springs & Spa, Jjimjilbang
|
|
|
|
# 다국어 ContentTypeId. 국문과 코드가 다르다.
|
|
# 반경은 목적에 따라 다르다. 회복기에 매일 가는 곳(숙박·식사·스파)은 가깝게, 관광은 넓게 본다.
|
|
SEASON = {3: "spring", 4: "spring", 5: "spring", 6: "summer", 7: "summer", 8: "summer",
|
|
9: "autumn", 10: "autumn", 11: "autumn", 12: "winter", 1: "winter", 2: "winter"}
|
|
|
|
# ContentTypeId 는 다국어와 국문이 다르다. 같은 코드로 부르면 엉뚱한 것이 온다.
|
|
TYPES_EN = {
|
|
"76": ("attraction", "10000"),
|
|
"78": ("culture", "10000"),
|
|
"79": ("shopping", "10000"),
|
|
"80": ("stay", "5000"),
|
|
"82": ("restaurant", "5000"),
|
|
}
|
|
TYPES_KO = {
|
|
"12": ("attraction", "10000"),
|
|
"14": ("culture", "10000"),
|
|
"38": ("shopping", "10000"),
|
|
"32": ("stay", "5000"),
|
|
"39": ("restaurant", "5000"),
|
|
}
|
|
# 언어별 타입 표는 collect() 가 lang 에 맞게 고른다.
|
|
|
|
|
|
def ssl_ctx():
|
|
try:
|
|
import certifi; return ssl.create_default_context(cafile=certifi.where())
|
|
except Exception:
|
|
return ssl.create_default_context()
|
|
|
|
|
|
def load_env(path):
|
|
""".env 파일을 읽고, 같은 이름의 환경변수가 있으면 그 값을 우선한다.
|
|
워커(run.mjs)나 클라우드 실행은 파일 없이 환경변수만 줄 수 있다."""
|
|
d = {}
|
|
if os.path.exists(path):
|
|
for line in io.open(path, encoding="utf-8"):
|
|
line = line.strip()
|
|
if line and not line.startswith("#") and "=" in line:
|
|
k, v = line.split("=", 1); d[k.strip()] = v.strip().strip('"').strip("'")
|
|
for k, v in os.environ.items():
|
|
if k.startswith(("TOUR_API_", "NAVER_CLIENT_")) and v:
|
|
d[k] = v
|
|
return d
|
|
|
|
|
|
def haversine_km(lat1, lng1, lat2, lng2):
|
|
r = 6371.0
|
|
p1, p2 = math.radians(lat1), math.radians(lat2)
|
|
dp, dl = math.radians(lat2 - lat1), math.radians(lng2 - lng1)
|
|
a = math.sin(dp / 2) ** 2 + math.cos(p1) * math.cos(p2) * math.sin(dl / 2) ** 2
|
|
return 2 * r * math.asin(math.sqrt(a))
|
|
|
|
|
|
def call(env, op, extra, rows=100, retries=3, page=1):
|
|
base, svc, key = env["TOUR_API_BASE"], env["TOUR_API_SERVICE"], env["TOUR_API_KEY"]
|
|
p = {"MobileOS": "ETC", "MobileApp": "INFINITH", "_type": "json",
|
|
"numOfRows": str(rows), "pageNo": str(page), **extra}
|
|
url = f"{base}/{svc}/{op}?serviceKey={key}&" + urllib.parse.urlencode(p)
|
|
for i in range(retries):
|
|
try:
|
|
with urllib.request.urlopen(url, timeout=30, context=ssl_ctx()) as r:
|
|
body = r.read().decode("utf-8", "replace")
|
|
if body.lstrip().startswith("<"):
|
|
raise RuntimeError(f"XML 응답(인증·한도 확인): {body[:160]}")
|
|
j = json.loads(body)["response"]
|
|
if j["header"].get("resultCode") != "0000":
|
|
raise RuntimeError(f"{j['header'].get('resultCode')} {j['header'].get('resultMsg')}")
|
|
b = j.get("body", {}); items = b.get("items")
|
|
return b.get("totalCount", 0), ([] if not items or items == "" else items["item"])
|
|
except Exception as e:
|
|
if i == retries - 1: raise
|
|
time.sleep(1.5 * (i + 1))
|
|
|
|
|
|
def call_all(env, op, extra, max_pages=6, rows=100):
|
|
"""전 페이지를 모은다. 관광지는 거리순 상위 100건이 거의 다 의료관광업소라 한 페이지로는 부족하다."""
|
|
got, total = [], None
|
|
for page in range(1, max_pages + 1):
|
|
total, items = call(env, op, extra, rows=rows, page=page)
|
|
got += items
|
|
if not items or len(got) >= (total or 0): break
|
|
time.sleep(0.2)
|
|
return total, got
|
|
|
|
|
|
def walk_min(m):
|
|
"""도보 분. 80m/분(4.8km/h) 기준. 참고 사이트의 표기를 역산해 맞춘 값이다
|
|
(135m→2분, 248m→3분, 297m→4분, 566m→7분)."""
|
|
return max(1, round(m / 80))
|
|
|
|
|
|
def drive_min(km):
|
|
"""차 이동 분. 직선거리 km x 1.2. 역시 역산값이다(25km→30분, 33km→40분)."""
|
|
return max(1, round(km * 1.2))
|
|
|
|
|
|
def travel(m):
|
|
"""이동 수단과 소요 시간.
|
|
2km 까지는 걸을 수 있는 거리로 본다(약 25분). 그 위는 차·지하철이다.
|
|
짧은 거리의 차 시간은 주차와 신호 때문에 의미가 없으므로 최소 5분으로 둔다."""
|
|
if m is None: return None
|
|
if m <= 2000: return {"mode": "walk", "minutes": walk_min(m), "label": f"{walk_min(m)} min walk"}
|
|
mins = max(5, drive_min(m / 1000))
|
|
return {"mode": "drive", "minutes": mins, "label": f"{mins} min by car"}
|
|
|
|
|
|
def fetch_overview(env, content_id, cache):
|
|
"""detailCommon2 의 overview(설명문)와 homepage. locationBasedList2 는 주지 않는다.
|
|
설명문이 없으면 카드가 이름과 주소뿐이라 사람이 고를 수 없다."""
|
|
if content_id in cache: return cache[content_id]
|
|
try:
|
|
_, rows = call(env, "detailCommon2", {"contentId": content_id}, rows=1)
|
|
except Exception:
|
|
cache[content_id] = ("", ""); return cache[content_id]
|
|
if not rows: cache[content_id] = ("", ""); return cache[content_id]
|
|
r = rows[0]
|
|
ov = re.sub(r"<[^>]+>", " ", r.get("overview") or "")
|
|
ov = re.sub(r"\s+", " ", ov).strip()
|
|
hp = re.sub(r"<[^>]+>", " ", r.get("homepage") or "").strip()
|
|
hp = (re.search(r"https?://\S+", hp) or [None])[0] if hp else ""
|
|
cache[content_id] = (ov, hp or "")
|
|
return cache[content_id]
|
|
|
|
|
|
def fetch_image(env, content_id, cache):
|
|
"""locationBasedList2 의 firstimage 가 비어 있을 때 detailImage2 로 한 번 더 찾는다.
|
|
이미지 전용 오퍼레이션이라 목록 API 에 없는 사진이 여기 있다(도산공원 7장, 몽마르뜨공원 5장).
|
|
실측: 사진 없던 38건 중 15건을 여기서 살렸다. 나머지 23건은 TourAPI 에 사진이 없다."""
|
|
key = ("img", content_id)
|
|
if key in cache: return cache[key]
|
|
try:
|
|
_, rows = call(env, "detailImage2", {"contentId": content_id, "imageYN": "Y"}, rows=5)
|
|
except Exception:
|
|
cache[key] = ""; return ""
|
|
url = ""
|
|
for r in rows or []:
|
|
u = (r.get("originimgurl") or r.get("smallimageurl") or "").strip()
|
|
if u: url = u.replace("http://", "https://", 1); break
|
|
cache[key] = url
|
|
return url
|
|
|
|
|
|
def call_kor(env, op, extra, rows=10):
|
|
"""국문 서비스(KorService2). 영문 항목에 사진이 없어도 국문 항목에는 있는 경우가 있다.
|
|
같은 한국관광공사 사진이라 출처가 달라지지 않는다."""
|
|
p = {"MobileOS": "ETC", "MobileApp": "INFINITH", "_type": "json",
|
|
"numOfRows": str(rows), "pageNo": "1", **extra}
|
|
url = f"{env['TOUR_API_BASE']}/KorService2/{op}?serviceKey={env['TOUR_API_KEY']}&" + urllib.parse.urlencode(p)
|
|
try:
|
|
with urllib.request.urlopen(url, timeout=25, context=ssl_ctx()) as r:
|
|
body = r.read().decode("utf-8", "replace")
|
|
if body.lstrip().startswith("<"): return []
|
|
j = json.loads(body)["response"]
|
|
if j["header"].get("resultCode") != "0000": return []
|
|
it = j["body"].get("items")
|
|
return [] if not it or it == "" else it["item"]
|
|
except Exception:
|
|
return []
|
|
|
|
|
|
def korean_name(title):
|
|
m = re.search(r"\(([^)]+)\)\s*$", title or "")
|
|
return (m.group(1) if m else (title or "")).strip()
|
|
|
|
|
|
def fetch_image_kor(env, p):
|
|
"""국문 서비스에서 같은 이름의 항목을 찾아 사진을 가져온다. 실측 23건 중 10건이 여기서 나왔다."""
|
|
ko = korean_name(p.get("title"))
|
|
if not re.search(r"[가-힣]", ko): return ""
|
|
rows = []
|
|
if p.get("lat"):
|
|
rows = call_kor(env, "locationBasedList2",
|
|
{"mapX": str(p["lng"]), "mapY": str(p["lat"]), "radius": "300", "keyword": ko})
|
|
if not rows: rows = call_kor(env, "searchKeyword2", {"keyword": ko})
|
|
key = ko.replace(" ", "")
|
|
for r in rows:
|
|
if key in (r.get("title") or "").replace(" ", ""):
|
|
u = (r.get("firstimage") or r.get("firstimage2") or "").strip()
|
|
if u: return u.replace("http://", "https://", 1)
|
|
return ""
|
|
|
|
|
|
def fetch_image_og(url):
|
|
"""업체 공식 홈페이지의 대표 이미지(og:image). 업체가 스스로 올린 사진이라 목록에 쓰기에 무리가 없다.
|
|
검색 결과 이미지를 긁어 쓰면 식당·블로그·언론사의 저작물이라 환자용 페이지에 올릴 수 없다."""
|
|
if not url or not url.startswith("http"): return ""
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
|
with urllib.request.urlopen(req, timeout=12, context=ssl_ctx()) as r:
|
|
head = r.read(120000).decode("utf-8", "replace")
|
|
base = r.geturl()
|
|
except Exception:
|
|
return ""
|
|
m = re.search(r'<meta[^>]+(?:property|name)=["\']og:image["\'][^>]*content=["\']([^"\']+)', head, re.I) \
|
|
or re.search(r'<meta[^>]+content=["\']([^"\']+)["\'][^>]+(?:property|name)=["\']og:image', head, re.I)
|
|
if not m: return ""
|
|
u = m.group(1).strip()
|
|
if u.startswith("//"): u = "https:" + u
|
|
elif u.startswith("/"):
|
|
from urllib.parse import urlparse
|
|
pr = urlparse(base); u = f"{pr.scheme}://{pr.netloc}{u}"
|
|
return u if u.startswith("https://") else ""
|
|
|
|
|
|
def usable_image(url):
|
|
"""실제로 이미지가 오는지 열어 본다. og:image 가 이미지가 아니거나(빈 content-type)
|
|
인스타 CDN 처럼 핫링크를 403 으로 막는 경우가 있어 화면에서 빈 칸이 된다."""
|
|
if not url: return False
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
|
with urllib.request.urlopen(req, timeout=10, context=ssl_ctx()) as r:
|
|
return r.status == 200 and (r.headers.get("content-type") or "").startswith("image")
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def fetch_image_naver(env, p):
|
|
"""마지막 폴백. Naver 이미지 검색 API.
|
|
주의: 결과는 제3자 저작물이다. 내부 시연 목적으로만 쓰고, 공개 배포 전에는
|
|
imageSource 가 'naver' 인 항목을 걷어내거나 라이선스가 있는 사진으로 교체한다.
|
|
구글 HTML 크롤링 대신 이걸 쓰는 이유는 공식 API 라 차단되지 않고 결과가 일정해서다."""
|
|
cid, sec = env.get("NAVER_CLIENT_ID"), env.get("NAVER_CLIENT_SECRET")
|
|
if not (cid and sec): return ""
|
|
ko = korean_name(p.get("title"))
|
|
if not ko: return ""
|
|
# 지역을 붙이면 정확도가 오르지만 결과가 아예 없는 경우가 있다. 넓혀 가며 시도한다.
|
|
plain = re.sub(r"\s*\(.*\)\s*", "", p.get("title") or "").strip()
|
|
for q in (f"{ko} 서울", ko, plain):
|
|
if not q: continue
|
|
qs = urllib.parse.urlencode({"query": q, "display": "3", "sort": "sim"})
|
|
req = urllib.request.Request("https://openapi.naver.com/v1/search/image.json?" + qs,
|
|
headers={"X-Naver-Client-Id": cid, "X-Naver-Client-Secret": sec})
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=15, context=ssl_ctx()) as r:
|
|
items = json.loads(r.read().decode()).get("items", [])
|
|
except Exception:
|
|
continue
|
|
for it in items:
|
|
u = (it.get("link") or "").strip()
|
|
if u.startswith("https://"): return u
|
|
return ""
|
|
|
|
|
|
def bucket(it):
|
|
"""분류코드로 범주를 정한다. None 이면 버린다."""
|
|
cat3 = (it.get("cat3") or "").strip()
|
|
if cat3 == CAT_MEDICAL_TOURISM: return None
|
|
if cat3 in CAT_WELLNESS: return "wellness"
|
|
return "keep"
|
|
|
|
|
|
def clean(it, lat, lng):
|
|
"""TourAPI 원본에서 화면이 쓸 필드만 남긴다. mapx=경도, mapy=위도 (이름과 반대라 자주 틀린다)."""
|
|
try:
|
|
ilng, ilat = float(it.get("mapx") or 0), float(it.get("mapy") or 0)
|
|
except ValueError:
|
|
ilng = ilat = 0.0
|
|
dist = it.get("dist")
|
|
return {
|
|
"id": it.get("contentid"),
|
|
"title": (it.get("title") or "").strip(),
|
|
"address": (it.get("addr1") or "").strip(),
|
|
# TourAPI 는 이미지 URL 을 http 로 준다. https 사이트에서 혼합 콘텐츠로 차단되므로 올린다.
|
|
# tong.visitkorea.or.kr 은 https 로도 같은 파일을 준다(실측 확인).
|
|
"image": (lambda u: u.replace("http://", "https://", 1) if u else None)(
|
|
it.get("firstimage") or it.get("firstimage2") or None),
|
|
"imageSource": "tourapi" if (it.get("firstimage") or it.get("firstimage2")) else "",
|
|
"tel": (it.get("tel") or "").strip() or None,
|
|
"lat": ilat or None, "lng": ilng or None,
|
|
"distanceM": round(float(dist)) if dist else (
|
|
round(haversine_km(lat, lng, ilat, ilng) * 1000) if ilat and ilng else None),
|
|
"travel": travel(round(float(dist)) if dist else (
|
|
round(haversine_km(lat, lng, ilat, ilng) * 1000) if ilat and ilng else None)),
|
|
# detailCommon2 로 따로 채운다(locationBasedList2 는 설명문을 주지 않는다).
|
|
"overview": "", "homepage": "",
|
|
# TourAPI 가 주지 않는 값. Google Places 승인 후 채운다. 지금은 비워 둔다.
|
|
"rating": None, "reviewCount": None, "openingHours": None,
|
|
}
|
|
|
|
|
|
# ---------- 기준 좌표 ----------
|
|
|
|
def road_key(address):
|
|
"""주소에서 도로명과 건물번호를 뽑는다. "서울 강남구 선릉로 612 한일빌딩" → ("선릉로", "612").
|
|
비교에만 쓴다. 없으면 (None, None)."""
|
|
m = re.search(r"([가-힣A-Za-z0-9]*?(?:대로|로|길))\s*(\d+(?:-\d+)?)(?!\s*길)", address or "")
|
|
return (m.group(1), m.group(2)) if m else (None, None)
|
|
|
|
|
|
def addr_match(want, got):
|
|
"""검색 결과 주소가 준 주소와 같은 도로명·번지인지. 도로명과 번지 사이 공백은 무시하고,
|
|
번지 바로 뒤에 숫자나 '-' 가 붙으면(612 와 6120, 612-1) 다른 곳으로 본다.
|
|
"봉은사로 107 1, 3-5층" 처럼 공백 뒤에 층이 오는 것은 같은 곳이다."""
|
|
road, num = road_key(want)
|
|
if not road: return False
|
|
return re.search(re.escape(road) + r"\s*" + re.escape(num) + r"(?![\d-])", got or "") is not None
|
|
|
|
|
|
def strip_tags(s):
|
|
return re.sub(r"<[^>]+>", "", s or "").strip()
|
|
|
|
|
|
def geocode_naver(env, name, address):
|
|
"""네이버 지역검색으로 상호를 찾아 좌표를 얻는다. mapx/mapy 는 WGS84 경위도 x 10^7 이다.
|
|
주소만으로는 결과가 없다(실측). 상호를 넣고, 결과의 도로명 주소가 준 주소와 같을 때만 받는다."""
|
|
cid, sec = env.get("NAVER_CLIENT_ID"), env.get("NAVER_CLIENT_SECRET")
|
|
if not (cid and sec): return None, "NAVER_CLIENT_ID/SECRET 없음"
|
|
road, num = road_key(address)
|
|
queries = []
|
|
if road: queries.append(f"{name} {road} {num}")
|
|
queries += [f"{name} {address}", name]
|
|
tried = []
|
|
for q in queries:
|
|
qs = urllib.parse.urlencode({"query": q, "display": "5", "sort": "random"})
|
|
req = urllib.request.Request("https://openapi.naver.com/v1/search/local.json?" + qs,
|
|
headers={"X-Naver-Client-Id": cid, "X-Naver-Client-Secret": sec})
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=15, context=ssl_ctx()) as r:
|
|
items = json.loads(r.read().decode()).get("items", [])
|
|
except Exception as e:
|
|
tried.append(f"'{q}': 요청 실패 {e}"); continue
|
|
for it in items:
|
|
got = it.get("roadAddress") or it.get("address") or ""
|
|
if not addr_match(address, got): continue
|
|
try:
|
|
lng, lat = int(it["mapx"]) / 1e7, int(it["mapy"]) / 1e7
|
|
except (KeyError, ValueError):
|
|
continue
|
|
if not (33 <= lat <= 39 and 124 <= lng <= 132): continue
|
|
return {"lat": lat, "lng": lng, "matched": strip_tags(it.get("title")), "matchedAddress": got,
|
|
"coordSource": f"네이버 지역검색 '{strip_tags(it.get('title'))}' ({got}) {time.strftime('%Y-%m-%d')}"}, ""
|
|
tried.append(f"'{q}': {len(items)}건, 주소 일치 없음")
|
|
return None, "; ".join(tried)
|
|
|
|
|
|
def geocode_tourapi(env, name, address):
|
|
"""TourAPI 키워드 검색(국문·영문)으로 병원 항목을 찾는다. 등재된 병원만 나오므로 보조 수단이다."""
|
|
tried = []
|
|
for svc in ("KorService2", "EngService2"):
|
|
e = dict(env, TOUR_API_SERVICE=svc)
|
|
try:
|
|
_, rows = call(e, "searchKeyword2", {"keyword": name}, rows=20, retries=1)
|
|
except Exception as ex:
|
|
tried.append(f"{svc}: 요청 실패 {ex}"); continue
|
|
for it in rows:
|
|
got = it.get("addr1") or ""
|
|
if not addr_match(address, got): continue
|
|
try:
|
|
lng, lat = float(it.get("mapx") or 0), float(it.get("mapy") or 0)
|
|
except ValueError:
|
|
continue
|
|
if not (33 <= lat <= 39 and 124 <= lng <= 132): continue
|
|
return {"lat": lat, "lng": lng, "matched": (it.get("title") or "").strip(), "matchedAddress": got,
|
|
"coordSource": f"한국관광공사 TourAPI {svc} contentid {it.get('contentid')} ({got}) {time.strftime('%Y-%m-%d')}"}, ""
|
|
tried.append(f"{svc}: {len(rows)}건, 주소 일치 없음")
|
|
return None, "; ".join(tried)
|
|
|
|
|
|
def resolve_origin(env, a):
|
|
"""기준 좌표와 출처. 못 정하면 (None, 이유)."""
|
|
if a.lat is not None or a.lng is not None:
|
|
if a.lat is None or a.lng is None: return None, "--lat 와 --lng 는 함께 줘야 합니다"
|
|
return {"lat": a.lat, "lng": a.lng, "address": a.address or "",
|
|
"coordSource": f"명령 인자 --lat/--lng {time.strftime('%Y-%m-%d')}"}, ""
|
|
if a.address:
|
|
if not a.name: return None, "--address 에는 --name(검색할 상호)이 필요합니다. 주소만으로는 지역검색 결과가 없습니다"
|
|
road, num = road_key(a.address)
|
|
if not road: return None, f"주소에서 도로명·번지를 찾지 못했습니다: {a.address}"
|
|
found, why1 = geocode_naver(env, a.name, a.address)
|
|
if found: return dict(found, address=a.address), ""
|
|
found, why2 = geocode_tourapi(env, a.name, a.address)
|
|
if found: return dict(found, address=a.address), ""
|
|
return None, f"주소를 좌표로 바꾸지 못했습니다. 네이버 지역검색: {why1} / TourAPI: {why2}"
|
|
lat, lng = env.get("TOUR_API_ORIGIN_LAT"), env.get("TOUR_API_ORIGIN_LNG")
|
|
if not (lat and lng): return None, "기준 좌표가 없습니다. --address --name 또는 --lat --lng 를 주거나 .env 의 TOUR_API_ORIGIN_LAT/LNG 를 채우세요"
|
|
return {"lat": float(lat), "lng": float(lng), "address": "",
|
|
"coordSource": f".env TOUR_API_ORIGIN_LAT/LNG {time.strftime('%Y-%m-%d')}"}, ""
|
|
|
|
|
|
# ---------- 수집 ----------
|
|
|
|
def collect(env, lang, origin, a):
|
|
"""한 언어의 데이터를 모아 dict 로 돌려준다. 파일은 쓰지 않는다."""
|
|
types = TYPES_KO if lang == "ko" else TYPES_EN
|
|
env = dict(env, TOUR_API_SERVICE="KorService2" if lang == "ko" else "EngService2")
|
|
lat, lng = origin["lat"], origin["lng"]
|
|
print(f"\n[{lang}] {env['TOUR_API_SERVICE']} · 기준 {lat}, {lng}")
|
|
|
|
out = {"places": {}, "festivals": [], "filters": {}, "meta": {}}
|
|
filtered_log = {}
|
|
|
|
wellness = []
|
|
for ct, (name, radius) in types.items():
|
|
total, rows = call_all(env, "locationBasedList2",
|
|
{"mapX": str(lng), "mapY": str(lat), "radius": radius,
|
|
"contentTypeId": ct, "arrange": "E"})
|
|
kept, dropped = [], []
|
|
for it in rows:
|
|
b = bucket(it)
|
|
if b is None:
|
|
dropped.append((it.get("title") or "").strip()); continue
|
|
(wellness if b == "wellness" else kept).append(clean(it, lat, lng))
|
|
kept.sort(key=lambda x: x["distanceM"] if x["distanceM"] is not None else 10 ** 9)
|
|
out["places"][name] = kept[:a.per_type]
|
|
filtered_log[name] = {"apiTotal": total, "fetched": len(rows), "radiusM": int(radius),
|
|
"droppedMedicalTourism": len(dropped), "kept": len(kept),
|
|
"shown": len(out["places"][name]), "droppedSample": dropped[:5]}
|
|
print(f" {name:<11} r={radius:>5}m · API {total:>4} · 받음 {len(rows):>3} · 의료관광업소 제외 {len(dropped):>3} · 담음 {len(out['places'][name])}")
|
|
wellness.sort(key=lambda x: x["distanceM"] if x["distanceM"] is not None else 10 ** 9)
|
|
out["places"]["wellness"] = wellness[:a.per_type]
|
|
filtered_log["wellness"] = {"note": "관광지에서 스파·찜질방 분류만 분리", "shown": len(out["places"]["wellness"])}
|
|
print(f" {'wellness':<11} 스파·찜질방 분리 · 담음 {len(out['places']['wellness'])}")
|
|
|
|
# 영문 축제는 areacode 필드가 비어 있어 지역 필터를 쓸 수 없다. 좌표로 거리를 계산해 거른다.
|
|
today = time.strftime("%Y%m%d")
|
|
total, rows = call(env, "searchFestival2", {"eventStartDate": today, "arrange": "A"}, rows=300)
|
|
near, nocoord = [], 0
|
|
for it in rows:
|
|
c = clean(it, lat, lng)
|
|
if c["lat"] is None: nocoord += 1; continue
|
|
km = haversine_km(lat, lng, c["lat"], c["lng"])
|
|
if km > a.festival_km: continue
|
|
sd = it.get("eventstartdate") or ""
|
|
c.update(startDate=sd, endDate=it.get("eventenddate"),
|
|
distanceKm=round(km, 1), month=(int(sd[4:6]) if len(sd) == 8 else None),
|
|
season=SEASON.get(int(sd[4:6]), "") if len(sd) == 8 else "")
|
|
near.append(c)
|
|
near.sort(key=lambda x: x["startDate"] or "")
|
|
out["festivals"] = near[:a.per_type]
|
|
print(f" festival API {total:>4}건 · 받음 {len(rows):>3} · 좌표없음 {nocoord} · {a.festival_km:.0f}km 이내 {len(near)}")
|
|
|
|
# 설명문은 항목마다 detailCommon2 를 한 번씩 부른다. 카드에 이름과 주소만 있으면 고를 수 없다.
|
|
cache: dict = {}
|
|
targets = [p for v in out["places"].values() for p in v] + out["festivals"]
|
|
print(f" 설명문 수집 {len(targets)}건", end="", flush=True)
|
|
for i, p in enumerate(targets):
|
|
if not p.get("id"): continue
|
|
ov, hp = fetch_overview(env, p["id"], cache)
|
|
p["overview"] = ov[:600]
|
|
p["homepage"] = hp
|
|
if not p.get("image"):
|
|
# 1) 같은 언어 detailImage2 → 2) 국문 서비스 → 3) 업체 홈페이지 og:image → 4) Naver 이미지 검색
|
|
src = ""
|
|
img = fetch_image(env, p["id"], cache)
|
|
if img: src = "tourapi"
|
|
if not img and lang == "en":
|
|
img = fetch_image_kor(env, p); src = "tourapi-kor" if img else src
|
|
if not img:
|
|
cand = fetch_image_og(p.get("homepage") or "")
|
|
if cand and usable_image(cand): img, src = cand, "homepage"
|
|
if not img and not a.no_naver_images:
|
|
img = fetch_image_naver(env, p); src = "naver" if img else src
|
|
p["image"] = img or None
|
|
p["imageSource"] = src
|
|
if i % 20 == 0: print(".", end="", flush=True)
|
|
time.sleep(0.12)
|
|
got = sum(1 for p in targets if p["overview"])
|
|
pic = sum(1 for p in targets if p.get("image"))
|
|
print(f" 완료 · 설명문 {got}/{len(targets)} · 사진 {pic}/{len(targets)}")
|
|
|
|
# 도보 시간 구간별 건수. 화면의 필터 칩이 이 수를 쓴다.
|
|
allp = [p for v in out["places"].values() for p in v]
|
|
# 누적 기준이다. "15분 이내"는 5분 이내도 포함한다. 사람이 고르는 방식에 맞춘다.
|
|
def walk_upto(mins):
|
|
return sum(1 for p in allp if p.get("travel") and p["travel"]["mode"] == "walk"
|
|
and p["travel"]["minutes"] <= mins)
|
|
out["filters"] = {
|
|
"total": len(allp),
|
|
"walk5": walk_upto(5),
|
|
"walk15": walk_upto(15),
|
|
"walkAll": sum(1 for p in allp if p.get("travel") and p["travel"]["mode"] == "walk"),
|
|
"drive": sum(1 for p in allp if not p.get("travel") or p["travel"]["mode"] != "walk"),
|
|
}
|
|
|
|
out["meta"] = {
|
|
"lang": lang,
|
|
"clinic": a.clinic or "",
|
|
"source": "한국관광공사 TourAPI " + env["TOUR_API_SERVICE"],
|
|
"fetchedAt": time.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
"origin": {"lat": lat, "lng": lng, "label": a.label or origin.get("matched") or origin.get("address") or "",
|
|
"address": origin.get("address", ""), "coordSource": origin["coordSource"]},
|
|
"radiusByType": {n: int(r) for n, r in types.values()}, "festivalRadiusKm": a.festival_km,
|
|
"counts": filtered_log,
|
|
# 화면이 한계를 그대로 표시할 수 있게 데이터에 적어 둔다. 없는 값을 채우지 않는다.
|
|
"limits": [
|
|
"영문 서비스는 도심 숙박·음식점 커버리지가 국문보다 얇다.",
|
|
"관광지 항목 중 분류코드 A02020500 \"Medical Tourism Sites\"(병원·유치업체)는 제외했다.",
|
|
"평점·리뷰·영업시간은 TourAPI 가 제공하지 않는다. Google Places 승인 후 채운다.",
|
|
"영문 축제는 areacode 필드가 비어 있어 좌표 거리로 걸렀다.",
|
|
"수술 전후 식이 적합성은 병원이 입력한 가이드로만 표시한다. 이 파일에는 없다.",
|
|
],
|
|
}
|
|
return out
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description="병원 주변 관광 데이터 수집 (TourAPI)")
|
|
ap.add_argument("--clinic", default="", help="병원 id (meta.clinic)")
|
|
ap.add_argument("--name", default="", help="지역검색에 쓸 상호. --address 와 함께 준다")
|
|
ap.add_argument("--address", default="", help="병원 주소. 네이버 지역검색으로 좌표를 찾는다")
|
|
ap.add_argument("--lat", type=float, default=None); ap.add_argument("--lng", type=float, default=None)
|
|
ap.add_argument("--label", default="", help="화면에 보일 기준점 이름 (meta.origin.label)")
|
|
ap.add_argument("--out-en", default="", help="영문 JSON 경로 (절대 또는 현재 폴더 기준)")
|
|
ap.add_argument("--out-ko", default="", help="국문 JSON 경로")
|
|
ap.add_argument("--festival-km", type=float, default=60.0, help="이 거리 안의 축제만 담는다")
|
|
ap.add_argument("--per-type", type=int, default=24, help="타입별 최대 건수")
|
|
ap.add_argument("--no-naver-images", action="store_true", help="네이버 이미지 검색 폴백을 끈다 (공개 배포용)")
|
|
ap.add_argument("--lang", choices=["en", "ko"], default="en", help="(이전 방식) --out 하나만 낼 때의 언어")
|
|
ap.add_argument("--out", default="", help="(이전 방식) 저장소 루트 기준 경로")
|
|
a = ap.parse_args()
|
|
|
|
env = load_env(os.path.join(ROOT, ".env"))
|
|
for k in ("TOUR_API_KEY", "TOUR_API_BASE"):
|
|
if not env.get(k): sys.exit(f"{k} 가 비어 있습니다 (.env 또는 환경변수)")
|
|
|
|
origin, why = resolve_origin(env, a)
|
|
if not origin:
|
|
print(f"기준 좌표 실패: {why}", file=sys.stderr); sys.exit(1)
|
|
print(f"기준 좌표 {origin['lat']}, {origin['lng']} · {origin['coordSource']}")
|
|
|
|
jobs = []
|
|
if a.out_en: jobs.append(("en", os.path.abspath(a.out_en)))
|
|
if a.out_ko: jobs.append(("ko", os.path.abspath(a.out_ko)))
|
|
if not jobs:
|
|
rel = a.out or ("supporters/src/data/medicalTourism.json" if a.lang == "en" else "supporters/src/data/medicalTourismKo.json")
|
|
jobs.append((a.lang, os.path.join(ROOT, rel)))
|
|
|
|
for lang, path in jobs:
|
|
out = collect(env, lang, origin, a)
|
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
io.open(path, "w", encoding="utf-8").write(json.dumps(out, ensure_ascii=False, indent=2))
|
|
print(f"→ {path} · 장소 {out['filters']['total']} · 축제 {len(out['festivals'])}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|