여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
80 lines
3.9 KiB
Python
80 lines
3.9 KiB
Python
"""확정된 NOL 링크의 수집 원문에서 공개 안내 섹션만 전달한다."""
|
|
import re
|
|
from urllib.parse import urlsplit
|
|
|
|
|
|
def nol_stay_guide(url: str, raw) -> dict:
|
|
try:
|
|
parsed = urlsplit(url)
|
|
except ValueError:
|
|
return {}
|
|
if parsed.scheme != "https" or parsed.hostname != "nol.yanolja.com":
|
|
return {}
|
|
if not re.fullmatch(r"/stay/domestic/\d+/?", parsed.path):
|
|
return {}
|
|
text = raw.get("text") if isinstance(raw, dict) else None
|
|
if not isinstance(text, str):
|
|
return {}
|
|
|
|
labels = {"시설/서비스": "service", "이용 안내": "policy", "예약 공지": "reservation"}
|
|
# 수집기가 붙인 경계로만 나눈다.
|
|
sections = re.split(r"(?m)^\[(숙소 소개|시설/서비스|이용 안내|예약 공지)\]\s*\n", text)
|
|
guide = {}
|
|
for label, body in zip(sections[1::2], sections[2::2]):
|
|
if label not in labels:
|
|
continue
|
|
lines = body.strip().splitlines()
|
|
if lines and lines[0].strip() == label:
|
|
lines.pop(0)
|
|
# 제목 중복과 UI 버튼만 제외한다.
|
|
body = "\n".join(line for line in lines if line.strip() != "전체보기").strip()
|
|
# NOL 은 원 플랫폼 브랜드다 — 사장님 사이트에 그대로 실으면 남의 고객센터·정책을 우리 것처럼 안내하게 된다.
|
|
body = re.sub(r"NOL\s*", "", body)
|
|
if body:
|
|
guide[labels[label]] = body
|
|
if guide:
|
|
guide["fields"] = structured_fields(guide)
|
|
return guide
|
|
|
|
|
|
def structured_fields(guide: dict) -> list[dict]:
|
|
"""확실한 표기만 구조화한다."""
|
|
policy = guide.get("policy", "")
|
|
service = guide.get("service", "")
|
|
reservation = guide.get("reservation", "")
|
|
lines = {line.strip().lstrip("- ") for line in (policy + "\n" + reservation).splitlines()}
|
|
facilities = {line.strip() for line in service.splitlines()}
|
|
fields = []
|
|
|
|
def add(key, label, value, group="rules", note=None):
|
|
fields.append(dict(key=key, label=label, value=value, group=group,
|
|
**({"note": note} if note else {})))
|
|
|
|
for token, key, label in (("체크인", "check_in_time", "체크인 시간"),
|
|
("체크아웃", "check_out_time", "체크아웃 시간")):
|
|
matches = set(re.findall(rf"{token}\s+([0-2]\d:[0-5]\d)(?!\d)", policy))
|
|
if len(matches) == 1:
|
|
value = matches.pop()
|
|
if int(value[:2]) < 24:
|
|
add(key, label, value)
|
|
fees = [re.fullmatch(r"전 연령 동일 1인당\s*([\d,]+)\s*(만)?원", line) for line in lines]
|
|
amounts = {int(m[1].replace(",", "")) * (10000 if m[2] else 1) for m in fees if m}
|
|
if len(amounts) == 1:
|
|
add("extra_person_fee", "인원 추가 요금", f"{amounts.pop():,}원", note="1인당 · 전 연령 동일")
|
|
if "반려동물 입실금지" in lines and not any("반려동물 입실가능" == line for line in lines):
|
|
add("pet_allowed", "반려동물 동반", "불가")
|
|
if "전 구역 금연" in lines:
|
|
add("smoking", "흡연", "불가", "facilities", "전 구역 금연")
|
|
for token, key, label in (("주차가능", "parking", "주차"), ("와이파이", "wifi", "와이파이"),
|
|
("취사가능", "cooking_allowed", "취사")):
|
|
if token in facilities:
|
|
restrictions = [line.strip().lstrip("- ") for line in reservation.splitlines()
|
|
if "조리금지" in line or "조리 금지" in line]
|
|
add(key, label, "가능", "facilities",
|
|
"\n".join(restrictions) if key == "cooking_allowed" and restrictions else None)
|
|
known = [name for name in ("욕조", "개별 화장실", "주방", "테라스/발코니", "OTT (스트리밍 서비스)",
|
|
"다이닝룸", "벽난로", "어메니티", "카페형룸") if name in facilities]
|
|
if known:
|
|
add("facilities", "부대시설", ", ".join(known), "facilities")
|
|
return fields
|