"""정적 HTML 어댑터 — **사장님이 확정한 자기 홈페이지** 전용.""" import asyncio import json import re import time from html import unescape from html.parser import HTMLParser from typing import Optional from urllib.parse import urljoin, urlparse from urllib.robotparser import RobotFileParser import httpx from common.enums import LinkChannel, PlaceCategory from common.logger import LOG from services.collector.base import CollectedFact, CollectedMedia, RawSource # 우리를 밝히는 UA. BOT_NAME = "o2o-web4ai-collector" USER_AGENT = f"{BOT_NAME}/1.0 (+owner-confirmed URL only; contact: owner of the listed business)" REQUEST_TIMEOUT = 20 MAX_BYTES = 3 * 1024 * 1024 # 본문 상한. MAX_MEDIA = 30 ROBOTS_TTL = 3600 # 호스트당 robots.txt 캐시 수명(초) HEADERS = { "User-Agent": USER_AGENT, "Accept": "text/html,application/xhtml+xml", "Accept-Language": "ko-KR,ko;q=0.9", } # 이 호스트들은 이 어댑터가 절대 건드리지 않는다. _DENY_HOSTS = ( "naver.com", "naver.me", # 플레이스·지도·블로그·예약 — robots Disallow: / "kakao.com", "daum.net", # 카카오맵 — robots Disallow, 내부 API 406 "yanolja.com", "goodchoice.kr", # OTA — 403 + 민사 10억 선례. "dailyhotel.com", "catchtable.co.kr", "airbnb.co.kr", "airbnb.com", "booking.com", "agoda.com", "expedia.co.kr", "instagram.com", "facebook.com", # Graph API(사장님 OAuth)로만 "baemin.com", "yogiyo.co.kr", "coupangeats.com", "diningcode.com", "siksinhot.com", "mangoplate.com", ) # schema.org 타입 → 우리가 관심 있는 업소인가. _LB_TYPES = { "localbusiness", "lodgingbusiness", "hotel", "motel", "resort", "hostel", "bedandbreakfast", "campground", "vacationrental", "restaurant", "cafeorcoffeeshop", "foodestablishment", "bakery", "barorpub", "touristattraction", "store", } # 부정 표현. _NEGATIONS = ("불가", "없음", "없슴", "미제공", "제공하지", "안됨", "안 됨", "불가능", "금지", "not available") # 편의시설 표기 → 업종 스키마의 bool key. _AMENITY_BOOL = { "주차": "parking", "발렛": "parking", "parking": "parking", "와이파이": "wifi", "무선인터넷": "wifi", "무선 인터넷": "wifi", "wifi": "wifi", "wi-fi": "wifi", "바비큐": "bbq_available", "바베큐": "bbq_available", "bbq": "bbq_available", "취사": "cooking_allowed", "조리": "cooking_allowed", "조식": "breakfast", "breakfast": "breakfast", "반려동물": "pet_allowed", "애견": "pet_allowed", "pet": "pet_allowed", "픽업": "pickup_service", "유아용품": "baby_amenities", "아기": "baby_amenities", "에어컨": "has_aircon", "냉방": "has_aircon", "주방": "has_kitchen", "취사시설": "has_kitchen", "테라스": "terrace", "포장": "takeout", "테이크아웃": "takeout", "배달": "delivery", "콘센트": "power_outlet", "휠체어": "wheelchair_accessible", "장애인": "wheelchair_accessible", } # 업종마다 "영업시간" 자리를 차지하는 key. _HOURS_KEYS = ("business_hours", "operating_hours", "reception_hours") # 본문에서 유일하게 허용하는 정규식 — 체크인·체크아웃. _CHECKIN = re.compile(r"체크\s*인[^0-9]{0,12}((?:오전|오후)?\s*\d{1,2}\s*[:시]\s*\d{0,2})") _CHECKOUT = re.compile(r"체크\s*아웃[^0-9]{0,12}((?:오전|오후)?\s*\d{1,2}\s*[:시]\s*\d{0,2})") _robots_cache: dict[str, tuple[float, Optional[RobotFileParser]]] = {} class _Extract(HTMLParser): """한 번 훑으면서 JSON-LD · meta · 본문 텍스트를 같이 걷는다.""" def __init__(self): super().__init__(convert_charrefs=True) self.ld_raw: list[str] = [] self.meta: dict[str, str] = {} self.title: str = "" self._text: list[str] = [] self._skip = 0 # script/style 안쪽 깊이 self._in_ld = False self._in_title = False def handle_starttag(self, tag, attrs): a = {k.lower(): (v or "") for k, v in attrs} if tag in ("script", "style", "noscript"): self._skip += 1 if tag == "script" and a.get("type", "").lower().strip() == "application/ld+json": self._in_ld = True self.ld_raw.append("") elif tag == "title": self._in_title = True elif tag == "meta": name = (a.get("property") or a.get("name") or "").lower().strip() content = a.get("content", "").strip() if name and content and name not in self.meta: self.meta[name] = content def handle_endtag(self, tag): if tag in ("script", "style", "noscript"): self._skip = max(0, self._skip - 1) self._in_ld = False elif tag == "title": self._in_title = False def handle_data(self, data): if self._in_ld: self.ld_raw[-1] += data elif self._in_title: self.title += data elif not self._skip: s = data.strip() if s: self._text.append(s) @property def text(self) -> str: return " ".join(self._text) class StaticHtmlAdapter: """사장님 확정 홈페이지 → fact·사진 후보.""" id = "static_html" def can_handle(self, url: str) -> bool: """http(s) 이고, 전용 어댑터·수집 불가 결론이 난 호스트가 아니면 손든다.""" u = (url or "").strip().lower() if not u.startswith(("http://", "https://")): return False host = urlparse(u).netloc.split(":")[0] if not host: return False return not any(host == d or host.endswith("." + d) for d in _DENY_HOSTS) async def fetch(self, url: str, category: Optional[PlaceCategory] = None) -> RawSource: channel = self._channel(url) allowed, why = await self._robots_allows(url) if not allowed: # 우회하지 않는다. return RawSource.failure(url, self.id, f"robots.txt 가 수집을 허용하지 않는다: {why}", channel) try: html = await self._get_html(url) except Exception as ex: return RawSource.failure(url, self.id, str(ex), channel) parser = _Extract() try: parser.feed(html) except Exception as ex: # 깨진 HTML 도 여기까지 온 만큼은 쓴다 LOG.w(f"[static_html] HTML 파싱 도중 중단: {ex} ({url})") nodes = self._ld_nodes(parser.ld_raw) biz = self._pick_business(nodes) facts = self._to_facts(biz, parser) media = self._to_media(biz, parser, url) if not facts and not media: return RawSource.failure( url, self.id, "구조화 데이터(JSON-LD·OpenGraph)가 없어 확신할 수 있는 값을 찾지 못했다", channel ) LOG.i(f"[static_html] {parser.title.strip()[:40] or url} — fact {len(facts)}건 · 사진 {len(media)}장") return RawSource( url=url, adapter_id=self.id, channel=channel, text=parser.text[:20000], facts=facts, media=media, ) # robots async def _robots_allows(self, url: str) -> tuple[bool, str]: """robots.txt 를 우리 UA 이름으로 판정한다.""" parts = urlparse(url) origin = f"{parts.scheme}://{parts.netloc}" now = time.monotonic() cached = _robots_cache.get(origin) if cached and now - cached[0] < ROBOTS_TTL: rp = cached[1] else: rp = await self._load_robots(origin) _robots_cache[origin] = (now, rp) if rp is None: return True, "" if rp.can_fetch(BOT_NAME, url): return True, "" return False, f"{origin}/robots.txt 가 {BOT_NAME} 에게 이 경로를 금지한다" async def _load_robots(self, origin: str) -> Optional[RobotFileParser]: try: async with httpx.AsyncClient(timeout=10, follow_redirects=True) as client: res = await client.get(f"{origin}/robots.txt", headers=HEADERS) if res.status_code != 200: return None # HTML 을 돌려주는 서버가 있다(robots.txt 가 없어서 SPA 로 떨어지는 경우). body = res.text if body.lstrip()[:15].lower().startswith((" str: """HTML 본문.""" try: async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT, follow_redirects=True) as client: async with client.stream("GET", url, headers=HEADERS) as res: if res.status_code == 429: raise RuntimeError("서버가 요청을 제한했다(429)") if res.status_code != 200: raise RuntimeError(f"HTTP {res.status_code}") ctype = res.headers.get("content-type", "").lower() if "html" not in ctype: raise RuntimeError(f"HTML 이 아니다: {ctype or '알 수 없음'}") chunks: list[bytes] = [] total = 0 async for chunk in res.aiter_bytes(): chunks.append(chunk) total += len(chunk) if total >= MAX_BYTES: break raw = b"".join(chunks) enc = res.encoding or "utf-8" return raw.decode(enc, errors="replace") except httpx.TimeoutException: raise RuntimeError(f"{REQUEST_TIMEOUT}s 안에 응답이 없다") except httpx.HTTPError as ex: raise RuntimeError(f"네트워크 오류: {ex}") # JSON-LD @staticmethod def _ld_nodes(raw_blocks: list[str]) -> list[dict]: """JSON-LD 블록들을 평평한 dict 목록으로.""" out: list[dict] = [] def walk(node): if isinstance(node, list): for x in node: walk(x) elif isinstance(node, dict): out.append(node) for key in ("@graph", "mainEntity", "itemListElement"): if key in node: walk(node[key]) for block in raw_blocks: block = block.strip() if not block: continue try: walk(json.loads(block)) except (json.JSONDecodeError, ValueError): continue # 깨진 블록 하나가 전체를 죽이지 않는다 return out @staticmethod def _pick_business(nodes: list[dict]) -> Optional[dict]: """업소를 가리키는 노드 하나.""" cands = [] for n in nodes: types = n.get("@type") types = [types] if isinstance(types, str) else (types or []) if any(str(t).lower() in _LB_TYPES for t in types): cands.append(n) return max(cands, key=len) if cands else None # fact def _to_facts(self, biz: Optional[dict], parser: _Extract) -> list[CollectedFact]: """스키마에 있는 key 만 만든다.""" facts: list[CollectedFact] = [] biz = biz or {} # 소개문(intro)은 수집하지 않는다 — allow_llm=True, LLM 의 출력 칸이다(절대규칙 7). # 영업시간. hours = self._hours(biz) if hours: for key in _HOURS_KEYS: facts.append(CollectedFact(key=key, value=hours)) # 체크인·체크아웃. for ld_key, our_key, pattern in ( ("checkinTime", "check_in_time", _CHECKIN), ("checkoutTime", "check_out_time", _CHECKOUT), ): value = self._clean(biz.get(ld_key)) if not value: m = pattern.search(parser.text) value = re.sub(r"\s+", "", m.group(1)) if m else "" if value: facts.append(CollectedFact(key=our_key, value=value[:40])) # 가격대 · 결제수단 — 값이 그대로 문자열인 것들. for ld_key, our_key in (("priceRange", "price_range"), ("paymentAccepted", "payment_methods")): value = self._clean(biz.get(ld_key)) if value: facts.append(CollectedFact(key=our_key, value=value[:200])) facts.extend(self._bool_facts(biz)) facts.extend(self._menu_facts(biz)) return facts def _bool_facts(self, biz: dict) -> list[CollectedFact]: """편의시설 → bool fact.""" labels: list[str] = [] for item in self._as_list(biz.get("amenityFeature")): if isinstance(item, dict): name = self._clean(item.get("name")) # LocationFeatureSpecification.value 가 false 면 '없다'는 뜻이다. if item.get("value") is False: name = f"{name} 없음" if name: labels.append(name) elif isinstance(item, str): labels.append(item) # schema.org 가 전용 필드로 주는 것들. for ld_key, our_key in (("petsAllowed", "pet_allowed"), ("smokingAllowed", "smoking")): raw = biz.get(ld_key) if isinstance(raw, bool): labels.append(f"{ld_key} {'' if raw else '없음'}") _AMENITY_BOOL.setdefault(ld_key.lower(), our_key) elif isinstance(raw, str) and raw.strip(): labels.append(f"{ld_key} {raw}") _AMENITY_BOOL.setdefault(ld_key.lower(), our_key) out: list[CollectedFact] = [] seen: set[str] = set() for text in labels: low = text.lower() negated = any(neg in low for neg in _NEGATIONS) for needle, key in _AMENITY_BOOL.items(): if needle not in low or key in seen: continue seen.add(key) out.append(CollectedFact(key=key, value="false" if negated else "true")) return out def _menu_facts(self, biz: dict) -> list[CollectedFact]: """hasMenu → 단위(메뉴) 스코프 fact.""" sections = self._as_list(biz.get("hasMenu")) + self._as_list(biz.get("menu")) items: list[dict] = [] for sec in sections: if isinstance(sec, dict): items += [x for x in self._as_list(sec.get("hasMenuItem")) if isinstance(x, dict)] for sub in self._as_list(sec.get("hasMenuSection")): if isinstance(sub, dict): items += [x for x in self._as_list(sub.get("hasMenuItem")) if isinstance(x, dict)] out: list[CollectedFact] = [] seen: list[str] = [] for item in items: name = self._clean(item.get("name")) if not name or name in seen: continue seen.append(name) out.append(CollectedFact(key="menu_name", value=name, scope="unit", unit_name=name)) offer = item.get("offers") offer = offer[0] if isinstance(offer, list) and offer else offer price = self._clean(offer.get("price")) if isinstance(offer, dict) else "" if price: out.append(CollectedFact(key="menu_price", value=price[:40], scope="unit", unit_name=name)) # menu_intro 도 allow_llm=True 라 수집하지 않는다 — 메뉴 설명 문장은 LLM 이 쓰는 칸이다. return out def _hours(self, biz: dict) -> str: """openingHours(문자열) / openingHoursSpecification(구조) 둘 다 받는다.""" parts: list[str] = [] for x in self._as_list(biz.get("openingHours")): s = self._clean(x) if s and s not in parts: parts.append(s) for spec in self._as_list(biz.get("openingHoursSpecification")): if not isinstance(spec, dict): continue days = ", ".join( str(d).rsplit("/", 1)[-1] for d in self._as_list(spec.get("dayOfWeek")) if d ) span = f"{self._clean(spec.get('opens'))}~{self._clean(spec.get('closes'))}".strip("~") text = f"{days} {span}".strip() if text and text not in parts: parts.append(text) return " · ".join(parts)[:500] # 사진 def _to_media(self, biz: Optional[dict], parser: _Extract, base_url: str) -> list[CollectedMedia]: """JSON-LD image 를 앞에, OpenGraph og:image 를 뒤에.""" raw: list[str] = [] for item in self._as_list((biz or {}).get("image")): if isinstance(item, str): raw.append(item) elif isinstance(item, dict): raw.append(str(item.get("url") or item.get("contentUrl") or "")) for key in ("og:image", "og:image:secure_url", "twitter:image"): if parser.meta.get(key): raw.append(parser.meta[key]) out: list[CollectedMedia] = [] seen: set[str] = set() for src in raw: src = (src or "").strip() if not src or src.startswith("data:"): continue absolute = urljoin(base_url, src) if absolute in seen: continue seen.add(absolute) out.append(CollectedMedia(origin_url=absolute)) if len(out) >= MAX_MEDIA: break return out # 잡동사니 @staticmethod def _channel(url: str) -> LinkChannel: """사장님 확정 URL 이므로 기본은 공식 홈페이지다.""" u = (url or "").lower() if any(b in u for b in ("blog.", "/blog", "tistory.com", "brunch.co.kr")): return LinkChannel.BLOG return LinkChannel.OFFICIAL_SITE @staticmethod def _as_list(value) -> list: if value is None: return [] return value if isinstance(value, list) else [value] @staticmethod def _clean(value) -> str: """문자열로 정규화.""" if value is None or isinstance(value, (dict, list)): return "" text = unescape(str(value)) return re.sub(r"\s+", " ", text).strip()