"""업소 조사 응답 해석 — 쓸 수 있는 항목만 남긴다.""" import json import re from common.logger import LOG _FENCE_RE = re.compile(r"^\s*```(?:json)?\s*|\s*```\s*$", re.MULTILINE) def _payload_text(payload: dict) -> str: choices = payload.get("choices") or [] if not choices or not isinstance(choices[0], dict): return "" return ((choices[0].get("message") or {}).get("content")) or "" def _clean_source(value) -> dict | None: if not isinstance(value, dict): return None url = (value.get("url") or "").strip() if not url.startswith("http"): return None return {"name": (value.get("name") or url).strip(), "url": url} def _name_tokens(name: str) -> list[str]: """상호를 대조에 쓸 조각으로.""" parts = [p for p in re.split(r"[\s,·・/|]+", name or "") if len(p) >= 2] return parts or ([name] if name else []) def parse_items(payload: dict, place_name: str, limit: int) -> tuple[list[dict], list[str]]: """(쓸 수 있는 항목, 버린 이유).""" text = _FENCE_RE.sub("", _payload_text(payload)).strip() if not text: return [], ["응답이 비었다"] try: parsed = json.loads(text) except json.JSONDecodeError as ex: LOG.w(f"[research] JSON 이 아니다: {ex}") return [], [f"JSON 파싱 실패: {ex}"] rows = parsed.get("items") if isinstance(parsed, dict) else parsed if not isinstance(rows, list): return [], ["items 배열이 없다"] tokens = _name_tokens(place_name) items: list[dict] = [] dropped: list[str] = [] for row in rows[: limit * 2]: # 버려질 것을 감안해 넉넉히 보되, 채택은 limit 까지다 if len(items) >= limit: break if not isinstance(row, dict): dropped.append("항목이 객체가 아니다") continue sentence = str(row.get("text") or "").strip() if not sentence: dropped.append("빈 문장") continue source = _clean_source(row.get("source")) if not source: dropped.append(f"출처 없음: {sentence[:30]}") continue # 상호 대조(머리주석). haystack = f"{sentence} {source['url']} {source['name']}".lower() if not any(tok.lower() in haystack for tok in tokens): dropped.append(f"상호가 없다: {sentence[:30]}") continue items.append({"text": sentence, "source": source}) return items, dropped