여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
74 lines
2.5 KiB
Python
74 lines
2.5 KiB
Python
"""업소 조사 응답 해석 — 쓸 수 있는 항목만 남긴다."""
|
|
import json
|
|
import re
|
|
|
|
from common.logger import LOG
|
|
|
|
_FENCE_RE = re.compile(r"^\s*```(?:json)?\s*|\s*```\s*$", re.MULTILINE)
|
|
|
|
|
|
def _payload_text(payload: dict) -> str:
|
|
choices = payload.get("choices") or []
|
|
if not choices or not isinstance(choices[0], dict):
|
|
return ""
|
|
return ((choices[0].get("message") or {}).get("content")) or ""
|
|
|
|
|
|
def _clean_source(value) -> dict | None:
|
|
if not isinstance(value, dict):
|
|
return None
|
|
url = (value.get("url") or "").strip()
|
|
if not url.startswith("http"):
|
|
return None
|
|
return {"name": (value.get("name") or url).strip(), "url": url}
|
|
|
|
|
|
def _name_tokens(name: str) -> list[str]:
|
|
"""상호를 대조에 쓸 조각으로."""
|
|
parts = [p for p in re.split(r"[\s,·・/|]+", name or "") if len(p) >= 2]
|
|
return parts or ([name] if name else [])
|
|
|
|
|
|
def parse_items(payload: dict, place_name: str, limit: int) -> tuple[list[dict], list[str]]:
|
|
"""(쓸 수 있는 항목, 버린 이유)."""
|
|
text = _FENCE_RE.sub("", _payload_text(payload)).strip()
|
|
if not text:
|
|
return [], ["응답이 비었다"]
|
|
|
|
try:
|
|
parsed = json.loads(text)
|
|
except json.JSONDecodeError as ex:
|
|
LOG.w(f"[research] JSON 이 아니다: {ex}")
|
|
return [], [f"JSON 파싱 실패: {ex}"]
|
|
|
|
rows = parsed.get("items") if isinstance(parsed, dict) else parsed
|
|
if not isinstance(rows, list):
|
|
return [], ["items 배열이 없다"]
|
|
|
|
tokens = _name_tokens(place_name)
|
|
items: list[dict] = []
|
|
dropped: list[str] = []
|
|
|
|
for row in rows[: limit * 2]: # 버려질 것을 감안해 넉넉히 보되, 채택은 limit 까지다
|
|
if len(items) >= limit:
|
|
break
|
|
if not isinstance(row, dict):
|
|
dropped.append("항목이 객체가 아니다")
|
|
continue
|
|
sentence = str(row.get("text") or "").strip()
|
|
if not sentence:
|
|
dropped.append("빈 문장")
|
|
continue
|
|
source = _clean_source(row.get("source"))
|
|
if not source:
|
|
dropped.append(f"출처 없음: {sentence[:30]}")
|
|
continue
|
|
# 상호 대조(머리주석).
|
|
haystack = f"{sentence} {source['url']} {source['name']}".lower()
|
|
if not any(tok.lower() in haystack for tok in tokens):
|
|
dropped.append(f"상호가 없다: {sentence[:30]}")
|
|
continue
|
|
items.append({"text": sentence, "source": source})
|
|
|
|
return items, dropped
|