여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
86 lines
2.9 KiB
Python
86 lines
2.9 KiB
Python
"""FAQ 목표 수 채우기 — 생성된 FAQ 가 모자라면 카탈로그에서 겹치지 않는 공통 질문을 고른다."""
|
|
import re
|
|
from dataclasses import dataclass
|
|
from typing import Iterable, Optional
|
|
|
|
from common.faq_catalog import FaqCatalog
|
|
|
|
# 사이트에 싣는 FAQ 목표 수.
|
|
FAQ_TARGET = 20
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExistingFaq:
|
|
question: str
|
|
source_fact_ids: Optional[list] = None
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class FillFaq:
|
|
catalog_id: str
|
|
question: str
|
|
answer: str
|
|
|
|
|
|
def _compact(text: str) -> str:
|
|
return re.sub(r"\s+", "", text or "").lower()
|
|
|
|
|
|
def _base_key(key: str) -> str:
|
|
"""객실 근거는 "A동:max_capacity" 로 적힌다 — 주제 비교에는 key 만 쓴다."""
|
|
return str(key).rsplit(":", 1)[-1]
|
|
|
|
|
|
def _with_topic_particle(topic: str) -> str:
|
|
"""주제 뒤에 은/는 을 붙인다."""
|
|
last = topic.strip()[-1:]
|
|
if "가" <= last <= "힣":
|
|
return topic + ("은" if (ord(last) - ord("가")) % 28 else "는")
|
|
return topic + "은(는)"
|
|
|
|
|
|
def fallback_answer(catalog: FaqCatalog, index: int, topic: str, phone: Optional[str]) -> str:
|
|
"""문의 안내 문구."""
|
|
phone = (phone or "").strip()
|
|
templates = catalog.fallback_with_contact if phone else catalog.fallback_without_contact
|
|
template = templates[index % len(templates)]
|
|
return template.format(topic=_with_topic_particle(topic), contact=f"전화({phone})")
|
|
|
|
|
|
def pick_fill_faqs(
|
|
catalog: FaqCatalog,
|
|
existing: Iterable[ExistingFaq],
|
|
fact_keys: Iterable[str],
|
|
phone: Optional[str] = None,
|
|
target: int = FAQ_TARGET,
|
|
) -> list[FillFaq]:
|
|
"""목표 수까지 모자란 만큼 카탈로그 질문을 고른다."""
|
|
existing = list(existing)
|
|
need = target - len(existing)
|
|
if need <= 0:
|
|
return []
|
|
|
|
asked = [_compact(faq.question) for faq in existing]
|
|
covered_keys = {_base_key(k) for faq in existing for k in (faq.source_fact_ids or [])}
|
|
known_keys = {_base_key(k) for k in fact_keys}
|
|
|
|
picked: list[FillFaq] = []
|
|
for item in catalog.items:
|
|
if len(picked) >= need:
|
|
break
|
|
if known_keys.intersection(item.fact_keys):
|
|
continue # 규칙 1
|
|
if covered_keys.intersection(item.fact_keys):
|
|
continue # 규칙 2 — 근거 key
|
|
if any(keyword in question for keyword in item.keywords for question in asked):
|
|
continue # 규칙 2 — 질문 낱말
|
|
answer = fallback_answer(catalog, len(picked), item.topic, phone)
|
|
picked.append(FillFaq(item.id, item.question, answer))
|
|
return picked
|
|
|
|
|
|
def suggested_questions(catalog: FaqCatalog, fact_keys: Iterable[str]) -> list[str]:
|
|
"""fact 로 답할 수 있는 카탈로그 질문 — 프롬프트에 실어 LLM 이 이 질문들부터 쓰게 한다."""
|
|
known = {_base_key(k) for k in fact_keys}
|
|
return [item.question for item in catalog.items if known.intersection(item.fact_keys)]
|