여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
112 lines
3.6 KiB
Python
112 lines
3.6 KiB
Python
"""채널 URL 발견 — 겹들을 엮어 결과를 만드는 자리."""
|
|
from dataclasses import dataclass, field
|
|
|
|
import httpx
|
|
|
|
from common.logger import LOG
|
|
from services.grounding.channels import (
|
|
DiscoveredLink,
|
|
classify_url,
|
|
collect_links,
|
|
filter_links,
|
|
search_count,
|
|
)
|
|
from services.llm.perplexity import (
|
|
DEFAULT_MAX_TOKENS,
|
|
DEFAULT_MODEL,
|
|
DEFAULT_TIMEOUT,
|
|
PerplexityError,
|
|
PerplexityNotConfigured,
|
|
call,
|
|
is_configured,
|
|
read_usage,
|
|
)
|
|
from services.prompts.channel_discovery import RESPONSE_SCHEMA, SYSTEM_PROMPT, build_prompt
|
|
|
|
# 후기 유입을 막기 위해 공식 채널 도메인만 검색한다.
|
|
SEARCH_DOMAIN_FILTER = [
|
|
"yanolja.com",
|
|
"goodchoice.kr",
|
|
"place.naver.com",
|
|
"map.naver.com",
|
|
"naver.me",
|
|
]
|
|
|
|
# 이 횟수를 넘으면 프롬프트·도메인 필터를 의심한다 — 검색 요금은 토큰 요금과 별도다.
|
|
SEARCH_COUNT_WARN_THRESHOLD = 8
|
|
|
|
|
|
|
|
@dataclass
|
|
class ChannelDiscovery:
|
|
"""채널 발견 결과."""
|
|
|
|
links: list[DiscoveredLink] = field(default_factory=list)
|
|
raw: dict = field(default_factory=dict)
|
|
search_count: int = 0
|
|
filtered_out: list[tuple[str, str]] = field(default_factory=list)
|
|
|
|
def reason_counts(self) -> dict[str, int]:
|
|
"""탈락 사유별 건수."""
|
|
counts: dict[str, int] = {}
|
|
for _url, reason in self.filtered_out:
|
|
counts[reason] = counts.get(reason, 0) + 1
|
|
return counts
|
|
|
|
|
|
async def discover_channels(
|
|
name: str,
|
|
address: str | None = None,
|
|
category_hint: str | None = None,
|
|
*,
|
|
model: str = DEFAULT_MODEL,
|
|
include_blogs: bool = False,
|
|
client: httpx.AsyncClient | None = None,
|
|
) -> ChannelDiscovery:
|
|
"""상호명으로 채널 URL 후보를 찾는다."""
|
|
if not (name or "").strip():
|
|
raise PerplexityError("상호명이 비어 있다")
|
|
|
|
body = {
|
|
"model": model,
|
|
"messages": [
|
|
{
|
|
"role": "system",
|
|
"content": SYSTEM_PROMPT,
|
|
},
|
|
{"role": "user", "content": build_prompt(name, address, category_hint)},
|
|
],
|
|
"max_tokens": DEFAULT_MAX_TOKENS,
|
|
"temperature": 0, # URL 수집이라 창의성이 해롭다
|
|
"response_format": RESPONSE_SCHEMA,
|
|
"search_domain_filter": SEARCH_DOMAIN_FILTER, # 야놀자·여기어때·네이버로 한정
|
|
}
|
|
payload = await call(body, client=client)
|
|
|
|
found, dropped_raw = collect_links(payload)
|
|
links, dropped_quality = filter_links(found, include_blogs)
|
|
filtered_out = dropped_raw + dropped_quality
|
|
searches = search_count(payload)
|
|
|
|
result = ChannelDiscovery(
|
|
links=links, raw=payload, search_count=searches, filtered_out=filtered_out
|
|
)
|
|
|
|
# 내부 검색 횟수는 품질·지연 관측값이다.
|
|
usage = read_usage(payload)
|
|
reasons = result.reason_counts()
|
|
reason_text = " ".join(f"{k}{v}" for k, v in sorted(reasons.items())) or "없음"
|
|
LOG.i(
|
|
f"[perplexity] '{name}' 검색={searches}회 발견={len(found)} 통과={len(links)} "
|
|
f"탈락={len(filtered_out)}({reason_text}) "
|
|
f"tokens in={usage.input_tokens} out={usage.output_tokens} · 약 ${usage.cost}"
|
|
)
|
|
if searches > SEARCH_COUNT_WARN_THRESHOLD:
|
|
LOG.w(
|
|
f"[perplexity] 검색 {searches}회 — 기준({SEARCH_COUNT_WARN_THRESHOLD}회) 초과. "
|
|
f"지연·오탐 후보가 늘 수 있으니 도메인 필터·프롬프트를 확인하라"
|
|
)
|
|
|
|
# raw 는 응답 전체를 그대로 둔다 — 나중에 환각 추적에 쓴다(사실 근거로는 쓰지 않는다).
|
|
return result
|