여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
99 lines
3.7 KiB
Python
99 lines
3.7 KiB
Python
"""업소 조사 — 소개문을 쓸 **재료**를 공개 웹에서 찾아 근거 자리에 넣는다."""
|
|
import uuid
|
|
|
|
import httpx
|
|
|
|
from common.database.db_session_manager import DB_SESSION_MNG
|
|
from common.database.model.models import place_channels
|
|
from common.enums import ErrorType, LinkChannel, SourceType
|
|
from common.logger import LOG
|
|
from common.utils.gtime import GTime
|
|
from crud.place_crud import PlaceCRUD
|
|
from services.grounding import place_research as grounding
|
|
from services.llm import perplexity
|
|
from services.prompts import place_research as prompts
|
|
|
|
_place_crud = PlaceCRUD()
|
|
|
|
# 검색을 동반해 느리다.
|
|
_TIMEOUT = 120.0
|
|
|
|
# raw 봉투의 표식.
|
|
RAW_KIND = "research"
|
|
|
|
|
|
def _envelope(items: list[dict]) -> dict:
|
|
"""근거 봉투."""
|
|
return {
|
|
"kind": RAW_KIND,
|
|
"text": "\n".join(f"- {row['text']} (출처: {row['source']['name']})" for row in items),
|
|
"sources": [row["source"] for row in items],
|
|
"collected_at": GTime.UTC().isoformat(),
|
|
}
|
|
|
|
|
|
async def research_place(place, place_id: str) -> dict:
|
|
"""업소 하나를 조사해 근거를 적재한다."""
|
|
name = (getattr(place, "name", None) or "").strip()
|
|
address = (getattr(place, "road_address", None) or getattr(place, "address", None) or "").strip()
|
|
if not name or not address:
|
|
return {"skipped": "상호·주소를 모른다"}
|
|
if not perplexity.is_configured():
|
|
return {"skipped": "PERPLEXITY_API_KEY 미설정"}
|
|
|
|
# 업종 이름은 업종 스키마가 단일 출처다(`common/category_schema`) — 여기에 표를 또 적으면 업종이 늘 때 한쪽만 늘어난다.
|
|
from common.category_schema import get_schema
|
|
|
|
category_label = get_schema(place.category).label
|
|
|
|
body = {
|
|
"model": perplexity.DEFAULT_MODEL,
|
|
"messages": [
|
|
{"role": "system", "content": prompts.SYSTEM_PROMPT},
|
|
{"role": "user", "content": prompts.build_prompt(name, address, category_label)},
|
|
],
|
|
"max_tokens": perplexity.DEFAULT_MAX_TOKENS,
|
|
}
|
|
|
|
try:
|
|
async with httpx.AsyncClient(timeout=_TIMEOUT) as client:
|
|
payload = await perplexity.call(body, client=client)
|
|
except perplexity.PerplexityNotConfigured:
|
|
return {"skipped": "PERPLEXITY_API_KEY 미설정"}
|
|
except perplexity.PerplexityError as ex:
|
|
LOG.w(f"[research] 호출 실패 place={place_id}: {ex}")
|
|
return {"error": str(ex)}
|
|
|
|
items, dropped = grounding.parse_items(payload, name, prompts.MAX_ITEMS)
|
|
usage = perplexity.read_usage(payload)
|
|
LOG.i(
|
|
f"[research] '{name}' 조사 {len(items)}건 채택, {len(dropped)}건 버림 · "
|
|
f"tokens in={usage.input_tokens} out={usage.output_tokens} · 약 ${usage.cost}"
|
|
)
|
|
if not items:
|
|
return {"items": 0, "dropped": dropped}
|
|
|
|
# 출처 주소 하나를 대표로 링크에 단다 — 없는 URL 을 만들지 않기 위해 첫 출처를 쓴다.
|
|
url = items[0]["source"]["url"]
|
|
row = place_channels(
|
|
link_id=uuid.uuid4(),
|
|
place_id=uuid.UUID(place_id),
|
|
channel=LinkChannel.ETC.value,
|
|
url=url,
|
|
title=f"{name} 조사 근거",
|
|
discovered_by=SourceType.API.value,
|
|
discovered_at=GTime.UTC(),
|
|
raw=_envelope(items),
|
|
)
|
|
err = await DB_SESSION_MNG.execute_lambda_run(
|
|
[place_channels.DBType()], [lambda s: _place_crud.add_link(s, row)],
|
|
)
|
|
if err != ErrorType.SUCCESS:
|
|
# 이미 같은 URL 이 있으면 raw 만 갱신한다 — 재조사가 행을 늘리면 안 된다.
|
|
await DB_SESSION_MNG.execute_lambda_claim(
|
|
place_channels.DBType(),
|
|
lambda s: _place_crud.set_link_raw(s, uuid.UUID(place_id), url, _envelope(items)),
|
|
)
|
|
|
|
return {"items": len(items), "dropped": dropped, "source": url}
|