여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
97 lines
3.3 KiB
Python
97 lines
3.3 KiB
Python
"""원문 텍스트 → fact 후보 추출 — 겹들을 엮어 결과를 만드는 자리."""
|
|
from dataclasses import dataclass, field
|
|
from typing import Optional
|
|
|
|
import httpx
|
|
|
|
from common.category_schema import get_schema
|
|
from common.enums import PlaceCategory
|
|
from common.logger import LOG
|
|
from services.collector.base import CollectedFact
|
|
from services.grounding.extract import verify
|
|
from services.llm import provider
|
|
from services.llm.errors import LlmInvalidOutput as GeminiInvalidOutput
|
|
from services.llm.errors import LlmNotConfigured as GeminiNotConfigured
|
|
from services.prompts.extract import RESPONSE_SCHEMA, build_prompt
|
|
|
|
# 이보다 짧은 원문은 호출하지 않는다.
|
|
MIN_SOURCE_CHARS = 80
|
|
|
|
|
|
@dataclass
|
|
class ExtractResult:
|
|
"""추출 결과."""
|
|
|
|
facts: list[CollectedFact] = field(default_factory=list)
|
|
rejected: list[tuple[str, str]] = field(default_factory=list)
|
|
|
|
@property
|
|
def ok(self) -> bool:
|
|
return bool(self.facts)
|
|
|
|
|
|
async def extract_facts(
|
|
place_name: str,
|
|
category: PlaceCategory,
|
|
source_text: str,
|
|
*,
|
|
source_url: str,
|
|
model: Optional[str] = None,
|
|
max_retries: int = 2,
|
|
client: Optional[httpx.AsyncClient] = None,
|
|
) -> ExtractResult:
|
|
"""원문 텍스트에서 업종 스키마 fact 를 뽑는다."""
|
|
llm = provider.active()
|
|
if not llm.is_configured():
|
|
raise GeminiNotConfigured("API 키가 설정되지 않았다")
|
|
if not (source_url or "").strip():
|
|
raise ValueError("source_url 이 비었다 — 출처 없는 추출은 하지 않는다")
|
|
model = model or llm.DEFAULT_MODEL
|
|
|
|
text = (source_text or "").strip()
|
|
if len(text) < MIN_SOURCE_CHARS:
|
|
LOG.i(f"[extract] '{place_name}' 원문 {len(text)}자 — 짧아서 호출하지 않는다")
|
|
return ExtractResult(rejected=[("(전체)", f"원문이 {len(text)}자로 너무 짧다 — 호출하지 않았다")])
|
|
|
|
owns_client = client is None
|
|
client = client or httpx.AsyncClient(timeout=httpx.Timeout(120.0, connect=10.0))
|
|
try:
|
|
llm_result = await llm.generate(
|
|
client, model, prompt=build_prompt(place_name, category, text),
|
|
# 0.0 — 옮겨 적는 작업이다.
|
|
response_schema=RESPONSE_SCHEMA, temperature=0.0, max_retries=max_retries,
|
|
)
|
|
finally:
|
|
if owns_client:
|
|
await client.aclose()
|
|
|
|
rows = llm_result.json.get("facts") if llm_result.json else None
|
|
if not isinstance(rows, list):
|
|
raise GeminiInvalidOutput(f"facts 가 배열이 아니다: {type(rows).__name__}")
|
|
|
|
# 여기가 관문이다.
|
|
passed, rejected = verify(rows, source_text=text, schema=get_schema(category))
|
|
|
|
facts = [
|
|
CollectedFact(
|
|
key=row["key"],
|
|
value=row["value"],
|
|
scope=row["scope"],
|
|
unit_name=row["unit_name"],
|
|
source_url=source_url,
|
|
)
|
|
for row in passed
|
|
]
|
|
|
|
usage = llm_result.usage
|
|
LOG.i(
|
|
f"[extract] '{place_name}' 추출 {len(rows)}건 → 통과 {len(facts)}건 · "
|
|
f"반려 {len(rejected)}건 · model={model} · "
|
|
f"tokens in={usage.input_tokens} out={usage.output_tokens} · 약 ${llm.price(model, usage)}"
|
|
)
|
|
if rejected:
|
|
for label, why in rejected[:10]:
|
|
LOG.w(f"[extract] 반려 {label} — {why}")
|
|
|
|
return ExtractResult(facts=facts, rejected=rejected)
|