여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
124 lines
4.7 KiB
Python
124 lines
4.7 KiB
Python
"""Gemini 호출 — 이 프로젝트에서 Gemini 로 나가는 **유일한 통로**."""
|
|
import asyncio
|
|
import base64
|
|
import json
|
|
|
|
import httpx
|
|
|
|
from config.server_configs import external_api_config
|
|
from services.llm.errors import LlmError as GeminiError
|
|
from services.llm.errors import LlmInvalidOutput
|
|
from services.llm.errors import LlmInvalidOutput as GeminiInvalidOutput # noqa: F401 (하위 호환 재노출)
|
|
from services.llm.errors import LlmNotConfigured as GeminiNotConfigured
|
|
from services.llm.types import ImagePart, LlmResult, Usage
|
|
|
|
_BASE_URL = "https://generativelanguage.googleapis.com/v1beta/models"
|
|
|
|
DEFAULT_MODEL = "gemini-3.7-flash"
|
|
|
|
# 100만 토큰당 USD.
|
|
_PRICE_PER_1M_INPUT = {"gemini-3.7-flash": 0.75, "gemini-3.6-flash": 0.75, "gemini-2.5-flash": 0.30}
|
|
_PRICE_PER_1M_OUTPUT = {"gemini-3.7-flash": 3.75, "gemini-3.6-flash": 3.75, "gemini-2.5-flash": 2.50}
|
|
|
|
# 일시적 장애만 재시도한다.
|
|
_RETRYABLE_STATUS = {408, 429, 500, 502, 503, 504}
|
|
|
|
|
|
def is_configured() -> bool:
|
|
"""키가 있는지 — 어댑터 등록/스킵 판단용."""
|
|
return bool(external_api_config.gemini_api_key)
|
|
|
|
|
|
async def call(
|
|
client: httpx.AsyncClient,
|
|
model: str,
|
|
body: dict,
|
|
max_retries: int = 2,
|
|
) -> dict:
|
|
"""LLM 이 실제로 불리는 지점."""
|
|
url = f"{_BASE_URL}/{model}:generateContent"
|
|
headers = {"Content-Type": "application/json", "x-goog-api-key": external_api_config.gemini_api_key}
|
|
last = None
|
|
for attempt in range(max_retries + 1):
|
|
try:
|
|
resp = await client.post(url, json=body, headers=headers)
|
|
except (httpx.TimeoutException, httpx.TransportError) as ex:
|
|
last = f"{type(ex).__name__}: {ex}"
|
|
else:
|
|
if resp.status_code == 200:
|
|
try:
|
|
return resp.json()
|
|
except ValueError as ex:
|
|
raise GeminiInvalidOutput(f"JSON 이 아닌 응답: {ex}") from ex
|
|
if resp.status_code in (401, 403):
|
|
raise GeminiNotConfigured(f"인증 실패 status={resp.status_code} — API 키를 확인하세요")
|
|
if resp.status_code not in _RETRYABLE_STATUS:
|
|
raise GeminiError(f"status={resp.status_code} body={resp.text[:200]}")
|
|
last = f"status={resp.status_code}"
|
|
|
|
if attempt < max_retries:
|
|
await asyncio.sleep(min(8.0, 1.0 * (2 ** attempt)))
|
|
raise GeminiError(f"{max_retries + 1}회 시도 실패: {last}")
|
|
|
|
|
|
def extract_text(payload: dict) -> str:
|
|
"""응답에서 텍스트 파트만 이어붙인다."""
|
|
candidates = payload.get("candidates") or []
|
|
if not candidates:
|
|
raise GeminiInvalidOutput("candidates 가 비었다(안전 필터 차단 가능)")
|
|
parts = (candidates[0].get("content") or {}).get("parts") or []
|
|
text = "".join(p["text"] for p in parts if isinstance(p, dict) and "text" in p)
|
|
if not text.strip():
|
|
raise GeminiInvalidOutput(f"텍스트 파트가 없다 finishReason={candidates[0].get('finishReason')}")
|
|
return text
|
|
|
|
|
|
def read_usage(payload: dict) -> Usage:
|
|
"""응답의 usageMetadata → Usage."""
|
|
meta = payload.get("usageMetadata") or {}
|
|
return Usage(
|
|
input_tokens=int(meta.get("promptTokenCount") or 0),
|
|
output_tokens=int(meta.get("candidatesTokenCount") or 0),
|
|
)
|
|
|
|
|
|
def price(model: str, usage: Usage) -> float:
|
|
"""USD."""
|
|
inp = _PRICE_PER_1M_INPUT.get(model, 0.0) * usage.input_tokens / 1_000_000
|
|
out = _PRICE_PER_1M_OUTPUT.get(model, 0.0) * usage.output_tokens / 1_000_000
|
|
return round(inp + out, 4)
|
|
|
|
|
|
async def generate(
|
|
client: httpx.AsyncClient,
|
|
model: str,
|
|
*,
|
|
prompt: str,
|
|
images: list[ImagePart] | None = None,
|
|
response_schema: dict | None = None,
|
|
temperature: float = 0.2,
|
|
max_retries: int = 2,
|
|
) -> LlmResult:
|
|
"""공급자 무관 인터페이스."""
|
|
parts: list[dict] = [{"text": prompt}]
|
|
for image in images or []:
|
|
if image.label:
|
|
parts.append({"text": image.label})
|
|
parts.append({"inline_data": {"mime_type": image.mime_type, "data": base64.b64encode(image.data).decode()}})
|
|
|
|
generation_config: dict = {"temperature": temperature}
|
|
if response_schema is not None:
|
|
generation_config["responseMimeType"] = "application/json"
|
|
generation_config["responseSchema"] = response_schema
|
|
|
|
body = {"contents": [{"role": "user", "parts": parts}], "generationConfig": generation_config}
|
|
payload = await call(client, model, body, max_retries)
|
|
text = extract_text(payload)
|
|
parsed = None
|
|
if response_schema is not None:
|
|
try:
|
|
parsed = json.loads(text)
|
|
except json.JSONDecodeError as ex:
|
|
raise LlmInvalidOutput(f"구조화 출력 파싱 실패: {ex}") from ex
|
|
return LlmResult(json=parsed, text=text, usage=read_usage(payload))
|