"""Experimental evidence extraction followed by short-summary compression.""" from __future__ import annotations import json from app.engine.summary_consensus import SHORT_SUMMARY_PROMPT, select_consensus EVIDENCE_PROMPT = ( "전기 문단 전체를 읽고 중심 인물·핵심 행동·결과를 가장 잘 드러내는 원문 구절을 고르세요. " "주변 묘사보다 문단의 중심 사건이나 논지를 우선하세요. 표현을 고쳐 쓰지 말고 " "원문에 연속해서 실제 존재하는 구절만 최대 3개, 각 120자 이내로 복사하세요. " 'JSON 객체 {"evidence": ["구절1", "구절2"]}만 출력하세요.' ) COMPRESSION_INSTRUCTION = ( "아래 핵심 구절은 원문에서 검증한 보조 단서입니다. 반드시 원문 전체와 대조하세요. " "핵심 구절에 등장하는 인물명·명사·동사를 임의의 동의어로 치환하지 마세요. " "단서를 기계적으로 나열하지 말고 중심 사건과 결과를 자연스럽게 연결하세요." ) def validated_evidence(text: str, payload: dict) -> list[str]: raw = payload.get("evidence", []) if not isinstance(raw, list): return [] return list(dict.fromkeys(s.strip() for s in raw if isinstance(s, str) and 1 <= len(s.strip()) <= 120 and s.strip() in text))[:3] def generate_grounded(client, text: str, count: int = 5) -> dict: if not text.strip(): raise ValueError("Source text must not be empty") if not 1 <= count <= 5: raise ValueError("Candidate count must be between 1 and 5") extraction = client.chat.completions.create( model="gpt-4o", temperature=0, max_tokens=400, response_format={"type": "json_object"}, messages=[{"role": "system", "content": "원문에서 근거 구절을 정확히 추출하세요."}, {"role": "user", "content": EVIDENCE_PROMPT + "\n\n" + text}], ) evidence = [] choice = extraction.choices[0] if choice.finish_reason == "stop": try: payload = json.loads(choice.message.content or "{}") evidence = validated_evidence(text, payload) if isinstance(payload, dict) else [] except (ValueError, TypeError): pass # Invalid extraction falls back to the full source; never invent evidence. prompt = SHORT_SUMMARY_PROMPT + "\n\n[원문]\n" + text if evidence: prompt += "\n\n" + COMPRESSION_INSTRUCTION + "\n[핵심 구절]\n" + "\n".join(evidence) response = client.chat.completions.create( model="gpt-4o", temperature=0.3, max_tokens=100, n=count, messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."}, {"role": "user", "content": prompt}], ) candidates = [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "", "finish_reason": c.finish_reason, "model": response.model, "response_id": response.id} for c in response.choices] return {"evidence": evidence, "evidence_response_id": extraction.id, "evidence_model": extraction.model, "evidence_finish_reason": choice.finish_reason, "evidence_fallback": not bool(evidence), "candidates": candidates} def summarize_grounded(client, text: str) -> dict: generated = generate_grounded(client, text) index = select_consensus([c["summary"] for c in generated["candidates"]], word_weight=0.5) return {**generated, "selected_index": index, "valid": index >= 0, "summary": generated["candidates"][index]["summary"] if index >= 0 else ""}