요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
69 lines
3.6 KiB
Python
69 lines
3.6 KiB
Python
"""Experimental evidence extraction followed by short-summary compression."""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from app.engine.summary_consensus import SHORT_SUMMARY_PROMPT, select_consensus
|
|
|
|
EVIDENCE_PROMPT = (
|
|
"전기 문단 전체를 읽고 중심 인물·핵심 행동·결과를 가장 잘 드러내는 원문 구절을 고르세요. "
|
|
"주변 묘사보다 문단의 중심 사건이나 논지를 우선하세요. 표현을 고쳐 쓰지 말고 "
|
|
"원문에 연속해서 실제 존재하는 구절만 최대 3개, 각 120자 이내로 복사하세요. "
|
|
'JSON 객체 {"evidence": ["구절1", "구절2"]}만 출력하세요.'
|
|
)
|
|
COMPRESSION_INSTRUCTION = (
|
|
"아래 핵심 구절은 원문에서 검증한 보조 단서입니다. 반드시 원문 전체와 대조하세요. "
|
|
"핵심 구절에 등장하는 인물명·명사·동사를 임의의 동의어로 치환하지 마세요. "
|
|
"단서를 기계적으로 나열하지 말고 중심 사건과 결과를 자연스럽게 연결하세요."
|
|
)
|
|
|
|
|
|
def validated_evidence(text: str, payload: dict) -> list[str]:
|
|
raw = payload.get("evidence", [])
|
|
if not isinstance(raw, list):
|
|
return []
|
|
return list(dict.fromkeys(s.strip() for s in raw if isinstance(s, str)
|
|
and 1 <= len(s.strip()) <= 120 and s.strip() in text))[:3]
|
|
|
|
|
|
def generate_grounded(client, text: str, count: int = 5) -> dict:
|
|
if not text.strip():
|
|
raise ValueError("Source text must not be empty")
|
|
if not 1 <= count <= 5:
|
|
raise ValueError("Candidate count must be between 1 and 5")
|
|
extraction = client.chat.completions.create(
|
|
model="gpt-4o", temperature=0, max_tokens=400,
|
|
response_format={"type": "json_object"},
|
|
messages=[{"role": "system", "content": "원문에서 근거 구절을 정확히 추출하세요."},
|
|
{"role": "user", "content": EVIDENCE_PROMPT + "\n\n" + text}],
|
|
)
|
|
evidence = []
|
|
choice = extraction.choices[0]
|
|
if choice.finish_reason == "stop":
|
|
try:
|
|
payload = json.loads(choice.message.content or "{}")
|
|
evidence = validated_evidence(text, payload) if isinstance(payload, dict) else []
|
|
except (ValueError, TypeError):
|
|
pass
|
|
# Invalid extraction falls back to the full source; never invent evidence.
|
|
prompt = SHORT_SUMMARY_PROMPT + "\n\n[원문]\n" + text
|
|
if evidence:
|
|
prompt += "\n\n" + COMPRESSION_INSTRUCTION + "\n[핵심 구절]\n" + "\n".join(evidence)
|
|
response = client.chat.completions.create(
|
|
model="gpt-4o", temperature=0.3, max_tokens=100, n=count,
|
|
messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
|
|
{"role": "user", "content": prompt}],
|
|
)
|
|
candidates = [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "",
|
|
"finish_reason": c.finish_reason, "model": response.model,
|
|
"response_id": response.id} for c in response.choices]
|
|
return {"evidence": evidence, "evidence_response_id": extraction.id,
|
|
"evidence_model": extraction.model, "evidence_finish_reason": choice.finish_reason,
|
|
"evidence_fallback": not bool(evidence), "candidates": candidates}
|
|
|
|
|
|
def summarize_grounded(client, text: str) -> dict:
|
|
generated = generate_grounded(client, text)
|
|
index = select_consensus([c["summary"] for c in generated["candidates"]], word_weight=0.5)
|
|
return {**generated, "selected_index": index, "valid": index >= 0,
|
|
"summary": generated["candidates"][index]["summary"] if index >= 0 else ""}
|