요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
76 lines
3.4 KiB
Python
76 lines
3.4 KiB
Python
"""Source-only short-summary consensus; never accepts reference summaries."""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from collections import Counter
|
|
|
|
SHORT_SUMMARY_PROMPT = (
|
|
"다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. "
|
|
"핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. "
|
|
"배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요."
|
|
)
|
|
|
|
|
|
def character_f1(a: str, b: str) -> float:
|
|
def grams(text):
|
|
text = re.sub(r"\s+", "", text)
|
|
return Counter(text[i:i + 2] for i in range(len(text) - 1))
|
|
x, y = grams(a), grams(b)
|
|
total = sum(x.values()) + sum(y.values())
|
|
return 2 * sum((x & y).values()) / total if total else 0.0
|
|
|
|
|
|
def word_f1(a: str, b: str) -> float:
|
|
def grams(text):
|
|
words = text.split()
|
|
return Counter(zip(words, words[1:]))
|
|
x, y = grams(a), grams(b)
|
|
total = sum(x.values()) + sum(y.values())
|
|
return 2 * sum((x & y).values()) / total if total else 0.0
|
|
|
|
|
|
def select_consensus(summaries: list[str], word_weight: float = 0.0) -> int:
|
|
"""Select a medoid among independently generated candidates, under 50 chars.
|
|
|
|
Consensus measures stability, not truth: systematic model errors can survive.
|
|
Return -1 if all candidates are empty/overlength instead of silently truncating.
|
|
"""
|
|
if not 0 <= word_weight <= 1:
|
|
raise ValueError("word_weight must be between 0 and 1")
|
|
eligible = [i for i, s in enumerate(summaries) if s.strip() and len(s.strip()) <= 50]
|
|
if not eligible:
|
|
return -1
|
|
return max(eligible, key=lambda i: (
|
|
sum((1 - word_weight) * character_f1(summaries[i], summaries[j])
|
|
+ word_weight * word_f1(summaries[i], summaries[j])
|
|
for j in eligible if j != i), -i))
|
|
|
|
|
|
def generate_candidates(client, text: str, count: int = 5) -> list[dict]:
|
|
if not text.strip():
|
|
raise ValueError("Source text must not be empty")
|
|
if not 1 <= count <= 5:
|
|
raise ValueError("Candidate count must be between 1 and 5")
|
|
response = client.chat.completions.create(
|
|
model="gpt-4o", temperature=0.3, max_tokens=100, n=count,
|
|
messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
|
|
{"role": "user", "content": SHORT_SUMMARY_PROMPT + "\n\n" + text}],
|
|
)
|
|
return [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "",
|
|
"finish_reason": c.finish_reason, "model": response.model,
|
|
"response_id": response.id} for c in response.choices]
|
|
|
|
|
|
def summarize_short(client, text: str, word_weight: float = 0.0) -> dict:
|
|
"""Opt-in 50-character summary; five completions incur additional API cost.
|
|
|
|
Return an explicit invalid result if no complete length-compliant summary exists.
|
|
Callers should inspect valid, not silently present an empty result as success.
|
|
"""
|
|
candidates = generate_candidates(client, text)
|
|
chosen = select_consensus([c["summary"] for c in candidates], word_weight=word_weight)
|
|
return {"summary": candidates[chosen]["summary"] if chosen >= 0 else "",
|
|
"valid": chosen >= 0, "selected_index": chosen,
|
|
"strategy": "source_only_consensus5", "word_weight": word_weight,
|
|
"candidates": candidates}
|