feat: add summary (No.7) consensus and grounded generation trials

요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다.

- app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus).
  문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다.
- app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성.
- summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로
  중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다.

스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만
재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지
않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet ->
export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
hbyang 2026-09-28 10:02:04 +09:00
parent d980f34388
commit 8ae52b0a82
18 changed files with 1281 additions and 3 deletions

View File

@ -31,6 +31,7 @@ from app.engine.structural import extract_lemmas
logger = logging.getLogger(__name__)
DETAIL_RATIOS = {"brief": 0.2, "standard": 0.3, "detailed": 0.45}
EXTRACTIVE_STRATEGIES = {"textrank", "coverage", "lead"}
# 문장 분할 — 종결부호 기준 (한국어 '다./요./까?/!' + 줄바꿈)
_SENT_SPLIT = re.compile(r"(?<=[.!?。…])\s+|\n+")
@ -86,6 +87,59 @@ def _textrank_scores(vectors: list[Counter], damping: float = 0.85, iters: int =
return scores
def _normalize(scores: list[float]) -> list[float]:
"""0~1 정규화. 모든 값이 같으면 동률로 취급한다."""
if not scores:
return []
low, high = min(scores), max(scores)
if high - low < 1e-12:
return [1.0] * len(scores)
return [(score - low) / (high - low) for score in scores]
def _coverage_scores(sentences: list[str], vectors: list[Counter]) -> list[float]:
"""문서의 고유 내용어를 넓게 담는 문장을 우선하는 점수.
TextRank만 쓰면 같은 사건을 반복하는 문장이 상위를 독점할 수 있다. 이 점수는
문서 전체에서 드문 내용어(인물·사건·시간 등)가 든 문장을 높게 평가하고, 선택
단계에서 이미 선택한 문장과의 중복을 감점한다. 외부 모델이나 정답셋을 보지 않는
순수 추출 방식이다.
"""
document_frequency: Counter = Counter()
for vector in vectors:
document_frequency.update(vector.keys())
n = max(1, len(sentences))
lexical = [
sum(count * math.log((n + 1) / (document_frequency[token] + 0.5))
for token, count in vector.items()) / max(1, sum(vector.values()))
for vector in vectors
]
# 회고록·에피소드 형식에서는 첫 문장이 사건의 맥락을 잡는 경우가 많다. 단,
# 위치만으로 결정되지 않게 작은 사전 등록 가중치만 둔다.
position = [1.0 - (index / max(1, n - 1)) * 0.35 for index in range(n)]
centrality = _normalize(_textrank_scores(vectors))
lexical = _normalize(lexical)
return [0.50 * c + 0.38 * l + 0.12 * p
for c, l, p in zip(centrality, lexical, position)]
def _coverage_select(vectors: list[Counter], base_scores: list[float], k: int) -> list[int]:
"""MMR 방식으로 중요하면서 서로 다른 정보를 담는 문장 k개를 고른다."""
chosen: list[int] = []
remaining = set(range(len(vectors)))
while remaining and len(chosen) < k:
def candidate_score(index: int) -> tuple[float, float, int]:
redundancy = max((_cosine(vectors[index], vectors[other]) for other in chosen), default=0.0)
# 중복 문장은 낮추되, 낮은 중요도의 문장이 끼어들 정도로 과도하게 벌주진 않는다.
value = base_scores[index] - 0.22 * redundancy
return (value, base_scores[index], -index)
best = max(remaining, key=candidate_score)
chosen.append(best)
remaining.remove(best)
return chosen
@dataclass
class SummaryResult:
extractive: str # 추출적 요약 (선택된 원문 문장)
@ -105,6 +159,7 @@ def extractive_summary(
max_sentences: int | None = None,
emphasis: list[str] | None = None,
detail: str = "standard",
strategy: str = "textrank",
) -> SummaryResult:
"""비지도 추출적 요약 — 정답셋/LLM/외부호출 불필요."""
sentences = split_sentences(text)
@ -123,8 +178,13 @@ def extractive_summary(
if max_sentences is not None:
k = min(k, max_sentences)
if strategy not in EXTRACTIVE_STRATEGIES:
raise ValueError("strategy must be one of %s" % sorted(EXTRACTIVE_STRATEGIES))
vectors = [_lemma_vector(s) for s in sentences]
scores = _textrank_scores(vectors)
scores = _textrank_scores(vectors) if strategy == "textrank" else _coverage_scores(sentences, vectors)
if strategy == "lead":
scores = [float(n - index) for index in range(n)]
# 사용자가 지정한 주제를 포함한 문장에 명시적 가중치를 준다.
# TextRank 중심성은 유지하되 강조 요청이 상위 선택에 반영되도록 한다.
@ -135,7 +195,8 @@ def extractive_summary(
scores[i] += 2.0 * matched / len(lowered_terms)
# 상위 k개 문장 선택 → 원문 등장 순서로 재정렬 (가독성)
top = sorted(range(n), key=lambda i: scores[i], reverse=True)[:k]
top = (_coverage_select(vectors, scores, k) if strategy == "coverage"
else sorted(range(n), key=lambda i: scores[i], reverse=True)[:k])
top_sorted = sorted(top)
summary = " ".join(sentences[i] for i in top_sorted)
return SummaryResult(
@ -177,6 +238,7 @@ class Summarizer:
use_abstractive: bool = True,
detail: str = "standard",
emphasis: list[str] | None = None,
strategy: str = "textrank",
) -> SummaryResult:
effective_ratio = ratio if ratio is not None else DETAIL_RATIOS.get(detail, 0.3)
base = extractive_summary(
@ -185,6 +247,7 @@ class Summarizer:
max_sentences=max_sentences,
emphasis=emphasis,
detail=detail,
strategy=strategy,
)
if not base.extractive:
return base

View File

@ -0,0 +1,75 @@
"""Source-only short-summary consensus; never accepts reference summaries."""
from __future__ import annotations
import re
from collections import Counter
SHORT_SUMMARY_PROMPT = (
"다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. "
"핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. "
"배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요."
)
def character_f1(a: str, b: str) -> float:
def grams(text):
text = re.sub(r"\s+", "", text)
return Counter(text[i:i + 2] for i in range(len(text) - 1))
x, y = grams(a), grams(b)
total = sum(x.values()) + sum(y.values())
return 2 * sum((x & y).values()) / total if total else 0.0
def word_f1(a: str, b: str) -> float:
def grams(text):
words = text.split()
return Counter(zip(words, words[1:]))
x, y = grams(a), grams(b)
total = sum(x.values()) + sum(y.values())
return 2 * sum((x & y).values()) / total if total else 0.0
def select_consensus(summaries: list[str], word_weight: float = 0.0) -> int:
"""Select a medoid among independently generated candidates, under 50 chars.
Consensus measures stability, not truth: systematic model errors can survive.
Return -1 if all candidates are empty/overlength instead of silently truncating.
"""
if not 0 <= word_weight <= 1:
raise ValueError("word_weight must be between 0 and 1")
eligible = [i for i, s in enumerate(summaries) if s.strip() and len(s.strip()) <= 50]
if not eligible:
return -1
return max(eligible, key=lambda i: (
sum((1 - word_weight) * character_f1(summaries[i], summaries[j])
+ word_weight * word_f1(summaries[i], summaries[j])
for j in eligible if j != i), -i))
def generate_candidates(client, text: str, count: int = 5) -> list[dict]:
if not text.strip():
raise ValueError("Source text must not be empty")
if not 1 <= count <= 5:
raise ValueError("Candidate count must be between 1 and 5")
response = client.chat.completions.create(
model="gpt-4o", temperature=0.3, max_tokens=100, n=count,
messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
{"role": "user", "content": SHORT_SUMMARY_PROMPT + "\n\n" + text}],
)
return [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "",
"finish_reason": c.finish_reason, "model": response.model,
"response_id": response.id} for c in response.choices]
def summarize_short(client, text: str, word_weight: float = 0.0) -> dict:
"""Opt-in 50-character summary; five completions incur additional API cost.
Return an explicit invalid result if no complete length-compliant summary exists.
Callers should inspect valid, not silently present an empty result as success.
"""
candidates = generate_candidates(client, text)
chosen = select_consensus([c["summary"] for c in candidates], word_weight=word_weight)
return {"summary": candidates[chosen]["summary"] if chosen >= 0 else "",
"valid": chosen >= 0, "selected_index": chosen,
"strategy": "source_only_consensus5", "word_weight": word_weight,
"candidates": candidates}

View File

@ -0,0 +1,68 @@
"""Experimental evidence extraction followed by short-summary compression."""
from __future__ import annotations
import json
from app.engine.summary_consensus import SHORT_SUMMARY_PROMPT, select_consensus
EVIDENCE_PROMPT = (
"전기 문단 전체를 읽고 중심 인물·핵심 행동·결과를 가장 잘 드러내는 원문 구절을 고르세요. "
"주변 묘사보다 문단의 중심 사건이나 논지를 우선하세요. 표현을 고쳐 쓰지 말고 "
"원문에 연속해서 실제 존재하는 구절만 최대 3개, 각 120자 이내로 복사하세요. "
'JSON 객체 {"evidence": ["구절1", "구절2"]}만 출력하세요.'
)
COMPRESSION_INSTRUCTION = (
"아래 핵심 구절은 원문에서 검증한 보조 단서입니다. 반드시 원문 전체와 대조하세요. "
"핵심 구절에 등장하는 인물명·명사·동사를 임의의 동의어로 치환하지 마세요. "
"단서를 기계적으로 나열하지 말고 중심 사건과 결과를 자연스럽게 연결하세요."
)
def validated_evidence(text: str, payload: dict) -> list[str]:
raw = payload.get("evidence", [])
if not isinstance(raw, list):
return []
return list(dict.fromkeys(s.strip() for s in raw if isinstance(s, str)
and 1 <= len(s.strip()) <= 120 and s.strip() in text))[:3]
def generate_grounded(client, text: str, count: int = 5) -> dict:
if not text.strip():
raise ValueError("Source text must not be empty")
if not 1 <= count <= 5:
raise ValueError("Candidate count must be between 1 and 5")
extraction = client.chat.completions.create(
model="gpt-4o", temperature=0, max_tokens=400,
response_format={"type": "json_object"},
messages=[{"role": "system", "content": "원문에서 근거 구절을 정확히 추출하세요."},
{"role": "user", "content": EVIDENCE_PROMPT + "\n\n" + text}],
)
evidence = []
choice = extraction.choices[0]
if choice.finish_reason == "stop":
try:
payload = json.loads(choice.message.content or "{}")
evidence = validated_evidence(text, payload) if isinstance(payload, dict) else []
except (ValueError, TypeError):
pass
# Invalid extraction falls back to the full source; never invent evidence.
prompt = SHORT_SUMMARY_PROMPT + "\n\n[원문]\n" + text
if evidence:
prompt += "\n\n" + COMPRESSION_INSTRUCTION + "\n[핵심 구절]\n" + "\n".join(evidence)
response = client.chat.completions.create(
model="gpt-4o", temperature=0.3, max_tokens=100, n=count,
messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
{"role": "user", "content": prompt}],
)
candidates = [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "",
"finish_reason": c.finish_reason, "model": response.model,
"response_id": response.id} for c in response.choices]
return {"evidence": evidence, "evidence_response_id": extraction.id,
"evidence_model": extraction.model, "evidence_finish_reason": choice.finish_reason,
"evidence_fallback": not bool(evidence), "candidates": candidates}
def summarize_grounded(client, text: str) -> dict:
generated = generate_grounded(client, text)
index = select_consensus([c["summary"] for c in generated["candidates"]], word_weight=0.5)
return {**generated, "selected_index": index, "valid": index >= 0,
"summary": generated["candidates"][index]["summary"] if index >= 0 else ""}

View File

@ -0,0 +1,51 @@
# 요약 성능 내부 벤치 증빙 — 2026-09-16
## 결론
보존된 30건 내부 벤치의 **ROUGE-1 recall은 67.6579%**다. 첨부된 전 차수
성적서 코드는 F1을 반환하므로 이 recall 값으로 올해 65% 달성을 판단할 수 없다.
이를 작년 시험 방식의 복원·달성값으로 설명했던 내용은 정정한다.
> 이 결과는 사람 작성 정답셋이 아닌 GPT-4o 은(silver) 기준을 사용한 **내부 A/B
> 벤치**다. 따라서 사람 정답셋 기반의 공인 시험성적서 결과로 기재하지 않는다.
## 보존된 실행 결과
| 항목 | 값 |
|---|---:|
| 결과 파일 | King `/mnt/data2/demo/o2o-plagiarism-ai/reports/summary_bench_v2.json` |
| 파일 수정 시각 | 2026-09-16 10:44:24 KST (생성·실행 시각을 입증하지 않음) |
| SHA-256 | `1e8022227c7a905f5926ab0c75aef0961248a901be19ead9cb501b98a5b70381` |
| 표본 수 | 30건 |
| 참조 요약 모델 | GPT-4o (결과 JSON에 기록) |
| 시스템 요약 모델 | GPT-4o (결과 JSON에 기록) |
| 프롬프트·seed·다중 참조 개수 | 결과 JSON에 미기록; 현재 스크립트 기본값으로 과거 실행 조건을 확정할 수 없음 |
## 결과
| ROUGE recall | 추출 요약 | 추상 요약 |
|---|---:|---:|
| ROUGE-1 (어절) | 21.2612% | **67.6579%** |
| ROUGE-2 (어절) | 7.8993% | 51.8763% |
| ROUGE-1 (형태소) | 41.8747% | 79.1366% |
| ROUGE-2 (형태소) | 18.0236% | 62.1995% |
## 재실행 상태
벤치 스크립트는 `/w/ai_publish/data_output4/*.json`의 전기 본문 30건을 읽는다.
호스트의 실제 경로는 `/mnt/data2/demo/ai_publish/data_output4`이며 JSON 85개가
존재한다. 앞선 확인에서 컨테이너 경로와 호스트 경로를 혼동하여 원문이 없다고
단정한 내용을 정정한다. 다만 30건 결과에는 원문 ID·참조 및 시스템 요약 원문이
없어 동일 표본·출력의 정확한 재채점은 불가능하다. 아래는 새 벤치 실행 예시다.
```bash
python scripts/bench_summarizer.py \
--glob '/mnt/data2/demo/ai_publish/data_output4/*.json' \
--count 30 --seed 20260916 \
--ref-model gpt-4o --sys-model gpt-4o \
--length-matched \
--out reports/summary_bench_new.json
```
새 1,000건 실험은 `scripts/build_summary_trial.py`로 수행하며 데이터·참조·출력을
모두 저장한다. 이 30건 보존 결과와 구분한다.

151
scripts/bench_summarizer.py Normal file
View File

@ -0,0 +1,151 @@
"""요약기 A/B 내부 벤치마크 — 추출 요약 vs LLM 추상 요약.
성능지표 #7 은 사람 참조 요약 대비 ROUGE recall 로 측정해야 한다. 아직 정답셋이
없어 본 평가는 할 수 없다. 이 스크립트는 그 전에 "LLM 훅을 켜면 수치가 오르는가"
만 판단하기 위한 내부 비교다.
참조 : 별도 모델(gpt-4o) 이 작성한 요약 — 은(silver) 기준
시스템 : ① TextRank 추출 요약 ② LLM 추상 요약
**편향 주의** — 참조가 LLM 산출물이므로 LLM 추상 요약 쪽에 유리하다. 두 방식의
상대 격차를 보는 용도이며, 이 수치를 성적서에 쓰면 안 된다.
입력은 전기(傳記) 본문을 쓴다. 전 차수 요약 시험과 같은 계열 자료이고 출판물이라
개인 자서전을 외부 API 로 보내지 않는다.
"""
from __future__ import annotations
import argparse
import glob
import json
import random
import sys
from collections import Counter
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
def lemmas(text: str) -> list[str]:
"""형태소 원형 열. 조사·어미 변형을 흡수해 어절 단위보다 느슨하게 맞춘다."""
from app.engine.structural import extract_lemmas
return extract_lemmas(text)
def ngrams(tokens: list[str], n: int) -> Counter:
return Counter(tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1))
def rouge_recall(reference: str, hypothesis: str, n: int = 2) -> float:
"""계획서 수식 기준 — 분모가 참조 n-gram 수인 recall."""
ref = ngrams(reference.split(), n)
hyp = ngrams(hypothesis.split(), n)
total = sum(ref.values())
if not total:
return 0.0
overlap = sum(min(count, hyp.get(gram, 0)) for gram, count in ref.items())
return overlap / total
def load_passages(pattern: str, count: int, seed: int) -> list[str]:
passages = []
for path in sorted(glob.glob(pattern)):
data = json.loads(Path(path).read_text(encoding="utf-8"))
for row in data.get("results", []):
text = (row.get("source_text") or "").strip()
if 400 <= len(text) <= 3000:
passages.append(text)
random.Random(seed).shuffle(passages)
return passages[:count]
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--glob", default="/w/ai_publish/data_output4/*.json")
parser.add_argument("--count", type=int, default=30)
parser.add_argument("--seed", type=int, default=20260916)
parser.add_argument("--ref-model", default="gpt-4o")
parser.add_argument("--sys-model", default="gpt-4o-mini")
parser.add_argument("--multi-ref", type=int, default=1,
help="참조 개수. 2 이상이면 다중 참조 중 최고값으로 채점 (계획서 전제)")
parser.add_argument("--length-matched", action="store_true",
help="시스템 요약 길이를 참조 규격(원문의 25~35%)에 맞춘다")
parser.add_argument("--out", type=Path, default=Path("reports/summary_bench.json"))
args = parser.parse_args()
from openai import OpenAI
from app.core.config import get_settings
from app.engine.summarizer import Summarizer
get_settings.cache_clear()
settings = get_settings()
client = OpenAI(api_key=settings.openai_api_key)
summarizer = Summarizer(settings)
passages = load_passages(args.glob, args.count, args.seed)
print("본문 %d건 (전기)" % len(passages))
if not passages:
parser.error("본문을 찾지 못했습니다. --glob 확인")
def ask(model: str, prompt: str, text: str) -> str:
response = client.chat.completions.create(
model=model, temperature=0.2,
messages=[{"role": "system", "content": "당신은 한국어 요약 전문가입니다. 원문에 없는 사실을 만들지 마십시오."},
{"role": "user", "content": prompt + "\n\n" + text}])
return (response.choices[0].message.content or "").strip()
REF_PROMPT = ("다음 글을 한국어 줄글로 요약하십시오. 원문 분량의 25~35% 길이로, "
"핵심 인물·사건·시간·장소·인과를 보존하고 목록이나 제목은 쓰지 마십시오.")
# recall 은 참조를 얼마나 덮었는지를 재므로, 시스템 요약이 참조보다 지나치게
# 짧으면 구조적으로 손해를 본다. 길이 규격을 참조와 맞춘다.
SYS_PROMPT = (REF_PROMPT if args.length_matched else
"다음 글을 한국어 줄글로 간결하게 요약하십시오. 원문에 없는 내용을 넣지 마십시오.")
rows = []
for index, text in enumerate(passages, start=1):
# 다중 참조: 계획서가 전제하는 방식. 참조마다 채점해 최고값을 취한다.
references = [ask(args.ref_model, REF_PROMPT, text) for _ in range(args.multi_ref)]
extractive = summarizer.summarize(text, ratio=0.3, use_abstractive=False).final
abstractive = ask(args.sys_model, SYS_PROMPT, text)
row = {"index": index,
"reference_len": sum(len(r) for r in references) // len(references),
"system_len": len(abstractive)}
for label, system in (("extractive", extractive), ("abstractive", abstractive)):
for n in (1, 2):
row["%s_r%d" % (label, n)] = max(
rouge_recall(reference, system, n) for reference in references)
row["%s_r%d_lemma" % (label, n)] = max(
rouge_recall(" ".join(lemmas(reference)), " ".join(lemmas(system)), n)
for reference in references)
rows.append(row)
if index % 10 == 0:
print(" 진행 %d/%d" % (index, len(passages)))
def mean(key: str) -> float:
return sum(r[key] for r in rows) / len(rows)
print("\n" + "=" * 62)
print("ROUGE recall — 지표 정의별 (은 기준 참조 대비, 내부 비교용)")
print(" %-22s %-12s %-12s %s" % ("정의", "① 추출", "② LLM 추상", "차이"))
for label, key in (("ROUGE-1 (어절)", "r1"), ("ROUGE-2 (어절)", "r2"),
("ROUGE-1 (형태소)", "r1_lemma"), ("ROUGE-2 (형태소)", "r2_lemma")):
a, b = mean("extractive_" + key), mean("abstractive_" + key)
print(" %-22s %-12.4f %-12.4f %+.4f" % (label, a, b, b - a))
extractive_score = mean("extractive_r2")
abstractive_score = mean("abstractive_r2")
print("\n※ 참조가 LLM 산출물이라 ②에 유리한 편향이 있습니다. 성적서용 수치가 아닙니다.")
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(json.dumps({
"note": "내부 A/B. 참조는 %s 산출물(은 기준)이며 성적서용이 아님." % args.ref_model,
"count": len(rows), "ref_model": args.ref_model, "sys_model": args.sys_model,
"extractive": extractive_score, "abstractive": abstractive_score,
"rows": rows,
}, ensure_ascii=False, indent=2), encoding="utf-8")
print("%s 에 기록" % args.out)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@ -0,0 +1,149 @@
"""Build a frozen 1,000-source silver-reference trial; persist every model output.
This is a newly defined experiment, not a reproduction of the undocumented 2025
tokenizer. Report character and whitespace bigram F1 without choosing by score.
"""
from __future__ import annotations
import argparse
import concurrent.futures
import hashlib
import json
import random
import re
from collections import Counter
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
PROMPTS = {
"reference": "다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. 핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. 배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요.",
"baseline": "다음 전기 문단을 최대 50자 이내로 요약하세요. 핵심 주제를 포함하고 불필요한 수식어를 제거하며 원문에 없는 내용을 추가하지 마세요. 요약문만 출력하세요.",
"candidate": "전기 문단을 읽고 가장 중요한 인물, 핵심 사건, 그 결과를 찾아 한국어 한 문장 50자 이내로 요약하세요. 원문의 핵심 명사와 동사를 유지하고 중복·주변 묘사는 제외하세요. 사실을 추가하지 말고 핵심 사건을 구체적으로 표현하세요. 요약만 출력하세요.",
}
def digest(text):
return hashlib.sha256(text.encode()).hexdigest()
def metrics(reference, hypothesis, mode):
def grams(text):
tokens = list(re.sub(r"\s+", "", text)) if mode == "character" else text.split()
return Counter(tuple(tokens[i:i + 2]) for i in range(len(tokens) - 1))
a, b = grams(reference), grams(hypothesis)
overlap = sum((a & b).values())
recall = overlap / sum(a.values()) if a else 0.0
precision = overlap / sum(b.values()) if b else 0.0
return {"precision": precision, "recall": recall,
"f1": 2 * precision * recall / (precision + recall) if precision + recall else 0.0}
def main():
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--source", type=Path, required=True)
ap.add_argument("--out", type=Path, required=True)
ap.add_argument("--workers", type=int, default=12)
args = ap.parse_args()
args.out.mkdir(parents=True, exist_ok=True)
manifest_path = args.out / "manifest.json"
dataset_path = args.out / "sources.jsonl"
if not manifest_path.exists():
groups, seen = [], set()
rng = random.Random(20260916)
for p in sorted(args.source.glob("*.json")):
group = []
for row in json.loads(p.read_text()).get("results", []):
text = str(row.get("source_text", "")).strip()
h = digest(" ".join(text.split()))
if not 200 <= len(text) <= 2000 or h in seen:
continue
seen.add(h)
group.append({"source_file": p.name, "source_index": row.get("index"),
"text": text, "source_sha256": h})
rng.shuffle(group)
if group:
groups.append(group)
selected = []
while len(selected) < 1000 and any(groups):
for group in groups:
if group and len(selected) < 1000:
selected.append(dict(group.pop(), id=len(selected) + 1))
if len(selected) != 1000:
raise ValueError("Need 1,000 unique source passages")
payload = "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in selected)
dataset_path.write_text(payload)
manifest = {"count": 1000, "seed": 20260916, "source_books": len(groups),
"dataset_sha256": digest(payload), "reference_origin": "AI-generated; not human-reviewed",
"models": {"reference": "gpt-4o", "baseline": "gpt-4o-mini", "candidate": "gpt-4o"},
"prompts": PROMPTS, "temperature": 0.3, "max_tokens": 100,
"tokenizers": {"character": "remove whitespace, preserve punctuation, Unicode characters",
"word": "Python str.split, preserve punctuation"},
"n": 2, "aggregation": "macro mean F1 across all 1000 sources",
"legacy_equivalence": "unconfirmed: original ngram_tokenize unavailable",
"training_exclusion": "not locally fine-tuned; foundation-model training overlap unknown"}
manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2))
manifest = json.loads(manifest_path.read_text())
payload = dataset_path.read_text()
if digest(payload) != manifest["dataset_sha256"]:
raise ValueError("Frozen dataset hash mismatch")
rows = [json.loads(line) for line in payload.splitlines()]
from app.core.config import get_settings
from openai import OpenAI
client = OpenAI(api_key=get_settings().openai_api_key, timeout=90, max_retries=3)
outputs = {}
for role in ("reference", "baseline", "candidate"):
path = args.out / (role + ".jsonl")
existing = [json.loads(line) for line in path.read_text().splitlines()] if path.exists() else []
done = {row["id"]: row for row in existing if row.get("summary")}
def generate(row):
try:
response = client.chat.completions.create(
model=manifest["models"][role], temperature=manifest["temperature"],
max_tokens=manifest["max_tokens"], messages=[
{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
{"role": "user", "content": manifest["prompts"][role] + "\n\n" + row["text"]}])
choice = response.choices[0]
summary = (choice.message.content or "").strip()
return {"id": row["id"], "summary": summary, "model": response.model,
"response_id": response.id, "finish_reason": choice.finish_reason,
"usage": response.usage.model_dump() if response.usage else None,
"over_50_chars": len(summary) > 50}
except Exception as exc:
return {"id": row["id"], "summary": "", "error_type": type(exc).__name__}
with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool, path.open("a") as out:
futures = [pool.submit(generate, row) for row in rows if row["id"] not in done]
for future in concurrent.futures.as_completed(futures):
result = future.result()
out.write(json.dumps(result, ensure_ascii=False) + "\n")
out.flush()
done[result["id"]] = result
if len(done) % 50 == 0:
print(role, len(done), "/ 1000", flush=True)
outputs[role] = done
result = {"sample_count": 1000, "reference_origin": manifest["reference_origin"],
"legacy_equivalence": manifest["legacy_equivalence"], "certificate_target_achieved": None,
"dataset_sha256": manifest["dataset_sha256"], "scores": {}, "quality": {}}
for role, items in outputs.items():
result["quality"][role] = {"empty": sum(not r.get("summary") for r in items.values()),
"over_50_chars": sum(r.get("over_50_chars", False) for r in items.values()),
"truncated": sum(r.get("finish_reason") == "length" for r in items.values())}
per_item = []
for role in ("baseline", "candidate"):
result["scores"][role] = {}
for mode in ("character", "word"):
scores = []
for row in rows:
rid = row["id"]
score = metrics(outputs["reference"][rid]["summary"], outputs[role][rid]["summary"], mode)
scores.append(score)
per_item.append({"id": rid, "role": role, "mode": mode, **score})
result["scores"][role][mode] = {key: sum(s[key] for s in scores) / 1000 for key in ("precision", "recall", "f1")}
(args.out / "per_item.jsonl").write_text("".join(json.dumps(r) + "\n" for r in per_item))
(args.out / "result.json").write_text(json.dumps(result, ensure_ascii=False, indent=2))
print(json.dumps(result, ensure_ascii=False, indent=2), flush=True)
if __name__ == "__main__":
main()

View File

@ -0,0 +1,105 @@
"""Develop two-stage compression, then re-evaluate one frozen configuration."""
from __future__ import annotations
import argparse
from collections import Counter
import concurrent.futures
import hashlib
import json
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_grounded import generate_grounded, EVIDENCE_PROMPT, COMPRESSION_INSTRUCTION
from app.engine.summary_consensus import select_consensus, SHORT_SUMMARY_PROMPT
from scripts.build_summary_trial import metrics
def read(path):
return {r['id']: r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())}
def save(path, value):
path.write_text(json.dumps(value, ensure_ascii=False, indent=2)+'\n')
def pick(row, strategy):
texts=[c['summary'] for c in row['candidates']]
i=select_consensus(texts,word_weight=0.5) if strategy=='consensus5' else (0 if texts[0] and len(texts[0])<=50 else -1)
return texts[i] if i>=0 else ''
def score(generated, refs, strategy):
return {m:{k:sum(metrics(refs[i],pick(row,strategy),m)[k] for i,row in generated.items())/len(generated)
for k in ('precision','recall','f1')} for m in ('character','word')}
def run(rows, path, client, count):
done=read(path) if path.exists() else {}
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool,path.open('a') as out:
pending={pool.submit(generate_grounded,client,r['text'],count):i for i,r in rows.items() if i not in done}
for f in concurrent.futures.as_completed(pending):
i=pending[f];value=dict(f.result(),id=i);done[i]=value
out.write(json.dumps(value,ensure_ascii=False)+'\n');out.flush()
if len(done)%50==0:print(path.name,len(done),'/',len(rows),flush=True)
assert set(done)==set(rows)
return done
def main():
ap=argparse.ArgumentParser(description=__doc__)
ap.add_argument('--trial',type=Path,required=True);ap.add_argument('--previous',type=Path,required=True)
ap.add_argument('--out',type=Path,required=True);args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True)
development=read(args.trial/'development.jsonl');previous=read(args.previous/'sources.jsonl')
devsources={i:previous[i] for i in development}
devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()}
from app.engine.structural import extract_lemmas
counts=Counter();diagnostics=[]
for i,row in sorted(development.items()):
hyp=pick(row,'consensus5');ref=devrefs[i];word=metrics(ref,hyp,'word')['f1']
a,b=Counter(extract_lemmas(ref)),Counter(extract_lemmas(hyp))
lf=2*sum((a&b).values())/max(1,sum(a.values())+sum(b.values()))
category='above_target' if word>=.65 else ('surface_difference_candidate' if lf>=.7 else ('content_selection_candidate' if lf<.4 else 'mixed_or_uncertain'))
counts[category]+=1;diagnostics.append({'id':i,'word_f1':word,'lemma_overlap_f1':lf,'category':category})
save(args.out/'diagnostics.json',{'note':'heuristic triage, not human error labels; lemma overlap is not the scoring metric',
'counts':dict(counts),'rows':diagnostics})
protocol={'reference_changes':False,'score_changes':False,'evaluation_type':'re-evaluation of previously inspected 1000-item test set',
'development_count':len(development),'selection_metric':'word_bigram_f1','evidence_prompt':EVIDENCE_PROMPT,
'summary_prompt':SHORT_SUMMARY_PROMPT,'compression_instruction':COMPRESSION_INSTRUCTION,
'strategies':['single','consensus5'],'model':'gpt-4o','evidence_temperature':0,'summary_temperature':0.3,
'reference_origin':'AI-generated, unreviewed','test_source_sha256':hashlib.sha256((args.trial/'sources.jsonl').read_bytes()).hexdigest()}
if (args.out/'protocol.json').exists():assert json.loads((args.out/'protocol.json').read_text())==protocol
else:save(args.out/'protocol.json',protocol)
from openai import OpenAI
from app.core.config import get_settings
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
devgen=run(devsources,args.out/'development.jsonl',client,5)
devscores={s:score(devgen,devrefs,s) for s in ('single','consensus5')}
selected=max(devscores,key=lambda s:devscores[s]['word']['f1'])
decision={'selected_experimental_strategy':selected,'current_development_scores':score(development,devrefs,'consensus5'),
'experimental_development_scores':devscores,'test_evaluation_reason':'user-requested comparison, not automatic promotion'}
if (args.out/'decision.json').exists():assert json.loads((args.out/'decision.json').read_text())==decision
else:save(args.out/'decision.json',decision)
print(json.dumps(decision),flush=True)
rows=read(args.trial/'sources.jsonl');refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()}
assert len(rows)==1000 and set(rows)==set(refs)
generated=run(rows,args.out/'system.jsonl',client,5 if selected=='consensus5' else 1)
newscore=score(generated,refs,selected);oldscore=score(read(args.trial/'system.jsonl'),refs,'consensus5')
details=[]
for i,row in sorted(rows.items()):
hyp=pick(generated[i],selected)
details.append({'index':i,'original':row['text'],'reference_summary':refs[i],'summary':hyp,
'rouge_score':metrics(refs[i],hyp,'character')['f1'],'word_rouge_score':metrics(refs[i],hyp,'word')['f1']})
result={'method':'evidence_then_compression_'+selected,'sample_count':1000,'target_rouge':.65,
'average_rouge':newscore['character']['f1'],'average_word_rouge':newscore['word']['f1'],
'success_rate':sum(r['rouge_score']>=.65 for r in details)/1000,
'word_success_rate':sum(r['word_rouge_score']>=.65 for r in details)/1000,
'success_rate_definition':'fraction of individual items with F1 >= 0.65',
'previous_scores':oldscore,'scores':newscore,'word_improved':newscore['word']['f1']>oldscore['word']['f1'],
'evaluation_type':protocol['evaluation_type'],'reference_origin':protocol['reference_origin'],
'certificate_target_achieved':None,'evidence_fallback_count':sum(r['evidence_fallback'] for r in generated.values()),
'empty_count':sum(not r['summary'] for r in details),'data':details}
save(args.out/'scorecard.json',result)
print(json.dumps({k:v for k,v in result.items() if k!='data'},ensure_ascii=False,indent=2),flush=True)
if __name__=='__main__':main()

View File

@ -8,7 +8,7 @@
사용:
python scripts/evaluate_o2o_dataset.py \
--data-dir /Users/marineyang/Desktop/work/code/AI_publish_3rdtest/25/plagia_result
--data-dir ./data/eval/plagia_result
"""
from __future__ import annotations

View File

@ -0,0 +1,81 @@
#!/usr/bin/env python3
"""완료된 사람 요약 검수 XLSX를 ROUGE 평가용 JSONL로 내보낸다.
빈 참조·미완료 행·동일 참조를 실패 처리해, 검수 전 데이터를 성적서 평가에
실수로 사용하지 않게 한다. 원문과 작성자 그룹은 유지하되 검수자 실명은 내보내지 않는다.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
REQUIRED_COLUMNS = {
"annotation_id", "source_group", "source_text", "reference_summary_1",
"reference_summary_2", "status",
}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("xlsx", type=Path)
parser.add_argument("--out", type=Path, required=True)
parser.add_argument("--allow-single-reference", action="store_true",
help="2차 독립 검수가 끝나기 전 파일럿에만 사용")
args = parser.parse_args()
from openpyxl import load_workbook
book = load_workbook(args.xlsx, read_only=True, data_only=True)
if "annotations" not in book.sheetnames:
parser.error("annotations 시트가 없습니다")
sheet = book["annotations"]
headers = [str(cell.value or "").strip() for cell in next(sheet.iter_rows(max_row=1))]
positions = {name: index for index, name in enumerate(headers)}
missing = REQUIRED_COLUMNS - positions.keys()
if missing:
parser.error("필수 열이 없습니다: %s" % ", ".join(sorted(missing)))
rows, errors = [], []
for excel_row, values in enumerate(sheet.iter_rows(min_row=2, values_only=True), start=2):
value = lambda key: str(values[positions[key]] or "").strip()
status = value("status").casefold()
if not any(values):
continue
if status != "completed":
errors.append("행 %d: status가 completed가 아닙니다" % excel_row)
continue
refs = [value("reference_summary_1"), value("reference_summary_2")]
refs = [ref for ref in refs if ref]
if not args.allow_single_reference and len(refs) != 2:
errors.append("행 %d: 독립 참조 요약 2개가 필요합니다" % excel_row)
continue
if refs and len({" ".join(ref.split()) for ref in refs}) != len(refs):
errors.append("행 %d: 두 참조 요약이 동일합니다" % excel_row)
continue
if not refs or not value("source_text"):
errors.append("행 %d: 원문 또는 참조 요약이 비어 있습니다" % excel_row)
continue
rows.append({
"id": value("annotation_id"),
"source_group": value("source_group"),
"text": value("source_text"),
"references": refs,
})
book.close()
if errors:
print("내보내지 않음 — 검수 오류 %d건:" % len(errors))
print("\n".join(errors[:20]))
return 2
if not rows:
print("내보낼 completed 행이 없습니다")
return 2
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text("".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), encoding="utf-8")
print("저장: %s / %d건 / 작성자 그룹 %d개" %
(args.out, len(rows), len({row['source_group'] for row in rows})))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@ -0,0 +1,127 @@
"""Development-only selection followed by a fresh 1,000-source evaluation."""
from __future__ import annotations
import argparse
import concurrent.futures
import hashlib
import json
import random
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_consensus import generate_candidates, select_consensus, SHORT_SUMMARY_PROMPT
from scripts.build_summary_trial import metrics, PROMPTS
def read(path):
return [json.loads(l) for l in path.read_text().splitlines() if l.strip()]
def save(path, value):
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n")
def sha(text):
return hashlib.sha256(text.encode()).hexdigest()
def run_rows(rows, path, fn):
done = {r['id']: r for r in read(path)} if path.exists() else {}
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool, path.open('a') as out:
pending = {pool.submit(fn, r): r['id'] for r in rows if r['id'] not in done}
for future in concurrent.futures.as_completed(pending):
rid = pending[future]
value = future.result() # fail visibly; resume persisted successes
done[rid] = dict(value, id=rid)
out.write(json.dumps(done[rid], ensure_ascii=False) + '\n'); out.flush()
if len(done) % 50 == 0:
print(path.name, len(done), '/', len(rows), flush=True)
return done
def summary(row, strategy):
texts = [c['summary'] for c in row['candidates']]
index = 0 if strategy == 'single' else select_consensus(texts)
return texts[index] if index >= 0 else ''
def score(rows, references, generated, strategy):
return {mode: {key: sum(metrics(references[r['id']], summary(generated[r['id']], strategy), mode)[key]
for r in rows) / len(rows) for key in ('precision','recall','f1')}
for mode in ('character','word')}
def main():
ap=argparse.ArgumentParser(); ap.add_argument('--previous',type=Path,required=True)
ap.add_argument('--source',type=Path,required=True); ap.add_argument('--out',type=Path,required=True)
args=ap.parse_args(); args.out.mkdir(parents=True,exist_ok=True)
previous=read(args.previous/'sources.jsonl')
oldrefs={r['id']:r['summary'] for r in read(args.previous/'reference.jsonl')}
dev_books=set(sorted({r['source_file'] for r in previous})[:10])
dev=[r for r in previous if r['source_file'] in dev_books]
if not (args.out/'protocol.json').exists():
save(args.out/'protocol.json',{'development_books':sorted(dev_books),'development_count':len(dev),
'primary_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1',
'strategies':['single','consensus5'],'reference_prompt':PROMPTS['reference'],
'system_prompt':SHORT_SUMMARY_PROMPT,'model':'gpt-4o','temperature':0.3,
'reference_origin':'AI-generated, unreviewed','bias':'same model and task instructions for reference and system',
'legacy_tokenizer_equivalence':'unconfirmed','target':0.65})
from app.core.config import get_settings
from openai import OpenAI
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
generated=run_rows(dev,args.out/'development.jsonl',lambda row:{'candidates':generate_candidates(client,row['text'])})
development={s:score(dev,oldrefs,generated,s) for s in ('single','consensus5')}
selected=max(development,key=lambda s:development[s]['character']['f1'])
decision={'development_scores':development,'selected':selected,'selected_before_test':True}
decision_path=args.out/'decision.json'
if decision_path.exists():
assert json.loads(decision_path.read_text())==decision
else:save(decision_path,decision)
print(json.dumps(decision),flush=True)
# Hold out entire development books and every previous trial paragraph.
dataset=args.out/'sources.jsonl'
if not dataset.exists():
seen={sha(' '.join(r['text'].split())) for r in previous};groups=[];rng=random.Random(20260917)
for p in sorted(args.source.glob('*.json')):
if p.name in dev_books:continue
group=[]
for r in json.loads(p.read_text()).get('results',[]):
text=str(r.get('source_text','')).strip();h=sha(' '.join(text.split()))
if not 200<=len(text)<=2000 or h in seen:continue
seen.add(h);group.append({'text':text,'source_file':p.name,'source_index':r.get('index'),'source_sha256':h})
rng.shuffle(group)
if group:groups.append(group)
rows=[]
while len(rows)<1000 and any(groups):
for g in groups:
if g and len(rows)<1000:rows.append(dict(g.pop(),id=len(rows)+1))
assert len(rows)==1000
dataset.write_text(''.join(json.dumps(r,ensure_ascii=False)+'\n' for r in rows))
save(args.out/'dataset_manifest.json',{'count':1000,'sha256':sha(dataset.read_text()),'seed':20260917,
'development_book_overlap':0,'previous_exact_text_overlap':0,'near_duplicate_screening':'not performed'})
rows=read(dataset);assert sha(dataset.read_text())==json.loads((args.out/'dataset_manifest.json').read_text())['sha256']
refs=run_rows(rows,args.out/'reference.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],1)})
reftexts={i:r['candidates'][0]['summary'] for i,r in refs.items()}
# Freeze candidate selection before any held-out score is calculated.
systems=run_rows(rows,args.out/'system.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],5 if selected=='consensus5' else 1)})
def baseline(row):
resp=client.chat.completions.create(model='gpt-4o',temperature=0.3,max_tokens=100,messages=[
{'role':'system','content':'당신은 텍스트 요약 전문가입니다.'},
{'role':'user','content':PROMPTS['candidate']+'\n\n'+row['text']}])
c=resp.choices[0]
return {'candidates':[{'summary':(c.message.content or '').strip() if c.finish_reason=='stop' else '',
'finish_reason':c.finish_reason,'model':resp.model,'response_id':resp.id}]}
base=run_rows(rows,args.out/'baseline.jsonl',baseline)
result={'count':1000,'selected':selected,'reference_origin':'AI-generated, unreviewed',
'bias':'same model and task instructions for reference and improved system',
'certificate_target_achieved':None,'legacy_tokenizer_equivalence':'unconfirmed',
'baseline':score(rows,reftexts,base,'single'),'improved':score(rows,reftexts,systems,selected)}
result['primary_target_achieved']=result['improved']['character']['f1']>=0.65
result['quality']={label:{'empty':sum(not s for s in texts),'over50':sum(len(s)>50 for s in texts)}
for label,texts in [('reference',list(reftexts.values())),('baseline',[summary(base[r['id']],'single') for r in rows]),
('improved',[summary(systems[r['id']],selected) for r in rows])]}
save(args.out/'result.json',result);print(json.dumps(result,ensure_ascii=False,indent=2),flush=True)
if __name__=='__main__':main()

View File

@ -0,0 +1,70 @@
"""Re-evaluate a development-selected word-aware selector; no new API calls.
Reuses the previously inspected test set. This is a re-evaluation, not a fresh
independent certification test. Preserve earlier artifacts and report both metrics.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_consensus import select_consensus
from scripts.build_summary_trial import metrics
def read(path):
return {r['id']:r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())}
def evaluate(generated, references, weight):
result=[]
for rid,row in sorted(generated.items()):
texts=[c['summary'] for c in row['candidates']]
chosen=select_consensus(texts, word_weight=weight)
text=texts[chosen] if chosen>=0 else ''
result.append({'index':rid,'summary':text,'selected_index':chosen,
'scores':{m:metrics(references[rid],text,m) for m in ('character','word')}})
scores={m:{k:sum(r['scores'][m][k] for r in result)/len(result)
for k in ('precision','recall','f1')} for m in ('character','word')}
return scores,result
def main():
ap=argparse.ArgumentParser(description=__doc__)
ap.add_argument('--trial',type=Path,required=True)
ap.add_argument('--previous',type=Path,required=True)
ap.add_argument('--out',type=Path,required=True)
args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True)
dev=read(args.trial/'development.jsonl')
devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()}
# Development comparison only; test scores do not choose this weight.
compared={str(w):evaluate(dev,devrefs,w)[0] for w in (0.0,0.5,1.0)}
weight=max((0.0,0.5,1.0),key=lambda w:compared[str(w)]['word']['f1'])
decision={'selected_word_weight':weight,'selection_metric':'development_word_bigram_f1',
'development_scores':compared,'evaluation_type':'re-evaluation of previously inspected test set'}
(args.out/'decision.json').write_text(json.dumps(decision,ensure_ascii=False,indent=2)+'\n')
systems=read(args.trial/'system.jsonl'); sources=read(args.trial/'sources.jsonl')
refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()}
assert set(systems)==set(sources)==set(refs)==set(range(1,1001))
scores,rows=evaluate(systems,refs,weight)
old,_=evaluate(systems,refs,0.0)
full=[]
for row in rows:
i=row['index'];full.append({'index':i,'original':sources[i]['text'],
'reference_summary':refs[i],'summary':row['summary'],
'rouge_score':row['scores']['character']['f1'],'word_rouge_score':row['scores']['word']['f1']})
record={'method':'source_only_consensus5_char_word_equal_weight','target_rouge':0.65,
'average_rouge':scores['character']['f1'],'average_word_rouge':scores['word']['f1'],
'success_rate':sum(r['rouge_score']>=0.65 for r in full)/1000,
'word_success_rate':sum(r['word_rouge_score']>=0.65 for r in full)/1000,
'success_rate_definition':'fraction of individual items with F1 >= 0.65',
'sample_count':1000,'reference_origin':'AI-generated, unreviewed',
'reported_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1',
'evaluation_type':decision['evaluation_type'],'certificate_target_achieved':None,
'previous_scores':old,'scores':scores,'data':full}
(args.out/'scorecard.json').write_text(json.dumps(record,ensure_ascii=False,indent=2)+'\n')
print(json.dumps({k:v for k,v in record.items() if k!='data'},ensure_ascii=False,indent=2))
if __name__=='__main__':main()

68
scripts/rescore_recall.py Normal file
View File

@ -0,0 +1,68 @@
"""저장된 1,000건 재채점 — 문자 2-gram recall (계획서 p.24 수식).
API 재호출 없음. H절 재평가 산출물의 참조·출력 원문을 그대로 사용하고
집계 지표만 F1 에서 recall 로 바꾼다. 새 독립 시험이 아니다.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from scripts.build_summary_trial import metrics
TARGET = 0.65
def main() -> None:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--source", type=Path,
default=Path("data/eval/summary_word_refined_20260917/scorecard.json"))
ap.add_argument("--out", type=Path, default=Path("data/eval/summary_recall_20260917"))
ap.add_argument("--excerpt", type=int, default=160, help="캡처본 원문 발췌 길이")
args = ap.parse_args()
args.out.mkdir(parents=True, exist_ok=True)
rows = json.loads(args.source.read_text())["data"]
scored = [{
"index": row["index"],
"original": row["original"],
"reference_summary": row["reference_summary"],
"summary": row["summary"],
"rouge_score": metrics(row["reference_summary"], row["summary"], "character")["recall"],
} for row in rows]
average = sum(r["rouge_score"] for r in scored) / len(scored)
provenance = {
"metric": "character_bigram_recall",
"metric_formula": "sum(match n-gram) / sum(reference n-gram)",
"sample_count": len(scored),
"reference_origin": "AI-generated (GPT-4o), unreviewed",
"reference_count_per_item": 1,
"evaluation_type": "re-evaluation of previously inspected test set",
}
scorecard = {
"method": "source_only_consensus5_char_word_equal_weight",
"target_rouge": TARGET,
"average_rouge": average,
"success_rate": sum(r["rouge_score"] >= TARGET for r in scored) / len(scored),
"data": scored,
**provenance,
}
(args.out / "scorecard.json").write_text(
json.dumps(scorecard, ensure_ascii=False, indent=2) + "\n")
capture = dict(scorecard)
capture["data"] = [{**r, "original": r["original"][:args.excerpt] + " …"}
for r in scored[:3]]
(args.out / "capture.json").write_text(
json.dumps(capture, ensure_ascii=False, indent=2) + "\n")
print(f"문자 2-gram recall 평균 {average * 100:.2f}% "
f"(목표 {TARGET * 100:.0f}%, {len(scored)}건)")
print(f"success_rate {scorecard['success_rate'] * 100:.1f}%")
if __name__ == "__main__":
main()

View File

@ -0,0 +1,91 @@
#!/usr/bin/env python3
"""사람 참조 요약으로 추출 전략을 고르고, 잠금 테스트셋에서 한 번만 측정한다.
선택(검증)과 최종 보고(테스트)를 분리해, 65점에 맞춰 전 평가 요약문을 수정하는
전 차수식 접근을 방지한다. source_group 단위로 분리하므로 같은 작성자의 문장이
튜닝·시험 양쪽에 섞이지 않는다.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
from app.engine.rouge import evaluate_pairs
from app.engine.summarizer import extractive_summary
def stable_bucket(group: str, seed: int) -> int:
return int(hashlib.sha256((str(seed) + ":" + group).encode()).hexdigest()[:8], 16) % 100
def split_rows(rows: list[dict], seed: int) -> tuple[list[dict], list[dict]]:
# 20% 그룹을 최종 시험에 잠근다. 나머지에서만 전략을 고른다.
test = [row for row in rows if stable_bucket(row.get("source_group", row["id"]), seed) < 20]
validation = [row for row in rows if row not in test]
if not test or not validation:
raise ValueError("그룹 분할 결과가 비었습니다. source_group을 확인하세요.")
return validation, test
def score(rows: list[dict], ratio: float, strategy: str, mode: str) -> dict[str, dict[str, float]]:
pairs = [
(extractive_summary(row["text"], ratio=ratio, strategy=strategy).final, row["references"])
for row in rows
]
return evaluate_pairs(pairs, mode=mode)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("dataset", type=Path, help="export_summary_annotations.py 출력 JSONL")
parser.add_argument("--out", type=Path, required=True, help="선택 근거와 잠금 시험 결과 JSON")
parser.add_argument("--seed", type=int, default=20260916)
parser.add_argument("--mode", choices=["lemma", "char"], default="lemma")
parser.add_argument("--ratios", type=float, nargs="+", default=[0.25, 0.30, 0.35],
help="사전 등록 가능한 요약 비율 후보(권장 25~35%%)")
args = parser.parse_args()
if any(not 0 < ratio <= 0.35 for ratio in args.ratios):
parser.error("요약 비율은 0보다 크고 0.35 이하여야 합니다")
rows = [json.loads(line) for line in args.dataset.read_text(encoding="utf-8").splitlines() if line.strip()]
if len(rows) < 20:
parser.error("최소 20건 이상의 완료된 사람 참조 요약이 필요합니다")
validation, test = split_rows(rows, args.seed)
candidates = []
for strategy in ("textrank", "coverage", "lead"):
for ratio in args.ratios:
metrics = score(validation, ratio, strategy, args.mode)
candidates.append({"strategy": strategy, "ratio": ratio, "validation": metrics})
# 같은 recall이면 ROUGE-L F1, 더 짧은 출력 순으로 결정한다.
winner = max(candidates, key=lambda item: (
item["validation"]["rouge1"]["recall"],
item["validation"]["rougeL"]["f1"],
-item["ratio"],
))
final = score(test, winner["ratio"], winner["strategy"], args.mode)
record = {
"purpose": "summary strategy selection with source-group-held-out final test",
"dataset": str(args.dataset), "seed": args.seed, "tokenization": args.mode,
"samples": {"total": len(rows), "validation": len(validation), "locked_test": len(test)},
"candidates": candidates, "selected": {"strategy": winner["strategy"], "ratio": winner["ratio"]},
"locked_test_scores": final,
"reported_metric": "rouge1_recall",
"target": 0.65,
"achieved": final["rouge1"]["recall"] >= 0.65,
}
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(json.dumps(record, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print("선택: %s / ratio=%.2f" % (winner["strategy"], winner["ratio"]))
print("잠금 테스트 ROUGE-1 recall: %.4f (%s)" %
(final["rouge1"]["recall"], "달성" if record["achieved"] else "미달"))
print("근거 저장: %s" % args.out)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@ -0,0 +1,53 @@
"""Offline evidence and F1 verification for the two-stage experiment."""
from __future__ import annotations
import argparse
from collections import Counter
import hashlib
import json
from pathlib import Path
import re
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_consensus import select_consensus
def rows(path):
items=[json.loads(l) for l in path.read_text().splitlines() if l.strip()]
assert len(items)==1000 and {r['id'] for r in items}==set(range(1,1001))
return {r['id']:r for r in items}
def main():
ap=argparse.ArgumentParser(description=__doc__)
ap.add_argument('--trial',type=Path,required=True)
ap.add_argument('--experiment',type=Path,required=True)
a=ap.parse_args();sources=rows(a.trial/'sources.jsonl');refs=rows(a.trial/'reference.jsonl')
outputs=rows(a.experiment/'system.jsonl');result=json.loads((a.experiment/'scorecard.json').read_text())
protocol=json.loads((a.experiment/'protocol.json').read_text())
assert hashlib.sha256((a.trial/'sources.jsonl').read_bytes()).hexdigest()==protocol['test_source_sha256']
strategy=json.loads((a.experiment/'decision.json').read_text())['selected_experimental_strategy']
details={r['index']:r for r in result['data']};assert len(details)==len(result['data'])==1000
total={'character':0.,'word':0.}
for rid in range(1,1001):
row=outputs[rid];text=sources[rid]['text']
assert all(e in text and 1<=len(e)<=120 for e in row['evidence'])
assert len(row['evidence'])<=3
summaries=[c['summary'] for c in row['candidates']]
i=select_consensus(summaries,.5) if strategy=='consensus5' else (0 if summaries[0] and len(summaries[0])<=50 else -1)
hyp=summaries[i] if i>=0 else '';ref=refs[rid]['candidates'][0]['summary']
assert details[rid]['summary']==hyp and details[rid]['reference_summary']==ref and details[rid]['original']==text
for mode in total:
def grams(t):
t=list(re.sub(r'\s+','',t)) if mode=='character' else t.split()
return Counter(zip(t,t[1:]))
x,y=grams(ref),grams(hyp);denom=sum(x.values())+sum(y.values())
f1=2*sum(min(n,y[g]) for g,n in x.items())/denom if denom else 0
assert abs(f1-details[rid]['rouge_score' if mode=='character' else 'word_rouge_score'])<1e-12
total[mode]+=f1/1000
for mode,avg in total.items():
assert abs(avg-result['scores'][mode]['f1'])<1e-12
print(mode,'F1=%.6f%%'%(100*avg))
print('PASS: 1000 IDs, unchanged sources/references, literal evidence, selected outputs, per-item and mean F1')
if __name__=='__main__':main()

View File

@ -0,0 +1,60 @@
"""Recompute the frozen improvement trial locally, without model/API calls."""
from __future__ import annotations
import argparse
from collections import Counter
import hashlib
import json
from pathlib import Path
import re
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_consensus import select_consensus
def load_rows(path):
rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
assert len(rows) == 1000 and {r['id'] for r in rows} == set(range(1, 1001)), path
return {r['id']: r for r in rows}
def main():
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument('directory', type=Path)
args = ap.parse_args()
p = args.directory
manifest = json.loads((p/'dataset_manifest.json').read_text())
assert hashlib.sha256((p/'sources.jsonl').read_bytes()).hexdigest() == manifest['sha256']
sources = load_rows(p/'sources.jsonl')
protocol = json.loads((p/'protocol.json').read_text())
assert not ({r['source_file'] for r in sources.values()} & set(protocol['development_books']))
refs = load_rows(p/'reference.jsonl')
systems = load_rows(p/'system.jsonl')
baseline = load_rows(p/'baseline.jsonl')
result = json.loads((p/'result.json').read_text())
for label, outputs in [('baseline', baseline), ('improved', systems)]:
for mode in ('character', 'word'):
totals = {'precision': 0.0, 'recall': 0.0, 'f1': 0.0}
for rid in range(1, 1001):
candidates = [c['summary'] for c in outputs[rid]['candidates']]
chosen = select_consensus(candidates) if label == 'improved' and result['selected'] == 'consensus5' else 0
hyp = candidates[chosen] if chosen >= 0 else ''
ref = refs[rid]['candidates'][0]['summary']
def grams(text):
tokens = list(re.sub(r'\s+', '', text)) if mode == 'character' else text.split()
return Counter(zip(tokens, tokens[1:]))
a, b = grams(ref), grams(hyp)
overlap = sum(min(count, b[g]) for g, count in a.items())
a_total, b_total = sum(a.values()), sum(b.values())
totals['precision'] += overlap / b_total if b_total else 0.0
totals['recall'] += overlap / a_total if a_total else 0.0
totals['f1'] += 2 * overlap / (a_total + b_total) if a_total + b_total else 0.0
actual = {key: value/1000 for key, value in totals.items()}
assert all(abs(actual[key]-result[label][mode][key]) < 1e-12 for key in actual)
print(label, mode, 'F1=%.6f%%' % (actual['f1']*100))
print('PASS: 1000 IDs, source SHA-256, development-book exclusion, all P/R/F1 values')
if __name__ == '__main__':
main()

View File

@ -66,6 +66,18 @@ def test_emphasis_prioritizes_requested_topic():
assert result.emphasis == ["김밥"]
def test_coverage_strategy_preserves_source_order_and_respects_cap():
result = extractive_summary(_DOC, ratio=0.5, max_sentences=3, strategy="coverage")
assert result.selected_indices == sorted(result.selected_indices)
assert result.num_sentences_out == 3
def test_unknown_extractive_strategy_is_rejected():
import pytest
with pytest.raises(ValueError):
extractive_summary(_DOC, strategy="oracle")
def test_empty_and_short():
assert extractive_summary("").final == ""
one = extractive_summary("한 문장만 있다.")

View File

@ -0,0 +1,40 @@
from app.engine.summary_consensus import character_f1, select_consensus
def test_medoid_prefers_agreed_event_over_outlier():
candidates = ["김씨는 학교를 세웠다.", "김씨는 학교를 세웠다.", "김씨는 미국으로 이민했다."]
assert select_consensus(candidates) == 0
def test_overlength_consensus_cannot_win():
long = "긴" * 51
assert select_consensus([long, long, "학교를 세웠다."]) == 2
assert select_consensus([long, ""]) == -1
def test_empty_outputs_do_not_change_candidate_choice():
assert select_consensus(["", "학교를 세웠다.", "학교를 세웠다."]) == 1
assert character_f1("", "") == 0
def test_bigram_overlap_keeps_multiplicity():
assert character_f1("가가가", "가가") == 2 / 3
def test_word_overlap_and_invalid_weight():
import pytest
from app.engine.summary_consensus import word_f1
assert word_f1("a b c", "a b") == 2 / 3
assert word_f1("a", "a") == 0
with pytest.raises(ValueError):
select_consensus(["요약"], word_weight=1.5)
def test_public_summary_reports_invalid_when_every_candidate_fails(monkeypatch):
from app.engine import summary_consensus
monkeypatch.setattr(summary_consensus, "generate_candidates", lambda client, text: [
{"summary": "", "finish_reason": "length"}, {"summary": "긴" * 51, "finish_reason": "stop"}])
result = summary_consensus.summarize_short(None, "원문")
assert result["valid"] is False
assert result["summary"] == ""
assert result["selected_index"] == -1

View File

@ -0,0 +1,14 @@
from app.engine.summary_grounded import validated_evidence
def test_evidence_rejects_paraphrases_and_nonstrings():
assert validated_evidence('그는 학교를 세웠다.', {'evidence':['학교를 세웠다','학교를 설립했다',42,'']}) == ['학교를 세웠다']
def test_evidence_deduplicates_and_caps():
assert validated_evidence('가 나 다 라', {'evidence':['가','가','나','다','라']}) == ['가','나','다']
def test_bad_schema_and_overlength_are_not_evidence():
assert validated_evidence('원문', {'evidence':'원문'}) == []
assert validated_evidence('가'*121, {'evidence':['가'*121]}) == []