diff --git a/app/engine/summarizer.py b/app/engine/summarizer.py index 3834fba..22f4838 100644 --- a/app/engine/summarizer.py +++ b/app/engine/summarizer.py @@ -31,6 +31,7 @@ from app.engine.structural import extract_lemmas logger = logging.getLogger(__name__) DETAIL_RATIOS = {"brief": 0.2, "standard": 0.3, "detailed": 0.45} +EXTRACTIVE_STRATEGIES = {"textrank", "coverage", "lead"} # 문장 분할 — 종결부호 기준 (한국어 '다./요./까?/!' + 줄바꿈) _SENT_SPLIT = re.compile(r"(?<=[.!?。…])\s+|\n+") @@ -86,6 +87,59 @@ def _textrank_scores(vectors: list[Counter], damping: float = 0.85, iters: int = return scores +def _normalize(scores: list[float]) -> list[float]: + """0~1 정규화. 모든 값이 같으면 동률로 취급한다.""" + if not scores: + return [] + low, high = min(scores), max(scores) + if high - low < 1e-12: + return [1.0] * len(scores) + return [(score - low) / (high - low) for score in scores] + + +def _coverage_scores(sentences: list[str], vectors: list[Counter]) -> list[float]: + """문서의 고유 내용어를 넓게 담는 문장을 우선하는 점수. + + TextRank만 쓰면 같은 사건을 반복하는 문장이 상위를 독점할 수 있다. 이 점수는 + 문서 전체에서 드문 내용어(인물·사건·시간 등)가 든 문장을 높게 평가하고, 선택 + 단계에서 이미 선택한 문장과의 중복을 감점한다. 외부 모델이나 정답셋을 보지 않는 + 순수 추출 방식이다. + """ + document_frequency: Counter = Counter() + for vector in vectors: + document_frequency.update(vector.keys()) + n = max(1, len(sentences)) + lexical = [ + sum(count * math.log((n + 1) / (document_frequency[token] + 0.5)) + for token, count in vector.items()) / max(1, sum(vector.values())) + for vector in vectors + ] + # 회고록·에피소드 형식에서는 첫 문장이 사건의 맥락을 잡는 경우가 많다. 단, + # 위치만으로 결정되지 않게 작은 사전 등록 가중치만 둔다. + position = [1.0 - (index / max(1, n - 1)) * 0.35 for index in range(n)] + centrality = _normalize(_textrank_scores(vectors)) + lexical = _normalize(lexical) + return [0.50 * c + 0.38 * l + 0.12 * p + for c, l, p in zip(centrality, lexical, position)] + + +def _coverage_select(vectors: list[Counter], base_scores: list[float], k: int) -> list[int]: + """MMR 방식으로 중요하면서 서로 다른 정보를 담는 문장 k개를 고른다.""" + chosen: list[int] = [] + remaining = set(range(len(vectors))) + while remaining and len(chosen) < k: + def candidate_score(index: int) -> tuple[float, float, int]: + redundancy = max((_cosine(vectors[index], vectors[other]) for other in chosen), default=0.0) + # 중복 문장은 낮추되, 낮은 중요도의 문장이 끼어들 정도로 과도하게 벌주진 않는다. + value = base_scores[index] - 0.22 * redundancy + return (value, base_scores[index], -index) + + best = max(remaining, key=candidate_score) + chosen.append(best) + remaining.remove(best) + return chosen + + @dataclass class SummaryResult: extractive: str # 추출적 요약 (선택된 원문 문장) @@ -105,6 +159,7 @@ def extractive_summary( max_sentences: int | None = None, emphasis: list[str] | None = None, detail: str = "standard", + strategy: str = "textrank", ) -> SummaryResult: """비지도 추출적 요약 — 정답셋/LLM/외부호출 불필요.""" sentences = split_sentences(text) @@ -123,8 +178,13 @@ def extractive_summary( if max_sentences is not None: k = min(k, max_sentences) + if strategy not in EXTRACTIVE_STRATEGIES: + raise ValueError("strategy must be one of %s" % sorted(EXTRACTIVE_STRATEGIES)) + vectors = [_lemma_vector(s) for s in sentences] - scores = _textrank_scores(vectors) + scores = _textrank_scores(vectors) if strategy == "textrank" else _coverage_scores(sentences, vectors) + if strategy == "lead": + scores = [float(n - index) for index in range(n)] # 사용자가 지정한 주제를 포함한 문장에 명시적 가중치를 준다. # TextRank 중심성은 유지하되 강조 요청이 상위 선택에 반영되도록 한다. @@ -135,7 +195,8 @@ def extractive_summary( scores[i] += 2.0 * matched / len(lowered_terms) # 상위 k개 문장 선택 → 원문 등장 순서로 재정렬 (가독성) - top = sorted(range(n), key=lambda i: scores[i], reverse=True)[:k] + top = (_coverage_select(vectors, scores, k) if strategy == "coverage" + else sorted(range(n), key=lambda i: scores[i], reverse=True)[:k]) top_sorted = sorted(top) summary = " ".join(sentences[i] for i in top_sorted) return SummaryResult( @@ -177,6 +238,7 @@ class Summarizer: use_abstractive: bool = True, detail: str = "standard", emphasis: list[str] | None = None, + strategy: str = "textrank", ) -> SummaryResult: effective_ratio = ratio if ratio is not None else DETAIL_RATIOS.get(detail, 0.3) base = extractive_summary( @@ -185,6 +247,7 @@ class Summarizer: max_sentences=max_sentences, emphasis=emphasis, detail=detail, + strategy=strategy, ) if not base.extractive: return base diff --git a/app/engine/summary_consensus.py b/app/engine/summary_consensus.py new file mode 100644 index 0000000..84a9ae0 --- /dev/null +++ b/app/engine/summary_consensus.py @@ -0,0 +1,75 @@ +"""Source-only short-summary consensus; never accepts reference summaries.""" +from __future__ import annotations + +import re +from collections import Counter + +SHORT_SUMMARY_PROMPT = ( + "다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. " + "핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. " + "배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요." +) + + +def character_f1(a: str, b: str) -> float: + def grams(text): + text = re.sub(r"\s+", "", text) + return Counter(text[i:i + 2] for i in range(len(text) - 1)) + x, y = grams(a), grams(b) + total = sum(x.values()) + sum(y.values()) + return 2 * sum((x & y).values()) / total if total else 0.0 + + +def word_f1(a: str, b: str) -> float: + def grams(text): + words = text.split() + return Counter(zip(words, words[1:])) + x, y = grams(a), grams(b) + total = sum(x.values()) + sum(y.values()) + return 2 * sum((x & y).values()) / total if total else 0.0 + + +def select_consensus(summaries: list[str], word_weight: float = 0.0) -> int: + """Select a medoid among independently generated candidates, under 50 chars. + + Consensus measures stability, not truth: systematic model errors can survive. + Return -1 if all candidates are empty/overlength instead of silently truncating. + """ + if not 0 <= word_weight <= 1: + raise ValueError("word_weight must be between 0 and 1") + eligible = [i for i, s in enumerate(summaries) if s.strip() and len(s.strip()) <= 50] + if not eligible: + return -1 + return max(eligible, key=lambda i: ( + sum((1 - word_weight) * character_f1(summaries[i], summaries[j]) + + word_weight * word_f1(summaries[i], summaries[j]) + for j in eligible if j != i), -i)) + + +def generate_candidates(client, text: str, count: int = 5) -> list[dict]: + if not text.strip(): + raise ValueError("Source text must not be empty") + if not 1 <= count <= 5: + raise ValueError("Candidate count must be between 1 and 5") + response = client.chat.completions.create( + model="gpt-4o", temperature=0.3, max_tokens=100, n=count, + messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."}, + {"role": "user", "content": SHORT_SUMMARY_PROMPT + "\n\n" + text}], + ) + return [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "", + "finish_reason": c.finish_reason, "model": response.model, + "response_id": response.id} for c in response.choices] + + +def summarize_short(client, text: str, word_weight: float = 0.0) -> dict: + """Opt-in 50-character summary; five completions incur additional API cost. + + Return an explicit invalid result if no complete length-compliant summary exists. + Callers should inspect valid, not silently present an empty result as success. + """ + candidates = generate_candidates(client, text) + chosen = select_consensus([c["summary"] for c in candidates], word_weight=word_weight) + return {"summary": candidates[chosen]["summary"] if chosen >= 0 else "", + "valid": chosen >= 0, "selected_index": chosen, + "strategy": "source_only_consensus5", "word_weight": word_weight, + "candidates": candidates} diff --git a/app/engine/summary_grounded.py b/app/engine/summary_grounded.py new file mode 100644 index 0000000..8c58efd --- /dev/null +++ b/app/engine/summary_grounded.py @@ -0,0 +1,68 @@ +"""Experimental evidence extraction followed by short-summary compression.""" +from __future__ import annotations + +import json +from app.engine.summary_consensus import SHORT_SUMMARY_PROMPT, select_consensus + +EVIDENCE_PROMPT = ( + "전기 문단 전체를 읽고 중심 인물·핵심 행동·결과를 가장 잘 드러내는 원문 구절을 고르세요. " + "주변 묘사보다 문단의 중심 사건이나 논지를 우선하세요. 표현을 고쳐 쓰지 말고 " + "원문에 연속해서 실제 존재하는 구절만 최대 3개, 각 120자 이내로 복사하세요. " + 'JSON 객체 {"evidence": ["구절1", "구절2"]}만 출력하세요.' +) +COMPRESSION_INSTRUCTION = ( + "아래 핵심 구절은 원문에서 검증한 보조 단서입니다. 반드시 원문 전체와 대조하세요. " + "핵심 구절에 등장하는 인물명·명사·동사를 임의의 동의어로 치환하지 마세요. " + "단서를 기계적으로 나열하지 말고 중심 사건과 결과를 자연스럽게 연결하세요." +) + + +def validated_evidence(text: str, payload: dict) -> list[str]: + raw = payload.get("evidence", []) + if not isinstance(raw, list): + return [] + return list(dict.fromkeys(s.strip() for s in raw if isinstance(s, str) + and 1 <= len(s.strip()) <= 120 and s.strip() in text))[:3] + + +def generate_grounded(client, text: str, count: int = 5) -> dict: + if not text.strip(): + raise ValueError("Source text must not be empty") + if not 1 <= count <= 5: + raise ValueError("Candidate count must be between 1 and 5") + extraction = client.chat.completions.create( + model="gpt-4o", temperature=0, max_tokens=400, + response_format={"type": "json_object"}, + messages=[{"role": "system", "content": "원문에서 근거 구절을 정확히 추출하세요."}, + {"role": "user", "content": EVIDENCE_PROMPT + "\n\n" + text}], + ) + evidence = [] + choice = extraction.choices[0] + if choice.finish_reason == "stop": + try: + payload = json.loads(choice.message.content or "{}") + evidence = validated_evidence(text, payload) if isinstance(payload, dict) else [] + except (ValueError, TypeError): + pass + # Invalid extraction falls back to the full source; never invent evidence. + prompt = SHORT_SUMMARY_PROMPT + "\n\n[원문]\n" + text + if evidence: + prompt += "\n\n" + COMPRESSION_INSTRUCTION + "\n[핵심 구절]\n" + "\n".join(evidence) + response = client.chat.completions.create( + model="gpt-4o", temperature=0.3, max_tokens=100, n=count, + messages=[{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."}, + {"role": "user", "content": prompt}], + ) + candidates = [{"summary": (c.message.content or "").strip() if c.finish_reason == "stop" else "", + "finish_reason": c.finish_reason, "model": response.model, + "response_id": response.id} for c in response.choices] + return {"evidence": evidence, "evidence_response_id": extraction.id, + "evidence_model": extraction.model, "evidence_finish_reason": choice.finish_reason, + "evidence_fallback": not bool(evidence), "candidates": candidates} + + +def summarize_grounded(client, text: str) -> dict: + generated = generate_grounded(client, text) + index = select_consensus([c["summary"] for c in generated["candidates"]], word_weight=0.5) + return {**generated, "selected_index": index, "valid": index >= 0, + "summary": generated["candidates"][index]["summary"] if index >= 0 else ""} diff --git a/reports/SUMMARY_BENCHMARK_EVIDENCE_2026.md b/reports/SUMMARY_BENCHMARK_EVIDENCE_2026.md new file mode 100644 index 0000000..f12a079 --- /dev/null +++ b/reports/SUMMARY_BENCHMARK_EVIDENCE_2026.md @@ -0,0 +1,51 @@ +# 요약 성능 내부 벤치 증빙 — 2026-09-16 + +## 결론 + +보존된 30건 내부 벤치의 **ROUGE-1 recall은 67.6579%**다. 첨부된 전 차수 +성적서 코드는 F1을 반환하므로 이 recall 값으로 올해 65% 달성을 판단할 수 없다. +이를 작년 시험 방식의 복원·달성값으로 설명했던 내용은 정정한다. + +> 이 결과는 사람 작성 정답셋이 아닌 GPT-4o 은(silver) 기준을 사용한 **내부 A/B +> 벤치**다. 따라서 사람 정답셋 기반의 공인 시험성적서 결과로 기재하지 않는다. + +## 보존된 실행 결과 + +| 항목 | 값 | +|---|---:| +| 결과 파일 | King `/mnt/data2/demo/o2o-plagiarism-ai/reports/summary_bench_v2.json` | +| 파일 수정 시각 | 2026-09-16 10:44:24 KST (생성·실행 시각을 입증하지 않음) | +| SHA-256 | `1e8022227c7a905f5926ab0c75aef0961248a901be19ead9cb501b98a5b70381` | +| 표본 수 | 30건 | +| 참조 요약 모델 | GPT-4o (결과 JSON에 기록) | +| 시스템 요약 모델 | GPT-4o (결과 JSON에 기록) | +| 프롬프트·seed·다중 참조 개수 | 결과 JSON에 미기록; 현재 스크립트 기본값으로 과거 실행 조건을 확정할 수 없음 | + +## 결과 + +| ROUGE recall | 추출 요약 | 추상 요약 | +|---|---:|---:| +| ROUGE-1 (어절) | 21.2612% | **67.6579%** | +| ROUGE-2 (어절) | 7.8993% | 51.8763% | +| ROUGE-1 (형태소) | 41.8747% | 79.1366% | +| ROUGE-2 (형태소) | 18.0236% | 62.1995% | + +## 재실행 상태 + +벤치 스크립트는 `/w/ai_publish/data_output4/*.json`의 전기 본문 30건을 읽는다. +호스트의 실제 경로는 `/mnt/data2/demo/ai_publish/data_output4`이며 JSON 85개가 +존재한다. 앞선 확인에서 컨테이너 경로와 호스트 경로를 혼동하여 원문이 없다고 +단정한 내용을 정정한다. 다만 30건 결과에는 원문 ID·참조 및 시스템 요약 원문이 +없어 동일 표본·출력의 정확한 재채점은 불가능하다. 아래는 새 벤치 실행 예시다. + +```bash +python scripts/bench_summarizer.py \ + --glob '/mnt/data2/demo/ai_publish/data_output4/*.json' \ + --count 30 --seed 20260916 \ + --ref-model gpt-4o --sys-model gpt-4o \ + --length-matched \ + --out reports/summary_bench_new.json +``` + +새 1,000건 실험은 `scripts/build_summary_trial.py`로 수행하며 데이터·참조·출력을 +모두 저장한다. 이 30건 보존 결과와 구분한다. diff --git a/scripts/bench_summarizer.py b/scripts/bench_summarizer.py new file mode 100644 index 0000000..6e0d0b4 --- /dev/null +++ b/scripts/bench_summarizer.py @@ -0,0 +1,151 @@ +"""요약기 A/B 내부 벤치마크 — 추출 요약 vs LLM 추상 요약. + +성능지표 #7 은 사람 참조 요약 대비 ROUGE recall 로 측정해야 한다. 아직 정답셋이 +없어 본 평가는 할 수 없다. 이 스크립트는 그 전에 "LLM 훅을 켜면 수치가 오르는가" +만 판단하기 위한 내부 비교다. + + 참조 : 별도 모델(gpt-4o) 이 작성한 요약 — 은(silver) 기준 + 시스템 : ① TextRank 추출 요약 ② LLM 추상 요약 + +**편향 주의** — 참조가 LLM 산출물이므로 LLM 추상 요약 쪽에 유리하다. 두 방식의 +상대 격차를 보는 용도이며, 이 수치를 성적서에 쓰면 안 된다. + +입력은 전기(傳記) 본문을 쓴다. 전 차수 요약 시험과 같은 계열 자료이고 출판물이라 +개인 자서전을 외부 API 로 보내지 않는다. +""" +from __future__ import annotations + +import argparse +import glob +import json +import random +import sys +from collections import Counter +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + + +def lemmas(text: str) -> list[str]: + """형태소 원형 열. 조사·어미 변형을 흡수해 어절 단위보다 느슨하게 맞춘다.""" + from app.engine.structural import extract_lemmas + return extract_lemmas(text) + + +def ngrams(tokens: list[str], n: int) -> Counter: + return Counter(tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1)) + + +def rouge_recall(reference: str, hypothesis: str, n: int = 2) -> float: + """계획서 수식 기준 — 분모가 참조 n-gram 수인 recall.""" + ref = ngrams(reference.split(), n) + hyp = ngrams(hypothesis.split(), n) + total = sum(ref.values()) + if not total: + return 0.0 + overlap = sum(min(count, hyp.get(gram, 0)) for gram, count in ref.items()) + return overlap / total + + +def load_passages(pattern: str, count: int, seed: int) -> list[str]: + passages = [] + for path in sorted(glob.glob(pattern)): + data = json.loads(Path(path).read_text(encoding="utf-8")) + for row in data.get("results", []): + text = (row.get("source_text") or "").strip() + if 400 <= len(text) <= 3000: + passages.append(text) + random.Random(seed).shuffle(passages) + return passages[:count] + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--glob", default="/w/ai_publish/data_output4/*.json") + parser.add_argument("--count", type=int, default=30) + parser.add_argument("--seed", type=int, default=20260916) + parser.add_argument("--ref-model", default="gpt-4o") + parser.add_argument("--sys-model", default="gpt-4o-mini") + parser.add_argument("--multi-ref", type=int, default=1, + help="참조 개수. 2 이상이면 다중 참조 중 최고값으로 채점 (계획서 전제)") + parser.add_argument("--length-matched", action="store_true", + help="시스템 요약 길이를 참조 규격(원문의 25~35%)에 맞춘다") + parser.add_argument("--out", type=Path, default=Path("reports/summary_bench.json")) + args = parser.parse_args() + + from openai import OpenAI + from app.core.config import get_settings + from app.engine.summarizer import Summarizer + + get_settings.cache_clear() + settings = get_settings() + client = OpenAI(api_key=settings.openai_api_key) + summarizer = Summarizer(settings) + + passages = load_passages(args.glob, args.count, args.seed) + print("본문 %d건 (전기)" % len(passages)) + if not passages: + parser.error("본문을 찾지 못했습니다. --glob 확인") + + def ask(model: str, prompt: str, text: str) -> str: + response = client.chat.completions.create( + model=model, temperature=0.2, + messages=[{"role": "system", "content": "당신은 한국어 요약 전문가입니다. 원문에 없는 사실을 만들지 마십시오."}, + {"role": "user", "content": prompt + "\n\n" + text}]) + return (response.choices[0].message.content or "").strip() + + REF_PROMPT = ("다음 글을 한국어 줄글로 요약하십시오. 원문 분량의 25~35% 길이로, " + "핵심 인물·사건·시간·장소·인과를 보존하고 목록이나 제목은 쓰지 마십시오.") + # recall 은 참조를 얼마나 덮었는지를 재므로, 시스템 요약이 참조보다 지나치게 + # 짧으면 구조적으로 손해를 본다. 길이 규격을 참조와 맞춘다. + SYS_PROMPT = (REF_PROMPT if args.length_matched else + "다음 글을 한국어 줄글로 간결하게 요약하십시오. 원문에 없는 내용을 넣지 마십시오.") + + rows = [] + for index, text in enumerate(passages, start=1): + # 다중 참조: 계획서가 전제하는 방식. 참조마다 채점해 최고값을 취한다. + references = [ask(args.ref_model, REF_PROMPT, text) for _ in range(args.multi_ref)] + extractive = summarizer.summarize(text, ratio=0.3, use_abstractive=False).final + abstractive = ask(args.sys_model, SYS_PROMPT, text) + row = {"index": index, + "reference_len": sum(len(r) for r in references) // len(references), + "system_len": len(abstractive)} + for label, system in (("extractive", extractive), ("abstractive", abstractive)): + for n in (1, 2): + row["%s_r%d" % (label, n)] = max( + rouge_recall(reference, system, n) for reference in references) + row["%s_r%d_lemma" % (label, n)] = max( + rouge_recall(" ".join(lemmas(reference)), " ".join(lemmas(system)), n) + for reference in references) + rows.append(row) + if index % 10 == 0: + print(" 진행 %d/%d" % (index, len(passages))) + + def mean(key: str) -> float: + return sum(r[key] for r in rows) / len(rows) + + print("\n" + "=" * 62) + print("ROUGE recall — 지표 정의별 (은 기준 참조 대비, 내부 비교용)") + print(" %-22s %-12s %-12s %s" % ("정의", "① 추출", "② LLM 추상", "차이")) + for label, key in (("ROUGE-1 (어절)", "r1"), ("ROUGE-2 (어절)", "r2"), + ("ROUGE-1 (형태소)", "r1_lemma"), ("ROUGE-2 (형태소)", "r2_lemma")): + a, b = mean("extractive_" + key), mean("abstractive_" + key) + print(" %-22s %-12.4f %-12.4f %+.4f" % (label, a, b, b - a)) + extractive_score = mean("extractive_r2") + abstractive_score = mean("abstractive_r2") + print("\n※ 참조가 LLM 산출물이라 ②에 유리한 편향이 있습니다. 성적서용 수치가 아닙니다.") + + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(json.dumps({ + "note": "내부 A/B. 참조는 %s 산출물(은 기준)이며 성적서용이 아님." % args.ref_model, + "count": len(rows), "ref_model": args.ref_model, "sys_model": args.sys_model, + "extractive": extractive_score, "abstractive": abstractive_score, + "rows": rows, + }, ensure_ascii=False, indent=2), encoding="utf-8") + print("%s 에 기록" % args.out) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/build_summary_trial.py b/scripts/build_summary_trial.py new file mode 100644 index 0000000..c977287 --- /dev/null +++ b/scripts/build_summary_trial.py @@ -0,0 +1,149 @@ +"""Build a frozen 1,000-source silver-reference trial; persist every model output. + +This is a newly defined experiment, not a reproduction of the undocumented 2025 +tokenizer. Report character and whitespace bigram F1 without choosing by score. +""" +from __future__ import annotations + +import argparse +import concurrent.futures +import hashlib +import json +import random +import re +from collections import Counter +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +PROMPTS = { + "reference": "다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. 핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. 배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요.", + "baseline": "다음 전기 문단을 최대 50자 이내로 요약하세요. 핵심 주제를 포함하고 불필요한 수식어를 제거하며 원문에 없는 내용을 추가하지 마세요. 요약문만 출력하세요.", + "candidate": "전기 문단을 읽고 가장 중요한 인물, 핵심 사건, 그 결과를 찾아 한국어 한 문장 50자 이내로 요약하세요. 원문의 핵심 명사와 동사를 유지하고 중복·주변 묘사는 제외하세요. 사실을 추가하지 말고 핵심 사건을 구체적으로 표현하세요. 요약만 출력하세요.", +} + + +def digest(text): + return hashlib.sha256(text.encode()).hexdigest() + + +def metrics(reference, hypothesis, mode): + def grams(text): + tokens = list(re.sub(r"\s+", "", text)) if mode == "character" else text.split() + return Counter(tuple(tokens[i:i + 2]) for i in range(len(tokens) - 1)) + a, b = grams(reference), grams(hypothesis) + overlap = sum((a & b).values()) + recall = overlap / sum(a.values()) if a else 0.0 + precision = overlap / sum(b.values()) if b else 0.0 + return {"precision": precision, "recall": recall, + "f1": 2 * precision * recall / (precision + recall) if precision + recall else 0.0} + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--source", type=Path, required=True) + ap.add_argument("--out", type=Path, required=True) + ap.add_argument("--workers", type=int, default=12) + args = ap.parse_args() + args.out.mkdir(parents=True, exist_ok=True) + manifest_path = args.out / "manifest.json" + dataset_path = args.out / "sources.jsonl" + if not manifest_path.exists(): + groups, seen = [], set() + rng = random.Random(20260916) + for p in sorted(args.source.glob("*.json")): + group = [] + for row in json.loads(p.read_text()).get("results", []): + text = str(row.get("source_text", "")).strip() + h = digest(" ".join(text.split())) + if not 200 <= len(text) <= 2000 or h in seen: + continue + seen.add(h) + group.append({"source_file": p.name, "source_index": row.get("index"), + "text": text, "source_sha256": h}) + rng.shuffle(group) + if group: + groups.append(group) + selected = [] + while len(selected) < 1000 and any(groups): + for group in groups: + if group and len(selected) < 1000: + selected.append(dict(group.pop(), id=len(selected) + 1)) + if len(selected) != 1000: + raise ValueError("Need 1,000 unique source passages") + payload = "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in selected) + dataset_path.write_text(payload) + manifest = {"count": 1000, "seed": 20260916, "source_books": len(groups), + "dataset_sha256": digest(payload), "reference_origin": "AI-generated; not human-reviewed", + "models": {"reference": "gpt-4o", "baseline": "gpt-4o-mini", "candidate": "gpt-4o"}, + "prompts": PROMPTS, "temperature": 0.3, "max_tokens": 100, + "tokenizers": {"character": "remove whitespace, preserve punctuation, Unicode characters", + "word": "Python str.split, preserve punctuation"}, + "n": 2, "aggregation": "macro mean F1 across all 1000 sources", + "legacy_equivalence": "unconfirmed: original ngram_tokenize unavailable", + "training_exclusion": "not locally fine-tuned; foundation-model training overlap unknown"} + manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2)) + manifest = json.loads(manifest_path.read_text()) + payload = dataset_path.read_text() + if digest(payload) != manifest["dataset_sha256"]: + raise ValueError("Frozen dataset hash mismatch") + rows = [json.loads(line) for line in payload.splitlines()] + from app.core.config import get_settings + from openai import OpenAI + client = OpenAI(api_key=get_settings().openai_api_key, timeout=90, max_retries=3) + outputs = {} + for role in ("reference", "baseline", "candidate"): + path = args.out / (role + ".jsonl") + existing = [json.loads(line) for line in path.read_text().splitlines()] if path.exists() else [] + done = {row["id"]: row for row in existing if row.get("summary")} + def generate(row): + try: + response = client.chat.completions.create( + model=manifest["models"][role], temperature=manifest["temperature"], + max_tokens=manifest["max_tokens"], messages=[ + {"role": "system", "content": "당신은 텍스트 요약 전문가입니다."}, + {"role": "user", "content": manifest["prompts"][role] + "\n\n" + row["text"]}]) + choice = response.choices[0] + summary = (choice.message.content or "").strip() + return {"id": row["id"], "summary": summary, "model": response.model, + "response_id": response.id, "finish_reason": choice.finish_reason, + "usage": response.usage.model_dump() if response.usage else None, + "over_50_chars": len(summary) > 50} + except Exception as exc: + return {"id": row["id"], "summary": "", "error_type": type(exc).__name__} + with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool, path.open("a") as out: + futures = [pool.submit(generate, row) for row in rows if row["id"] not in done] + for future in concurrent.futures.as_completed(futures): + result = future.result() + out.write(json.dumps(result, ensure_ascii=False) + "\n") + out.flush() + done[result["id"]] = result + if len(done) % 50 == 0: + print(role, len(done), "/ 1000", flush=True) + outputs[role] = done + result = {"sample_count": 1000, "reference_origin": manifest["reference_origin"], + "legacy_equivalence": manifest["legacy_equivalence"], "certificate_target_achieved": None, + "dataset_sha256": manifest["dataset_sha256"], "scores": {}, "quality": {}} + for role, items in outputs.items(): + result["quality"][role] = {"empty": sum(not r.get("summary") for r in items.values()), + "over_50_chars": sum(r.get("over_50_chars", False) for r in items.values()), + "truncated": sum(r.get("finish_reason") == "length" for r in items.values())} + per_item = [] + for role in ("baseline", "candidate"): + result["scores"][role] = {} + for mode in ("character", "word"): + scores = [] + for row in rows: + rid = row["id"] + score = metrics(outputs["reference"][rid]["summary"], outputs[role][rid]["summary"], mode) + scores.append(score) + per_item.append({"id": rid, "role": role, "mode": mode, **score}) + result["scores"][role][mode] = {key: sum(s[key] for s in scores) / 1000 for key in ("precision", "recall", "f1")} + (args.out / "per_item.jsonl").write_text("".join(json.dumps(r) + "\n" for r in per_item)) + (args.out / "result.json").write_text(json.dumps(result, ensure_ascii=False, indent=2)) + print(json.dumps(result, ensure_ascii=False, indent=2), flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/evaluate_grounded_summary.py b/scripts/evaluate_grounded_summary.py new file mode 100644 index 0000000..c7bee0b --- /dev/null +++ b/scripts/evaluate_grounded_summary.py @@ -0,0 +1,105 @@ +"""Develop two-stage compression, then re-evaluate one frozen configuration.""" +from __future__ import annotations +import argparse +from collections import Counter +import concurrent.futures +import hashlib +import json +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from app.engine.summary_grounded import generate_grounded, EVIDENCE_PROMPT, COMPRESSION_INSTRUCTION +from app.engine.summary_consensus import select_consensus, SHORT_SUMMARY_PROMPT +from scripts.build_summary_trial import metrics + + +def read(path): + return {r['id']: r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())} + + +def save(path, value): + path.write_text(json.dumps(value, ensure_ascii=False, indent=2)+'\n') + + +def pick(row, strategy): + texts=[c['summary'] for c in row['candidates']] + i=select_consensus(texts,word_weight=0.5) if strategy=='consensus5' else (0 if texts[0] and len(texts[0])<=50 else -1) + return texts[i] if i>=0 else '' + + +def score(generated, refs, strategy): + return {m:{k:sum(metrics(refs[i],pick(row,strategy),m)[k] for i,row in generated.items())/len(generated) + for k in ('precision','recall','f1')} for m in ('character','word')} + + +def run(rows, path, client, count): + done=read(path) if path.exists() else {} + with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool,path.open('a') as out: + pending={pool.submit(generate_grounded,client,r['text'],count):i for i,r in rows.items() if i not in done} + for f in concurrent.futures.as_completed(pending): + i=pending[f];value=dict(f.result(),id=i);done[i]=value + out.write(json.dumps(value,ensure_ascii=False)+'\n');out.flush() + if len(done)%50==0:print(path.name,len(done),'/',len(rows),flush=True) + assert set(done)==set(rows) + return done + + +def main(): + ap=argparse.ArgumentParser(description=__doc__) + ap.add_argument('--trial',type=Path,required=True);ap.add_argument('--previous',type=Path,required=True) + ap.add_argument('--out',type=Path,required=True);args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True) + development=read(args.trial/'development.jsonl');previous=read(args.previous/'sources.jsonl') + devsources={i:previous[i] for i in development} + devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()} + from app.engine.structural import extract_lemmas + counts=Counter();diagnostics=[] + for i,row in sorted(development.items()): + hyp=pick(row,'consensus5');ref=devrefs[i];word=metrics(ref,hyp,'word')['f1'] + a,b=Counter(extract_lemmas(ref)),Counter(extract_lemmas(hyp)) + lf=2*sum((a&b).values())/max(1,sum(a.values())+sum(b.values())) + category='above_target' if word>=.65 else ('surface_difference_candidate' if lf>=.7 else ('content_selection_candidate' if lf<.4 else 'mixed_or_uncertain')) + counts[category]+=1;diagnostics.append({'id':i,'word_f1':word,'lemma_overlap_f1':lf,'category':category}) + save(args.out/'diagnostics.json',{'note':'heuristic triage, not human error labels; lemma overlap is not the scoring metric', + 'counts':dict(counts),'rows':diagnostics}) + protocol={'reference_changes':False,'score_changes':False,'evaluation_type':'re-evaluation of previously inspected 1000-item test set', + 'development_count':len(development),'selection_metric':'word_bigram_f1','evidence_prompt':EVIDENCE_PROMPT, + 'summary_prompt':SHORT_SUMMARY_PROMPT,'compression_instruction':COMPRESSION_INSTRUCTION, + 'strategies':['single','consensus5'],'model':'gpt-4o','evidence_temperature':0,'summary_temperature':0.3, + 'reference_origin':'AI-generated, unreviewed','test_source_sha256':hashlib.sha256((args.trial/'sources.jsonl').read_bytes()).hexdigest()} + if (args.out/'protocol.json').exists():assert json.loads((args.out/'protocol.json').read_text())==protocol + else:save(args.out/'protocol.json',protocol) + from openai import OpenAI + from app.core.config import get_settings + client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3) + devgen=run(devsources,args.out/'development.jsonl',client,5) + devscores={s:score(devgen,devrefs,s) for s in ('single','consensus5')} + selected=max(devscores,key=lambda s:devscores[s]['word']['f1']) + decision={'selected_experimental_strategy':selected,'current_development_scores':score(development,devrefs,'consensus5'), + 'experimental_development_scores':devscores,'test_evaluation_reason':'user-requested comparison, not automatic promotion'} + if (args.out/'decision.json').exists():assert json.loads((args.out/'decision.json').read_text())==decision + else:save(args.out/'decision.json',decision) + print(json.dumps(decision),flush=True) + rows=read(args.trial/'sources.jsonl');refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()} + assert len(rows)==1000 and set(rows)==set(refs) + generated=run(rows,args.out/'system.jsonl',client,5 if selected=='consensus5' else 1) + newscore=score(generated,refs,selected);oldscore=score(read(args.trial/'system.jsonl'),refs,'consensus5') + details=[] + for i,row in sorted(rows.items()): + hyp=pick(generated[i],selected) + details.append({'index':i,'original':row['text'],'reference_summary':refs[i],'summary':hyp, + 'rouge_score':metrics(refs[i],hyp,'character')['f1'],'word_rouge_score':metrics(refs[i],hyp,'word')['f1']}) + result={'method':'evidence_then_compression_'+selected,'sample_count':1000,'target_rouge':.65, + 'average_rouge':newscore['character']['f1'],'average_word_rouge':newscore['word']['f1'], + 'success_rate':sum(r['rouge_score']>=.65 for r in details)/1000, + 'word_success_rate':sum(r['word_rouge_score']>=.65 for r in details)/1000, + 'success_rate_definition':'fraction of individual items with F1 >= 0.65', + 'previous_scores':oldscore,'scores':newscore,'word_improved':newscore['word']['f1']>oldscore['word']['f1'], + 'evaluation_type':protocol['evaluation_type'],'reference_origin':protocol['reference_origin'], + 'certificate_target_achieved':None,'evidence_fallback_count':sum(r['evidence_fallback'] for r in generated.values()), + 'empty_count':sum(not r['summary'] for r in details),'data':details} + save(args.out/'scorecard.json',result) + print(json.dumps({k:v for k,v in result.items() if k!='data'},ensure_ascii=False,indent=2),flush=True) + + +if __name__=='__main__':main() diff --git a/scripts/evaluate_o2o_dataset.py b/scripts/evaluate_o2o_dataset.py index 1a73cdf..6bcf126 100644 --- a/scripts/evaluate_o2o_dataset.py +++ b/scripts/evaluate_o2o_dataset.py @@ -8,7 +8,7 @@ 사용: python scripts/evaluate_o2o_dataset.py \ - --data-dir /Users/marineyang/Desktop/work/code/AI_publish_3rdtest/25/plagia_result + --data-dir ./data/eval/plagia_result """ from __future__ import annotations diff --git a/scripts/export_summary_annotations.py b/scripts/export_summary_annotations.py new file mode 100644 index 0000000..c54ff87 --- /dev/null +++ b/scripts/export_summary_annotations.py @@ -0,0 +1,81 @@ +#!/usr/bin/env python3 +"""완료된 사람 요약 검수 XLSX를 ROUGE 평가용 JSONL로 내보낸다. + +빈 참조·미완료 행·동일 참조를 실패 처리해, 검수 전 데이터를 성적서 평가에 +실수로 사용하지 않게 한다. 원문과 작성자 그룹은 유지하되 검수자 실명은 내보내지 않는다. +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +REQUIRED_COLUMNS = { + "annotation_id", "source_group", "source_text", "reference_summary_1", + "reference_summary_2", "status", +} + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("xlsx", type=Path) + parser.add_argument("--out", type=Path, required=True) + parser.add_argument("--allow-single-reference", action="store_true", + help="2차 독립 검수가 끝나기 전 파일럿에만 사용") + args = parser.parse_args() + + from openpyxl import load_workbook + book = load_workbook(args.xlsx, read_only=True, data_only=True) + if "annotations" not in book.sheetnames: + parser.error("annotations 시트가 없습니다") + sheet = book["annotations"] + headers = [str(cell.value or "").strip() for cell in next(sheet.iter_rows(max_row=1))] + positions = {name: index for index, name in enumerate(headers)} + missing = REQUIRED_COLUMNS - positions.keys() + if missing: + parser.error("필수 열이 없습니다: %s" % ", ".join(sorted(missing))) + + rows, errors = [], [] + for excel_row, values in enumerate(sheet.iter_rows(min_row=2, values_only=True), start=2): + value = lambda key: str(values[positions[key]] or "").strip() + status = value("status").casefold() + if not any(values): + continue + if status != "completed": + errors.append("행 %d: status가 completed가 아닙니다" % excel_row) + continue + refs = [value("reference_summary_1"), value("reference_summary_2")] + refs = [ref for ref in refs if ref] + if not args.allow_single_reference and len(refs) != 2: + errors.append("행 %d: 독립 참조 요약 2개가 필요합니다" % excel_row) + continue + if refs and len({" ".join(ref.split()) for ref in refs}) != len(refs): + errors.append("행 %d: 두 참조 요약이 동일합니다" % excel_row) + continue + if not refs or not value("source_text"): + errors.append("행 %d: 원문 또는 참조 요약이 비어 있습니다" % excel_row) + continue + rows.append({ + "id": value("annotation_id"), + "source_group": value("source_group"), + "text": value("source_text"), + "references": refs, + }) + book.close() + if errors: + print("내보내지 않음 — 검수 오류 %d건:" % len(errors)) + print("\n".join(errors[:20])) + return 2 + if not rows: + print("내보낼 completed 행이 없습니다") + return 2 + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text("".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), encoding="utf-8") + print("저장: %s / %d건 / 작성자 그룹 %d개" % + (args.out, len(rows), len({row['source_group'] for row in rows}))) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/improve_summary_trial.py b/scripts/improve_summary_trial.py new file mode 100644 index 0000000..8c3136d --- /dev/null +++ b/scripts/improve_summary_trial.py @@ -0,0 +1,127 @@ +"""Development-only selection followed by a fresh 1,000-source evaluation.""" +from __future__ import annotations + +import argparse +import concurrent.futures +import hashlib +import json +import random +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from app.engine.summary_consensus import generate_candidates, select_consensus, SHORT_SUMMARY_PROMPT +from scripts.build_summary_trial import metrics, PROMPTS + + +def read(path): + return [json.loads(l) for l in path.read_text().splitlines() if l.strip()] + + +def save(path, value): + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n") + + +def sha(text): + return hashlib.sha256(text.encode()).hexdigest() + + +def run_rows(rows, path, fn): + done = {r['id']: r for r in read(path)} if path.exists() else {} + with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool, path.open('a') as out: + pending = {pool.submit(fn, r): r['id'] for r in rows if r['id'] not in done} + for future in concurrent.futures.as_completed(pending): + rid = pending[future] + value = future.result() # fail visibly; resume persisted successes + done[rid] = dict(value, id=rid) + out.write(json.dumps(done[rid], ensure_ascii=False) + '\n'); out.flush() + if len(done) % 50 == 0: + print(path.name, len(done), '/', len(rows), flush=True) + return done + + +def summary(row, strategy): + texts = [c['summary'] for c in row['candidates']] + index = 0 if strategy == 'single' else select_consensus(texts) + return texts[index] if index >= 0 else '' + + +def score(rows, references, generated, strategy): + return {mode: {key: sum(metrics(references[r['id']], summary(generated[r['id']], strategy), mode)[key] + for r in rows) / len(rows) for key in ('precision','recall','f1')} + for mode in ('character','word')} + + +def main(): + ap=argparse.ArgumentParser(); ap.add_argument('--previous',type=Path,required=True) + ap.add_argument('--source',type=Path,required=True); ap.add_argument('--out',type=Path,required=True) + args=ap.parse_args(); args.out.mkdir(parents=True,exist_ok=True) + previous=read(args.previous/'sources.jsonl') + oldrefs={r['id']:r['summary'] for r in read(args.previous/'reference.jsonl')} + dev_books=set(sorted({r['source_file'] for r in previous})[:10]) + dev=[r for r in previous if r['source_file'] in dev_books] + if not (args.out/'protocol.json').exists(): + save(args.out/'protocol.json',{'development_books':sorted(dev_books),'development_count':len(dev), + 'primary_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1', + 'strategies':['single','consensus5'],'reference_prompt':PROMPTS['reference'], + 'system_prompt':SHORT_SUMMARY_PROMPT,'model':'gpt-4o','temperature':0.3, + 'reference_origin':'AI-generated, unreviewed','bias':'same model and task instructions for reference and system', + 'legacy_tokenizer_equivalence':'unconfirmed','target':0.65}) + from app.core.config import get_settings + from openai import OpenAI + client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3) + generated=run_rows(dev,args.out/'development.jsonl',lambda row:{'candidates':generate_candidates(client,row['text'])}) + development={s:score(dev,oldrefs,generated,s) for s in ('single','consensus5')} + selected=max(development,key=lambda s:development[s]['character']['f1']) + decision={'development_scores':development,'selected':selected,'selected_before_test':True} + decision_path=args.out/'decision.json' + if decision_path.exists(): + assert json.loads(decision_path.read_text())==decision + else:save(decision_path,decision) + print(json.dumps(decision),flush=True) + # Hold out entire development books and every previous trial paragraph. + dataset=args.out/'sources.jsonl' + if not dataset.exists(): + seen={sha(' '.join(r['text'].split())) for r in previous};groups=[];rng=random.Random(20260917) + for p in sorted(args.source.glob('*.json')): + if p.name in dev_books:continue + group=[] + for r in json.loads(p.read_text()).get('results',[]): + text=str(r.get('source_text','')).strip();h=sha(' '.join(text.split())) + if not 200<=len(text)<=2000 or h in seen:continue + seen.add(h);group.append({'text':text,'source_file':p.name,'source_index':r.get('index'),'source_sha256':h}) + rng.shuffle(group) + if group:groups.append(group) + rows=[] + while len(rows)<1000 and any(groups): + for g in groups: + if g and len(rows)<1000:rows.append(dict(g.pop(),id=len(rows)+1)) + assert len(rows)==1000 + dataset.write_text(''.join(json.dumps(r,ensure_ascii=False)+'\n' for r in rows)) + save(args.out/'dataset_manifest.json',{'count':1000,'sha256':sha(dataset.read_text()),'seed':20260917, + 'development_book_overlap':0,'previous_exact_text_overlap':0,'near_duplicate_screening':'not performed'}) + rows=read(dataset);assert sha(dataset.read_text())==json.loads((args.out/'dataset_manifest.json').read_text())['sha256'] + refs=run_rows(rows,args.out/'reference.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],1)}) + reftexts={i:r['candidates'][0]['summary'] for i,r in refs.items()} + # Freeze candidate selection before any held-out score is calculated. + systems=run_rows(rows,args.out/'system.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],5 if selected=='consensus5' else 1)}) + def baseline(row): + resp=client.chat.completions.create(model='gpt-4o',temperature=0.3,max_tokens=100,messages=[ + {'role':'system','content':'당신은 텍스트 요약 전문가입니다.'}, + {'role':'user','content':PROMPTS['candidate']+'\n\n'+row['text']}]) + c=resp.choices[0] + return {'candidates':[{'summary':(c.message.content or '').strip() if c.finish_reason=='stop' else '', + 'finish_reason':c.finish_reason,'model':resp.model,'response_id':resp.id}]} + base=run_rows(rows,args.out/'baseline.jsonl',baseline) + result={'count':1000,'selected':selected,'reference_origin':'AI-generated, unreviewed', + 'bias':'same model and task instructions for reference and improved system', + 'certificate_target_achieved':None,'legacy_tokenizer_equivalence':'unconfirmed', + 'baseline':score(rows,reftexts,base,'single'),'improved':score(rows,reftexts,systems,selected)} + result['primary_target_achieved']=result['improved']['character']['f1']>=0.65 + result['quality']={label:{'empty':sum(not s for s in texts),'over50':sum(len(s)>50 for s in texts)} + for label,texts in [('reference',list(reftexts.values())),('baseline',[summary(base[r['id']],'single') for r in rows]), + ('improved',[summary(systems[r['id']],selected) for r in rows])]} + save(args.out/'result.json',result);print(json.dumps(result,ensure_ascii=False,indent=2),flush=True) + + +if __name__=='__main__':main() diff --git a/scripts/refine_word_consensus.py b/scripts/refine_word_consensus.py new file mode 100644 index 0000000..c379345 --- /dev/null +++ b/scripts/refine_word_consensus.py @@ -0,0 +1,70 @@ +"""Re-evaluate a development-selected word-aware selector; no new API calls. + +Reuses the previously inspected test set. This is a re-evaluation, not a fresh +independent certification test. Preserve earlier artifacts and report both metrics. +""" +from __future__ import annotations +import argparse +import json +from pathlib import Path +import sys +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from app.engine.summary_consensus import select_consensus +from scripts.build_summary_trial import metrics + + +def read(path): + return {r['id']:r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())} + + +def evaluate(generated, references, weight): + result=[] + for rid,row in sorted(generated.items()): + texts=[c['summary'] for c in row['candidates']] + chosen=select_consensus(texts, word_weight=weight) + text=texts[chosen] if chosen>=0 else '' + result.append({'index':rid,'summary':text,'selected_index':chosen, + 'scores':{m:metrics(references[rid],text,m) for m in ('character','word')}}) + scores={m:{k:sum(r['scores'][m][k] for r in result)/len(result) + for k in ('precision','recall','f1')} for m in ('character','word')} + return scores,result + + +def main(): + ap=argparse.ArgumentParser(description=__doc__) + ap.add_argument('--trial',type=Path,required=True) + ap.add_argument('--previous',type=Path,required=True) + ap.add_argument('--out',type=Path,required=True) + args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True) + dev=read(args.trial/'development.jsonl') + devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()} + # Development comparison only; test scores do not choose this weight. + compared={str(w):evaluate(dev,devrefs,w)[0] for w in (0.0,0.5,1.0)} + weight=max((0.0,0.5,1.0),key=lambda w:compared[str(w)]['word']['f1']) + decision={'selected_word_weight':weight,'selection_metric':'development_word_bigram_f1', + 'development_scores':compared,'evaluation_type':'re-evaluation of previously inspected test set'} + (args.out/'decision.json').write_text(json.dumps(decision,ensure_ascii=False,indent=2)+'\n') + systems=read(args.trial/'system.jsonl'); sources=read(args.trial/'sources.jsonl') + refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()} + assert set(systems)==set(sources)==set(refs)==set(range(1,1001)) + scores,rows=evaluate(systems,refs,weight) + old,_=evaluate(systems,refs,0.0) + full=[] + for row in rows: + i=row['index'];full.append({'index':i,'original':sources[i]['text'], + 'reference_summary':refs[i],'summary':row['summary'], + 'rouge_score':row['scores']['character']['f1'],'word_rouge_score':row['scores']['word']['f1']}) + record={'method':'source_only_consensus5_char_word_equal_weight','target_rouge':0.65, + 'average_rouge':scores['character']['f1'],'average_word_rouge':scores['word']['f1'], + 'success_rate':sum(r['rouge_score']>=0.65 for r in full)/1000, + 'word_success_rate':sum(r['word_rouge_score']>=0.65 for r in full)/1000, + 'success_rate_definition':'fraction of individual items with F1 >= 0.65', + 'sample_count':1000,'reference_origin':'AI-generated, unreviewed', + 'reported_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1', + 'evaluation_type':decision['evaluation_type'],'certificate_target_achieved':None, + 'previous_scores':old,'scores':scores,'data':full} + (args.out/'scorecard.json').write_text(json.dumps(record,ensure_ascii=False,indent=2)+'\n') + print(json.dumps({k:v for k,v in record.items() if k!='data'},ensure_ascii=False,indent=2)) + + +if __name__=='__main__':main() diff --git a/scripts/rescore_recall.py b/scripts/rescore_recall.py new file mode 100644 index 0000000..c479c3d --- /dev/null +++ b/scripts/rescore_recall.py @@ -0,0 +1,68 @@ +"""저장된 1,000건 재채점 — 문자 2-gram recall (계획서 p.24 수식). + +API 재호출 없음. H절 재평가 산출물의 참조·출력 원문을 그대로 사용하고 +집계 지표만 F1 에서 recall 로 바꾼다. 새 독립 시험이 아니다. +""" +from __future__ import annotations +import argparse +import json +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from scripts.build_summary_trial import metrics + +TARGET = 0.65 + + +def main() -> None: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--source", type=Path, + default=Path("data/eval/summary_word_refined_20260917/scorecard.json")) + ap.add_argument("--out", type=Path, default=Path("data/eval/summary_recall_20260917")) + ap.add_argument("--excerpt", type=int, default=160, help="캡처본 원문 발췌 길이") + args = ap.parse_args() + args.out.mkdir(parents=True, exist_ok=True) + + rows = json.loads(args.source.read_text())["data"] + scored = [{ + "index": row["index"], + "original": row["original"], + "reference_summary": row["reference_summary"], + "summary": row["summary"], + "rouge_score": metrics(row["reference_summary"], row["summary"], "character")["recall"], + } for row in rows] + + average = sum(r["rouge_score"] for r in scored) / len(scored) + provenance = { + "metric": "character_bigram_recall", + "metric_formula": "sum(match n-gram) / sum(reference n-gram)", + "sample_count": len(scored), + "reference_origin": "AI-generated (GPT-4o), unreviewed", + "reference_count_per_item": 1, + "evaluation_type": "re-evaluation of previously inspected test set", + } + scorecard = { + "method": "source_only_consensus5_char_word_equal_weight", + "target_rouge": TARGET, + "average_rouge": average, + "success_rate": sum(r["rouge_score"] >= TARGET for r in scored) / len(scored), + "data": scored, + **provenance, + } + (args.out / "scorecard.json").write_text( + json.dumps(scorecard, ensure_ascii=False, indent=2) + "\n") + + capture = dict(scorecard) + capture["data"] = [{**r, "original": r["original"][:args.excerpt] + " …"} + for r in scored[:3]] + (args.out / "capture.json").write_text( + json.dumps(capture, ensure_ascii=False, indent=2) + "\n") + + print(f"문자 2-gram recall 평균 {average * 100:.2f}% " + f"(목표 {TARGET * 100:.0f}%, {len(scored)}건)") + print(f"success_rate {scorecard['success_rate'] * 100:.1f}%") + + +if __name__ == "__main__": + main() diff --git a/scripts/tune_summarizer.py b/scripts/tune_summarizer.py new file mode 100644 index 0000000..befee30 --- /dev/null +++ b/scripts/tune_summarizer.py @@ -0,0 +1,91 @@ +#!/usr/bin/env python3 +"""사람 참조 요약으로 추출 전략을 고르고, 잠금 테스트셋에서 한 번만 측정한다. + +선택(검증)과 최종 보고(테스트)를 분리해, 65점에 맞춰 전 평가 요약문을 수정하는 +전 차수식 접근을 방지한다. source_group 단위로 분리하므로 같은 작성자의 문장이 +튜닝·시험 양쪽에 섞이지 않는다. +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +from app.engine.rouge import evaluate_pairs +from app.engine.summarizer import extractive_summary + + +def stable_bucket(group: str, seed: int) -> int: + return int(hashlib.sha256((str(seed) + ":" + group).encode()).hexdigest()[:8], 16) % 100 + + +def split_rows(rows: list[dict], seed: int) -> tuple[list[dict], list[dict]]: + # 20% 그룹을 최종 시험에 잠근다. 나머지에서만 전략을 고른다. + test = [row for row in rows if stable_bucket(row.get("source_group", row["id"]), seed) < 20] + validation = [row for row in rows if row not in test] + if not test or not validation: + raise ValueError("그룹 분할 결과가 비었습니다. source_group을 확인하세요.") + return validation, test + + +def score(rows: list[dict], ratio: float, strategy: str, mode: str) -> dict[str, dict[str, float]]: + pairs = [ + (extractive_summary(row["text"], ratio=ratio, strategy=strategy).final, row["references"]) + for row in rows + ] + return evaluate_pairs(pairs, mode=mode) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("dataset", type=Path, help="export_summary_annotations.py 출력 JSONL") + parser.add_argument("--out", type=Path, required=True, help="선택 근거와 잠금 시험 결과 JSON") + parser.add_argument("--seed", type=int, default=20260916) + parser.add_argument("--mode", choices=["lemma", "char"], default="lemma") + parser.add_argument("--ratios", type=float, nargs="+", default=[0.25, 0.30, 0.35], + help="사전 등록 가능한 요약 비율 후보(권장 25~35%%)") + args = parser.parse_args() + if any(not 0 < ratio <= 0.35 for ratio in args.ratios): + parser.error("요약 비율은 0보다 크고 0.35 이하여야 합니다") + rows = [json.loads(line) for line in args.dataset.read_text(encoding="utf-8").splitlines() if line.strip()] + if len(rows) < 20: + parser.error("최소 20건 이상의 완료된 사람 참조 요약이 필요합니다") + validation, test = split_rows(rows, args.seed) + candidates = [] + for strategy in ("textrank", "coverage", "lead"): + for ratio in args.ratios: + metrics = score(validation, ratio, strategy, args.mode) + candidates.append({"strategy": strategy, "ratio": ratio, "validation": metrics}) + # 같은 recall이면 ROUGE-L F1, 더 짧은 출력 순으로 결정한다. + winner = max(candidates, key=lambda item: ( + item["validation"]["rouge1"]["recall"], + item["validation"]["rougeL"]["f1"], + -item["ratio"], + )) + final = score(test, winner["ratio"], winner["strategy"], args.mode) + record = { + "purpose": "summary strategy selection with source-group-held-out final test", + "dataset": str(args.dataset), "seed": args.seed, "tokenization": args.mode, + "samples": {"total": len(rows), "validation": len(validation), "locked_test": len(test)}, + "candidates": candidates, "selected": {"strategy": winner["strategy"], "ratio": winner["ratio"]}, + "locked_test_scores": final, + "reported_metric": "rouge1_recall", + "target": 0.65, + "achieved": final["rouge1"]["recall"] >= 0.65, + } + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(json.dumps(record, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + print("선택: %s / ratio=%.2f" % (winner["strategy"], winner["ratio"])) + print("잠금 테스트 ROUGE-1 recall: %.4f (%s)" % + (final["rouge1"]["recall"], "달성" if record["achieved"] else "미달")) + print("근거 저장: %s" % args.out) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/verify_grounded_summary.py b/scripts/verify_grounded_summary.py new file mode 100644 index 0000000..dec2a5e --- /dev/null +++ b/scripts/verify_grounded_summary.py @@ -0,0 +1,53 @@ +"""Offline evidence and F1 verification for the two-stage experiment.""" +from __future__ import annotations +import argparse +from collections import Counter +import hashlib +import json +from pathlib import Path +import re +import sys +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from app.engine.summary_consensus import select_consensus + + +def rows(path): + items=[json.loads(l) for l in path.read_text().splitlines() if l.strip()] + assert len(items)==1000 and {r['id'] for r in items}==set(range(1,1001)) + return {r['id']:r for r in items} + + +def main(): + ap=argparse.ArgumentParser(description=__doc__) + ap.add_argument('--trial',type=Path,required=True) + ap.add_argument('--experiment',type=Path,required=True) + a=ap.parse_args();sources=rows(a.trial/'sources.jsonl');refs=rows(a.trial/'reference.jsonl') + outputs=rows(a.experiment/'system.jsonl');result=json.loads((a.experiment/'scorecard.json').read_text()) + protocol=json.loads((a.experiment/'protocol.json').read_text()) + assert hashlib.sha256((a.trial/'sources.jsonl').read_bytes()).hexdigest()==protocol['test_source_sha256'] + strategy=json.loads((a.experiment/'decision.json').read_text())['selected_experimental_strategy'] + details={r['index']:r for r in result['data']};assert len(details)==len(result['data'])==1000 + total={'character':0.,'word':0.} + for rid in range(1,1001): + row=outputs[rid];text=sources[rid]['text'] + assert all(e in text and 1<=len(e)<=120 for e in row['evidence']) + assert len(row['evidence'])<=3 + summaries=[c['summary'] for c in row['candidates']] + i=select_consensus(summaries,.5) if strategy=='consensus5' else (0 if summaries[0] and len(summaries[0])<=50 else -1) + hyp=summaries[i] if i>=0 else '';ref=refs[rid]['candidates'][0]['summary'] + assert details[rid]['summary']==hyp and details[rid]['reference_summary']==ref and details[rid]['original']==text + for mode in total: + def grams(t): + t=list(re.sub(r'\s+','',t)) if mode=='character' else t.split() + return Counter(zip(t,t[1:])) + x,y=grams(ref),grams(hyp);denom=sum(x.values())+sum(y.values()) + f1=2*sum(min(n,y[g]) for g,n in x.items())/denom if denom else 0 + assert abs(f1-details[rid]['rouge_score' if mode=='character' else 'word_rouge_score'])<1e-12 + total[mode]+=f1/1000 + for mode,avg in total.items(): + assert abs(avg-result['scores'][mode]['f1'])<1e-12 + print(mode,'F1=%.6f%%'%(100*avg)) + print('PASS: 1000 IDs, unchanged sources/references, literal evidence, selected outputs, per-item and mean F1') + + +if __name__=='__main__':main() diff --git a/scripts/verify_summary_improvement.py b/scripts/verify_summary_improvement.py new file mode 100644 index 0000000..cbc4eb5 --- /dev/null +++ b/scripts/verify_summary_improvement.py @@ -0,0 +1,60 @@ +"""Recompute the frozen improvement trial locally, without model/API calls.""" +from __future__ import annotations + +import argparse +from collections import Counter +import hashlib +import json +from pathlib import Path +import re +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from app.engine.summary_consensus import select_consensus + + +def load_rows(path): + rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()] + assert len(rows) == 1000 and {r['id'] for r in rows} == set(range(1, 1001)), path + return {r['id']: r for r in rows} + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument('directory', type=Path) + args = ap.parse_args() + p = args.directory + manifest = json.loads((p/'dataset_manifest.json').read_text()) + assert hashlib.sha256((p/'sources.jsonl').read_bytes()).hexdigest() == manifest['sha256'] + sources = load_rows(p/'sources.jsonl') + protocol = json.loads((p/'protocol.json').read_text()) + assert not ({r['source_file'] for r in sources.values()} & set(protocol['development_books'])) + refs = load_rows(p/'reference.jsonl') + systems = load_rows(p/'system.jsonl') + baseline = load_rows(p/'baseline.jsonl') + result = json.loads((p/'result.json').read_text()) + for label, outputs in [('baseline', baseline), ('improved', systems)]: + for mode in ('character', 'word'): + totals = {'precision': 0.0, 'recall': 0.0, 'f1': 0.0} + for rid in range(1, 1001): + candidates = [c['summary'] for c in outputs[rid]['candidates']] + chosen = select_consensus(candidates) if label == 'improved' and result['selected'] == 'consensus5' else 0 + hyp = candidates[chosen] if chosen >= 0 else '' + ref = refs[rid]['candidates'][0]['summary'] + def grams(text): + tokens = list(re.sub(r'\s+', '', text)) if mode == 'character' else text.split() + return Counter(zip(tokens, tokens[1:])) + a, b = grams(ref), grams(hyp) + overlap = sum(min(count, b[g]) for g, count in a.items()) + a_total, b_total = sum(a.values()), sum(b.values()) + totals['precision'] += overlap / b_total if b_total else 0.0 + totals['recall'] += overlap / a_total if a_total else 0.0 + totals['f1'] += 2 * overlap / (a_total + b_total) if a_total + b_total else 0.0 + actual = {key: value/1000 for key, value in totals.items()} + assert all(abs(actual[key]-result[label][mode][key]) < 1e-12 for key in actual) + print(label, mode, 'F1=%.6f%%' % (actual['f1']*100)) + print('PASS: 1000 IDs, source SHA-256, development-book exclusion, all P/R/F1 values') + + +if __name__ == '__main__': + main() diff --git a/tests/test_summarizer.py b/tests/test_summarizer.py index 0eeed10..8ac5387 100644 --- a/tests/test_summarizer.py +++ b/tests/test_summarizer.py @@ -66,6 +66,18 @@ def test_emphasis_prioritizes_requested_topic(): assert result.emphasis == ["김밥"] +def test_coverage_strategy_preserves_source_order_and_respects_cap(): + result = extractive_summary(_DOC, ratio=0.5, max_sentences=3, strategy="coverage") + assert result.selected_indices == sorted(result.selected_indices) + assert result.num_sentences_out == 3 + + +def test_unknown_extractive_strategy_is_rejected(): + import pytest + with pytest.raises(ValueError): + extractive_summary(_DOC, strategy="oracle") + + def test_empty_and_short(): assert extractive_summary("").final == "" one = extractive_summary("한 문장만 있다.") diff --git a/tests/test_summary_consensus.py b/tests/test_summary_consensus.py new file mode 100644 index 0000000..0377d42 --- /dev/null +++ b/tests/test_summary_consensus.py @@ -0,0 +1,40 @@ +from app.engine.summary_consensus import character_f1, select_consensus + + +def test_medoid_prefers_agreed_event_over_outlier(): + candidates = ["김씨는 학교를 세웠다.", "김씨는 학교를 세웠다.", "김씨는 미국으로 이민했다."] + assert select_consensus(candidates) == 0 + + +def test_overlength_consensus_cannot_win(): + long = "긴" * 51 + assert select_consensus([long, long, "학교를 세웠다."]) == 2 + assert select_consensus([long, ""]) == -1 + + +def test_empty_outputs_do_not_change_candidate_choice(): + assert select_consensus(["", "학교를 세웠다.", "학교를 세웠다."]) == 1 + assert character_f1("", "") == 0 + + +def test_bigram_overlap_keeps_multiplicity(): + assert character_f1("가가가", "가가") == 2 / 3 + + +def test_word_overlap_and_invalid_weight(): + import pytest + from app.engine.summary_consensus import word_f1 + assert word_f1("a b c", "a b") == 2 / 3 + assert word_f1("a", "a") == 0 + with pytest.raises(ValueError): + select_consensus(["요약"], word_weight=1.5) + + +def test_public_summary_reports_invalid_when_every_candidate_fails(monkeypatch): + from app.engine import summary_consensus + monkeypatch.setattr(summary_consensus, "generate_candidates", lambda client, text: [ + {"summary": "", "finish_reason": "length"}, {"summary": "긴" * 51, "finish_reason": "stop"}]) + result = summary_consensus.summarize_short(None, "원문") + assert result["valid"] is False + assert result["summary"] == "" + assert result["selected_index"] == -1 diff --git a/tests/test_summary_grounded.py b/tests/test_summary_grounded.py new file mode 100644 index 0000000..226a1a3 --- /dev/null +++ b/tests/test_summary_grounded.py @@ -0,0 +1,14 @@ +from app.engine.summary_grounded import validated_evidence + + +def test_evidence_rejects_paraphrases_and_nonstrings(): + assert validated_evidence('그는 학교를 세웠다.', {'evidence':['학교를 세웠다','학교를 설립했다',42,'']}) == ['학교를 세웠다'] + + +def test_evidence_deduplicates_and_caps(): + assert validated_evidence('가 나 다 라', {'evidence':['가','가','나','다','라']}) == ['가','나','다'] + + +def test_bad_schema_and_overlength_are_not_evidence(): + assert validated_evidence('원문', {'evidence':'원문'}) == [] + assert validated_evidence('가'*121, {'evidence':['가'*121]}) == []