요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
150 lines
8.3 KiB
Python
150 lines
8.3 KiB
Python
"""Build a frozen 1,000-source silver-reference trial; persist every model output.
|
|
|
|
This is a newly defined experiment, not a reproduction of the undocumented 2025
|
|
tokenizer. Report character and whitespace bigram F1 without choosing by score.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import concurrent.futures
|
|
import hashlib
|
|
import json
|
|
import random
|
|
import re
|
|
from collections import Counter
|
|
from pathlib import Path
|
|
import sys
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
|
|
PROMPTS = {
|
|
"reference": "다음 전기 문단의 중심 사건 또는 주제를 한국어 한 문장, 50자 이내로 요약하세요. 핵심 인물과 행동 또는 원인·결과를 보존하고 원문의 표현을 가능한 한 유지하세요. 배경 설명과 수식은 줄이고 원문에 없는 사실은 쓰지 마세요. 제목·목록 없이 요약만 출력하세요.",
|
|
"baseline": "다음 전기 문단을 최대 50자 이내로 요약하세요. 핵심 주제를 포함하고 불필요한 수식어를 제거하며 원문에 없는 내용을 추가하지 마세요. 요약문만 출력하세요.",
|
|
"candidate": "전기 문단을 읽고 가장 중요한 인물, 핵심 사건, 그 결과를 찾아 한국어 한 문장 50자 이내로 요약하세요. 원문의 핵심 명사와 동사를 유지하고 중복·주변 묘사는 제외하세요. 사실을 추가하지 말고 핵심 사건을 구체적으로 표현하세요. 요약만 출력하세요.",
|
|
}
|
|
|
|
|
|
def digest(text):
|
|
return hashlib.sha256(text.encode()).hexdigest()
|
|
|
|
|
|
def metrics(reference, hypothesis, mode):
|
|
def grams(text):
|
|
tokens = list(re.sub(r"\s+", "", text)) if mode == "character" else text.split()
|
|
return Counter(tuple(tokens[i:i + 2]) for i in range(len(tokens) - 1))
|
|
a, b = grams(reference), grams(hypothesis)
|
|
overlap = sum((a & b).values())
|
|
recall = overlap / sum(a.values()) if a else 0.0
|
|
precision = overlap / sum(b.values()) if b else 0.0
|
|
return {"precision": precision, "recall": recall,
|
|
"f1": 2 * precision * recall / (precision + recall) if precision + recall else 0.0}
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument("--source", type=Path, required=True)
|
|
ap.add_argument("--out", type=Path, required=True)
|
|
ap.add_argument("--workers", type=int, default=12)
|
|
args = ap.parse_args()
|
|
args.out.mkdir(parents=True, exist_ok=True)
|
|
manifest_path = args.out / "manifest.json"
|
|
dataset_path = args.out / "sources.jsonl"
|
|
if not manifest_path.exists():
|
|
groups, seen = [], set()
|
|
rng = random.Random(20260916)
|
|
for p in sorted(args.source.glob("*.json")):
|
|
group = []
|
|
for row in json.loads(p.read_text()).get("results", []):
|
|
text = str(row.get("source_text", "")).strip()
|
|
h = digest(" ".join(text.split()))
|
|
if not 200 <= len(text) <= 2000 or h in seen:
|
|
continue
|
|
seen.add(h)
|
|
group.append({"source_file": p.name, "source_index": row.get("index"),
|
|
"text": text, "source_sha256": h})
|
|
rng.shuffle(group)
|
|
if group:
|
|
groups.append(group)
|
|
selected = []
|
|
while len(selected) < 1000 and any(groups):
|
|
for group in groups:
|
|
if group and len(selected) < 1000:
|
|
selected.append(dict(group.pop(), id=len(selected) + 1))
|
|
if len(selected) != 1000:
|
|
raise ValueError("Need 1,000 unique source passages")
|
|
payload = "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in selected)
|
|
dataset_path.write_text(payload)
|
|
manifest = {"count": 1000, "seed": 20260916, "source_books": len(groups),
|
|
"dataset_sha256": digest(payload), "reference_origin": "AI-generated; not human-reviewed",
|
|
"models": {"reference": "gpt-4o", "baseline": "gpt-4o-mini", "candidate": "gpt-4o"},
|
|
"prompts": PROMPTS, "temperature": 0.3, "max_tokens": 100,
|
|
"tokenizers": {"character": "remove whitespace, preserve punctuation, Unicode characters",
|
|
"word": "Python str.split, preserve punctuation"},
|
|
"n": 2, "aggregation": "macro mean F1 across all 1000 sources",
|
|
"legacy_equivalence": "unconfirmed: original ngram_tokenize unavailable",
|
|
"training_exclusion": "not locally fine-tuned; foundation-model training overlap unknown"}
|
|
manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2))
|
|
manifest = json.loads(manifest_path.read_text())
|
|
payload = dataset_path.read_text()
|
|
if digest(payload) != manifest["dataset_sha256"]:
|
|
raise ValueError("Frozen dataset hash mismatch")
|
|
rows = [json.loads(line) for line in payload.splitlines()]
|
|
from app.core.config import get_settings
|
|
from openai import OpenAI
|
|
client = OpenAI(api_key=get_settings().openai_api_key, timeout=90, max_retries=3)
|
|
outputs = {}
|
|
for role in ("reference", "baseline", "candidate"):
|
|
path = args.out / (role + ".jsonl")
|
|
existing = [json.loads(line) for line in path.read_text().splitlines()] if path.exists() else []
|
|
done = {row["id"]: row for row in existing if row.get("summary")}
|
|
def generate(row):
|
|
try:
|
|
response = client.chat.completions.create(
|
|
model=manifest["models"][role], temperature=manifest["temperature"],
|
|
max_tokens=manifest["max_tokens"], messages=[
|
|
{"role": "system", "content": "당신은 텍스트 요약 전문가입니다."},
|
|
{"role": "user", "content": manifest["prompts"][role] + "\n\n" + row["text"]}])
|
|
choice = response.choices[0]
|
|
summary = (choice.message.content or "").strip()
|
|
return {"id": row["id"], "summary": summary, "model": response.model,
|
|
"response_id": response.id, "finish_reason": choice.finish_reason,
|
|
"usage": response.usage.model_dump() if response.usage else None,
|
|
"over_50_chars": len(summary) > 50}
|
|
except Exception as exc:
|
|
return {"id": row["id"], "summary": "", "error_type": type(exc).__name__}
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as pool, path.open("a") as out:
|
|
futures = [pool.submit(generate, row) for row in rows if row["id"] not in done]
|
|
for future in concurrent.futures.as_completed(futures):
|
|
result = future.result()
|
|
out.write(json.dumps(result, ensure_ascii=False) + "\n")
|
|
out.flush()
|
|
done[result["id"]] = result
|
|
if len(done) % 50 == 0:
|
|
print(role, len(done), "/ 1000", flush=True)
|
|
outputs[role] = done
|
|
result = {"sample_count": 1000, "reference_origin": manifest["reference_origin"],
|
|
"legacy_equivalence": manifest["legacy_equivalence"], "certificate_target_achieved": None,
|
|
"dataset_sha256": manifest["dataset_sha256"], "scores": {}, "quality": {}}
|
|
for role, items in outputs.items():
|
|
result["quality"][role] = {"empty": sum(not r.get("summary") for r in items.values()),
|
|
"over_50_chars": sum(r.get("over_50_chars", False) for r in items.values()),
|
|
"truncated": sum(r.get("finish_reason") == "length" for r in items.values())}
|
|
per_item = []
|
|
for role in ("baseline", "candidate"):
|
|
result["scores"][role] = {}
|
|
for mode in ("character", "word"):
|
|
scores = []
|
|
for row in rows:
|
|
rid = row["id"]
|
|
score = metrics(outputs["reference"][rid]["summary"], outputs[role][rid]["summary"], mode)
|
|
scores.append(score)
|
|
per_item.append({"id": rid, "role": role, "mode": mode, **score})
|
|
result["scores"][role][mode] = {key: sum(s[key] for s in scores) / 1000 for key in ("precision", "recall", "f1")}
|
|
(args.out / "per_item.jsonl").write_text("".join(json.dumps(r) + "\n" for r in per_item))
|
|
(args.out / "result.json").write_text(json.dumps(result, ensure_ascii=False, indent=2))
|
|
print(json.dumps(result, ensure_ascii=False, indent=2), flush=True)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|