요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
54 lines
2.6 KiB
Python
54 lines
2.6 KiB
Python
"""Offline evidence and F1 verification for the two-stage experiment."""
|
|
from __future__ import annotations
|
|
import argparse
|
|
from collections import Counter
|
|
import hashlib
|
|
import json
|
|
from pathlib import Path
|
|
import re
|
|
import sys
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
from app.engine.summary_consensus import select_consensus
|
|
|
|
|
|
def rows(path):
|
|
items=[json.loads(l) for l in path.read_text().splitlines() if l.strip()]
|
|
assert len(items)==1000 and {r['id'] for r in items}==set(range(1,1001))
|
|
return {r['id']:r for r in items}
|
|
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument('--trial',type=Path,required=True)
|
|
ap.add_argument('--experiment',type=Path,required=True)
|
|
a=ap.parse_args();sources=rows(a.trial/'sources.jsonl');refs=rows(a.trial/'reference.jsonl')
|
|
outputs=rows(a.experiment/'system.jsonl');result=json.loads((a.experiment/'scorecard.json').read_text())
|
|
protocol=json.loads((a.experiment/'protocol.json').read_text())
|
|
assert hashlib.sha256((a.trial/'sources.jsonl').read_bytes()).hexdigest()==protocol['test_source_sha256']
|
|
strategy=json.loads((a.experiment/'decision.json').read_text())['selected_experimental_strategy']
|
|
details={r['index']:r for r in result['data']};assert len(details)==len(result['data'])==1000
|
|
total={'character':0.,'word':0.}
|
|
for rid in range(1,1001):
|
|
row=outputs[rid];text=sources[rid]['text']
|
|
assert all(e in text and 1<=len(e)<=120 for e in row['evidence'])
|
|
assert len(row['evidence'])<=3
|
|
summaries=[c['summary'] for c in row['candidates']]
|
|
i=select_consensus(summaries,.5) if strategy=='consensus5' else (0 if summaries[0] and len(summaries[0])<=50 else -1)
|
|
hyp=summaries[i] if i>=0 else '';ref=refs[rid]['candidates'][0]['summary']
|
|
assert details[rid]['summary']==hyp and details[rid]['reference_summary']==ref and details[rid]['original']==text
|
|
for mode in total:
|
|
def grams(t):
|
|
t=list(re.sub(r'\s+','',t)) if mode=='character' else t.split()
|
|
return Counter(zip(t,t[1:]))
|
|
x,y=grams(ref),grams(hyp);denom=sum(x.values())+sum(y.values())
|
|
f1=2*sum(min(n,y[g]) for g,n in x.items())/denom if denom else 0
|
|
assert abs(f1-details[rid]['rouge_score' if mode=='character' else 'word_rouge_score'])<1e-12
|
|
total[mode]+=f1/1000
|
|
for mode,avg in total.items():
|
|
assert abs(avg-result['scores'][mode]['f1'])<1e-12
|
|
print(mode,'F1=%.6f%%'%(100*avg))
|
|
print('PASS: 1000 IDs, unchanged sources/references, literal evidence, selected outputs, per-item and mean F1')
|
|
|
|
|
|
if __name__=='__main__':main()
|