o2o-plagiarism-ai/scripts/evaluate_grounded_summary.py
hbyang 8ae52b0a82 feat: add summary (No.7) consensus and grounded generation trials
요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다.

- app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus).
  문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다.
- app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성.
- summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로
  중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다.

스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만
재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지
않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet ->
export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-28 10:02:04 +09:00

106 lines
6.4 KiB
Python

"""Develop two-stage compression, then re-evaluate one frozen configuration."""
from __future__ import annotations
import argparse
from collections import Counter
import concurrent.futures
import hashlib
import json
from pathlib import Path
import sys
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_grounded import generate_grounded, EVIDENCE_PROMPT, COMPRESSION_INSTRUCTION
from app.engine.summary_consensus import select_consensus, SHORT_SUMMARY_PROMPT
from scripts.build_summary_trial import metrics
def read(path):
return {r['id']: r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())}
def save(path, value):
path.write_text(json.dumps(value, ensure_ascii=False, indent=2)+'\n')
def pick(row, strategy):
texts=[c['summary'] for c in row['candidates']]
i=select_consensus(texts,word_weight=0.5) if strategy=='consensus5' else (0 if texts[0] and len(texts[0])<=50 else -1)
return texts[i] if i>=0 else ''
def score(generated, refs, strategy):
return {m:{k:sum(metrics(refs[i],pick(row,strategy),m)[k] for i,row in generated.items())/len(generated)
for k in ('precision','recall','f1')} for m in ('character','word')}
def run(rows, path, client, count):
done=read(path) if path.exists() else {}
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool,path.open('a') as out:
pending={pool.submit(generate_grounded,client,r['text'],count):i for i,r in rows.items() if i not in done}
for f in concurrent.futures.as_completed(pending):
i=pending[f];value=dict(f.result(),id=i);done[i]=value
out.write(json.dumps(value,ensure_ascii=False)+'\n');out.flush()
if len(done)%50==0:print(path.name,len(done),'/',len(rows),flush=True)
assert set(done)==set(rows)
return done
def main():
ap=argparse.ArgumentParser(description=__doc__)
ap.add_argument('--trial',type=Path,required=True);ap.add_argument('--previous',type=Path,required=True)
ap.add_argument('--out',type=Path,required=True);args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True)
development=read(args.trial/'development.jsonl');previous=read(args.previous/'sources.jsonl')
devsources={i:previous[i] for i in development}
devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()}
from app.engine.structural import extract_lemmas
counts=Counter();diagnostics=[]
for i,row in sorted(development.items()):
hyp=pick(row,'consensus5');ref=devrefs[i];word=metrics(ref,hyp,'word')['f1']
a,b=Counter(extract_lemmas(ref)),Counter(extract_lemmas(hyp))
lf=2*sum((a&b).values())/max(1,sum(a.values())+sum(b.values()))
category='above_target' if word>=.65 else ('surface_difference_candidate' if lf>=.7 else ('content_selection_candidate' if lf<.4 else 'mixed_or_uncertain'))
counts[category]+=1;diagnostics.append({'id':i,'word_f1':word,'lemma_overlap_f1':lf,'category':category})
save(args.out/'diagnostics.json',{'note':'heuristic triage, not human error labels; lemma overlap is not the scoring metric',
'counts':dict(counts),'rows':diagnostics})
protocol={'reference_changes':False,'score_changes':False,'evaluation_type':'re-evaluation of previously inspected 1000-item test set',
'development_count':len(development),'selection_metric':'word_bigram_f1','evidence_prompt':EVIDENCE_PROMPT,
'summary_prompt':SHORT_SUMMARY_PROMPT,'compression_instruction':COMPRESSION_INSTRUCTION,
'strategies':['single','consensus5'],'model':'gpt-4o','evidence_temperature':0,'summary_temperature':0.3,
'reference_origin':'AI-generated, unreviewed','test_source_sha256':hashlib.sha256((args.trial/'sources.jsonl').read_bytes()).hexdigest()}
if (args.out/'protocol.json').exists():assert json.loads((args.out/'protocol.json').read_text())==protocol
else:save(args.out/'protocol.json',protocol)
from openai import OpenAI
from app.core.config import get_settings
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
devgen=run(devsources,args.out/'development.jsonl',client,5)
devscores={s:score(devgen,devrefs,s) for s in ('single','consensus5')}
selected=max(devscores,key=lambda s:devscores[s]['word']['f1'])
decision={'selected_experimental_strategy':selected,'current_development_scores':score(development,devrefs,'consensus5'),
'experimental_development_scores':devscores,'test_evaluation_reason':'user-requested comparison, not automatic promotion'}
if (args.out/'decision.json').exists():assert json.loads((args.out/'decision.json').read_text())==decision
else:save(args.out/'decision.json',decision)
print(json.dumps(decision),flush=True)
rows=read(args.trial/'sources.jsonl');refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()}
assert len(rows)==1000 and set(rows)==set(refs)
generated=run(rows,args.out/'system.jsonl',client,5 if selected=='consensus5' else 1)
newscore=score(generated,refs,selected);oldscore=score(read(args.trial/'system.jsonl'),refs,'consensus5')
details=[]
for i,row in sorted(rows.items()):
hyp=pick(generated[i],selected)
details.append({'index':i,'original':row['text'],'reference_summary':refs[i],'summary':hyp,
'rouge_score':metrics(refs[i],hyp,'character')['f1'],'word_rouge_score':metrics(refs[i],hyp,'word')['f1']})
result={'method':'evidence_then_compression_'+selected,'sample_count':1000,'target_rouge':.65,
'average_rouge':newscore['character']['f1'],'average_word_rouge':newscore['word']['f1'],
'success_rate':sum(r['rouge_score']>=.65 for r in details)/1000,
'word_success_rate':sum(r['word_rouge_score']>=.65 for r in details)/1000,
'success_rate_definition':'fraction of individual items with F1 >= 0.65',
'previous_scores':oldscore,'scores':newscore,'word_improved':newscore['word']['f1']>oldscore['word']['f1'],
'evaluation_type':protocol['evaluation_type'],'reference_origin':protocol['reference_origin'],
'certificate_target_achieved':None,'evidence_fallback_count':sum(r['evidence_fallback'] for r in generated.values()),
'empty_count':sum(not r['summary'] for r in details),'data':details}
save(args.out/'scorecard.json',result)
print(json.dumps({k:v for k,v in result.items() if k!='data'},ensure_ascii=False,indent=2),flush=True)
if __name__=='__main__':main()