요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
71 lines
3.7 KiB
Python
71 lines
3.7 KiB
Python
"""Re-evaluate a development-selected word-aware selector; no new API calls.
|
|
|
|
Reuses the previously inspected test set. This is a re-evaluation, not a fresh
|
|
independent certification test. Preserve earlier artifacts and report both metrics.
|
|
"""
|
|
from __future__ import annotations
|
|
import argparse
|
|
import json
|
|
from pathlib import Path
|
|
import sys
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
from app.engine.summary_consensus import select_consensus
|
|
from scripts.build_summary_trial import metrics
|
|
|
|
|
|
def read(path):
|
|
return {r['id']:r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())}
|
|
|
|
|
|
def evaluate(generated, references, weight):
|
|
result=[]
|
|
for rid,row in sorted(generated.items()):
|
|
texts=[c['summary'] for c in row['candidates']]
|
|
chosen=select_consensus(texts, word_weight=weight)
|
|
text=texts[chosen] if chosen>=0 else ''
|
|
result.append({'index':rid,'summary':text,'selected_index':chosen,
|
|
'scores':{m:metrics(references[rid],text,m) for m in ('character','word')}})
|
|
scores={m:{k:sum(r['scores'][m][k] for r in result)/len(result)
|
|
for k in ('precision','recall','f1')} for m in ('character','word')}
|
|
return scores,result
|
|
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument('--trial',type=Path,required=True)
|
|
ap.add_argument('--previous',type=Path,required=True)
|
|
ap.add_argument('--out',type=Path,required=True)
|
|
args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True)
|
|
dev=read(args.trial/'development.jsonl')
|
|
devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()}
|
|
# Development comparison only; test scores do not choose this weight.
|
|
compared={str(w):evaluate(dev,devrefs,w)[0] for w in (0.0,0.5,1.0)}
|
|
weight=max((0.0,0.5,1.0),key=lambda w:compared[str(w)]['word']['f1'])
|
|
decision={'selected_word_weight':weight,'selection_metric':'development_word_bigram_f1',
|
|
'development_scores':compared,'evaluation_type':'re-evaluation of previously inspected test set'}
|
|
(args.out/'decision.json').write_text(json.dumps(decision,ensure_ascii=False,indent=2)+'\n')
|
|
systems=read(args.trial/'system.jsonl'); sources=read(args.trial/'sources.jsonl')
|
|
refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()}
|
|
assert set(systems)==set(sources)==set(refs)==set(range(1,1001))
|
|
scores,rows=evaluate(systems,refs,weight)
|
|
old,_=evaluate(systems,refs,0.0)
|
|
full=[]
|
|
for row in rows:
|
|
i=row['index'];full.append({'index':i,'original':sources[i]['text'],
|
|
'reference_summary':refs[i],'summary':row['summary'],
|
|
'rouge_score':row['scores']['character']['f1'],'word_rouge_score':row['scores']['word']['f1']})
|
|
record={'method':'source_only_consensus5_char_word_equal_weight','target_rouge':0.65,
|
|
'average_rouge':scores['character']['f1'],'average_word_rouge':scores['word']['f1'],
|
|
'success_rate':sum(r['rouge_score']>=0.65 for r in full)/1000,
|
|
'word_success_rate':sum(r['word_rouge_score']>=0.65 for r in full)/1000,
|
|
'success_rate_definition':'fraction of individual items with F1 >= 0.65',
|
|
'sample_count':1000,'reference_origin':'AI-generated, unreviewed',
|
|
'reported_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1',
|
|
'evaluation_type':decision['evaluation_type'],'certificate_target_achieved':None,
|
|
'previous_scores':old,'scores':scores,'data':full}
|
|
(args.out/'scorecard.json').write_text(json.dumps(record,ensure_ascii=False,indent=2)+'\n')
|
|
print(json.dumps({k:v for k,v in record.items() if k!='data'},ensure_ascii=False,indent=2))
|
|
|
|
|
|
if __name__=='__main__':main()
|