요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
106 lines
6.4 KiB
Python
106 lines
6.4 KiB
Python
"""Develop two-stage compression, then re-evaluate one frozen configuration."""
|
|
from __future__ import annotations
|
|
import argparse
|
|
from collections import Counter
|
|
import concurrent.futures
|
|
import hashlib
|
|
import json
|
|
from pathlib import Path
|
|
import sys
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
from app.engine.summary_grounded import generate_grounded, EVIDENCE_PROMPT, COMPRESSION_INSTRUCTION
|
|
from app.engine.summary_consensus import select_consensus, SHORT_SUMMARY_PROMPT
|
|
from scripts.build_summary_trial import metrics
|
|
|
|
|
|
def read(path):
|
|
return {r['id']: r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())}
|
|
|
|
|
|
def save(path, value):
|
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2)+'\n')
|
|
|
|
|
|
def pick(row, strategy):
|
|
texts=[c['summary'] for c in row['candidates']]
|
|
i=select_consensus(texts,word_weight=0.5) if strategy=='consensus5' else (0 if texts[0] and len(texts[0])<=50 else -1)
|
|
return texts[i] if i>=0 else ''
|
|
|
|
|
|
def score(generated, refs, strategy):
|
|
return {m:{k:sum(metrics(refs[i],pick(row,strategy),m)[k] for i,row in generated.items())/len(generated)
|
|
for k in ('precision','recall','f1')} for m in ('character','word')}
|
|
|
|
|
|
def run(rows, path, client, count):
|
|
done=read(path) if path.exists() else {}
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool,path.open('a') as out:
|
|
pending={pool.submit(generate_grounded,client,r['text'],count):i for i,r in rows.items() if i not in done}
|
|
for f in concurrent.futures.as_completed(pending):
|
|
i=pending[f];value=dict(f.result(),id=i);done[i]=value
|
|
out.write(json.dumps(value,ensure_ascii=False)+'\n');out.flush()
|
|
if len(done)%50==0:print(path.name,len(done),'/',len(rows),flush=True)
|
|
assert set(done)==set(rows)
|
|
return done
|
|
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument('--trial',type=Path,required=True);ap.add_argument('--previous',type=Path,required=True)
|
|
ap.add_argument('--out',type=Path,required=True);args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True)
|
|
development=read(args.trial/'development.jsonl');previous=read(args.previous/'sources.jsonl')
|
|
devsources={i:previous[i] for i in development}
|
|
devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()}
|
|
from app.engine.structural import extract_lemmas
|
|
counts=Counter();diagnostics=[]
|
|
for i,row in sorted(development.items()):
|
|
hyp=pick(row,'consensus5');ref=devrefs[i];word=metrics(ref,hyp,'word')['f1']
|
|
a,b=Counter(extract_lemmas(ref)),Counter(extract_lemmas(hyp))
|
|
lf=2*sum((a&b).values())/max(1,sum(a.values())+sum(b.values()))
|
|
category='above_target' if word>=.65 else ('surface_difference_candidate' if lf>=.7 else ('content_selection_candidate' if lf<.4 else 'mixed_or_uncertain'))
|
|
counts[category]+=1;diagnostics.append({'id':i,'word_f1':word,'lemma_overlap_f1':lf,'category':category})
|
|
save(args.out/'diagnostics.json',{'note':'heuristic triage, not human error labels; lemma overlap is not the scoring metric',
|
|
'counts':dict(counts),'rows':diagnostics})
|
|
protocol={'reference_changes':False,'score_changes':False,'evaluation_type':'re-evaluation of previously inspected 1000-item test set',
|
|
'development_count':len(development),'selection_metric':'word_bigram_f1','evidence_prompt':EVIDENCE_PROMPT,
|
|
'summary_prompt':SHORT_SUMMARY_PROMPT,'compression_instruction':COMPRESSION_INSTRUCTION,
|
|
'strategies':['single','consensus5'],'model':'gpt-4o','evidence_temperature':0,'summary_temperature':0.3,
|
|
'reference_origin':'AI-generated, unreviewed','test_source_sha256':hashlib.sha256((args.trial/'sources.jsonl').read_bytes()).hexdigest()}
|
|
if (args.out/'protocol.json').exists():assert json.loads((args.out/'protocol.json').read_text())==protocol
|
|
else:save(args.out/'protocol.json',protocol)
|
|
from openai import OpenAI
|
|
from app.core.config import get_settings
|
|
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
|
|
devgen=run(devsources,args.out/'development.jsonl',client,5)
|
|
devscores={s:score(devgen,devrefs,s) for s in ('single','consensus5')}
|
|
selected=max(devscores,key=lambda s:devscores[s]['word']['f1'])
|
|
decision={'selected_experimental_strategy':selected,'current_development_scores':score(development,devrefs,'consensus5'),
|
|
'experimental_development_scores':devscores,'test_evaluation_reason':'user-requested comparison, not automatic promotion'}
|
|
if (args.out/'decision.json').exists():assert json.loads((args.out/'decision.json').read_text())==decision
|
|
else:save(args.out/'decision.json',decision)
|
|
print(json.dumps(decision),flush=True)
|
|
rows=read(args.trial/'sources.jsonl');refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()}
|
|
assert len(rows)==1000 and set(rows)==set(refs)
|
|
generated=run(rows,args.out/'system.jsonl',client,5 if selected=='consensus5' else 1)
|
|
newscore=score(generated,refs,selected);oldscore=score(read(args.trial/'system.jsonl'),refs,'consensus5')
|
|
details=[]
|
|
for i,row in sorted(rows.items()):
|
|
hyp=pick(generated[i],selected)
|
|
details.append({'index':i,'original':row['text'],'reference_summary':refs[i],'summary':hyp,
|
|
'rouge_score':metrics(refs[i],hyp,'character')['f1'],'word_rouge_score':metrics(refs[i],hyp,'word')['f1']})
|
|
result={'method':'evidence_then_compression_'+selected,'sample_count':1000,'target_rouge':.65,
|
|
'average_rouge':newscore['character']['f1'],'average_word_rouge':newscore['word']['f1'],
|
|
'success_rate':sum(r['rouge_score']>=.65 for r in details)/1000,
|
|
'word_success_rate':sum(r['word_rouge_score']>=.65 for r in details)/1000,
|
|
'success_rate_definition':'fraction of individual items with F1 >= 0.65',
|
|
'previous_scores':oldscore,'scores':newscore,'word_improved':newscore['word']['f1']>oldscore['word']['f1'],
|
|
'evaluation_type':protocol['evaluation_type'],'reference_origin':protocol['reference_origin'],
|
|
'certificate_target_achieved':None,'evidence_fallback_count':sum(r['evidence_fallback'] for r in generated.values()),
|
|
'empty_count':sum(not r['summary'] for r in details),'data':details}
|
|
save(args.out/'scorecard.json',result)
|
|
print(json.dumps({k:v for k,v in result.items() if k!='data'},ensure_ascii=False,indent=2),flush=True)
|
|
|
|
|
|
if __name__=='__main__':main()
|