요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
128 lines
7.0 KiB
Python
128 lines
7.0 KiB
Python
"""Development-only selection followed by a fresh 1,000-source evaluation."""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import concurrent.futures
|
|
import hashlib
|
|
import json
|
|
import random
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
from app.engine.summary_consensus import generate_candidates, select_consensus, SHORT_SUMMARY_PROMPT
|
|
from scripts.build_summary_trial import metrics, PROMPTS
|
|
|
|
|
|
def read(path):
|
|
return [json.loads(l) for l in path.read_text().splitlines() if l.strip()]
|
|
|
|
|
|
def save(path, value):
|
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n")
|
|
|
|
|
|
def sha(text):
|
|
return hashlib.sha256(text.encode()).hexdigest()
|
|
|
|
|
|
def run_rows(rows, path, fn):
|
|
done = {r['id']: r for r in read(path)} if path.exists() else {}
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool, path.open('a') as out:
|
|
pending = {pool.submit(fn, r): r['id'] for r in rows if r['id'] not in done}
|
|
for future in concurrent.futures.as_completed(pending):
|
|
rid = pending[future]
|
|
value = future.result() # fail visibly; resume persisted successes
|
|
done[rid] = dict(value, id=rid)
|
|
out.write(json.dumps(done[rid], ensure_ascii=False) + '\n'); out.flush()
|
|
if len(done) % 50 == 0:
|
|
print(path.name, len(done), '/', len(rows), flush=True)
|
|
return done
|
|
|
|
|
|
def summary(row, strategy):
|
|
texts = [c['summary'] for c in row['candidates']]
|
|
index = 0 if strategy == 'single' else select_consensus(texts)
|
|
return texts[index] if index >= 0 else ''
|
|
|
|
|
|
def score(rows, references, generated, strategy):
|
|
return {mode: {key: sum(metrics(references[r['id']], summary(generated[r['id']], strategy), mode)[key]
|
|
for r in rows) / len(rows) for key in ('precision','recall','f1')}
|
|
for mode in ('character','word')}
|
|
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser(); ap.add_argument('--previous',type=Path,required=True)
|
|
ap.add_argument('--source',type=Path,required=True); ap.add_argument('--out',type=Path,required=True)
|
|
args=ap.parse_args(); args.out.mkdir(parents=True,exist_ok=True)
|
|
previous=read(args.previous/'sources.jsonl')
|
|
oldrefs={r['id']:r['summary'] for r in read(args.previous/'reference.jsonl')}
|
|
dev_books=set(sorted({r['source_file'] for r in previous})[:10])
|
|
dev=[r for r in previous if r['source_file'] in dev_books]
|
|
if not (args.out/'protocol.json').exists():
|
|
save(args.out/'protocol.json',{'development_books':sorted(dev_books),'development_count':len(dev),
|
|
'primary_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1',
|
|
'strategies':['single','consensus5'],'reference_prompt':PROMPTS['reference'],
|
|
'system_prompt':SHORT_SUMMARY_PROMPT,'model':'gpt-4o','temperature':0.3,
|
|
'reference_origin':'AI-generated, unreviewed','bias':'same model and task instructions for reference and system',
|
|
'legacy_tokenizer_equivalence':'unconfirmed','target':0.65})
|
|
from app.core.config import get_settings
|
|
from openai import OpenAI
|
|
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
|
|
generated=run_rows(dev,args.out/'development.jsonl',lambda row:{'candidates':generate_candidates(client,row['text'])})
|
|
development={s:score(dev,oldrefs,generated,s) for s in ('single','consensus5')}
|
|
selected=max(development,key=lambda s:development[s]['character']['f1'])
|
|
decision={'development_scores':development,'selected':selected,'selected_before_test':True}
|
|
decision_path=args.out/'decision.json'
|
|
if decision_path.exists():
|
|
assert json.loads(decision_path.read_text())==decision
|
|
else:save(decision_path,decision)
|
|
print(json.dumps(decision),flush=True)
|
|
# Hold out entire development books and every previous trial paragraph.
|
|
dataset=args.out/'sources.jsonl'
|
|
if not dataset.exists():
|
|
seen={sha(' '.join(r['text'].split())) for r in previous};groups=[];rng=random.Random(20260917)
|
|
for p in sorted(args.source.glob('*.json')):
|
|
if p.name in dev_books:continue
|
|
group=[]
|
|
for r in json.loads(p.read_text()).get('results',[]):
|
|
text=str(r.get('source_text','')).strip();h=sha(' '.join(text.split()))
|
|
if not 200<=len(text)<=2000 or h in seen:continue
|
|
seen.add(h);group.append({'text':text,'source_file':p.name,'source_index':r.get('index'),'source_sha256':h})
|
|
rng.shuffle(group)
|
|
if group:groups.append(group)
|
|
rows=[]
|
|
while len(rows)<1000 and any(groups):
|
|
for g in groups:
|
|
if g and len(rows)<1000:rows.append(dict(g.pop(),id=len(rows)+1))
|
|
assert len(rows)==1000
|
|
dataset.write_text(''.join(json.dumps(r,ensure_ascii=False)+'\n' for r in rows))
|
|
save(args.out/'dataset_manifest.json',{'count':1000,'sha256':sha(dataset.read_text()),'seed':20260917,
|
|
'development_book_overlap':0,'previous_exact_text_overlap':0,'near_duplicate_screening':'not performed'})
|
|
rows=read(dataset);assert sha(dataset.read_text())==json.loads((args.out/'dataset_manifest.json').read_text())['sha256']
|
|
refs=run_rows(rows,args.out/'reference.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],1)})
|
|
reftexts={i:r['candidates'][0]['summary'] for i,r in refs.items()}
|
|
# Freeze candidate selection before any held-out score is calculated.
|
|
systems=run_rows(rows,args.out/'system.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],5 if selected=='consensus5' else 1)})
|
|
def baseline(row):
|
|
resp=client.chat.completions.create(model='gpt-4o',temperature=0.3,max_tokens=100,messages=[
|
|
{'role':'system','content':'당신은 텍스트 요약 전문가입니다.'},
|
|
{'role':'user','content':PROMPTS['candidate']+'\n\n'+row['text']}])
|
|
c=resp.choices[0]
|
|
return {'candidates':[{'summary':(c.message.content or '').strip() if c.finish_reason=='stop' else '',
|
|
'finish_reason':c.finish_reason,'model':resp.model,'response_id':resp.id}]}
|
|
base=run_rows(rows,args.out/'baseline.jsonl',baseline)
|
|
result={'count':1000,'selected':selected,'reference_origin':'AI-generated, unreviewed',
|
|
'bias':'same model and task instructions for reference and improved system',
|
|
'certificate_target_achieved':None,'legacy_tokenizer_equivalence':'unconfirmed',
|
|
'baseline':score(rows,reftexts,base,'single'),'improved':score(rows,reftexts,systems,selected)}
|
|
result['primary_target_achieved']=result['improved']['character']['f1']>=0.65
|
|
result['quality']={label:{'empty':sum(not s for s in texts),'over50':sum(len(s)>50 for s in texts)}
|
|
for label,texts in [('reference',list(reftexts.values())),('baseline',[summary(base[r['id']],'single') for r in rows]),
|
|
('improved',[summary(systems[r['id']],selected) for r in rows])]}
|
|
save(args.out/'result.json',result);print(json.dumps(result,ensure_ascii=False,indent=2),flush=True)
|
|
|
|
|
|
if __name__=='__main__':main()
|