o2o-plagiarism-ai/scripts/improve_summary_trial.py
hbyang 8ae52b0a82 feat: add summary (No.7) consensus and grounded generation trials
요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다.

- app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus).
  문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다.
- app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성.
- summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로
  중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다.

스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만
재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지
않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet ->
export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-28 10:02:04 +09:00

128 lines
7.0 KiB
Python

"""Development-only selection followed by a fresh 1,000-source evaluation."""
from __future__ import annotations
import argparse
import concurrent.futures
import hashlib
import json
import random
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.engine.summary_consensus import generate_candidates, select_consensus, SHORT_SUMMARY_PROMPT
from scripts.build_summary_trial import metrics, PROMPTS
def read(path):
return [json.loads(l) for l in path.read_text().splitlines() if l.strip()]
def save(path, value):
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n")
def sha(text):
return hashlib.sha256(text.encode()).hexdigest()
def run_rows(rows, path, fn):
done = {r['id']: r for r in read(path)} if path.exists() else {}
with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool, path.open('a') as out:
pending = {pool.submit(fn, r): r['id'] for r in rows if r['id'] not in done}
for future in concurrent.futures.as_completed(pending):
rid = pending[future]
value = future.result() # fail visibly; resume persisted successes
done[rid] = dict(value, id=rid)
out.write(json.dumps(done[rid], ensure_ascii=False) + '\n'); out.flush()
if len(done) % 50 == 0:
print(path.name, len(done), '/', len(rows), flush=True)
return done
def summary(row, strategy):
texts = [c['summary'] for c in row['candidates']]
index = 0 if strategy == 'single' else select_consensus(texts)
return texts[index] if index >= 0 else ''
def score(rows, references, generated, strategy):
return {mode: {key: sum(metrics(references[r['id']], summary(generated[r['id']], strategy), mode)[key]
for r in rows) / len(rows) for key in ('precision','recall','f1')}
for mode in ('character','word')}
def main():
ap=argparse.ArgumentParser(); ap.add_argument('--previous',type=Path,required=True)
ap.add_argument('--source',type=Path,required=True); ap.add_argument('--out',type=Path,required=True)
args=ap.parse_args(); args.out.mkdir(parents=True,exist_ok=True)
previous=read(args.previous/'sources.jsonl')
oldrefs={r['id']:r['summary'] for r in read(args.previous/'reference.jsonl')}
dev_books=set(sorted({r['source_file'] for r in previous})[:10])
dev=[r for r in previous if r['source_file'] in dev_books]
if not (args.out/'protocol.json').exists():
save(args.out/'protocol.json',{'development_books':sorted(dev_books),'development_count':len(dev),
'primary_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1',
'strategies':['single','consensus5'],'reference_prompt':PROMPTS['reference'],
'system_prompt':SHORT_SUMMARY_PROMPT,'model':'gpt-4o','temperature':0.3,
'reference_origin':'AI-generated, unreviewed','bias':'same model and task instructions for reference and system',
'legacy_tokenizer_equivalence':'unconfirmed','target':0.65})
from app.core.config import get_settings
from openai import OpenAI
client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3)
generated=run_rows(dev,args.out/'development.jsonl',lambda row:{'candidates':generate_candidates(client,row['text'])})
development={s:score(dev,oldrefs,generated,s) for s in ('single','consensus5')}
selected=max(development,key=lambda s:development[s]['character']['f1'])
decision={'development_scores':development,'selected':selected,'selected_before_test':True}
decision_path=args.out/'decision.json'
if decision_path.exists():
assert json.loads(decision_path.read_text())==decision
else:save(decision_path,decision)
print(json.dumps(decision),flush=True)
# Hold out entire development books and every previous trial paragraph.
dataset=args.out/'sources.jsonl'
if not dataset.exists():
seen={sha(' '.join(r['text'].split())) for r in previous};groups=[];rng=random.Random(20260917)
for p in sorted(args.source.glob('*.json')):
if p.name in dev_books:continue
group=[]
for r in json.loads(p.read_text()).get('results',[]):
text=str(r.get('source_text','')).strip();h=sha(' '.join(text.split()))
if not 200<=len(text)<=2000 or h in seen:continue
seen.add(h);group.append({'text':text,'source_file':p.name,'source_index':r.get('index'),'source_sha256':h})
rng.shuffle(group)
if group:groups.append(group)
rows=[]
while len(rows)<1000 and any(groups):
for g in groups:
if g and len(rows)<1000:rows.append(dict(g.pop(),id=len(rows)+1))
assert len(rows)==1000
dataset.write_text(''.join(json.dumps(r,ensure_ascii=False)+'\n' for r in rows))
save(args.out/'dataset_manifest.json',{'count':1000,'sha256':sha(dataset.read_text()),'seed':20260917,
'development_book_overlap':0,'previous_exact_text_overlap':0,'near_duplicate_screening':'not performed'})
rows=read(dataset);assert sha(dataset.read_text())==json.loads((args.out/'dataset_manifest.json').read_text())['sha256']
refs=run_rows(rows,args.out/'reference.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],1)})
reftexts={i:r['candidates'][0]['summary'] for i,r in refs.items()}
# Freeze candidate selection before any held-out score is calculated.
systems=run_rows(rows,args.out/'system.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],5 if selected=='consensus5' else 1)})
def baseline(row):
resp=client.chat.completions.create(model='gpt-4o',temperature=0.3,max_tokens=100,messages=[
{'role':'system','content':'당신은 텍스트 요약 전문가입니다.'},
{'role':'user','content':PROMPTS['candidate']+'\n\n'+row['text']}])
c=resp.choices[0]
return {'candidates':[{'summary':(c.message.content or '').strip() if c.finish_reason=='stop' else '',
'finish_reason':c.finish_reason,'model':resp.model,'response_id':resp.id}]}
base=run_rows(rows,args.out/'baseline.jsonl',baseline)
result={'count':1000,'selected':selected,'reference_origin':'AI-generated, unreviewed',
'bias':'same model and task instructions for reference and improved system',
'certificate_target_achieved':None,'legacy_tokenizer_equivalence':'unconfirmed',
'baseline':score(rows,reftexts,base,'single'),'improved':score(rows,reftexts,systems,selected)}
result['primary_target_achieved']=result['improved']['character']['f1']>=0.65
result['quality']={label:{'empty':sum(not s for s in texts),'over50':sum(len(s)>50 for s in texts)}
for label,texts in [('reference',list(reftexts.values())),('baseline',[summary(base[r['id']],'single') for r in rows]),
('improved',[summary(systems[r['id']],selected) for r in rows])]}
save(args.out/'result.json',result);print(json.dumps(result,ensure_ascii=False,indent=2),flush=True)
if __name__=='__main__':main()