"""Development-only selection followed by a fresh 1,000-source evaluation.""" from __future__ import annotations import argparse import concurrent.futures import hashlib import json import random import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from app.engine.summary_consensus import generate_candidates, select_consensus, SHORT_SUMMARY_PROMPT from scripts.build_summary_trial import metrics, PROMPTS def read(path): return [json.loads(l) for l in path.read_text().splitlines() if l.strip()] def save(path, value): path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n") def sha(text): return hashlib.sha256(text.encode()).hexdigest() def run_rows(rows, path, fn): done = {r['id']: r for r in read(path)} if path.exists() else {} with concurrent.futures.ThreadPoolExecutor(max_workers=12) as pool, path.open('a') as out: pending = {pool.submit(fn, r): r['id'] for r in rows if r['id'] not in done} for future in concurrent.futures.as_completed(pending): rid = pending[future] value = future.result() # fail visibly; resume persisted successes done[rid] = dict(value, id=rid) out.write(json.dumps(done[rid], ensure_ascii=False) + '\n'); out.flush() if len(done) % 50 == 0: print(path.name, len(done), '/', len(rows), flush=True) return done def summary(row, strategy): texts = [c['summary'] for c in row['candidates']] index = 0 if strategy == 'single' else select_consensus(texts) return texts[index] if index >= 0 else '' def score(rows, references, generated, strategy): return {mode: {key: sum(metrics(references[r['id']], summary(generated[r['id']], strategy), mode)[key] for r in rows) / len(rows) for key in ('precision','recall','f1')} for mode in ('character','word')} def main(): ap=argparse.ArgumentParser(); ap.add_argument('--previous',type=Path,required=True) ap.add_argument('--source',type=Path,required=True); ap.add_argument('--out',type=Path,required=True) args=ap.parse_args(); args.out.mkdir(parents=True,exist_ok=True) previous=read(args.previous/'sources.jsonl') oldrefs={r['id']:r['summary'] for r in read(args.previous/'reference.jsonl')} dev_books=set(sorted({r['source_file'] for r in previous})[:10]) dev=[r for r in previous if r['source_file'] in dev_books] if not (args.out/'protocol.json').exists(): save(args.out/'protocol.json',{'development_books':sorted(dev_books),'development_count':len(dev), 'primary_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1', 'strategies':['single','consensus5'],'reference_prompt':PROMPTS['reference'], 'system_prompt':SHORT_SUMMARY_PROMPT,'model':'gpt-4o','temperature':0.3, 'reference_origin':'AI-generated, unreviewed','bias':'same model and task instructions for reference and system', 'legacy_tokenizer_equivalence':'unconfirmed','target':0.65}) from app.core.config import get_settings from openai import OpenAI client=OpenAI(api_key=get_settings().openai_api_key,timeout=120,max_retries=3) generated=run_rows(dev,args.out/'development.jsonl',lambda row:{'candidates':generate_candidates(client,row['text'])}) development={s:score(dev,oldrefs,generated,s) for s in ('single','consensus5')} selected=max(development,key=lambda s:development[s]['character']['f1']) decision={'development_scores':development,'selected':selected,'selected_before_test':True} decision_path=args.out/'decision.json' if decision_path.exists(): assert json.loads(decision_path.read_text())==decision else:save(decision_path,decision) print(json.dumps(decision),flush=True) # Hold out entire development books and every previous trial paragraph. dataset=args.out/'sources.jsonl' if not dataset.exists(): seen={sha(' '.join(r['text'].split())) for r in previous};groups=[];rng=random.Random(20260917) for p in sorted(args.source.glob('*.json')): if p.name in dev_books:continue group=[] for r in json.loads(p.read_text()).get('results',[]): text=str(r.get('source_text','')).strip();h=sha(' '.join(text.split())) if not 200<=len(text)<=2000 or h in seen:continue seen.add(h);group.append({'text':text,'source_file':p.name,'source_index':r.get('index'),'source_sha256':h}) rng.shuffle(group) if group:groups.append(group) rows=[] while len(rows)<1000 and any(groups): for g in groups: if g and len(rows)<1000:rows.append(dict(g.pop(),id=len(rows)+1)) assert len(rows)==1000 dataset.write_text(''.join(json.dumps(r,ensure_ascii=False)+'\n' for r in rows)) save(args.out/'dataset_manifest.json',{'count':1000,'sha256':sha(dataset.read_text()),'seed':20260917, 'development_book_overlap':0,'previous_exact_text_overlap':0,'near_duplicate_screening':'not performed'}) rows=read(dataset);assert sha(dataset.read_text())==json.loads((args.out/'dataset_manifest.json').read_text())['sha256'] refs=run_rows(rows,args.out/'reference.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],1)}) reftexts={i:r['candidates'][0]['summary'] for i,r in refs.items()} # Freeze candidate selection before any held-out score is calculated. systems=run_rows(rows,args.out/'system.jsonl',lambda r:{'candidates':generate_candidates(client,r['text'],5 if selected=='consensus5' else 1)}) def baseline(row): resp=client.chat.completions.create(model='gpt-4o',temperature=0.3,max_tokens=100,messages=[ {'role':'system','content':'당신은 텍스트 요약 전문가입니다.'}, {'role':'user','content':PROMPTS['candidate']+'\n\n'+row['text']}]) c=resp.choices[0] return {'candidates':[{'summary':(c.message.content or '').strip() if c.finish_reason=='stop' else '', 'finish_reason':c.finish_reason,'model':resp.model,'response_id':resp.id}]} base=run_rows(rows,args.out/'baseline.jsonl',baseline) result={'count':1000,'selected':selected,'reference_origin':'AI-generated, unreviewed', 'bias':'same model and task instructions for reference and improved system', 'certificate_target_achieved':None,'legacy_tokenizer_equivalence':'unconfirmed', 'baseline':score(rows,reftexts,base,'single'),'improved':score(rows,reftexts,systems,selected)} result['primary_target_achieved']=result['improved']['character']['f1']>=0.65 result['quality']={label:{'empty':sum(not s for s in texts),'over50':sum(len(s)>50 for s in texts)} for label,texts in [('reference',list(reftexts.values())),('baseline',[summary(base[r['id']],'single') for r in rows]), ('improved',[summary(systems[r['id']],selected) for r in rows])]} save(args.out/'result.json',result);print(json.dumps(result,ensure_ascii=False,indent=2),flush=True) if __name__=='__main__':main()