"""Re-evaluate a development-selected word-aware selector; no new API calls. Reuses the previously inspected test set. This is a re-evaluation, not a fresh independent certification test. Preserve earlier artifacts and report both metrics. """ from __future__ import annotations import argparse import json from pathlib import Path import sys sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from app.engine.summary_consensus import select_consensus from scripts.build_summary_trial import metrics def read(path): return {r['id']:r for r in (json.loads(l) for l in path.read_text().splitlines() if l.strip())} def evaluate(generated, references, weight): result=[] for rid,row in sorted(generated.items()): texts=[c['summary'] for c in row['candidates']] chosen=select_consensus(texts, word_weight=weight) text=texts[chosen] if chosen>=0 else '' result.append({'index':rid,'summary':text,'selected_index':chosen, 'scores':{m:metrics(references[rid],text,m) for m in ('character','word')}}) scores={m:{k:sum(r['scores'][m][k] for r in result)/len(result) for k in ('precision','recall','f1')} for m in ('character','word')} return scores,result def main(): ap=argparse.ArgumentParser(description=__doc__) ap.add_argument('--trial',type=Path,required=True) ap.add_argument('--previous',type=Path,required=True) ap.add_argument('--out',type=Path,required=True) args=ap.parse_args();args.out.mkdir(parents=True,exist_ok=True) dev=read(args.trial/'development.jsonl') devrefs={i:r['summary'] for i,r in read(args.previous/'reference.jsonl').items()} # Development comparison only; test scores do not choose this weight. compared={str(w):evaluate(dev,devrefs,w)[0] for w in (0.0,0.5,1.0)} weight=max((0.0,0.5,1.0),key=lambda w:compared[str(w)]['word']['f1']) decision={'selected_word_weight':weight,'selection_metric':'development_word_bigram_f1', 'development_scores':compared,'evaluation_type':'re-evaluation of previously inspected test set'} (args.out/'decision.json').write_text(json.dumps(decision,ensure_ascii=False,indent=2)+'\n') systems=read(args.trial/'system.jsonl'); sources=read(args.trial/'sources.jsonl') refs={i:r['candidates'][0]['summary'] for i,r in read(args.trial/'reference.jsonl').items()} assert set(systems)==set(sources)==set(refs)==set(range(1,1001)) scores,rows=evaluate(systems,refs,weight) old,_=evaluate(systems,refs,0.0) full=[] for row in rows: i=row['index'];full.append({'index':i,'original':sources[i]['text'], 'reference_summary':refs[i],'summary':row['summary'], 'rouge_score':row['scores']['character']['f1'],'word_rouge_score':row['scores']['word']['f1']}) record={'method':'source_only_consensus5_char_word_equal_weight','target_rouge':0.65, 'average_rouge':scores['character']['f1'],'average_word_rouge':scores['word']['f1'], 'success_rate':sum(r['rouge_score']>=0.65 for r in full)/1000, 'word_success_rate':sum(r['word_rouge_score']>=0.65 for r in full)/1000, 'success_rate_definition':'fraction of individual items with F1 >= 0.65', 'sample_count':1000,'reference_origin':'AI-generated, unreviewed', 'reported_metric':'character_bigram_f1','secondary_metric':'word_bigram_f1', 'evaluation_type':decision['evaluation_type'],'certificate_target_achieved':None, 'previous_scores':old,'scores':scores,'data':full} (args.out/'scorecard.json').write_text(json.dumps(record,ensure_ascii=False,indent=2)+'\n') print(json.dumps({k:v for k,v in record.items() if k!='data'},ensure_ascii=False,indent=2)) if __name__=='__main__':main()