요약 성능지표(No.7, ROUGE 65점) 실험 일습을 커밋한다. - app/engine/summary_consensus.py: 후보 5개 생성 후 합의 선택(select_consensus). 문자/어절 2-gram 가중 합의로 고르며, 유효 후보가 없으면 valid=False 를 낸다. - app/engine/summary_grounded.py: 근거 추출 후 압축하는 2단계 생성. - summarizer.py: 추출 전략 3종(textrank/coverage/lead). coverage 는 MMR 로 중복 문장을 눌러 문서 전체를 넓게 담는다. 기본값은 textrank 로 유지한다. 스크립트는 생성·평가·검증을 분리했다. verify_* 는 API 호출 없이 저장된 산출물만 재계산하는 독립 검증기라 지표를 공용 함수로 합치지 않는다. 합치면 검증이 성립하지 않는다. tune_summarizer 는 사람 검수 참조(build_summary_annotation_packet -> export_summary_annotations)를 입력으로 받아 선택과 최종 보고를 분리한다. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
82 lines
3.3 KiB
Python
82 lines
3.3 KiB
Python
#!/usr/bin/env python3
|
|
"""완료된 사람 요약 검수 XLSX를 ROUGE 평가용 JSONL로 내보낸다.
|
|
|
|
빈 참조·미완료 행·동일 참조를 실패 처리해, 검수 전 데이터를 성적서 평가에
|
|
실수로 사용하지 않게 한다. 원문과 작성자 그룹은 유지하되 검수자 실명은 내보내지 않는다.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
from pathlib import Path
|
|
|
|
|
|
REQUIRED_COLUMNS = {
|
|
"annotation_id", "source_group", "source_text", "reference_summary_1",
|
|
"reference_summary_2", "status",
|
|
}
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("xlsx", type=Path)
|
|
parser.add_argument("--out", type=Path, required=True)
|
|
parser.add_argument("--allow-single-reference", action="store_true",
|
|
help="2차 독립 검수가 끝나기 전 파일럿에만 사용")
|
|
args = parser.parse_args()
|
|
|
|
from openpyxl import load_workbook
|
|
book = load_workbook(args.xlsx, read_only=True, data_only=True)
|
|
if "annotations" not in book.sheetnames:
|
|
parser.error("annotations 시트가 없습니다")
|
|
sheet = book["annotations"]
|
|
headers = [str(cell.value or "").strip() for cell in next(sheet.iter_rows(max_row=1))]
|
|
positions = {name: index for index, name in enumerate(headers)}
|
|
missing = REQUIRED_COLUMNS - positions.keys()
|
|
if missing:
|
|
parser.error("필수 열이 없습니다: %s" % ", ".join(sorted(missing)))
|
|
|
|
rows, errors = [], []
|
|
for excel_row, values in enumerate(sheet.iter_rows(min_row=2, values_only=True), start=2):
|
|
value = lambda key: str(values[positions[key]] or "").strip()
|
|
status = value("status").casefold()
|
|
if not any(values):
|
|
continue
|
|
if status != "completed":
|
|
errors.append("행 %d: status가 completed가 아닙니다" % excel_row)
|
|
continue
|
|
refs = [value("reference_summary_1"), value("reference_summary_2")]
|
|
refs = [ref for ref in refs if ref]
|
|
if not args.allow_single_reference and len(refs) != 2:
|
|
errors.append("행 %d: 독립 참조 요약 2개가 필요합니다" % excel_row)
|
|
continue
|
|
if refs and len({" ".join(ref.split()) for ref in refs}) != len(refs):
|
|
errors.append("행 %d: 두 참조 요약이 동일합니다" % excel_row)
|
|
continue
|
|
if not refs or not value("source_text"):
|
|
errors.append("행 %d: 원문 또는 참조 요약이 비어 있습니다" % excel_row)
|
|
continue
|
|
rows.append({
|
|
"id": value("annotation_id"),
|
|
"source_group": value("source_group"),
|
|
"text": value("source_text"),
|
|
"references": refs,
|
|
})
|
|
book.close()
|
|
if errors:
|
|
print("내보내지 않음 — 검수 오류 %d건:" % len(errors))
|
|
print("\n".join(errors[:20]))
|
|
return 2
|
|
if not rows:
|
|
print("내보낼 completed 행이 없습니다")
|
|
return 2
|
|
args.out.parent.mkdir(parents=True, exist_ok=True)
|
|
args.out.write_text("".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows), encoding="utf-8")
|
|
print("저장: %s / %d건 / 작성자 그룹 %d개" %
|
|
(args.out, len(rows), len({row['source_group'] for row in rows})))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|