From 99f0ac9b88bdaf12929819e295a0e1d4d5316352 Mon Sep 17 00:00:00 2001 From: hbyang Date: Wed, 16 Sep 2026 16:26:03 +0900 Subject: [PATCH] feat: add plagiarism testset sample viewer for scorecard capture MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 성적서 '1. 데이터 준비' 는 비표절/표절 데이터를 각각 첫 건과 마지막 건만 {line_number, original_text} 형태로 싣고 중간을 ... 로 생략한 뒤 총 건수를 적는다. 그 형식으로 출력한다. pairs.jsonl 의 실제 레코드를 읽으며 데이터를 새로 만들지 않는다. 표절 쪽 original_text 는 변형이 적용된 derived_text 다. 판정 대상이 그 글이기 때문이다. Co-Authored-By: Claude Opus 5 (1M context) --- scripts/show_plagiarism_samples.py | 93 ++++++++++++++++++++++++++++++ 1 file changed, 93 insertions(+) create mode 100644 scripts/show_plagiarism_samples.py diff --git a/scripts/show_plagiarism_samples.py b/scripts/show_plagiarism_samples.py new file mode 100644 index 0000000..bf853b5 --- /dev/null +++ b/scripts/show_plagiarism_samples.py @@ -0,0 +1,93 @@ +"""성능지표 #4 시험 데이터 예시를 성적서 형식으로 출력한다. + +전 차수 성적서(GERI `GERIR.CE.2511-02.009`) 12페이지 "1. 데이터 준비" 는 +비표절/표절 데이터를 각각 첫 건과 마지막 건만 `{line_number, original_text}` +형태로 싣고 중간을 `...` 로 생략한 뒤 총 건수를 적는다. 그 형식으로 낸다. + +시험셋(`pairs.jsonl`)의 실제 레코드를 읽으며, 데이터를 새로 만들지 않는다. +표절 데이터의 `original_text` 는 변형이 적용된 질의문(`derived_text`)이다. +판정 대상이 되는 글이 곧 그것이기 때문이다. + +사용: + python scripts/show_plagiarism_samples.py + python scripts/show_plagiarism_samples.py --head 2 --tail 2 + python scripts/show_plagiarism_samples.py --max-chars 300 --out samples.txt +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +def load(testset: Path) -> tuple[list[dict], list[dict]]: + rows = [json.loads(line) for line + in (testset / "pairs.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip()] + legitimate = [r for r in rows if not r["is_plagiarism"]] + plagiarism = [r for r in rows if r["is_plagiarism"]] + return legitimate, plagiarism + + +def record(line_number: int, text: str, max_chars: int | None) -> list[str]: + body = " ".join(text.split()) # 줄바꿈·연속 공백 정리 + if max_chars and len(body) > max_chars: + body = body[:max_chars] + " …" + return [ + " {", + ' "line_number": %d,' % line_number, + ' "original_text": "%s",' % body, + " },", + ] + + +def block(label: str, rows: list[dict], head: int, tail: int, + max_chars: int | None) -> list[str]: + total = len(rows) + lines = ["%s 데이터 예시 :" % label] + + indices = list(range(min(head, total))) + tail_start = max(len(indices), total - tail) + indices += list(range(tail_start, total)) + + previous = -1 + for i in indices: + if previous >= 0 and i != previous + 1: + lines.append("...") + lines.extend(record(i + 1, rows[i]["derived_text"], max_chars)) + previous = i + + lines.append("") + lines.append("%s 데이터 총 %d건" % (label, total)) + return lines + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--testset", type=Path, default=Path("data/eval/testset_v2")) + parser.add_argument("--head", type=int, default=1, help="앞에서 N건") + parser.add_argument("--tail", type=int, default=1, help="뒤에서 N건") + parser.add_argument("--max-chars", type=int, default=None, + help="본문 길이 제한 (미지정 시 전문)") + parser.add_argument("--out", help="파일로 저장 (미지정 시 표준출력)") + args = parser.parse_args() + + legitimate, plagiarism = load(args.testset) + + lines = ["1. 데이터 준비"] + lines.extend(block("비표절", legitimate, args.head, args.tail, args.max_chars)) + lines.append("") + lines.extend(block("표절", plagiarism, args.head, args.tail, args.max_chars)) + lines.append("전체 데이터 총 %d건" % (len(legitimate) + len(plagiarism))) + + text = "\n".join(lines) + if args.out: + Path(args.out).write_text(text + "\n", encoding="utf-8") + print("%s 에 기록" % args.out) + else: + print(text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())