from __future__ import annotations import re from collections import Counter def ngram_tokenize(text: str, n: int = 2) -> list[tuple[str, ...]]: """공백을 제거한 문자 단위 n-gram 목록""" tokens = list(re.sub(r"\s+", "", text)) return [tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1)] def calculate_rouge_n(reference: str, hypothesis: str, n: int = 2) -> float: """N-gram 기반 ROUGE 점수 계산 (계획서 p.24 수식: recall)""" ref_ngrams = ngram_tokenize(reference, n) hyp_ngrams = ngram_tokenize(hypothesis, n) if not ref_ngrams or not hyp_ngrams: return 0.0 ref_counter = Counter(ref_ngrams) hyp_counter = Counter(hyp_ngrams) overlap = 0 for ngram in hyp_counter: overlap += min(hyp_counter[ngram], ref_counter.get(ngram, 0)) recall = overlap / len(ref_ngrams) if len(ref_ngrams) > 0 else 0 return recall