30 lines
913 B
Python
30 lines
913 B
Python
from __future__ import annotations
|
|
|
|
import re
|
|
from collections import Counter
|
|
|
|
|
|
def ngram_tokenize(text: str, n: int = 2) -> list[tuple[str, ...]]:
|
|
"""공백을 제거한 문자 단위 n-gram 목록"""
|
|
tokens = list(re.sub(r"\s+", "", text))
|
|
return [tuple(tokens[i:i + n]) for i in range(len(tokens) - n + 1)]
|
|
|
|
|
|
def calculate_rouge_n(reference: str, hypothesis: str, n: int = 2) -> float:
|
|
"""N-gram 기반 ROUGE 점수 계산 (계획서 p.24 수식: recall)"""
|
|
ref_ngrams = ngram_tokenize(reference, n)
|
|
hyp_ngrams = ngram_tokenize(hypothesis, n)
|
|
|
|
if not ref_ngrams or not hyp_ngrams:
|
|
return 0.0
|
|
|
|
ref_counter = Counter(ref_ngrams)
|
|
hyp_counter = Counter(hyp_ngrams)
|
|
|
|
overlap = 0
|
|
for ngram in hyp_counter:
|
|
overlap += min(hyp_counter[ngram], ref_counter.get(ngram, 0))
|
|
|
|
recall = overlap / len(ref_ngrams) if len(ref_ngrams) > 0 else 0
|
|
return recall
|