以下代码实现了 BLEU 评测指标的计算,[1] 和 [2] 处应分别填入:
import math from collections import Counter
def compute_bleu(reference, candidate, max_n=4):
># 计算 BLEU 分数(简化版)
reference: 参考文本(分词后的列表),如 ["the", "cat", "sat", "on", "the", "mat"]
candidate: 候选文本(分词后的列表),如 ["the", "cat", "on", "the", "mat"]
max_n: 最大 n-gram 阶数
return:
BLEU 分数 (float)
precisions = []
for n in range(1, max_n + 1): # 生成 n-gram ref_ngrams = [tuple(reference[i:i+n]) for i in range(len(reference) - n + 1)] cand_ngrams = [tuple(candidate[i:i+n]) for i in range(len(candidate) - n + 1)]
>ref_counts = Counter(ref_ngrams)
cand_counts = Counter(cand_ngrams)
clipped_count = 0 for ngram, count in cand_counts.items():
>total_count = len(cand_ngrams)
if total_count == 0:
precisions.append(0)
[1] clipped_count += min(count, ref_counts.get(ngram, 0))
[2] bp = 0.0 if len(candidate) == 0 else (math.exp(1 - len(reference) / len(candidate))) if len(candidate) < len(reference) else 1.0
[1] clipped_count += max(count, ref_counts.get(ngram, 0))
[2] bp = 0.0 if len(candidate) == 0 else (math.exp(1 - len(reference) / len(candidate))) if len(candidate) < len(reference) else 1.0
[1] clipped_count += min(count, ref_counts.get(ngram, 0))
[2] bp = len(candidate) / len(reference)
[1] clipped_count += count
[2] bp = math.exp(1 - len(reference) / len(candidate)) if len(candidate) < len(reference) else 1.0