@InProceedings{ploeger-EtAl:2026:wmt,
  author    = {Ploeger, Esther  and  Bjerva, Johannes  and  Nguyen, Dong  and  Östling, Robert},
  title     = {An Analysis of Source-Side Diversity in Machine Translation Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {561--580},
  abstract  = {Benchmarks are central to assessing progress in machine translation (MT), yet the content of widely used MT test sets is increasingly scrutinized. Concerns about data contamination and benchmark difficulty have gained traction, but the role of source‑side diversity remains largely overlooked. Many popular benchmarks contain overlapping source items and even exact duplicates, raising the question of how such repetition affects evaluation quality. We empirically examine how source-side diversity shapes MT evaluation. First we measure the inter-source diversity of nine popular MT datasets. Next, our downstream analysis, based on WMT24, suggests that benchmarks with low diversity may be less discriminative and may provide artificially narrow confidence intervals. Ultimately, we call for greater attention to inter‑source diversity in general MT benchmark design.},
  url       = {https://aclanthology.org/2026.wmt-1.31}
}

