@InProceedings{otmar-bojar:2026:wmt,
  author    = {Otmar, Antonín  and  Bojar, Ondřej},
  title     = {How many raters do we need to recognize a bad translation via perplexity?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {537--547},
  abstract  = {We evaluated candidate translation pairs with various kinds of damage using perplexity assigned to them by LLMs to see how it holds up as a fluency and adequacy metric. We also explored how the discriminating power of perplexity can be improved by ensembling.},
  url       = {https://aclanthology.org/2026.wmt-1.29}
}

