@InProceedings{lavie-EtAl:2026:wmt,
  author    = {Lavie, Alon  and  Hanneman, Greg  and  Perrella, Stefano  and  Ding, Shuoyang  and  Avramidis, Eleftherios  and  Proietti, Lorenzo  and  Lo, Chi-kiu  and  Shurtz, Ammon  and  Zerva, Chrysoula  and  Sindhujan, Archchana  and  Zouhar, Vilem  and  Kanojia, Diptesh  and  Blain, Frederic  and  Thompson, Brian  and  Filandrianos, Giorgos  and  Menis Mastromichalakis, Orfeas  and  Kocmi, Tom  and  Gupta, Pranav},
  title     = {Findings of the WMT26 Shared Task on Automated Translation Quality Evaluation Systems: Compact Open Models are Competitive Quality Evaluators},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {931--1030},
  abstract  = {We present the findings of the WMT26 Shared Task on Automated Translation Quality Evaluation Systems, continuing last year's unification of the earlier separate WMT Metrics and Quality Estimation shared tasks. This year we evaluated three complementary views of segment-level translation quality on a common test set: fine-grained error-span detection, continuous quality-score prediction, and a new task on identifying error-free translations. Submissions to upstream tasks were also converted automatically to downstream predictions. The evaluation covered 21 translation directions using human cESA judgments from the WMT26 General Machine Translation task, with optional reference translations generated as either native, post-edited, or pseudo-references, depending on the translation direction. Official evaluation data was complemented by five submitted challenge sets. Across all three primary tasks, unsupervised LLM-as-a-judge approaches outperformed all traditional and supervised metrics, while open-weight models such as Gemma 4 were found to be largely competitive with proprietary frontier models. Reference-free LLM judges are highly competitive, and the benefit of adding a reference largely depends on its provenance and on the evaluator model family.},
  url       = {https://aclanthology.org/2026.wmt-1.49}
}

