@InProceedings{kocmi-EtAl:2026:wmt1,
  author    = {Kocmi, Tom  and  Artemova, Ekaterina  and  Avramidis, Eleftherios  and  Bawden, Rachel  and  Bojar, Ondřej  and  Dukanov, Sergey  and  Dvorkovich, Anton  and  Fishel, Mark  and  Freitag, Markus  and  Frontull, Samuel  and  Gowda, Thamme  and  Grundkiewicz, Roman  and  Haddow, Barry  and  Kharevich, Stan  and  Koehn, Philipp  and  Li, Zheng  and  Maillard, Jean  and  Monz, Christof  and  Murauski, Alexander  and  Murray, Kenton  and  Nagata, Masaaki  and  Perrella, Stefano  and  Popel, Martin  and  Popović, Maja  and  Proietti, Lorenzo  and  Rajaee, Sara  and  Riley, Parker  and  Shmatova, Mariya  and  Steingrímsson, Steinþór  and  Yankovskaya, Lisa  and  Zouhar, Vilem},
  title     = {Findings of the WMT26 General Machine Translation Shared Task: Contrastive Dynamic Human Evaluation at Scale},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {880--930},
  abstract  = {This paper presents the results of the General Machine Translation Task organized under the 2026 Conference on Machine Translation (WMT). Participants build systems for any of the 23 language pairs spanning four to five domains. This year we brought major changes to the human evaluation: (1) new annotation platform Pearmut for better reproducibility, (2) new human protocol "contrastive Error Span Annotation" for higher quality and side-by-side evaluation, and (3) Dynamic human evaluation that allows assessing all submitted models while evaluating higher-performing models more frequently. Beyond human evaluation, we (4) extended difficulty sampling with a human-driven stage, (5) added two new domains, (6) made the test sets fully document-level without requiring segment-level alignment and relying on HTML or JSON structure, (7) prepared some human references by post-editing open-weight model outputs, and (8) introduced contextual instructions governing formality and structural style. We evaluated 46 systems in total: 30 submitted by participants and 16 consisting of translations from large language models (LLMs) and industry translation providers.},
  url       = {https://aclanthology.org/2026.wmt-1.48}
}

