@InProceedings{servajean-EtAl:2026:wmt,
  author    = {Servajean, Richard  and  Sohrab, Mohammad Golam  and  Kunitomo-Jacquin, Lucie  and  Rikters, Matiss},
  title     = {Leveraging Verbalized Confidence in LLM-as-a-Judge for Automated Translation Quality Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1737--1745},
  abstract  = {This paper presents the AIST AIRC team's contributions to subtasks 1 (segment-level error detection and span annotation), 2 (segment-level quality score prediction) and 3 (detection of error-free segments) of the automated translation quality evaluation systems shared task in WMT 2026. Our approach to address each of the 3 subtasks relies on an LLM-as-a-judge system. Specifically, for subtasks 2, we follow an LLM-as-a-judge approach within the Multidimensional Quality Metrics (MQM) error typology, with the novel addition of prompting the large language model (LLM) to verbalize its confidence in each error type assessment. We then train a Multilayer Perceptron (MLP) Regressor on error severities with and without associated confidence ratings to predict human annotations. First, our results provide further evidence for the relevance of integrating LLMs into automated translation quality evaluation pipelines. Second, this work aims to advance ongoing efforts to make these pipelines uncertainty-aware through the integration of verbalized confidence. Our code is freely available at https://github.com/sshrichard/WMT-Evaluation-Shared-Task-2026.},
  url       = {https://aclanthology.org/2026.wmt-1.107}
}

