@InProceedings{jang-cho-choi:2026:wmt,
  author    = {Jang, Jisoo  and  Cho, Haewon  and  Choi, Seungtaek},
  title     = {HUFS-DILAB at WMT 2026 Automated Translation Quality Evaluation Task 1: Language-Pair Prompt Routing for Error Span Annotation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1665--1673},
  abstract  = {This paper describes HUFS-DILAB's submission to Task 1, segment-level error detection and span annotation, of the WMT26 Automated Translation Quality Evaluation task. We investigate how far an off-the-shelf LLM can be taken as a judge without task-specific training: Gemma-4-31B-it is prompted to annotate error spans, omissions, and severities as constrained JSON given the source, the reference, and the translation, with each language pair routed to a prompt variant specialized for that pair. This design is motivated by the observation that, even with the same prompt, the judge exhibits different severity and error-density distributions across language pairs. These pair-specific patterns recur in at least 24 of the 26 MT systems, suggesting systematic tendencies of the judge rather than effects specific to individual MT systems. To address these tendencies, we calibrate each prompt variant against human annotations where available. We also correct a rule-based wrong-language filter whose errors on closely related languages were silently overriding the judge's annotations. Our submission achieves a score of 0.5854 on the official metric, averaged over the eight prioritized language pairs.},
  url       = {https://aclanthology.org/2026.wmt-1.99}
}

