@InProceedings{huidrom-EtAl:2026:wmt1,
  author    = {Huidrom, Rudali  and  Kumar, Vikas  and  Pangsatabam, Hoomexsun  and  Das, Pinaki  and  Singh, Kshetrimayum Boynao  and  Konjengbam, Anand},
  title     = {Which Metric for Which Script? A Quantified Meta-Evaluation of Automatic Metrics and LLM Judges for Low-Resource Indic Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {288--310},
  abstract  = {Machine translation shared tasks publish several automatic metrics and rank systems on one of them, assuming the choice is inconsequential. We test this assumption on the WMT 2025 and WMT 2026 Low-Resource Indic Language Translation tasks, where pre-trained multilingual encoders are unavailable for several target scripts and languages. As no human gold standard evaluation exists at scale, we adapt the Quantified Reproducibility Assessment (QRA) framework across 20 WMT26 and 14 WMT25 directions, to measure agreement between six automatic metrics and two LLM judges as independent measuring instruments. We show that QRA's Type I measure fails under this repurposing when instruments sit on different scales, and suggest a rank-based repair. Using a script-controlled natural experiment (Manipuri evaluated in both Bengali and Meitei Mayek) and a null-output validity test, we observe that neural metrics inflate scores for text in scripts outside the encoder's coverage and compress it into a narrow discriminative band. Switching to Meitei Mayek raises BERTScore and COMET by 21.6 and 23.3 points, respectively, though language and content are unchanged, and no metric preserves system ranking; on degenerate output the two retain 0.81 and 0.61 points on English-to-Indic directions. The effect reverses in the opposite direction. Metric choice should follow script coverage, not language identity. Lacking human judgments, our claims concern instrument behaviour, not human preference.},
  url       = {https://aclanthology.org/2026.wmt-1.17}
}

