@InProceedings{mishra-sharma-khetarpaul:2026:wmt,
  author    = {Mishra, Animesh  and  Sharma, Krishang  and  Khetarpaul, Sonia},
  title     = {Translation Metrics Cannot Judge What Their Tokeniser Deletes: LIGATUR and AEGIS Submissions to the WMT26 Shared Task on Automated Translation Quality Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1707--1717},
  abstract  = {The WMT26 shared task on automated translation quality evaluation asks participants both to build metrics (Task 2) and to submit challenge sets that probe them (Task 4). We describe one submission to each, and a single finding that connects them. Modern machine translation (MT) metrics cannot detect a class of invisible Unicode corruption, and the reason is not that they judge it wrongly: their tokenisers delete the evidence first. Four such corruptions — a non-breaking space, a narrow non-breaking space, an ideographic space, and a leading byte-order mark — normalise to token sequences identical to the clean text. This holds under both XLM-R and mT5, the encoders behind COMET-22 and MetricX-24. The metric therefore receives the same input and returns the same score, and no amount of training can separate the pair. Perturbations that instead fragment tokenisation are caught reliably. LIGATUR (Task 4) is the 171-item contrastive challenge set that isolates this effect for English-German (en-de) and English-Hindi (en-hi); a blind re-evaluation with held-out labels also overturns our own pilot claim that judges ignore Devanagari conjunct breaks. AEGIS (Task 2) acts on the diagnosis, combining a reference-based ensemble with deterministic hygiene penalties for exactly what the encoders cannot see, and scores all 26,885 test items. An ablation on 34,174 WMT24 judgements shows the two-family ensemble is our strongest component and that the reference-free branch we added as insurance costs accuracy where references are good, which we report as a calibration error.},
  url       = {https://aclanthology.org/2026.wmt-1.103}
}

