@InProceedings{thavarasa-thevakumar-sukumar:2026:wmt,
  author    = {Thavarasa, Luxshan  and  Thevakumar, Jubeerathan  and  Sukumar, Sivasuthan},
  title     = {Obligatory Slots: Under Reference-Free Evaluation, Dropping a Distinction the Source Never Made Is Free},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {734--753},
  abstract  = {When the target language must mark a distinction the source does not, a translation system must still pick a value. English they came becomes one Tamil verb if those who came are people and another if they are not; English does not say which. We ask what automatic evaluation charges when the choice is wrong, or never made. We release TamilLingBench, an English→Tamil challenge set for three verb-agreement distinctions, checked by a morphological analyser: rationality (திணை tiṇai), gender, and number as a control. We pre-registered a prediction that neural metrics would not notice a single wrong morpheme. It is wrong for the neural reference-based metrics: COMET-22 charges 8.52× the score difference that human raters reliably notice. On the systems' own errors, both reference-free metrics rank the wrong form first more often than not, and a legitimate Tamil form that leaves the distinction out is not charged at all. Under reference-free evaluation, dropping the distinction is free; reference-based metrics do charge for it. xCOMET finds the error's region but not the morpheme. Inside one model, activation patching finds the same asymmetry: a distinction English marks is settled by the middle of the network, one it leaves unmarked only near the output.},
  url       = {https://aclanthology.org/2026.wmt-1.41}
}

