@InProceedings{silchenko:2026:wmt,
  author    = {Silchenko, Maksim},
  title     = {QEbreak at WMT26: A Pre-Registered Audit of Omission and Addition Asymmetry in Machine Translation Quality Metrics},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1752--1763},
  abstract  = {Reference-free quality estimation (QE) metrics see only the source and the hypothesis, so a translation that silently drops source content leaves nothing for them to flag. QEbreak, a pre-registered contrastive challenge set submitted to the WMT26 Automated Translation Quality Evaluation task (Subtask 4), measures this blind spot: 2,886 segments over 10 language directions, built as matched {base, omission, addition} triples, a fact-free verbosity ladder, and numeric-contradiction pairs as a positive control. Every hypothesis, endpoint, and equivalence margin was frozen in git-timestamped commits before any score existed. Scores from 31 systems confirm the deficit and refute our mechanism. Reference-free QE metrics under-penalize omission (family asymmetry A = −0.51; a current QE baseline scores the omission at or above its own base in 19.5 percent of matched contests, strictly above in 14.3), and the edit-direction by reference-availability interaction is significant (β = −0.069, 95\% CI [−0.081, −0.057]). But reference-based neural metrics are asymmetric in the same direction (A = −0.42), outside the pre-registered ±0.30 equivalence margin, so omission under-penalization belongs to learned metrics as a class; reference access softens it but does not remove it. Every returned metric penalizes pure verbosity, and two systems score corrupted numerals above correct ones.},
  url       = {https://aclanthology.org/2026.wmt-1.109}
}

