@InProceedings{thompson:2026:wmt,
  author    = {Thompson, Isaac},
  title     = {AmanaMT: Translation Direction Predicts Automatic Metric Reliability},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {754--771},
  abstract  = {When machine translation (MT) metrics fail in low-resource settings, the field typically blames data scarcity. We show this diagnosis is often wrong using AmanaMT, a unified benchmark spanning 730,748 segments, 33 languages, three annotation tiers, and 10 metrics: the dominant predictor of metric unreliability is not resource level but domain mismatch and translation direction. Direction is the single largest predictor of between-language variance in metric reliability (η2dir = 0.309, nearly a third of between-language variance); morphological type is a substantially smaller main effect (η2morph = 0.073), though the joint direction×morphology cells account for η2cell = 0.618, capturing direction-specific morphological variation. Ukrainian's near-zero COMET correlation is fully explained by a domain label (other, not news): a genre confound, not a language problem. Russian's weak aggregate dissolves into clean per-year signals once stratified by WMT year, a textbook Simpson's Paradox from pooling heterogeneous system generations. Neural metrics (MetricX-24, xCOMET, COMET) substantially outperform n-gram baselines at every tier; the gap is largest in morphologically complex and non-news settings where surface overlap is an especially poor proxy for adequacy.},
  url       = {https://aclanthology.org/2026.wmt-1.42}
}

