@InProceedings{griesbeck-geierhos:2026:wmt,
  author    = {Griesbeck, Lena  and  Geierhos, Michaela},
  title     = {AGGREE: An Aggregation-Grounded Challenge Set for Segment-Level MT Metric Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1651--1658},
  abstract  = {This paper introduces AGGREE, a WMT26 Metrics shared task challenge set, which is designed for the diagnostic evaluation of automatic machine translation quality evaluation systems. It is motivated by the observation that metric agreement can be inflated by aggregation as metrics may correlate well at a system level, but disagree at an individual segment level. AGGREE therefore focuses on naturally occurring system outputs where automatic metrics disagree locally or diverge from available WMT human judgments. The challenge set contains 350 single-hypothesis examples across six language pairs drawn from WMT23 and WMT24 system outputs. The submitted set was mined using diagnostics for metric disagreement and metric-human mismatch, then filtered using strict automated quality control. This WMT26 analysis serves only to describe the behavior of the metrics in AGGREE and does not provide an assessment of the performance degradation compared to standard test data.},
  url       = {https://aclanthology.org/2026.wmt-1.97}
}

