@InProceedings{eckhardt-glembek-stewart:2026:wmt,
  author    = {Eckhardt, Alan  and  Glembek, Ondrej  and  Stewart, Craig},
  title     = {Phrase at WMT26 Automated MT Evaluation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1622--1630},
  abstract  = {We describe the Phrase submission to WMT26's shared task on automated MT quality evaluation, covering all three of its tasks: segment-level error detection and span annotation (T1), segment-level quality score prediction (T2), and detection of error-free segments (T3). Treating independent LLMs as a panel of ESA (Error Span Annotation)-style error-span judges, we measure a ceiling: a perfect filter over the panel's pooled spans would raise MPP-F (our span-matching F-score) by roughly two thirds over the strongest single judge. An exhaustive search for how to realize that headroom (diversity, rank-weighted aggregation, a per-judge reliability prior, and boundary refinement) never beats the strongest single judge on MPP-F (T1), because the judges' shared recall is too low for any gold-free span selection to close the gap. The two tasks derived from the same spans behave differently: a static reliability prior wins the segment score (T2), and simple agreement wins the binary error-free decision (T3). Our primary submission, Phrase01, follows this split; a secondary submission, Phrase02, forces in a third judge (UFAL's cat-v4) across all tasks as an explicit ablation, despite evidence it does not help. The lever for span-level detection is annotation coverage through generation, not aggregation; for the derived tasks, it is matching the aggregation mechanism to the metric.},
  url       = {https://aclanthology.org/2026.wmt-1.94}
}

