@InProceedings{ding-duan:2026:wmt,
  author    = {Ding, Jie  and  Duan, Xiangyu},
  title     = {FECE: A Challenge Set for Evaluating Fine-Grained Translation Error Correction},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1631--1644},
  abstract  = {Existing machine translation evaluation metrics assess translation quality only at the segment level and therefore cannot assess fine-grained translation quality. For translation error correction systems, this means that the quality of fine-grained corrections cannot be evaluated using existing metrics. To highlight this evaluation challenge, we propose the FECE dataset for evaluating fine-grained correction quality. Experiments on FECE show that existing metrics do not always reflect fine-grained correction performance. To enable the evaluation of fine-grained correction performance, we further propose COMET\_Fine, a fine-grained evaluation metric. COMET\_Fine can effectively assess fine-grained correction quality while also revealing inconsistencies between existing metrics and COMET\_Fine.},
  url       = {https://aclanthology.org/2026.wmt-1.95}
}

