@InProceedings{kanojia-EtAl:2026:wmt1,
  author    = {Kanojia, Diptesh  and  Sindhujan, Archchana  and  Deoghare, Sourabh Dattatray  and  Sokova, Daria  and  Qian, Shenbin  and  Koushik, Girish  and  Ranasinghe, Tharindu  and  Orasan, Constantin  and  Zerva, Chrysoula  and  Rei, Ricardo  and  Blain, Frederic  and  Martins, André  and  Turchi, Marco  and  Negri, Matteo  and  Kunchukuttan, Anoop  and  Khapra, Mitesh M.  and  Bhattacharyya, Pushpak},
  title     = {IndicQE-APE: A Consolidated Benchmark for Quality Estimation and Automatic Post-Editing over Indic Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {325--351},
  abstract  = {Indic quality estimation (QE) and automatic post-editing (APE) data is spread across separate releases, so no single resource supports training and evaluation across tasks and language pairs on one footing. We consolidate the WMT 2020-2024 shared-task lineage with an extended English-Malayalam resource into IndicQE-APE: 126,754 instances over nine directional pairs, with up to four label types aligned on the same segment, a direct assessment, a human post-edit, word-level tags and an error explanation, and a test set stratified over four difficulty axes. We benchmark six prompted LLMs and three COMET metrics on segment-level QE, and three systems on APE. Two of the axes are defined partly on direct assessment and select a compressed slice of it. Segments whose segment-level and token-level signals disagree are ranked below equally scored segments of the same language. Four-shot prompting costs every model at or below 3.4B both correlation and output-format compliance. Unedited MT beats every APE system we run on three of the four pairs. The benchmark and code are released.},
  url       = {https://aclanthology.org/2026.wmt-1.19}
}

