@InProceedings{koo:2026:wmt,
  author    = {Koo, Zi Chen},
  title     = {From Error Detection to Credit Assignment: Do xCOMET Error Spans Provide Useful Policy Credit for Chinese–Malay Translation?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {375--386},
  abstract  = {Sequence-level rewards assign the same group-relative advantage to every generated token, even when only a short translation span is erroneous. We ask whether xCOMET error spans provide useful token-level policy credit for Group Relative Policy Optimization (GRPO) in Chinese–Malay translation. Signed Lexical-Residual GRPO (SLR) adds a bounded, zero-mean local residual to the sequence advantage. Across three matched seeds, SLR improves reward-aligned COMET and MetricX-24, a learned metric not used in training, while chrF++ and BLEU differences remain unresolved. A blinded 300-item evaluation by the bilingual author does not resolve an adequacy or fluency preference and finds a higher omission rate under SLR (9.33\% to 16.67\%). A repeated span audit finds high sentence-level error detection but weak target-word localization; fewer than half of consistently marked error words receive negative residual credit. In this setting, error detection, localization, and policy-credit actionability are distinct: explainable metric spans change optimization behavior without automatically providing human-aligned token supervision.},
  url       = {https://aclanthology.org/2026.wmt-1.21}
}

