@InProceedings{aditya-ekbal:2026:wmt,
  author    = {Aditya, Aditya  and  Ekbal, Asif},
  title     = {ADI-IITP: Reward-Guided Preference Optimization for Low-Resource English-Assamese Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2085--2092},
  abstract  = {This paper describes our submission to the WMT 2026 shared task on Low-Resource Indic Language Translation for the English-Assamese pair. Our systems build on the compact, distilled 200M-parameter IndicTrans2 model, adapted with parameter-efficient Low-Rank Adaptation (LoRA). Beyond standard supervised fine-tuning (SFT), we built a preference-optimization pipeline that generates candidate translations, scores them with a composite reward combining a GEMBA-style LLM-as-judge score, a reference-free CometKiwi quality-estimation signal, and an xCOMET score, and then trains with Direct Preference Optimization (DPO) on the resulting preference pairs. We used this pipeline to guide our system choices: for Assamese-to-English the SFT-only model scored best on our development set, so we submitted it as primary and the SFT-then-DPO model as a contrastive system; for English-to-Assamese we went with SFT-then-DPO as primary. On the official WMT 2026 blind test set, the Assamese-to-English primary system scored 24.08 BLEU (25.11 for the contrastive system), and the English-to-Assamese primary system scored 15.57 BLEU. We also report METEOR, TER, chrF++, BERTScore, and COMET.},
  url       = {https://aclanthology.org/2026.wmt-1.146}
}

