@InProceedings{ruan-EtAl:2026:wmt,
  author    = {Ruan, Lu  and  jia, shaoying  and  ning, fan  and  wang, wei  and  cai, zhengzhen  and  hu, fei  and  hu, heng  and  wang, chenzi},
  title     = {An LLM-Driven Pipeline for Extremely Low-Resource English-to-Indic Translation --- From Vocabulary Injection to Post-Training and Decoding},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2230--2237},
  abstract  = {This submission addresses WMT26 Category 2 for five extremely low-resource English-to-X translation directions: Bodo, Karbi, Kokborok, Nagamese, and Tagin. We develop an LLM-driven pipeline in which large language models are used both to construct bilingual lexical resources and as translation models. First, an English seed dictionary and qwen3-max are used to build word-level bilingual vocabularies, whose top candidates are injected into translation prompts as a "Helpful Vocabulary" block. On an 80/20 train-held-out split of the Tagin data, vocabulary injection is associated with a 7.09-point BLEU improvement and a 49.35-point TER reduction. Second, we compare LoRA adaptation of Qwen2.5-32B-Instruct with full-parameter fine-tuning of Hunyuan-MT-7B, with optional DPO. Hunyuan performs better in the reported Bodo and Kokborok comparisons, including an 11.87-point ChrF advantage on Bodo. Third, post-hoc repeated-ngram and length truncation, combined with per-sentence selection of the shorter of two decoding outputs, mitigates generation degeneration and improves Bodo BLEU from 19.27 to 27.18 without retraining. We submit two contrastive systems per direction: a take-shorter system with output truncation for Bodo, and augmented and baseline SFT systems for the other directions. Code and configurations are publicly available.},
  url       = {https://aclanthology.org/2026.wmt-1.162}
}

