@InProceedings{wang-EtAl:2026:wmt2,
  author    = {Wang, Hao  and  Liu, Yangyang  and  Zhao, Xiaohu  and  Liu, Heng  and  Shang, Zifu  and  Li, Tianhao  and  Gao, Ruize  and  Tang, Jialong  and  Shao, Shiao  and  Wei, Haoran  and  Yang, Baosong  and  Xu, Linlong  and  Wang, Longyue  and  Luo, Weihua},
  title     = {Lumen at WMT2026: Instruction-Specialized Translation via Quality-Aware Training and On-Policy Distillation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1437--1446},
  abstract  = {We present Lumen, our submission to the WMT2026 General Machine Translation Shared Task. We focus on instruction-specialized translation, where systems must produce accurate translations while satisfying fine-grained requirements on style, terminology, and structured output. Lumen follows a four-stage pipeline. First, continued pre-training and supervised fine-tuning jointly strengthen multilingual translation competence and the ability to follow translation-specific instructions, such as terminology, style, and format constraints. Second, we extend the originally offline M\^{}{2}PO preference-optimization framework into an online preference-optimization stage: the current policy samples candidate translations, an LLM judge assigns continuous direct-assessment scores, and the resulting multi-pair preferences are used to update the model. Third, on-policy distillation transfers stronger translation-instruction-following behavior from a larger teacher model to the Qwen3-14B student, improving translation-instruction adherence while retaining a 14B primary translation backbone. Finally, structure-aware inference and postprocessing preserve required formats, while selective repair addresses clear structural or decoding failures before submission. We participate in 23 translation directions. Under an absolute 1--5 LLM-judge rubric that evaluates translation-instruction following and translation quality separately, Lumen reaches an overall mean score of 4.60. Despite using a 14B translation backbone, it achieves performance comparable to leading proprietary systems in both translation quality and translation-instruction following.},
  url       = {https://aclanthology.org/2026.wmt-1.81}
}

