@InProceedings{hara-EtAl:2026:wmt,
  author    = {Hara, Nagito  and  Nowakowski, Karol  and  Ptaszynski, Michal  and  Overacker, Lloyd Nicholas  and  Toyoura, Masahiro},
  title     = {Low-resource machine translation using a general-purpose LLM informed by a dedicated NMT model and lexical resources: A case study on Ainu–Japanese translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {227--245},
  abstract  = {Previous methods for post-editing the output of a dedicated Neural Machine Translation (NMT) system using a general-purpose Large Language Model (LLM) rely on correcting a single hypothesis generated by the NMT model's decoder. However, in low-resource scenarios -- such as Ainu-to-Japanese translation -- the low accuracy of initial NMT outputs often leads to error propagation, where LLMs fail to override systemic NMT hallucinations or syntactic errors. In this paper, we propose a novel machine translation framework that incorporates information from an NMT model into an LLM, while bypassing the limitations of single-sentence correction. Instead of using raw NMT text, our method extracts token-level probability distributions from the NMT decoder and combines them with entries obtained by looking up a bilingual dictionary, having the LLM construct the final translation from these two sources of information. By grounding the LLM in both the NMT's internal confidence signals and external linguistic knowledge, our approach effectively mitigates the bias toward poor-quality NMT outputs. Empirical evaluations on Ainu-Japanese parallel data using chrF, semantic similarity, and perplexity demonstrate that our method outperforms six baseline approaches in specific domains in the Ainu-to-Japanese direction, that is, when translating into the high-resource language. In the opposite direction, the method yields no improvement, as its effectiveness is bounded by the LLM's ability to generate the target language. Furthermore, perplexity analysis confirms that LLM-driven translation substantially enhances linguistic fluency compared to traditional NMT outputs.},
  url       = {https://aclanthology.org/2026.wmt-1.13}
}

