@InProceedings{riosgaona:2026:wmt,
  author    = {Rios Gaona, Miguel Angel},
  title     = {UniVie-HAITrans at WMT26 Terminology Translation Task: Terminology-Guided Data Filtering and Few-Shot Fine-Tuning for Lightweight LLMs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1880--1887},
  abstract  = {Specialised domain translation using Large Language Models remains challenging due to terminology constraints, scarce in-domain parallel data, and the tendency of fine-tuning to degrade in-context learning performance. We present our submission for the English–Polish and Spanish–Basque translation in the Medical and Engineering \& Technology domains. Our method follows three stages: i) We extract in-domain terms from a terminology database and prompt a multilingual LLM to generate contextual sentences for each term. ii) We build an efficient search index over large-scale, unconstrained web-crawled and medical parallel corpora. Using the synthetic sentences as queries, we retrieve and filter the most semantically relevant pairs to build an in-domain fine-tuning dataset. iii) We structure our training instances with a mixture of 0-shot and semantically retrieved 5-shot prompts, and fine-tune lightweight multilingual LLMs (Gemma-3-4B, and Tiny-Aya-Global) via QLoRA. Fine-tuned Gemma-3 achieves the best results on held-out data across all metrics on English–Polish (Medical) and on Spanish–Basque (Engineering \& Technology), showing that few-shot fine-tuning enhances specialised domain translation quality and preserves in-context adaptation.},
  url       = {https://aclanthology.org/2026.wmt-1.122}
}

