@InProceedings{giraud-gargett:2026:wmt1,
  author    = {Giraud, Jurgi  and  Gargett, Andrew},
  title     = {Compact Models for Machine Translation in the Bioinformatics Domain: Synthetic Data, Domain Adaptation, and Expert Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {175--189},
  abstract  = {Bioinformatics is almost invisible in Machine Translation (MT) research, despite being a hard case: dense specialised terminology, a high density of Multiword Expressions (MWEs), and an acute scarcity of parallel data. We address this gap for English-French. We compile a bioinformatics parallel corpus (14,467 sentence pairs) and expand it to 236k pairs through four augmentation strategies, each profiled for lexical diversity before use, then fine-tune five compact models (223M-8B parameters) under an incremental data ablation. We evaluate with automatic metrics and with a three-rater MQM evaluation involving professional life-sciences translators and using an error typology extended with a dedicated MWE category. Domain adaptation improves every system on every metric, with fine-tuned models surpassing much larger state-of-the-art commercial models on our domain test set. Expert judgements reveal a more structured picture with Accuracy errors falling by up to 68.8\% and MWE errors by up to 51.1\%, while also highlighting an accuracy-fluency trade-off that the automatic metrics conceal. Notably, the least lexically diverse augmentation sources contribute gains comparable to the most diverse ones, which we advance as one candidate account of the trade-off. We release all corpora, synthetic data, and adapted models.},
  url       = {https://aclanthology.org/2026.wmt-1.10}
}

