@InProceedings{papek-schmidtova:2026:wmt,
  author    = {Papáček, Aleš Manuel Manuel  and  Schmidtova, Patricia},
  title     = {CUNI-UFAL at WMT26 Multilingual Instruction Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1956--1965},
  abstract  = {We describe our submission to the WMT26 Multilingual Instruction Shared Task, which evaluates sub-10B-parameter models on multilingual question answering, open-ended generation, and cross-lingual summarization. We compare four open-weight models and vary the system prompt, reasoning mode, and decoding parameters. On our development set, the tested prompt and reasoning changes have a larger effect on LLM-judge scores than the difference between the two finalist models in their best settings. Gemma 4 E4B obtains the highest overall judge score, whereas Qwen 3.5 9B follows hard constraints more reliably on our internal benchmark. We therefore submit Qwen as our primary system and Gemma as a secondary system. We also introduce a 1,283-example diagnostic benchmark for grounded QA, constrained generation, and abstract writing from ACL papers in 24 languages. Using it for QLoRA fine-tuning does not work in our setup, it lowers validation loss but does not improve quality under our LLM judge. In the official automatic evaluation, our Gemma system ranks first overall and our Qwen system second; Qwen wins QA-context, while Gemma wins QA-OEG and places second in summarization.},
  url       = {https://aclanthology.org/2026.wmt-1.130}
}

