@InProceedings{jon-EtAl:2026:wmt,
  author    = {Jon, Josef  and  Bondok, Rawan  and  Hrabal, Miroslav  and  Bojar, Ondřej},
  title     = {CUNI-AR Team at WMT26 General Translation Task for Egyptian Arabic},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1278--1290},
  abstract  = {We describe the Charles University CUNI-AR submission to the WMT26 General Translation shared task for English into Egyptian Arabic. Our pipeline is built around GEMBA (LLM-as-judge quality) estimator that scores translations independently for accuracy and dialect authenticity. We use it throughout our pipeline: to filter training data, to construct preference pairs for DPO, to select best checkpoints, and to pick the best hypothesis per segment at inference time. Our primary translation model is Gemma-4-12B-it finetuned via QLoRA, we also finetune Jais-2-8B-Chat and Aya-expanse-8B. Gemma-4-12B-it provides a high-quality Egyptian Arabic translation, so SFT yields only marginal quality gains in automatic metrics; DPO is able to improve the metrics it optimizes for, but the real gain in translation quality is to be assessed by human evaluation. The final submission selects the best-scoring translation per segment from 31 candidate checkpoints, with the unmodified base model contributing 31\% of the output.},
  url       = {https://aclanthology.org/2026.wmt-1.68}
}

