@InProceedings{boudraa:2026:wmt,
  author    = {Boudraa, Hossam},
  title     = {MAGADIR at WMT 2026: Cross-System QE Re-ranking and Entity-Aware Selection for Low-Resource Arabic-English Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2287--2304},
  abstract  = {We present MAGADIR's submissions to the WMT 2026 Low-Resource Arabic–Asian Language Translation Shared Task for English–Arabic translation. Our systems combine a small official parallel corpus, parameter-efficient adaptation of open-weight multilingual models, and segment-level, reference-free quality-estimation re-ranking over heterogeneous cross-system candidate pools. For Arabic-to-English translation, the pool includes supervised LoRA-adapted NLLB and Hunyuan-family models alongside zero-shot Hunyuan/HY-MT and Gemma models. For English-to-Arabic, we select among zero-shot Hunyuan/HY-MT, Gemma, Aya, and NLLB candidates. The Arabic-to-English primary system uses MetricX-24-QE together with a Wikidata-derived entity-recall floor that vetoes candidates omitting recognized proper names before quality estimation determines the final output; the English-to-Arabic primary uses COMETKiwi without an entity constraint. On the official Challenge Test, the Arabic-to-English primary achieves 27.45 BLEU, 56.61 chrF, and 81.94 COMET, while the English-to-Arabic primary obtains 14.59 BLEU, 51.48 chrF, and 83.88 COMET. Relative to our own single-model contrastives, QE-guided cross-system selection improves COMET by up to 1.94 points for Arabic-to-English and 0.81 points for English-to-Arabic, although some contrastive systems retain stronger lexical-overlap scores. This exposes a persistent trade-off between neural adequacy objectives and reference-based surface metrics. A post-hoc matched control separates the Arabic-to-English entity floor from the otherwise confounded change in QE utility. The floor accounts for 83\% of the entity-recall gain while changing only 7\% of outputs and imposing effectively no COMETKiwi cost (+0.0002); the switch from COMETKiwi to MetricX rewrites 74\% of outputs and accounts for the full −0.0085 COMETKiwi difference. Our audit also identifies an important limitation: on approximately 65\% of entity-bearing segments, no candidate preserves every recognized entity in canonical form, leaving the veto with no feasible alternative. These findings motivate routine reference-free entity auditing, more diverse candidate pools, expanded gazetteer coverage, and transliteration-aware entity matching for QE-selected translation systems.},
  url       = {https://aclanthology.org/2026.wmt-1.169}
}

