@InProceedings{zundanovi-EtAl:2026:wmt,
  author    = {Zundanović, Dragana  and  Leventić, Hrvoje  and  Božić Lenard, Dragana  and  Romić, Krešimir},
  title     = {The Croatian Dataset Seed Submission to the WMT26 Open Language Data Initiative Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2570--2582},
  abstract  = {We present a new English–Croatian parallel corpus for the WMT26 Open Language Data Initiative (OLDI) Seed shared task, consisting of 6,193 sentence pairs. To ensure high quality, the resource was produced via machine translation followed by a two-stage verification process (post-editing and independent revision) conducted by professional translators. We validate the corpus by fine-tuning NLLB-200, TranslateGemma, and Gemma-4-31B. The dataset improves every dedicated translation model in both directions and the strongest open LLMs in the Croatian→English direction. Through a training-data ablation, we demonstrate that the benefit of human verification increases with model capacity and is visible only to neural metrics, not to surfaceoverlap metrics such as chrF++. Finally, we analyze the quality ceiling for the strongest models, identifying a trade-off between naturalness and surface overlap in the English→Croatian direction.},
  url       = {https://aclanthology.org/2026.wmt-1.200}
}

