@InProceedings{signoroni-rychly:2026:wmt,
  author    = {Signoroni, Edoardo  and  Rychly, Pavel},
  title     = {FIÙR: A Benchmark Dataset for Eastern Lombard Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2546--2559},
  abstract  = {While recent initiatives have expanded machine translation benchmarks to include low-resource and minoritized languages, regional dialect continuums remain severely underrepresented. Current datasets for the Lombard language predominantly feature Western Lombard, with Eastern Lombard varieties almost absent from the digital landscape. In this paper, we introduce, a new benchmark dataset providing an Eastern Lombard translation of the FLORES+ dev and devtest splits. We detail the methodology of translating a largely oral, unstandardized language, including orthographic regularization and the handling of Italian loanwords. Finally, we establish zero-shot baselines by prompting current LLMs and MT models, demonstrating that existing systems struggle with Lombard varieties overall, and exhibit a strong Western Lombard bias, struggling even more to accurately generate Eastern Lombard text.},
  url       = {https://aclanthology.org/2026.wmt-1.198}
}

