@InProceedings{han-EtAl:2026:wmt,
  author    = {Han, Lifeng  and  Liang, Jiahui  and  Latusek, Anna  and  El Haff, Karim  and  Haddad Haddad, Amal  and  Höfgen, Josua  and  Evang, Kilian  and  Ma, Min  and  Zhyrko, Maryia},
  title     = {Mind the Gap: Exposing LLM Translation Blind Spots Using the AlphaMWE Multilingual Parallel Corpus},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1540--1561},
  abstract  = {LLMs' performance on machine translation (MT) tasks is often dependent on the data availability in the specific domains and language pairs that they are trained upon. To examine if multiword expressions (MWEs) still present a bottleneck for LLMs regarding language understanding and translation, we report the performance of systems from the WMT 2026 Test Suites shared task, using portions of the publicly available multilingual parallel corpus AlphaMWE as test suites. We received 31 MT systems' outputs covering English to Chinese (zh), Polish (pl), German (de), and Arabic (ar) including Modern Standard Arabic (MSA) and two dialectal ones (Egyptian and Tunisian Arabic). We carried out automatic evaluations using BLEU, ChrF, and BERT-score to select the top 3 systems per language pair, followed up with human evaluations on the selected systems. Our findings show that figurative and MWE-related phenomena remain challenging for contemporary MT systems, automatic metrics sometimes disagree in system ranking, and human evaluation uncovers language-specific errors that remain hidden by aggregate scores. Inter-annotator analysis further reveals challenges in consistently identifying and calibrating linguistically subtle translation errors, highlighting the need for explicit evaluation guidelines and careful annotator calibration.},
  url       = {https://aclanthology.org/2026.wmt-1.88}
}

