@InProceedings{schmidtova-EtAl:2026:wmt,
  author    = {Schmidtova, Patricia  and  Artemova, Ekaterina  and  Aycock, Seth  and  Bafna, Niyati  and  Banga, Shobhit  and  Kaur, Manmeet  and  Kocmi, Tom  and  Koehn, Philipp  and  Liu, Danni  and  Luu, Nam  and  Papi, Sara  and  Savoldi, Beatrice  and  Shmatova, Mariya  and  Sidh, Hanuman  and  Zerminova, Evfrosiniya  and  Zouhar, Vilem  and  Züfle, Maike  and  Chen, Pinzhen},
  title     = {Findings of the WMT26 Multilingual Instruction Shared Task: Small Models Are Not Yet Polyglots},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1031--1052},
  abstract  = {We present the findings of the WMT26 Multilingual Instruction Shared Task (MIST), evaluating small (<=10B parameter) open-weight language models across 24 languages on three sub-tasks: context-based QA, cross-lingual summarization, and open-ended generation. Submissions from eight teams explore fine-tuning, knowledge distillation, and prompt engineering. Our findings demonstrate that small models are not yet polyglots: while leading systems perform well on high-resource languages, performance drops sharply on low-resource varieties, and comprehension degrades significantly when transferring across non-English language pairs. In open-ended generation, human evaluation and automatic rule-based verification exhibit strong overall rank correlation (0.79) but diverge locally at the top: human evaluators favor distilled pipelines with superior fluency and cross-lingual equity, whereas automatic verifiers reward concise prompt-engineered systems. Furthermore, multi-stage pipelines suffer from prompt-language inertia on cross-lingual instructions, and summarization exhibits severe English leakage unless penalized by language gating. Finally, while baseline fluency is largely achieved by modern small models, multi-constraint instruction following and unanswerable question refusal remain the primary performance bottlenecks. We release all test sets, system outputs, human judgments, and evaluation code under a permissive license.},
  url       = {https://aclanthology.org/2026.wmt-1.50}
}

