@InProceedings{pakray-EtAl:2026:wmt,
  author    = {Pakray, Partha  and  Pal, Santanu  and  Vetagiri, Advaitha  and  Singh, Kshetrimayum Boynao  and  Dash, Sandeep Kumar  and  Maji, Arnab Kumar  and  Lyngdoh, Saralin A.  and  Laitonjam, Lenin  and  Jamatia, Anupam  and  Das, Ajit  and  Warjri, Sunita  and  Sharma, Uzzal  and  Sambyo, Koj  and  Manna, Riyanka},
  title     = {Findings of WMT 2026 shared task on Low-resource Indic Languages Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1098--1123},
  abstract  = {This paper presents the findings of the Lowresource Indic Languages Translation Shared Task organized at the Eleventh Conference on Machine Translation (WMT 2026). We evaluated machine translation systems across ten English–Indic language pairs spanning three major language families of Northeast India, grouped into moderately resourced and extremely low-resource categories. Built upon the expanded INDICNE-CORP 2.0 dataset, this edition introduced a rigorous, bidirectional multidomain benchmark covering the Healthcare, Political, Travel, Sports, and Entertainment domains to assess real-world generalizability. The task attracted 24 participating teams who deployed diverse methodologies, including parameter-efficient fine-tuning (PEFT) of multilingual foundations, retrieval-augmented generation (RAG) using frontier large language models (LLMs), and hybrid neuro-symbolic decoding constraints. The systems were evaluated using a comprehensive suite of lexical metrics (BLEU, METEOR, TER, ChrF++) and semantic metrics (BERTScore, COMET). Our analysis reveals a persistent performance cliff under extreme data scarcity and a pronounced directional asymmetry, demonstrating that while current architectures readily decode Indic representations into English, they struggle to generate morphologically complex Indic target text. By publicly releasing these datasets, evaluation protocols, and competitive baselines, this shared task establishes a foundation for advancing research in translation technologies for underrepresented and indigenous languages.},
  url       = {https://aclanthology.org/2026.wmt-1.53}
}

