@InProceedings{carraro-EtAl:2026:wmt,
  author    = {Carraro, Fabrício  and  Zevallos, Rodolfo Joel  and  Gonçalves de Souza, Rodrigo  and  Faustino da Silva, Caio Henrique  and  Ortega, John E.},
  title     = {The Constitution Speaks Nheengatu: An Open MT System and Reproducible Corpus Pipeline for the Amazonian Língua Geral},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {47--63},
  abstract  = {Nheengatu (yrl), the Amazonian Língua Geral, has 6-10 thousand speakers and is co-official in São Gabriel da Cachoeira (Brazil), yet remains poorly served by evaluated Portuguese-Nheengatu MT resources. We present the largest parallel corpus assembled for the language (14.5k Portuguese-Nheengatu pairs), anchored by the 2023 official translation of the Brazilian Federal Constitution; the first MT evaluation set built from transcribed spontaneous speech; and, to our knowledge, the first open-weight MT models dedicated to bidirectional Portuguese-Nheengatu translation. Three findings organize the paper. First, translation into Nheengatu on held-out constitutional articles plateaus on development data regardless of training length or model size, and much of the remaining error sits in conventional legal terms that the written sources do not attest consistently; a terminology success rate over a frozen lexicon shows that glossary mechanisms at training time and lexical constraints at decoding each raise term realization, the latter at a cost in repetitive output that chrF++ barely registers. Second, a corpus expansion that adds transcribed conversation raises speech chrF++ by 24-27 points while leaving written domains unchanged, a register gap that written-only benchmarks cannot see. Third, a larger NLLB model helps translation into Portuguese but not into Nheengatu on legal or speech text. For Portuguese-to-Nheengatu translation, our final systems reach 31.7-32.8 chrF++ on article-disjoint constitutional text, 64.1-64.5 on the near-domain set (62.1 after removing items with target-side content overlap), and 45.2-47.5 on speech.},
  url       = {https://aclanthology.org/2026.wmt-1.4}
}

