@InProceedings{rittikar-ramanna:2026:wmt,
  author    = {Rittikar, Sujay Uday  and  Ramanna, Sheela},
  title     = {Winterpeg: Do Neural Cellular Automata Help Where Pretraining Ends?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2220--2229},
  abstract  = {Machine translation for the low-resource languages of North-East India faces an uneven landscape: some are supported by large multilingual models, while others are absent from every pretrained system and usable subword vocabulary. For the covered pairs, fine-tuning a pretrained language model is a strong baseline, thus, our fine-tuned No Language Left Behind model (NLLB-200) ranks first in the WMT-2026 Indic MT shared task on English to Manipuri (Bengali-Assamese) at 11.95 BLEU and 44.67 chrF++. This paper addresses the remaining languages, for which no pretrained coverage exists. We train a vocabulary-free, UTF-8 byte-level, decoder-only language model from scratch and adapt it into a prompted translator that represents any script without tokenizer engineering. Operating at the byte level removes the vocabulary problem but shifts the cost onto the depth, as the model must compose byte sequences into the units, which a tokenizer would otherwise supply. We therefore introduce a causal Neural Cellular Automaton (NCA) as a front end, with a single local update rule iterated in place providing the required depth at the cost of a single set of shared weights. A front-end ablation at matched parameters attributes the observed gains to the iterated rule rather than to the addition of layers, while an equal-depth stack of independently parameterised layers recovers almost none of the improvement. Our causal NCA model places second on English to Mizo among primary submissions, degrades on Khasi, and reaches its limit on Meitei-Mayek, delineating a clear resource-and-script boundary from scratch byte-level modelling. Our code is available at https://github.com/sujayrittikar/wmt\_2026\_indic\_task.},
  url       = {https://aclanthology.org/2026.wmt-1.161}
}

