@InProceedings{popov-EtAl:2026:wmt,
  author    = {Popov, Dmitrii  and  Kozlova, Elizaveta  and  Taracheva, Ekaterina  and  Makhotina, Elizaveta  and  Mekhraliev, Artem  and  Baratelia, Miron  and  Enikeeva, Ekaterina  and  Karpachev, Nikolay},
  title     = {Yandex at WMT26: Translation Adaptation at 235B and Hard-Label Distillation to 8B},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1404--1411},
  abstract  = {We describe Yandex and Yandex-8B, our unconstrained and constrained submissions to the WMT26 General Machine Translation task. Our previous WMT25 submission, a 7B translation specialist, shows strong Fluency but comparatively weaker Accuracy. To combine its translation behavior with the capacity of a much larger model without replaying the full 7B training pipeline at 235B scale, we add two forms of supervision to the instruction-tuning data of an in-house 235B pretrained model: direct translation targets generated by the 7B specialist and in-house translation-related tasks. The resulting Yandex model improves both Accuracy and Fluency and reduces the LLM-based error score relative to the same 235B model tuned only on general instructions. Relative to the 7B specialist, it improves Accuracy and reduces the error score, while their Fluency point estimates remain similar. Yandex then provides hard targets for Yandex-8B, which is trained for English-to-Russian, English-to-Belarusian, English-to-Kazakh, and English-to-Armenian using direct and instruction-conditioned translation data. Across four directions, Yandex-8B obtains the highest paragraph-level ChrF++ and the highest Accuracy and Fluency under our LLM-based evaluation among five public baselines.},
  url       = {https://aclanthology.org/2026.wmt-1.77}
}

