@InProceedings{dukanov-tsyurupa-dvorkovich:2026:wmt,
  author    = {Dukanov, Sergey  and  Tsyurupa, Maria  and  Dvorkovich, Anton},
  title     = {Dubformer at WMT26 General MT: Translating the Video, Not the Transcript},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1238--1241},
  abstract  = {We describe the system Dubformer entered in the WMT26 General MT shared task, an un- constrained system built on our video-dubbing pipeline. It rests on two commitments. A doc- ument is never translated as raw text: it is first decomposed into a structure chosen for its do- main — scenes carrying speaker and audio- visual context for spoken dialogue, markup- masked segments for web documents, value- only segments for localization resources — so that non-linguistic material is protected by construction and everything needed for consis- tency is in front of the model at once. And no translation is taken on trust: every segment is checked by a second model from a different family for hallucination and for drift from the meaning of the source, and a segment it rejects is re-translated with the reason attached rather than filled in from a fallback.},
  url       = {https://aclanthology.org/2026.wmt-1.63}
}

