@InProceedings{ghimire-mahato:2026:wmt,
  author    = {Ghimire, Prajwal  and  Mahato, Aashish},
  title     = {SubVision@WMT26: QLoRA Fine-Tuning of Hy-MT2-1.8B for Chinese to English Video Subtitle Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2591--2606},
  abstract  = {We describe \textbf{SubVision}, our submission to the WMT26 Video Subtitle Translation shared task. The system fine-tunes the 1.8B-parameter multilingual translation model Hy-MT2-1.8B \citep{zheng2026hymt2familyfastefficient} using 4-bit QLoRA \citep{dettmers2023qloraefficientfinetuningquantized} on 60,000 sentence pairs sampled from the TVsub Chinese-to-English subtitle corpus \citep{DBLP:journals/corr/abs-1801-03257}. The training pipeline uses a single instruction style prompt template, LoRA adapters (rank 16, $\alpha=32$) on all attention and MLP projections, and early stopping based on dev set sacreBLEU. We fix the decoding with beam search with beam size 4, repetition penalty 1.15, and no-repeat 3-gram constraints. The fine-tuned model on a 200 pair TVsub test split with multiple references achieves sacreBLEU 37.63, chrF 52.76, and TER 53.93, compared with zero-shot NF4 baselines of sacreBLEU 11.60 for Hy-MT2-1.8B and 15.79 for Hy-MT2-7B on the same test pairs, prompt, and decoding configuration. The same configuration is used for official inference, translating subtitle lines independently while preserving SRT timing. We report our full training and inference configuration, describe a data processing detail in parsing the corpus's multi-document, multi-reference SGM files, and compare our final configuration against nine additional trained configurations that differ in training pair count, sequence-length budget, LoRA rank, and decoding generation length. The paired bootstrap analysis over this comparison shows that given the 200 segment test set size, the nominal ranking of the submitted configuration is not statistically distinguishable from several of the alternatives. We report this analysis along with the point estimates rather than presenting the ranking as an established result. We note that evaluation uses an in-domain TVsub split. The code is available at: \url{https://github.com/praaajg/wmt26-hymt2}.},
  url       = {https://aclanthology.org/2026.wmt-1.202}
}

