@InProceedings{ponce-etchegoyhen:2026:wmt,
  author    = {Ponce, David  and  Etchegoyhen, Thierry},
  title     = {Vicomtech@WMT 2026: Mask-based Pruning for Model Compression},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2045--2057},
  abstract  = {We describe Vicomtech's participation in the constrained track of the WMT 2026 Shared Task on Model Compression. We addressed all three translation directions of the task, namely Czech to German, English to Simplified Chinese, and English to Egyptian Arabic, using Gemma 3 12B IT as our base model. Our approach performs task-aware structured post-training pruning, which learns differentiable masks to prune model components using machine translation calibration data and an objective combining next-token cross-entropy, knowledge distillation, and intermediate representation matching. We systematically evaluated pruning at two target compression ratios (0.25 and 0.50) and explored different structured pruning configurations, combined with supervised fine-tuning to recover performance after pruning. Additionally, we explored language-specific vocabulary pruning and 4-bit AWQ quantization to achieve further reductions in model size and memory requirements. Our experimental results demonstrate that the combination of structured pruning, targeted post-training, vocabulary reduction, and quantization can achieve substantial model compression while maintaining competitive translation quality across all language directions.},
  url       = {https://aclanthology.org/2026.wmt-1.142}
}

