@InProceedings{balaga-EtAl:2026:wmt,
  author    = {Balaga, Havish  and  Racherla, Anish  and  Kiran, Kolupoti Navadeep  and  Yadav, Saumitra  and  Shrivastava, Manish},
  title     = {Importance-Guided Structural Pruning of Aya Expanse 8B and Gemma-3-12B for English–Chinese Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1986--1992},
  abstract  = {Large language models have significantly improved machine translation quality but remain expensive to deploy due to high computational and memory requirements. This paper describes our submissions to the WMT 2026 Model Compression Shared Task, targeting English–Chinese translation through structural compression of two multilingual models: Aya Expanse 8B and Gemma-3-12B. Both pipelines share a common 6-stage data curation process that filters 25 million raw WMT English–Chinese sentence pairs into a quality and diverse training corpus for fine-tuning. For Aya Expanse 8B, we use COMET-QE to identify and prune four layers with the least impact on translation quality, followed by QLoRA fine-tuning and INT8 inference. For Gemma-3-12B, we apply Fisher Information-based importance scoring to guide a two-step compression, pruning 8 transformer layers followed by feed-forward neuron pruning, recovered through QLoRA fine-tuning and INT4 inference. Results on the FLORES-200 benchmark show both pipelines achieve substantial reductions in GPU memory usage and gains in inference throughput while maintaining competitive translation quality. Our findings demonstrate that importance-guided structural pruning, combined with parameter-efficient fine-tuning and quantization, can achieve substantial model size reduction while maintaining comparable English–Chinese translation quality.},
  url       = {https://aclanthology.org/2026.wmt-1.134}
}

