@InProceedings{fakhrutdinov:2026:wmt,
  author    = {Fakhrutdinov, Nail},
  title     = {Pare4Bit: Quantization, Depth Pruning, and Model Selection for the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1993--1997},
  abstract  = {In the constrained track we compress google/gemma-3-12b-it with a data-light pipeline—vision-tower removal followed by GPTQ W4A16 quantization - reaching 43.2 mean chrF at 7.1 GB, a 3.4× size reduction for a 0.3 chrF drop. We further explore depth pruning with LoRA knowledge-distillation healing, which recovers a collapsed pruned model to near-parity and adds an engine-agnostic speedup. In the unconstrained track we treat model selection as a first-class compression decision: the newer Gemma-4-12B, deployed via quantization-aware-trained INT4, outperforms the constrained Gemma-3 baseline by +2.5 mean chrF.},
  url       = {https://aclanthology.org/2026.wmt-1.135}
}

