@InProceedings{rayarios-EtAl:2026:wmt,
  author    = {Raya-Rios, Vania  and  Klimchuk, Aleksandr  and  Cuadrado Avila, Nicolas M.  and  Gollini Navarrete, Ivo  and  Horvath, Samuel  and  Takác, Martin},
  title     = {Tiny Titans at WMT26: Task-Calibrated Quantization and Activation-Aware Low-Rank Compression of Gemma 3 12B},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2037--2044},
  abstract  = {We present the TinyTitans submissions for the WMT26 Model Compression shared task, which focuses on Czech-to-German translation. Starting with the instruction-tuned Gemma 3 12B model, we compare two approaches: 4-bit GPTQ, which uses task-specific calibration examples, and a more aggressive low-rank factorization method that replaces attention and feed-forward projections with truncated Singular Value Decomposition (SVD) factors. On a 300-sentence development set, the calibrated GPTQ retains 99.7\% of the dense model's COMET score while reducing GPU memory usage from 26.6 GiB to 17.6 GiB. The recovered SVD model reduces the number of parameters by 46.6\% and reaches 0.7804 COMET in a separate evaluation using 1,997 sentence pairs. In an external evaluation using 457 WMT25 paragraphs, the GPTQ, SVD, and SVD-plus-quantization submissions received COMET scores of 0.5751, 0.4263, and 0.3354, respectively. These findings suggest that while post-training quantization effectively reduces memory usage, combining low-rank truncation with low-bit quantization may increase approximation errors without clear efficiency gains.},
  url       = {https://aclanthology.org/2026.wmt-1.141}
}

