@InProceedings{suman-bentivogli:2026:wmt,
  author    = {Suman, Dhairya  and  Bentivogli, Luisa},
  title     = {FBK's Submission to WMT26 Model Compression Task: Improving Inference Speed for Quantized Models},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2068--2073},
  abstract  = {This paper presents FBK's submission to the WMT26 Shared Task on Model Compression. The task requires compressing the Gemma 3 12B model for machine translation on the Czech-German language pair under the task's constrained track. Post-training quantization (PTQ) methods such as GPTQ have become an industry standard for model compression, but in their standard, weight-only form they reduce only the on-disk size of the model; leaving the activations are in full-precision. As part of this submission, we quantize both weights and activations to 8-bits (W8A8), aiming to also improve inference speed in addition to memory savings, while preserving translation quality. We submit two systems: both quantizing activations d using round-to-nearest quantization, one of which additionally applies SmoothQuant's channel-wise smoothing before quantization. Our experiments demonstrate that both the W8A8 systems increase throughput by more than 20\% while only slightly reducing in quality relative to the full precision model.},
  url       = {https://aclanthology.org/2026.wmt-1.144}
}

