@InProceedings{kronis-EtAl:2026:wmt,
  author    = {Kronis, Martins  and  Bergmanis, Toms  and  Pretkalniņš, Ingus Jānis  and  Pinnis, Marcis},
  title     = {WMT26 Model Compression Task: Quantising and Distilling TildeOpen},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2006--2015},
  abstract  = {This paper describes Tilde's submission to the unconstrained track of the WMT~2026 Model Compression task in the Czech-German direction. Our baseline is TildeOpen-15B-64k, a model distilled from the TildeOpen-30B-64k European foundation model. We adapt it for CS-DE translation, using supervised fine-tuning followed by GRPO-based reinforcement learning. From this baseline, we explore two axes of compression: post-training quantisation and knowledge distillation. For quantisation, we use LLM Compressor to produce a GPTQ-calibrated model with 4-bit NVFP4 weights (10.1\,GB, $\times$3 smaller than the 30.3\,GB footprint of the 15B baseline) and a data-free RTN model with FP8 weights and activations (16.3\,GB). In parallel, we prune and distil the base 15B model into an 8B student and fine-tune it with the same supervised fine-tuning recipe. Applying the two quantisation schemes to the 8B model yields our smallest models at 8.9\,GB and 5.7\,GB. The latter is over $\times$5 more compact and $\times$1.8 faster than the 15B baseline. Evaluation on four test sets across a variety of metrics shows that quantisation and distillation have minimal impact on translation quality. The NVFP4-quantised 15B model offers the best quality/size/speed trade-off and is our primary submission for this competition. All six models are publicly released.},
  url       = {https://aclanthology.org/2026.wmt-1.137}
}

