@InProceedings{gowda-EtAl:2026:wmt,
  author    = {Gowda, Thamme  and  Gaido, Marco  and  Grundkiewicz, Roman  and  Negri, Matteo},
  title     = {Findings of the WMT 2026 Shared Task on Model Compression: No Free Lunch at Extreme Compression},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1087--1097},
  abstract  = {We present the second edition of the WMT 2026 Model Compression (WMT26MC) shared task, which studies how large language models can be made efficient for machine translation under model-size and latency constraints. WMT26MC offers a constrained track based on Gemma~3~12B and an unconstrained track for compressing models below 20B. Both tracks cover cs-de, en-zh, and en-ar. Participants submit complete runnable systems, which we run in a standardized environment on single H100 GPU. Evaluation jointly considers translation quality, model size on disk, and decoding speed, analyzing the resulting quality--size and quality--speed trade-offs. We received 41 submissions from 13 teams. The proposed systems span weight and activation quantization, structured and expert pruning, low-rank factorization, knowledge distillation, vocabulary and vision-component removal, mixed precision, and inference-time reranking. Evaluating with reference-free quality-estimation metrics on the blind test sets, we find that post-training quantization is the reliable lever: INT4 and FP8 systems shrink the constrained base to roughly a quarter of its on-disk size and run up to an order of magnitude faster while staying within quality-estimation noise of the uncompressed model, whereas aggressive low-rank and structural pruning collapse quality.},
  url       = {https://aclanthology.org/2026.wmt-1.52}
}

