@InProceedings{gowda:2026:wmt,
  author    = {Gowda, Thamme},
  title     = {Quantize, Qualify, Rerank: A Recipe for Compressing LLMs Without Losing Quality},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1998--2005},
  abstract  = {Compressing large language models lowers serving cost, but a single automatic metric may miss quality loss. For the WMT26 Model Compression constrained track, we compress `google/gemma-3-12b-it` (24.4 GB in bf16) for three translation directions on one H100. Paired bootstrap analysis exposes metric disagreement on 4-bit quantization; guided by WMT24 human meta-evaluation, we use MetricX24-XXL as primary and CometKiwi-XXL as a cross-check. Removing the unused vision stack and non-task vocabulary and using int8 weights or fp8 activations causes no measurable loss, whereas int4 leaves a small residual. QLoRA healing overfits its calibration domain. A cheap Cometoid best-of-4 reranker at `T=0.3` improves both the 8-bit and 4-bit systems under held-out CometKiwi-XXL and reference-based MetricX24-XXL. Both surpass greedy bf16 on CometKiwi-XXL; the 8-bit system also surpasses it on MetricX24-XXL, while the 4-bit system nearly matches it.},
  url       = {https://aclanthology.org/2026.wmt-1.136}
}

