@InProceedings{bajwa:2026:wmt,
  author    = {Bajwa, Angad Ripudaman Singh},
  title     = {COMET Sensitivity-Guided Mixed Precision Quantization for the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1980--1985},
  abstract  = {The WMT26 Model Compression shared task asks participants to reduce the footprint of a general-purpose large language model for machine translation (MT) deployment while preserving translation quality, under a fixed evaluation budget of model size, VRAM usage, and inference speed. We report a systematic comparison of post-training compression strategies applied to the constrained track base model, google/gemma-3-12b-it, across the three required language directions (Czech–German, English–Chinese (Simplified), English–Arabic (Egyptian)). We evaluate an uncompressed BF16 baseline and a vLLM-served engine-level control against six compression strategies: two bitsandbytes baselines (8-bit and 4-bit), two calibrated weight-only quantization methods (AWQ and GPTQ, both W4A16), vLLM's native dynamic FP8 weight quantization, and a novel COMET-sensitivity-guided mixed-precision scheme that assigns each of the 48 decoder layers an independent BF16/INT8/INT4 tier via a size-budget knapsack. Quality is measured with chrF and three COMET variants. Speed is full-process wall-clock time (including model load) on a held-out blind test set, and size is the on-disk safetensors footprint. Our mixed-precision submission achieves the best quality-size-speed trade-off in our pool: it is the fastest system measured (1.76 sentences/sec, batch 16), within 0.4 chrF and a few COMET points of the uncompressed baseline across all three language pairs, at roughly 40\% of the baseline's on-disk size (9.15GB vs 23GB).},
  url       = {https://aclanthology.org/2026.wmt-1.133}
}

