@InProceedings{palomino:2026:wmt,
  author    = {Palomino, Alonso},
  title     = {Layer-Aware Native Quantization for Compact Machine Translation: A WMT26 English-to-Chinese Submission},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2032--2036},
  abstract  = {This paper describes a submission to the constrained English-to-Simplified-Chinese track of the WMT26 Model Compression shared task. Starting from Gemma~3 12B, the system applies native BitsAndBytes NF4 with double quantization to the 144 gate, up, and down projections in the model's 48 text-transformer MLP blocks. Attention projections, embeddings, normalization layers, the tied language-model head, and vision modules remain in BF16. The method requires no calibration or fine-tuning data and produces a mixed checkpoint whose quantized weights remain low-bit at inference. On the official 945-segment blind set, the system scores 0.7669 CometKiwi-XXL and 2.7836 MetricX-24-XXL, slightly improving on the organizer global-q4 baseline's 0.7665 and 2.8147, respectively. Its 11.8~GB artifact is 48.4\% of BF16 size, and H100 throughput is essentially tied with global q4 (866.9 versus 867.9 source characters/s). On 332 reference-backed WMT25 segments, its chrF and BLEU gains over global q4 are significant under paired bootstrap resampling. A transferred selected-layer q3+Zstd codec reaches statistically indistinguishable local quality but expands to BF16-like memory after decoding.},
  url       = {https://aclanthology.org/2026.wmt-1.140}
}

