@InProceedings{sofianopoulos-prokopidis:2026:wmt,
  author    = {Sofianopoulos, Sokratis  and  Prokopidis, Prokopis},
  title     = {The ILSP/ARC submission to the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2058--2067},
  abstract  = {This paper describes our submission to the constrained track of the WMT26 Model Compression task, filed under the team name arc/ilsp. The task requires the compression of google/gemma-3-12b-it for translation in cs→de, en→zh, and en→ar in the Egyptian register. Our systems combine the removal of the vision tower, the pruning of the vocabulary by Unicode script, and GPTQ W4A16 quantization. The two checkpoints we file as primary models occupy 6.64 and 7.08 GiB, i.e. 29\% and 31\% respectively of the 22.7 GiB disk footprint of the bf16 original. No interval separates the pruned vocabulary from the full-vocabulary INT4 checkpoint on any direction, and at the paragraph granularity on which the task scores, our vocab-INT4 primary costs 0.0048 COMET against bf16 on cs→de and nothing measurable on the other two directions. We also submit two contrastive systems: an FP8 system that is faster than the bf16 original at lower memory, and a self-MBR decoder that adds roughly 0.009 COMET on en→ar without any additional parameters. We do not submit a depthpruned system; a default generation cap had clipped the distillation targets and taught the student to end a turn mid-paragraph, a defect whose repair recovers a quarter of the deficit against full depth.},
  url       = {https://aclanthology.org/2026.wmt-1.143}
}

