@InProceedings{ono:2026:wmt,
  author    = {Ono, Nobutaka},
  title     = {TMU-onono at WMT26: DiBA-Based Model Compression for Czech-to-German Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2024--2031},
  abstract  = {We describe TMU-onono's Czech-to-German systems submitted to the constrained track of the WMT26 Model Compression shared task. We apply DiBA (Diagonal and Binary Matrix Approximation) (Ono, 2026), which approximates a dense matrix $A$ as $D_1B_1D_2B_2D_3$, where $D_1$, $D_2$, and $D_3$ are real diagonal matrices and $B_1$ and $B_2$ are 0/1 binary matrices. We replaced 337 Gemma 3 12B weight matrices, including the tied embedding/lm-head, and retuned only diagonal entries on translations generated by the original model. Each 2.51 GB system package excludes base-model files used during setup and provides 9.72× compression relative to the original 24.37 GB BF16 weights. The primary diba-triton\_direct directly applies bitpacked factors for low-memory inference; the contrastive diba-cached\_unpacked uses unpacked caches for higher throughput. Both were slower than the original. To avoid out-of-memory errors, we used relatively short sequences for retuning. The submitted systems also limited inputs to 384 tokens including the prompt and outputs to 128 new tokens. Primary achieved 17.73 BLEU and 48.44 chrF on 128 held-out short segments, versus 8.25 and 32.16 on 256 WMT25 development paragraphs. Relaxing inference limits alone did not consistently improve quality, suggesting that short-sequence retuning may have limited performance on longer translation units.},
  url       = {https://aclanthology.org/2026.wmt-1.139}
}

