@InProceedings{charkiewicz-nowakowski:2026:wmt,
  author    = {Charkiewicz, Adrian  and  Nowakowski, Artur},
  title     = {Laniqo at WMT26 Multilingual Instruction Shared Task (MIST): Full-Parameter Knowledge Distillation with Task-Conditional Pivot Decoding Under a 10B-Parameter Constraint},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1913--1923},
  abstract  = {We describe our submission to the WMT26 Multilingual Instruction Shared Task (MIST). Our system is google/gemma-4-E4B-it (8B total / 4.5B effective parameters), fully fine-tuned via knowledge distillation from a larger teacher model (Gemma-4-31B-it) on teacher-generated responses spanning three sub-tasks and 24 languages. We find that (i) a two-judge quality-agreement filter on the distillation data provides no measurable benefit over using the teacher's outputs unfiltered, at both LoRA and full-parameter training scale, and (ii) a task- conditional decoding strategy (using two-turn English drafting for open-ended generation and summarization, versus single-turn direct decod- ing for context-grounded question answering) significantly improves quality where applied and is harmless where it is not. Our final submission combines full-parameter distillation with this task-conditional decoding scheme.},
  url       = {https://aclanthology.org/2026.wmt-1.126}
}

