@Book{wmt:2026,
  editor    = {Haddow, Barry  and  Kocmi, Tom  and  Koehn, Philipp  and  Monz, Christof},
  title     = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.wmt-1}
}

@InProceedings{aliane-semmar-aliane:2026:wmt,
  author    = {Aliane, Ahmed Amine  and  Semmar, Nasredine  and  Aliane, Hassina},
  title     = {Efficient Multilingual Neural Machine Translation via Corpus-Driven Vocabulary Pruning: An English-Arabic Case Study.},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1--13},
  abstract  = {The adoption of large pre-trained multilingual models for neural machine translation (MNMT) faces a major challenge: excessive memory and computational consumption due to overly large vocabularies and embedding layers. Although existing compression methods like pruning, quantization and knowledge distillation reduce parameter redundancy, they mainly preserve the structure of the original vocabulary, leaving a major source of inefficiency unresolved. We propose in this paper a general optimization framework combining a vocabulary pruning method with a targeted finetuning protocol for MNMT models. We evaluate the proposed framework using three models (M2M100, NLLB-200, mBART-50) on the English-Arabic language pair. Our approach reduces the vocabulary size by 82-87\% depending on the architecture, without any loss in performance. Results show that optimized multilingual models can match or exceed the performance of dedicated bilingual baselines. In particular, the pruned and fine-tuned M2M100 model achieves a competitive BLEU score of 42.04 (against 44.59 for the OPUS-MT-en-ar bilingual model) while it significantly outperforms it on the COMET metric (0.8730 vs 0.7911) revealing superior semantic adequacy and fluency.},
  url       = {https://aclanthology.org/2026.wmt-1.1}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-4685-6053
Author{3}{Orcid}:
@InProceedings{giraud-gargett:2026:wmt1,
  author    = {Giraud, Jurgi  and  Gargett, Andrew},
  title     = {Compact Models for Machine Translation in the Bioinformatics Domain: Synthetic Data, Domain Adaptation, and Expert Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {175--189},
  abstract  = {Bioinformatics is almost invisible in Machine Translation (MT) research, despite being a hard case: dense specialised terminology, a high density of Multiword Expressions (MWEs), and an acute scarcity of parallel data. We address this gap for English-French. We compile a bioinformatics parallel corpus (14,467 sentence pairs) and expand it to 236k pairs through four augmentation strategies, each profiled for lexical diversity before use, then fine-tune five compact models (223M-8B parameters) under an incremental data ablation. We evaluate with automatic metrics and with a three-rater MQM evaluation involving professional life-sciences translators and using an error typology extended with a dedicated MWE category. Domain adaptation improves every system on every metric, with fine-tuned models surpassing much larger state-of-the-art commercial models on our domain test set. Expert judgements reveal a more structured picture with Accuracy errors falling by up to 68.8\% and MWE errors by up to 51.1\%, while also highlighting an accuracy-fluency trade-off that the automatic metrics conceal. Notably, the least lexically diverse augmentation sources contribute gains comparable to the most diverse ones, which we advance as one candidate account of the trade-off. We release all corpora, synthetic data, and adapted models.},
  url       = {https://aclanthology.org/2026.wmt-1.10}
}

Author{1}{Orcid}:https://orcid.org/0009-0000-2614-3382
Author{2}{Orcid}:
@InProceedings{kim-park-kim:2026:wmt,
  author    = {Kim, Ahrii  and  Park, Chanjun  and  Kim, Seong-heum},
  title     = {FACET at WMT 2026 Automated Translation Quality Evaluation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1674--1684},
  abstract  = {Different error types in machine translation require different evidence. Whether meaning is preserved can be judged only against the source, while whether the target is well-formed, or whether it names one entity consistently, can be judged from the target alone. We present FACET, our reference-free submission to the WMT26 Automated Translation Quality Evaluation Task, which decomposes evaluation into Fluency, Accuracy, and Consistency passes and gives each pass only the context its error type requires. A single fixed model is prompted three times, and the merged error spans yield the three task outputs, error spans, quality scores, and error-free labels, with no trained components. We also submit FACET−C, which omits the Consistency pass. Without gold labels, we characterize the predictions of FACET. Its system rankings place post-edited human translation first, and the Consistency pass changes about a tenth of segment scores while leaving the ranking nearly unchanged.},
  url       = {https://aclanthology.org/2026.wmt-1.100}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-2989-3220
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{lee:2026:wmt2,
  author    = {Lee, Soyoung},
  title     = {PragmaSpan at WMT26: Taxonomy-Guided Few-Shot Error Span Detection and an Empirical Analysis of MPP},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1685--1695},
  abstract  = {We present PragmaSpan, a training-free, few-shot GPT-5.5 error-span detector with a seven-axis diagnostic taxonomy for WMT26 Task 1, wrapped in a reliability-oriented harness using incomplete-response-aware retries, deterministic parsing, offset recovery, exact-span deduplication, and independent artifact validation. The taxonomy provides a domain-sensitive diagnostic representation: its interpersonal and discourse axes track domain conversationality and match gold errors at least as often as its traditional axes, with no significant detection-F difference from a standard Multidimensional Quality Metrics (MQM) category tree. We use the system to characterize the severity-weighted Match with Partial overlap and Partial credit (MPP) metric. Across prompt and model comparisons, human-agreement calibration, and post-hoc refinement, we find (i) prompt variants do not differ significantly on our sample, while GPT-4.1-to-GPT-5.5 gains exceed differences among variants; (ii) human–human MPP is low on our multi-annotator surrogate sample, and system–human agreement is not significantly different from this empirical baseline; (iii) improvements in matched-pair severity or boundary accuracy can coincide with lower MPP, demonstrating the need to interpret MPP jointly with matched-pair diagnostics and match coverage; and (iv) a calibration gain on a single-annotator MQM benchmark does not replicate on a multi-annotator Error Span Annotation benchmark. Interpreting error-span improvements therefore requires complementary metrics and annotation-regime checks.},
  url       = {https://aclanthology.org/2026.wmt-1.101}
}

Author{1}{Orcid}:
@InProceedings{menismastromichalakis-EtAl:2026:wmt,
  author    = {Menis Mastromichalakis, Orfeas  and  Filandrianos, Giorgos  and  Mohammed, Wafaa  and  Attanasio, Giuseppe  and  Zerva, Chrysoula},
  title     = {Benchmarking Gender Bias in Machine Translation Evaluation Metrics across Occupations},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1696--1706},
  abstract  = {Gender bias remains a persistent concern in machine translation (MT), affecting both generated translations and their automatic evaluation. When a source text leaves a person's gender unspecified, translations may realize that person using masculine or feminine forms, and both MT systems and evaluation metrics may exhibit systematic preferences between these alternatives despite the source providing no basis for such a distinction. We study this behavior in the WMT 2026 Automated Translation Quality Evaluation Systems Shared Task using an occupation-balanced subset of GAMBIT+. We consider seven English-source language pairs, six from the original dataset, targeting Arabic, Czech, Greek, Icelandic, Russian, and Ukrainian, and extend the original resource with German. The subset contains 1,308 masculine/feminine translation pairs per target language, with three examples for each of the 436 ISCO-08 occupational groups. We evaluate shared-task submissions and baselines for score prediction and error annotation, examining the direction, magnitude, and frequency of gender-related differences. We find an overall tendency for masculine translations to receive higher scores, as well as differences per occupation following stereotypical gender representations, although the strength and consistency of this preference vary considerably across evaluators and languages. Our results show that gender bias remains present in MT evaluation, but that capturing its extent requires looking beyond a single aggregate measure to complementary dimensions of evaluator behavior.},
  url       = {https://aclanthology.org/2026.wmt-1.102}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/ 0000-0002-7015-7746
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0001-6945-3698
Author{5}{Orcid}:https://orcid.org/0000-0002-4031-9492
@InProceedings{mishra-sharma-khetarpaul:2026:wmt,
  author    = {Mishra, Animesh  and  Sharma, Krishang  and  Khetarpaul, Sonia},
  title     = {Translation Metrics Cannot Judge What Their Tokeniser Deletes: LIGATUR and AEGIS Submissions to the WMT26 Shared Task on Automated Translation Quality Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1707--1717},
  abstract  = {The WMT26 shared task on automated translation quality evaluation asks participants both to build metrics (Task 2) and to submit challenge sets that probe them (Task 4). We describe one submission to each, and a single finding that connects them. Modern machine translation (MT) metrics cannot detect a class of invisible Unicode corruption, and the reason is not that they judge it wrongly: their tokenisers delete the evidence first. Four such corruptions — a non-breaking space, a narrow non-breaking space, an ideographic space, and a leading byte-order mark — normalise to token sequences identical to the clean text. This holds under both XLM-R and mT5, the encoders behind COMET-22 and MetricX-24. The metric therefore receives the same input and returns the same score, and no amount of training can separate the pair. Perturbations that instead fragment tokenisation are caught reliably. LIGATUR (Task 4) is the 171-item contrastive challenge set that isolates this effect for English-German (en-de) and English-Hindi (en-hi); a blind re-evaluation with held-out labels also overturns our own pilot claim that judges ignore Devanagari conjunct breaks. AEGIS (Task 2) acts on the diagnosis, combining a reference-based ensemble with deterministic hygiene penalties for exactly what the encoders cannot see, and scores all 26,885 test items. An ablation on 34,174 WMT24 judgements shows the two-family ensemble is our strongest component and that the reference-free branch we added as insurance costs accuracy where references are good, which we report as a calibration error.},
  url       = {https://aclanthology.org/2026.wmt-1.103}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{mongush-pukhov:2026:wmt,
  author    = {Mongush, Airana  and  Pukhov, Dmitrii},
  title     = {Algebras FluencyScore at WMT26 Automated Translation Quality Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1718--1722},
  abstract  = {We describe the Algebras FluencyScore submissions to the WMT26 Automated Translation Quality Evaluation shared task (Tasks 1-2). FluencyScore is a reference-free structured LLM judge: five fluency sub-dimensions, a fixed-weight aggregate, and a diagnostic error record (issue type, severity, problem phrase). This system paper adapts that judge rather than replacing it. For Task 1 the diagnostic record is specialized into a multi-error character-span annotator with deterministic phrase-to-offset alignment, merge / whole-segment post-processing, and a conservative add-only overlay. For Task 2 the calibrated scalar is mapped onto the cESA scale with a WMT25-fitted affine adapter and moment matching (aira\_mo). Submitted systems use Gemini 3.5 Flash and cover all 21 official pairs plus challenge sets. Official human-cESA rankings were not available at camera-ready time; we report no ranks and no Codabench verification scores.},
  url       = {https://aclanthology.org/2026.wmt-1.104}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{nakhl-EtAl:2026:wmt,
  author    = {Nakhlé, Mariam  and  Qader, Raheel  and  Dinarelli, Marco  and  Blanchon, Hervé},
  title     = {Vertical: Quality Estimation of Machine Translation from Université Grenoble Alpes},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1723--1729},
  abstract  = {This paper describes the submission of Université Grenoble Alpes and OVHai LLM to the Eleventh Conference on Machine Translation (WMT26) Shared Task on Automated Translation Quality Evaluation Systems. We participate in the Task 2 - Segment-Level Quality Score Prediction. We present our system that is trained to predict a single quality score per input segment. Our contributions are the following: 1) we present a metric that uses a Large Language Model (LLM) as backbone, namely gemma-3-1b-it, thus leveraging its long context size and its large-scale training on 2 trillion tokens, 2) we propose a method to adapt a decoder-only model to a downstream task by training a special token and using it as the sequence representation and 3) we propose a training curriculum designed to boost metric performance on long inputs. Our metric is trained using publicly available data and it can predict a score with or without a reference. Preliminary results show that our method outperforms baseline models when provided with a reference on the system level. The reference-free variant ranks second on the segment level.},
  url       = {https://aclanthology.org/2026.wmt-1.105}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{pong:2026:wmt,
  author    = {Pong, Benjamin},
  title     = {Post-hoc Correction of Machine Translation Error Span Predictions},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1730--1736},
  abstract  = {This paper presents an unsupervised quality estimation metric for machine translation that combines both neural and LLM models to identify and classify error spans for machine translations. As a submission to the WMT2026 Automated Evaluation Shared Task 1, this system builds on XCOMET-XL by applying offsets to the logits produced by the neural metric, post-hoc, to address the long-tailed probability problem, and uses an LLM as a second-stage to refine these predictions. Results show that XCOMET-XL with logits offset alone shows a boost in recall and character-level F1 scores for error span classifications across multiple language pairs, surpassing state-of-the-art baseline (i.e XCOMET-XL) and LLM-based approaches. This system also comprises a second-stage where a reasoning LLM-judge is employed to audit the error spans predicted by the logits-adjusted XCOMET metric, whose goal is to reduce spurious error classifications and improve overall precision. Results show that using LLM-judge to refine error spans provides no measurable improvements over predictions of the logits-adjusted XCOMET metric.},
  url       = {https://aclanthology.org/2026.wmt-1.106}
}

Author{1}{Orcid}:
@InProceedings{servajean-EtAl:2026:wmt,
  author    = {Servajean, Richard  and  Sohrab, Mohammad Golam  and  Kunitomo-Jacquin, Lucie  and  Rikters, Matiss},
  title     = {Leveraging Verbalized Confidence in LLM-as-a-Judge for Automated Translation Quality Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1737--1745},
  abstract  = {This paper presents the AIST AIRC team's contributions to subtasks 1 (segment-level error detection and span annotation), 2 (segment-level quality score prediction) and 3 (detection of error-free segments) of the automated translation quality evaluation systems shared task in WMT 2026. Our approach to address each of the 3 subtasks relies on an LLM-as-a-judge system. Specifically, for subtasks 2, we follow an LLM-as-a-judge approach within the Multidimensional Quality Metrics (MQM) error typology, with the novel addition of prompting the large language model (LLM) to verbalize its confidence in each error type assessment. We then train a Multilayer Perceptron (MLP) Regressor on error severities with and without associated confidence ratings to predict human annotations. First, our results provide further evidence for the relevance of integrating LLMs into automated translation quality evaluation pipelines. Second, this work aims to advance ongoing efforts to make these pipelines uncertainty-aware through the integration of verbalized confidence. Our code is freely available at https://github.com/sshrichard/WMT-Evaluation-Shared-Task-2026.},
  url       = {https://aclanthology.org/2026.wmt-1.107}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-5540-7834
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-3530-6873
@InProceedings{siani:2026:wmt,
  author    = {Siani, Assaf},
  title     = {Reference-Free Consensus Evaluation of Machine Translation: Ensembling LLM Error-Span Judgments with Neutral Cross-Verification and STAPLE Reliability Fusion},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1746--1751},
  abstract  = {We present a reference-free system for the WMT 2026 automated translation-quality-evaluation shared task, covering all three subtasks: character-level error-span detection with Minor/Major severity, segment level quality scoring on a 0–100 scale, and binary error-free classification. Rather than treating a single large language model (LLM) as an oracle, the system treats LLMs as fallible annotators. Three heterogeneous judges independently propose target-side error spans; a neutral cross-verification round converts these open-ended proposals into a complete, anonymized candidate-by-judge vote matrix; a character-level adaptation of STAPLE estimates latent error truth and per-judge reliability without references; and a two-state HMM imposes minimal span structure. Omissions are handled by a dedicated source-coverage pass rather than by scanning the target. Calibrating the single free prior against the official metric with human gold yields several findings: plain majority voting matches or beats STAPLE; the error rate prior barely matters once outputs are schema-conformant; precision—judge over-flagging at roughly 3× the human rate—is the binding constraint; and LLM-derived silver labels cannot calibrate the prior without circularity. The system was deployed across six language pairs with partial-failure-tolerant, resumable execution.},
  url       = {https://aclanthology.org/2026.wmt-1.108}
}

Author{1}{Orcid}:
@InProceedings{silchenko:2026:wmt,
  author    = {Silchenko, Maksim},
  title     = {QEbreak at WMT26: A Pre-Registered Audit of Omission and Addition Asymmetry in Machine Translation Quality Metrics},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1752--1763},
  abstract  = {Reference-free quality estimation (QE) metrics see only the source and the hypothesis, so a translation that silently drops source content leaves nothing for them to flag. QEbreak, a pre-registered contrastive challenge set submitted to the WMT26 Automated Translation Quality Evaluation task (Subtask 4), measures this blind spot: 2,886 segments over 10 language directions, built as matched {base, omission, addition} triples, a fact-free verbosity ladder, and numeric-contradiction pairs as a positive control. Every hypothesis, endpoint, and equivalence margin was frozen in git-timestamped commits before any score existed. Scores from 31 systems confirm the deficit and refute our mechanism. Reference-free QE metrics under-penalize omission (family asymmetry A = −0.51; a current QE baseline scores the omission at or above its own base in 19.5 percent of matched contests, strictly above in 14.3), and the edit-direction by reference-availability interaction is significant (β = −0.069, 95\% CI [−0.081, −0.057]). But reference-based neural metrics are asymmetric in the same direction (A = −0.42), outside the pre-registered ±0.30 equivalence margin, so omission under-penalization belongs to learned metrics as a class; reference access softens it but does not remove it. Every returned metric penalizes pure verbosity, and two systems score corrupted numerals above correct ones.},
  url       = {https://aclanthology.org/2026.wmt-1.109}
}

Author{1}{Orcid}:
@InProceedings{goldner-gramazio:2026:wmt,
  author    = {Goldner, Owen Nicholas  and  Gramazio, Connor Casey},
  title     = {ChainAlign: Cross-Lingual Sentence Alignment via Sparse Co-linear Chaining},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {190--205},
  abstract  = {Cross-lingual sentence alignment establishes correspondence between a document and its translation, where correspondence is not always complete, one-to-one, or monotonic. Current embedding-based aligners score sentence pairs with a multilingual encoder, then trace a single monotonic path across the full alignment grid. This forced path cascades errors through non-parallel or reordered content. We present ChainAlign, which instead adapts *seed-chain-extend* from computational genomics: it seeds multi-resolution candidate anchors from multilingual embeddings, selects the maximum-weight non-crossing subset by co-linear chaining in *O(N log N)* via a Fenwick tree, and fills the residual gaps with local dynamic time warping. Because the chain imposes no coverage requirement, non-parallel content becomes unaligned gaps rather than cascading errors, and multi-chain extraction handles reordering that a monotonic path cannot represent. On three benchmarks across four language pairs, ChainAlign achieves the highest strict and lax F1, ahead of Bertalign, SentAlign, and Vecalign. Evaluation shows the sparse formulation outperforms exhaustive dynamic programming using the same edge weight function.},
  url       = {https://aclanthology.org/2026.wmt-1.11}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{xu:2026:wmt,
  author    = {Xu, Jia},
  title     = {SpinPop: A Fast Spin Metric for the WMT26 Metrics Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1764--1769},
  abstract  = {We describe our submission to the WMT26 Shared Task on Automated Translation Quality Evaluation. Our primary system is a per-language-pair, z-normalized, weight-free ensemble that combines six complementary signals: our novel metric SpinPop, COMET-22, and four large language model judges. SpinPop is training-free, constant-cost, tokenization-agnostic, broadly multilingual, and deterministic. Within its encoder's language coverage it scores every segment, with or without a reference, using a single frozen-encoder pass that runs locally and requires no paid API calls. Our system achieves a mean per-language-pair Pearson correlation of $0.8534$, ranking second on the WMT26 Task~2 leaderboard. The results demonstrate that a simple unsupervised ensemble of a spin-code metric, a neural metric, and multiple LLM judges can achieve top-tier performance with equal weights.},
  url       = {https://aclanthology.org/2026.wmt-1.110}
}

Author{1}{Orcid}:
@InProceedings{zhu-ziminapoirot-froeliger:2026:wmt,
  author    = {Zhu, Lichao  and  Zimina-Poirot, Maria  and  Froeliger, Nicolas},
  title     = {Beyond Rankings: Insights from ALTAE's PTT-MULTITAN Submission to the WMT 2026 Shared Task on Automated Translation Quality Evaluation Systems},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1770--1778},
  abstract  = {We describe PTT-MULTITAN, our submission to the WMT26 MT Evaluation shared task, Task 3 (error-free segment detection). The task requires a binary decision — does a machine translation contain any error? — without ac- cess to reference translations, and is scored with the Matthews Correlation Coefficient (MCC). Our system combines a supervised classifier fitted on WMT25 human error-span annotations with three families of reference- free indicators: structural and phraseological cues, corpus-linguistic association measures computed over an LLM-as-judge model, and format/locale conformity checks. We have im- proved our system submissions, whose MCC rises from 0.159 to 0.168. We submitted six language pairs from the official WMT26 test set. The pairs were from English into Belaru- sian, German, Russian, Ukrainian, Simplified Chinese, and Traditional Chinese.},
  url       = {https://aclanthology.org/2026.wmt-1.111}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-4432-0236
Author{2}{Orcid}:0000-0002-0892-2531
Author{3}{Orcid}:
@InProceedings{cambraguinea-alfieri:2026:wmt,
  author    = {Cambra Guinea, Jon  and  Alfieri, Andrea},
  title     = {RWS at WMT26 Terminology Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1779--1785},
  abstract  = {This submission is a prompt-based terminology and segment injection strategy for terminology translation across domains. We evaluate the technique on Basque to Spanish, English to Polish, and Traditional Chinese to English for WMT 2026 Terminology Shared Task 1 \& 2. We demonstrate that injecting bilingual terminology and bi-text examples into recent Large Language Models' (LLM) prompts improves both overall translation quality and terminology accuracy. Our results show that strong instruction-following ability allows a system to adapt to terminology and style constraints without the need for fine-tuning.},
  url       = {https://aclanthology.org/2026.wmt-1.112}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{castaldo-EtAl:2026:wmt,
  author    = {Castaldo, Antonio  and  Speranza, Giulia  and  Staiano, Maria Carmen  and  Giommarelli, Petra  and  Monti, Johanna  and  Di Buono, Maria Pia},
  title     = {UniOR-MT: An Agentic Pipeline for Terminology Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1786--1797},
  abstract  = {We present UniOR-MT, an agentic terminology-aware machine translation pipeline developed for the WMT 2026 Terminology Translation Shared Task. We compare five progressively enriched translation paths, each adding one component to isolate its individual contribution. We additionally submit a mixed system that selects, among the five candidate paths, the best candidate using a combined measure of terminology accuracy and document-level quality estimation. Evaluation results show that incorporating a bilingual glossary provides the most consistent benefit, while additional post-editing steps yield limited improvements, and can even degrade translation quality for Spanish--Basque in the official evaluation. The final system, which includes iterative QA feedback and a stronger fallback model, achieves the highest docCOMET scores in our internal evaluation.},
  url       = {https://aclanthology.org/2026.wmt-1.113}
}

Author{1}{Orcid}:https://orcid.org/0009-0008-3325-787X
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{giraud-gargett:2026:wmt2,
  author    = {Giraud, Jurgi  and  Gargett, Andrew},
  title     = {Agenteak: A Multi-Agent Pipeline for Terminology-Constrained Spanish-Basque Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1798--1807},
  abstract  = {This paper describes the OpenU's submission to the WMT26 Terminology Translation Task for Spanish into Basque (Track 1), covering the automotive and energy domains. We present AGENTEAK, a multi-agent system built on LangGraph in which three compact open-weight models (4-8B parameters) cooperate: a reasoning model that selects and disambiguates the terminology relevant to each segment, a domain fine-tuned translation model, and a verification model that checks and repairs terminology in the draft translation. To adapt the translator, we mined in-domain Spanish-Basque bitext from Wikipedia using multilingual sentence embeddings and complemented the data with in-domain synthetic parallel paragraphs generated by an instruction-tuned Basque LLM. Domain-tagged fine-tuning improves the base model by up to 10.4 BLEU and 3.9 COMET points on our in-domain test sets. On the shared task data, supplying the pipeline with correct terminology raises terminology success rate from 0.613 to 0.757 (automotive) and from 0.601 to 0.774 (energy).},
  url       = {https://aclanthology.org/2026.wmt-1.114}
}

Author{1}{Orcid}:https://orcid.org/0009-0000-2614-3382
Author{2}{Orcid}:
@InProceedings{huang-EtAl:2026:wmt,
  author    = {Huang, Boqi  and  Wei, Daimeng  and  GUO, Jiaxin  and  Luo, Yuanchang  and  Shang, Hengchao  and  Li, Zongyao  and  Yang, Jinlong  and  Wu, Zhanglin  and  He, Yu  and  Lan, Xiaoqing},
  title     = {HW-TSC's Submission to the WMT26 Terminology Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1808--1813},
  abstract  = {We describe HW-TSC's submissions to both tracks and all language pairs of the WMT26 Terminology Translation Shared Task. Our primary system uses Qwen3.7-Max for source-term extraction, bilingual term alignment, and document translation. In Track 1, it matches the current document or translation chunk against the supplied dictionary and places at most 200 applicable term pairs in the translation prompt. In Track 2, it first induces a bilingual glossary from the seed bitexts and then reuses the Track 1 translation module. Induction combines open source-term extraction with seed-source expression matching, followed by LLM alignment to the seed targets. We optimize the extraction prompt with SkillOpt on a train/validation/test split of the WMT25 English–Chinese finance data and use Qwen3.7-Max to rewrite it for directions without labeled optimization data. Extraction recall rises from 0.6250 to 0.6344 on validation and from 0.6451 to 0.6674 on the held-out test split. On the same held-out test documents with Qwen3.7-Max, glossary induction raises TSR from 0.5212 (no glossary) to 0.7910 with the initial prompt and to 0.8026 after SkillOpt, with chrF++ of 57.45, 67.05, and 67.36; the official glossary yields TSR 0.9019 and chrF++ 63.94.},
  url       = {https://aclanthology.org/2026.wmt-1.115}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:https://orcid.org/0000-0002-5211-3221
Author{6}{Orcid}:https://orcid.org/my-orcid?orcid=0000-0001-7196-2782
Author{7}{Orcid}:
Author{8}{Orcid}:https://orcid.org/0000-0002-2920-0773
Author{9}{Orcid}:
Author{10}{Orcid}:
@InProceedings{liao-melero:2026:wmt,
  author    = {Liao, Xixian  and  Melero, Maite},
  title     = {SalamandraTA at WMT 2026 Terminology Shared Task: Hard Examples Are Better Teachers},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1814--1825},
  abstract  = {Terminology-aware translation asks for more than a correct translation: the output must use the exact terms a glossary prescribes. The standard recipe, fine-tuning on glossary-annotated translation pairs, hides an inefficiency: for most examples the glossary prescribes exactly what the model would have produced anyway, so they teach nothing about following a glossary. We therefore keep only the examples where the model's own translation contradicts the glossary. In a controlled study at fixed data volume, this selection alone raises term accuracy from 78.7\% to 89.9\%. The filtered data, built by a two-way synthetic pipeline on open models, is part of the instruction-tuning mixture of our public release SalamandraTA-7b-instruct v3.0, which, used exactly as released and wrapped in a document-level inference pipeline, forms the BSC submission to the WMT26 Terminology Shared Task Track 1. At the official WMT26 evaluation, our system achieves 94.2\% term success at 74.6 chrF++, with only two of the twenty-two submissions outperforming it on both metrics. On last year's benchmark, it also surpasses our GRPO-based system, despite being trained solely with ordinary supervised fine-tuning.},
  url       = {https://aclanthology.org/2026.wmt-1.116}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0001-9933-3224
@InProceedings{lpezsnchez-EtAl:2026:wmt,
  author    = {López-Sánchez, Gonzalo  and  Oliver, Antoni  and  cozzini, marta  and  Vàzquez, Mercè  and  Morales-Hurtado, Patricia  and  Alvarez-Vidal, Sergi},
  title     = {GRIAL-TA participation in the WMT26 Terminology Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1826--1836},
  abstract  = {This submission tackles both tracks (in English-Polish) of the shared task using an instruction-aligned Gemma 4 model fine-tuned on highly curated, domain-specific corpora. For Track 2, we overcome the lack of explicit dictionaries by employing a combined BERT and POS-based automatic term extraction process.},
  url       = {https://aclanthology.org/2026.wmt-1.117}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0001-8399-3770
Author{3}{Orcid}:0009-0004-1992-9132
Author{4}{Orcid}:https://orcid.org/0000-0002-7983-4029
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{marchetti-EtAl:2026:wmt,
  author    = {Marchetti, Guilherme Aren  and  Rocha, Gil  and  Lopes Cardoso, Henrique  and  Sousa-Silva, Rui},
  title     = {STaR-MT: Select, Translate, and Revise Pipeline for Terminology Aware Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1837--1846},
  abstract  = {To assess the progress in terminology-aware machine translation, the Terminology Shared Task is hosted alongside the Conference on Machine Translation (WMT). In this paper, we describe our submission to the shared task: STaR-MT. This is a three-step pipeline composed of (S)election, (T)ranslation, (a)nd (R)evision, focused on terminology-aware machine translation (MT). Our methodology adopts a lightweight agent-based design that selects relevant entries or examples and presents them to the translation agent, which uses a general-purpose LLM with a large context window to produce the initial output. Next, the revision agent uses language-specific models to improve translation quality for selected target languages. Experiments on the FLORES+ dataset with different model sizes in the pipeline suggest that machine translation pipelines can benefit from combining large general-purpose translation models with smaller, language-specific revision models, at least in some language pairs. While simple in architecture, this approach allows us to generate translations at the document level, with no need for pre-processing or paragraph splitting, and with minimal errors across most languages.},
  url       = {https://aclanthology.org/2026.wmt-1.118}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-8252-7292
Author{3}{Orcid}:http://orcid.org/0000-0003-1252-7515
Author{4}{Orcid}:0000-0002-5249-0617
@InProceedings{mhaskar-EtAl:2026:wmt,
  author    = {Mhaskar, Shivam Ratnakant  and  Sukhadia, Vrunda Nileshkumar  and  Deshmukh, Anurag  and  Sharma, Manan  and  Rajpoot, Pawan Kumar},
  title     = {TARL: A Terminology-Aware Agentic Repair Loop for Language-Agnostic Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1847--1855},
  abstract  = {We describe TARL (Terminology-Aware Agentic Repair Loop), our submission to the WMT26 Terminology Translation Task, which targets document-level, terminology-constrained translation for morphologically rich and lower-resource directions (Spanish→Basque, English→Polish) and Traditional Chinese→English. Our system performs no fine-tuning: it is a sequential pipeline of four role-specialized LLM agents (translate, verify, review, and post-edit) in which a required target term, once inserted, is treated as locked so that later agents improve the translation around the terminology rather than altering it. Relevant glossary terms are retrieved per sentence by an exact, fuzzy n-gram matcher (with substring matching for CJK source), and for the sample-only Track 2 we first induce a glossary by extracting bilingual term pairs from the provided parallel data. The agents' instructions are optimized automatically with GEPA (Genetic-Pareto), an automated prompt optimization framework that uses natural language reflection and multi-objective Pareto evolutionary search to tune LLM prompts, rather than being hand-written. On the official WMT26 Track 1 evaluation, TARL attains the highest lemmatized terminology success rate of any submitted system (95.7\%, averaged over English→Polish and Spanish→Basque), while remaining competitive on translation quality (COMET-22: 88.3, XCOMET-XXL: 83.6). On Track 2, TARL achieves 80.4\% term success and 85.8 COMET-22 across three directions. A terminology-mode analysis confirms that the gains come from genuine use of the supplied dictionary rather than from instruction-following alone.},
  url       = {https://aclanthology.org/2026.wmt-1.119}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{haemmerl-bretschner-wuebker:2026:wmt,
  author    = {Haemmerl, Kathy  and  Bretschner, Gabriel  and  Wuebker, Joern},
  title     = {LocQE: Principled Domain Adaptation for Localisation Quality Estimation by Leveraging Post-Edits},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {206--226},
  abstract  = {Learned quality estimation (QE) models such as COMETKiwi are widespread and work well for general machine translation evaluation. However, they are known to struggle on unseen domains, limiting their performance in a real-world localisation context. We show that they are insensitive to some important factors in localisation, such as whether numbers are translated accurately, or even whether the correct number of spaces and punctuation are preserved in a translation. Further, a key capability for optimisation of machine translation is the ability of QE models to accurately rank different translations of a single segment, which suffers significantly from the domain transfer. In the absence of large-scale direct assessment data, we propose principled fine-tuning approaches to reduce the domain gap with even small amounts of post-editing data. Using a multi-task fine-tuning approach and a simple tokeniser intervention, we create a QE model which proves markedly better at distinguishing preferred post-edits from rejected initial translations in a localisation context. We show that preferences and artificial continuous scores stabilise each other, and argue that to calibrate metrics both in terms of their absolute scores and comparisons between translation of the same source, both types of signal are needed.},
  url       = {https://aclanthology.org/2026.wmt-1.12}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-8780-5287
@InProceedings{papek-sourada:2026:wmt,
  author    = {Papáček, Aleš Manuel Manuel  and  Sourada, Tomáš},
  title     = {CUNI-ÚFAL at the WMT26 Terminology Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1856--1866},
  abstract  = {This paper describes the CUNI-ÚFAL submission to the WMT26 Terminology Translation Task. We present a model-agnostic pipeline that selects document-relevant terminology for a general-purpose LLM, instantiated with GPT-5.6-Sol for both tracks. For Track1 (explicit dictionary) we compare glossary placement, inline annotation, dictionary filtering, and terminology-aware revision: filtering the dictionary per document improves judged coverage and, slightly, translation quality. Inline term insertion buys a small coverage gain at the cost of fluency. For Track2 (in-domain seed bitexts) we compare presenting them as translation history, extracting a dictionary and reusing the Track1 pipeline, and their hybrid. Lacking development references, we evaluated with surface- and lemma-based term matching plus an LLM judge. Among submitted systems in the official evaluation, our Track1 submission ties for the highest COMET-22 score and obtains the best XCOMET-XXL and MetricX-24 XL scores, and our primary Track2 submission achieves the highest term success.},
  url       = {https://aclanthology.org/2026.wmt-1.120}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{pinnis-EtAl:2026:wmt,
  author    = {Pinnis, Marcis  and  Kronis, Martins  and  Grims, Emils  and  Jakovelis, Ervins  and  Rozis, Roberts  and  Bergmanis, Toms},
  title     = {Tilde's Submission for the WMT2026 Shared Task on Terminology Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1867--1879},
  abstract  = {This paper describes Tilde's submissions to the WMT 2026 Shared Task on Terminology. For both subtasks, we use TildeOpen-30B, which we instruction-tune for translation using terminology glossaries and translation memories, and we detail the construction of the training data augmented with these resources. For Task 1, we compare three term recognition methods aimed at reducing large term collections to subsets relevant for translation: fuzzy search, stemming with exact-match search, and stemming with fuzzy search. For Task 2, we compare extracting a glossary from the provided translation memory and integrating it in context against using retrieved translation-memory entries directly as few-shot examples. As the shared task provides neither reference translations nor a development set, we validate our design decisions with an LLM-as-a-judge protocol. In-domain term collections reduce translation errors by up to 22\% relative to translating without terminology, whereas out-of-domain collections yield only marginal gains, and glossaries extracted from the translation memory outperform few-shot translation with retrieved entries.},
  url       = {https://aclanthology.org/2026.wmt-1.121}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-6832-5600
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{riosgaona:2026:wmt,
  author    = {Rios Gaona, Miguel Angel},
  title     = {UniVie-HAITrans at WMT26 Terminology Translation Task: Terminology-Guided Data Filtering and Few-Shot Fine-Tuning for Lightweight LLMs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1880--1887},
  abstract  = {Specialised domain translation using Large Language Models remains challenging due to terminology constraints, scarce in-domain parallel data, and the tendency of fine-tuning to degrade in-context learning performance. We present our submission for the English–Polish and Spanish–Basque translation in the Medical and Engineering \& Technology domains. Our method follows three stages: i) We extract in-domain terms from a terminology database and prompt a multilingual LLM to generate contextual sentences for each term. ii) We build an efficient search index over large-scale, unconstrained web-crawled and medical parallel corpora. Using the synthetic sentences as queries, we retrieve and filter the most semantically relevant pairs to build an in-domain fine-tuning dataset. iii) We structure our training instances with a mixture of 0-shot and semantically retrieved 5-shot prompts, and fine-tune lightweight multilingual LLMs (Gemma-3-4B, and Tiny-Aya-Global) via QLoRA. Fine-tuned Gemma-3 achieves the best results on held-out data across all metrics on English–Polish (Medical) and on Spanish–Basque (Engineering \& Technology), showing that few-shot fine-tuning enhances specialised domain translation quality and preserves in-context adaptation.},
  url       = {https://aclanthology.org/2026.wmt-1.122}
}

Author{1}{Orcid}:
@InProceedings{velda-hwang-kim:2026:wmt,
  author    = {Velda, Vania  and  Hwang, Yigyu  and  Kim, Yongho},
  title     = {STRA-MT: Deterministic Terminology Control for Document-Level Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1888--1898},
  abstract  = {STRA-MT is a translation pipeline with deter- ministic terminology control built around small open-weight language models. The pipeline (1) reuses exact repetitions and guides near- duplicates using translations from earlier occur- rences, (2) maintains a growing terminology ledger initialized from either the provided glos- sary or a termbase mined from sample bitext, and (3) verifies every required term and, when one is missing, regenerates the paragraph once with the missing terms named. On the WMT25 zh-Hant→en development sets, STRA-MT im- proves mean terminology success over glossary prompting from 87.3 to 92.4 under the official WMT25 evaluation, while maintaining com- petitive BLEU, chrF++, and COMET scores. Official WMT26 test sets shows that term de- livery on en→pl comes close to the reference ceiling while the translation quality depends on the backbones.},
  url       = {https://aclanthology.org/2026.wmt-1.123}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{zhang-EtAl:2026:wmt1,
  author    = {Zhang, Fan  and  Mei, Tu  and  Mengchao, Zhang  and  Wu, Jinting  and  Zhang, Bowbjut.edu.cnen},
  title     = {TaT at WMT26 Terminology Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1899--1905},
  abstract  = {The accurate translation of domain-specific terminology remains a critical bottleneck for ensuring the fidelity and professionalism of machine translation systems. This paper presents our submission to the WMT26 Terminology Translation Shared Task: the TaT (Terminology-aware Translation) system. Built upon a structured, pipelined architecture, TaT integrates five specialized components designed to seamlessly ingest, process, and translate ambiguous textual inputs while strictly adhering to targeted terminology dictionary. Specifically, the framework comprises a sentence splitter, a terminology retriever, a lexical disambiguator, a baseline translator, and a dedicated terminology extractor tailored for Track 2 constraints. Both internal benchmarking and ablation studies demonstrate that our decoupled pipeline delivers robust performance, successfully meeting design expectations and substantially mitigating terminology omission and misalignment errors in domain-specific translations.},
  url       = {https://aclanthology.org/2026.wmt-1.124}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{alkhder-EtAl:2026:wmt,
  author    = {Alkhder, Hasan  and  Hanini, Maria  and  Hocine, Imane  and  Pasa, Maher  and  Najjar, Amro},
  title     = {A Monolingual LoRA-Tuned Qwen2.5-3B System for Arabic Context-Based Question Answering},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1906--1912},
  abstract  = {In this paper, we describe our submission to the WMT26 Multilingual Instruction Shared Task (MIST), Sub-task 1 (context-based question answering). We fine-tune Qwen2.5-3B-Instruct with a LoRA adapter on 3,112 extractive-QA examples drawn from the answerable subset of TyDi QA as released in the wmt26-mist-sample data. We submit outputs for the monolingual Arabic (question and context both in Arabic) instances of the official qa-context test set. Our system reaches a mean token accuracy of 0.9643 and produces accurate extractive spans. On the held-out split, fine-tuning roughly doubles both exact match (23.81 → 53.97) and token-level F1 (37.00 → 68.52) relative to the base model. The adapter corrects over-refusal tendency in addition to improving span precision. The base model emitted the prescribed no-answer string in 60\% of passages with answers, versus 0\% for the fine-tuned model. We report error analysis identifying recurring failures, and discuss the model's weaker behaviour on cross-lingual context instances, which make up the large majority of the official test set, as a direct consequence of training exclusively on monolingual data.},
  url       = {https://aclanthology.org/2026.wmt-1.125}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{charkiewicz-nowakowski:2026:wmt,
  author    = {Charkiewicz, Adrian  and  Nowakowski, Artur},
  title     = {Laniqo at WMT26 Multilingual Instruction Shared Task (MIST): Full-Parameter Knowledge Distillation with Task-Conditional Pivot Decoding Under a 10B-Parameter Constraint},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1913--1923},
  abstract  = {We describe our submission to the WMT26 Multilingual Instruction Shared Task (MIST). Our system is google/gemma-4-E4B-it (8B total / 4.5B effective parameters), fully fine-tuned via knowledge distillation from a larger teacher model (Gemma-4-31B-it) on teacher-generated responses spanning three sub-tasks and 24 languages. We find that (i) a two-judge quality-agreement filter on the distillation data provides no measurable benefit over using the teacher's outputs unfiltered, at both LoRA and full-parameter training scale, and (ii) a task- conditional decoding strategy (using two-turn English drafting for open-ended generation and summarization, versus single-turn direct decod- ing for context-grounded question answering) significantly improves quality where applied and is harmless where it is not. Our final submission combines full-parameter distillation with this task-conditional decoding scheme.},
  url       = {https://aclanthology.org/2026.wmt-1.126}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0003-4473-0008
@InProceedings{egertonidehen:2026:wmt,
  author    = {Egerton-Idehen, Asher Uyiosa},
  title     = {SentinelQA: Task-Level Routing and Output Verification for Multilingual Instruction Following},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1924--1931},
  abstract  = {This submission presents Team Sentinel's approach to the WMT26 MIST shared task on multilingual instruction following. Built on Gemma-3-4B-IT, the system allocates its parameter budget across base and fine-tuned models, using a fixed task-level lookup to select between them. A verification layer checks output language, length constraints, repetition, and chat-template artefacts, with retries and repairs to improve output compliance.},
  url       = {https://aclanthology.org/2026.wmt-1.127}
}

Author{1}{Orcid}:
@InProceedings{hou-temesgen-fraser:2026:wmt,
  author    = {Hou, Jen-Chien  and  Temesgen, Tsedeniya Kinfe  and  Fraser, Alexander},
  title     = {TUM-Heilbronn: WMT26 Multilingual Instruction Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1932--1939},
  abstract  = {This paper describes the TUM-Heilbronn sys- tem submission to the WMT26 Multilingual Instruction Shared Task (MIST). The MIST covers three tasks: context-based question an- swering, open-ended generation, and cross- lingual summarization to evaluate the multi- lingual instruction-following ability of large language models. We fine-tune Qwen3.5-9B- Instruct on 27 diverse languages, across 10 dif- ferent writing scripts. Our model outperformed 6 of the 15 models submitted to this shared task. Moreover, we ranked first in truthfulness (i.e., decline to answer when the context lacks sufficient information) on the context-based question-answering task. Cross-lingual sum- marization remains a challenging task for all submitted models, with the lowest performance observed among the three tasks.},
  url       = {https://aclanthology.org/2026.wmt-1.128}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{krause-EtAl:2026:wmt,
  author    = {Krause, Lea  and  Schouten, Stefan F.  and  van der Meer, Michiel  and  Vossen, Piek T.J.M.},
  title     = {Probe Before you Answer! Answerability Probing for Abstention in Cross-Lingual Context-Based Question Answering},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1940--1955},
  abstract  = {We separate answering from abstention in a context-based question answering task, where a language model answers a question about a document in a possibly different language or abstains when the document contains no answer. A LoRA adapter on Qwen3.5-9B, trained only on answerable questions, generates the answer, while a linear probe on the frozen base model's activations decides answerability independently and returns the required refusal string when it predicts none exists. This isolates the model's answerability signal from its instruction-following. The probe's ranking of answerable against unanswerable questions transfers across languages, corpora, and negative constructions it was not trained on, but its decision threshold does not; training on in-domain multilingual data, rather than recalibrating the threshold, closes most of that gap. For the adapter, how the answer target is constructed trades off automatic correctness metrics against well-formedness and answer length.},
  url       = {https://aclanthology.org/2026.wmt-1.129}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-7187-5224
Author{2}{Orcid}:0000-0001-9839-9985
Author{3}{Orcid}:https://orcid.org/0000-0003-1877-6002
Author{4}{Orcid}:https://orcid.org/0000-0002-6238-5941
@InProceedings{hara-EtAl:2026:wmt,
  author    = {Hara, Nagito  and  Nowakowski, Karol  and  Ptaszynski, Michal  and  Overacker, Lloyd Nicholas  and  Toyoura, Masahiro},
  title     = {Low-resource machine translation using a general-purpose LLM informed by a dedicated NMT model and lexical resources: A case study on Ainu–Japanese translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {227--245},
  abstract  = {Previous methods for post-editing the output of a dedicated Neural Machine Translation (NMT) system using a general-purpose Large Language Model (LLM) rely on correcting a single hypothesis generated by the NMT model's decoder. However, in low-resource scenarios -- such as Ainu-to-Japanese translation -- the low accuracy of initial NMT outputs often leads to error propagation, where LLMs fail to override systemic NMT hallucinations or syntactic errors. In this paper, we propose a novel machine translation framework that incorporates information from an NMT model into an LLM, while bypassing the limitations of single-sentence correction. Instead of using raw NMT text, our method extracts token-level probability distributions from the NMT decoder and combines them with entries obtained by looking up a bilingual dictionary, having the LLM construct the final translation from these two sources of information. By grounding the LLM in both the NMT's internal confidence signals and external linguistic knowledge, our approach effectively mitigates the bias toward poor-quality NMT outputs. Empirical evaluations on Ainu-Japanese parallel data using chrF, semantic similarity, and perplexity demonstrate that our method outperforms six baseline approaches in specific domains in the Ainu-to-Japanese direction, that is, when translating into the high-resource language. In the opposite direction, the method yields no improvement, as its effectiveness is bounded by the LLM's ability to generate the target language. Furthermore, perplexity analysis confirms that LLM-driven translation substantially enhances linguistic fluency compared to traditional NMT outputs.},
  url       = {https://aclanthology.org/2026.wmt-1.13}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0001-7435-4061
Author{3}{Orcid}:https://orcid.org/0000-0002-1910-9183
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{papek-schmidtova:2026:wmt,
  author    = {Papáček, Aleš Manuel Manuel  and  Schmidtova, Patricia},
  title     = {CUNI-UFAL at WMT26 Multilingual Instruction Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1956--1965},
  abstract  = {We describe our submission to the WMT26 Multilingual Instruction Shared Task, which evaluates sub-10B-parameter models on multilingual question answering, open-ended generation, and cross-lingual summarization. We compare four open-weight models and vary the system prompt, reasoning mode, and decoding parameters. On our development set, the tested prompt and reasoning changes have a larger effect on LLM-judge scores than the difference between the two finalist models in their best settings. Gemma 4 E4B obtains the highest overall judge score, whereas Qwen 3.5 9B follows hard constraints more reliably on our internal benchmark. We therefore submit Qwen as our primary system and Gemma as a secondary system. We also introduce a 1,283-example diagnostic benchmark for grounded QA, constrained generation, and abstract writing from ACL papers in 24 languages. Using it for QLoRA fine-tuning does not work in our setup, it lowers validation loss but does not improve quality under our LLM judge. In the official automatic evaluation, our Gemma system ranks first overall and our Qwen system second; Qwen wins QA-context, while Gemma wins QA-OEG and places second in summarization.},
  url       = {https://aclanthology.org/2026.wmt-1.130}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0009-0008-5516-798X
@InProceedings{pulipaka:2026:wmt,
  author    = {Pulipaka, Srikar Kashyap},
  title     = {PSK at WMT 2026 MIST: Task-Specialized QLoRA Adapters for Multilingual Summarization and Question Answering},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1966--1971},
  abstract  = {We describe the PSK submission to the WMT 2026 Multilingual Instruction Shared Task. Our system uses the 3.35B-parameter Tiny Aya Global model with three QLoRA adapters, one for each task. The adapters are trained on multilingual document–summary pairs, passage-based question answering, and filtered standalone question answering. The summarization data also includes scientific papers with their author-written abstracts. On our held-out split, the context and summarization adapters perform better than our multitask adapter, while results for open QA are mixed. Official evaluation ranks PSK sixth overall and fourth for summarization. Context QA and open QA remain weaker, and the multitask open-QA route outperforms both specialized open-QA adapters.},
  url       = {https://aclanthology.org/2026.wmt-1.131}
}

Author{1}{Orcid}:https://orcid.org/0009-0002-0107-3319
@InProceedings{sant-luqueserrano-escolanopeinado:2026:wmt,
  author    = {Sant, Aleix  and  Luque Serrano, Jordi  and  Escolano Peinado, Carlos},
  title     = {Task-Preserving Multilingual Instruction Tuning for Open-Ended QA at WMT 2026 MIST},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1972--1979},
  abstract  = {We describe our submission to the WMT 2026 Multilingual Instruction Shared Task. We adapt Qwen3-8B with LoRA using organiser-provided data, EuroAlpaca, a multilingual instruction dataset created through task-preserving localisation, and CrossEuroAlpaca, a cross-lingual augmentation that assigns different source and target languages to instructions, contexts and responses. Our system targets open-ended question answering (QA-OEG) in ten shared-task languages represented in the auxiliary data. Under matched decoding, adaptation improves the mean automatic Gemma-4 judge score on target-language QA-OEG by 7.8\% relative to the base model. On the 27-language Aya Evaluation Suite, it yields relative gains of 41.5\% in macro-averaged ROUGE-L and 2.9\% in BERTScore-F1, with larger improvements in the target languages. However, these gains are task- and language-specific. Gemma-4 scores improve for QA-OEG and summarisation in the target languages but decline for context-grounded QA within this set and across all three tasks outside it.},
  url       = {https://aclanthology.org/2026.wmt-1.132}
}

Author{1}{Orcid}:https://orcid.org/0009-0007-7259-656X
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{bajwa:2026:wmt,
  author    = {Bajwa, Angad Ripudaman Singh},
  title     = {COMET Sensitivity-Guided Mixed Precision Quantization for the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1980--1985},
  abstract  = {The WMT26 Model Compression shared task asks participants to reduce the footprint of a general-purpose large language model for machine translation (MT) deployment while preserving translation quality, under a fixed evaluation budget of model size, VRAM usage, and inference speed. We report a systematic comparison of post-training compression strategies applied to the constrained track base model, google/gemma-3-12b-it, across the three required language directions (Czech–German, English–Chinese (Simplified), English–Arabic (Egyptian)). We evaluate an uncompressed BF16 baseline and a vLLM-served engine-level control against six compression strategies: two bitsandbytes baselines (8-bit and 4-bit), two calibrated weight-only quantization methods (AWQ and GPTQ, both W4A16), vLLM's native dynamic FP8 weight quantization, and a novel COMET-sensitivity-guided mixed-precision scheme that assigns each of the 48 decoder layers an independent BF16/INT8/INT4 tier via a size-budget knapsack. Quality is measured with chrF and three COMET variants. Speed is full-process wall-clock time (including model load) on a held-out blind test set, and size is the on-disk safetensors footprint. Our mixed-precision submission achieves the best quality-size-speed trade-off in our pool: it is the fastest system measured (1.76 sentences/sec, batch 16), within 0.4 chrF and a few COMET points of the uncompressed baseline across all three language pairs, at roughly 40\% of the baseline's on-disk size (9.15GB vs 23GB).},
  url       = {https://aclanthology.org/2026.wmt-1.133}
}

Author{1}{Orcid}:
@InProceedings{balaga-EtAl:2026:wmt,
  author    = {Balaga, Havish  and  Racherla, Anish  and  Kiran, Kolupoti Navadeep  and  Yadav, Saumitra  and  Shrivastava, Manish},
  title     = {Importance-Guided Structural Pruning of Aya Expanse 8B and Gemma-3-12B for English–Chinese Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1986--1992},
  abstract  = {Large language models have significantly improved machine translation quality but remain expensive to deploy due to high computational and memory requirements. This paper describes our submissions to the WMT 2026 Model Compression Shared Task, targeting English–Chinese translation through structural compression of two multilingual models: Aya Expanse 8B and Gemma-3-12B. Both pipelines share a common 6-stage data curation process that filters 25 million raw WMT English–Chinese sentence pairs into a quality and diverse training corpus for fine-tuning. For Aya Expanse 8B, we use COMET-QE to identify and prune four layers with the least impact on translation quality, followed by QLoRA fine-tuning and INT8 inference. For Gemma-3-12B, we apply Fisher Information-based importance scoring to guide a two-step compression, pruning 8 transformer layers followed by feed-forward neuron pruning, recovered through QLoRA fine-tuning and INT4 inference. Results on the FLORES-200 benchmark show both pipelines achieve substantial reductions in GPU memory usage and gains in inference throughput while maintaining competitive translation quality. Our findings demonstrate that importance-guided structural pruning, combined with parameter-efficient fine-tuning and quantization, can achieve substantial model size reduction while maintaining comparable English–Chinese translation quality.},
  url       = {https://aclanthology.org/2026.wmt-1.134}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:https://orcid.org/0000-0001-8705-6637
@InProceedings{fakhrutdinov:2026:wmt,
  author    = {Fakhrutdinov, Nail},
  title     = {Pare4Bit: Quantization, Depth Pruning, and Model Selection for the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1993--1997},
  abstract  = {In the constrained track we compress google/gemma-3-12b-it with a data-light pipeline—vision-tower removal followed by GPTQ W4A16 quantization - reaching 43.2 mean chrF at 7.1 GB, a 3.4× size reduction for a 0.3 chrF drop. We further explore depth pruning with LoRA knowledge-distillation healing, which recovers a collapsed pruned model to near-parity and adds an engine-agnostic speedup. In the unconstrained track we treat model selection as a first-class compression decision: the newer Gemma-4-12B, deployed via quantization-aware-trained INT4, outperforms the constrained Gemma-3 baseline by +2.5 mean chrF.},
  url       = {https://aclanthology.org/2026.wmt-1.135}
}

Author{1}{Orcid}:
@InProceedings{gowda:2026:wmt,
  author    = {Gowda, Thamme},
  title     = {Quantize, Qualify, Rerank: A Recipe for Compressing LLMs Without Losing Quality},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1998--2005},
  abstract  = {Compressing large language models lowers serving cost, but a single automatic metric may miss quality loss. For the WMT26 Model Compression constrained track, we compress `google/gemma-3-12b-it` (24.4 GB in bf16) for three translation directions on one H100. Paired bootstrap analysis exposes metric disagreement on 4-bit quantization; guided by WMT24 human meta-evaluation, we use MetricX24-XXL as primary and CometKiwi-XXL as a cross-check. Removing the unused vision stack and non-task vocabulary and using int8 weights or fp8 activations causes no measurable loss, whereas int4 leaves a small residual. QLoRA healing overfits its calibration domain. A cheap Cometoid best-of-4 reranker at `T=0.3` improves both the 8-bit and 4-bit systems under held-out CometKiwi-XXL and reference-based MetricX24-XXL. Both surpass greedy bf16 on CometKiwi-XXL; the 8-bit system also surpasses it on MetricX24-XXL, while the 4-bit system nearly matches it.},
  url       = {https://aclanthology.org/2026.wmt-1.136}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-5422-8674
@InProceedings{kronis-EtAl:2026:wmt,
  author    = {Kronis, Martins  and  Bergmanis, Toms  and  Pretkalniņš, Ingus Jānis  and  Pinnis, Marcis},
  title     = {WMT26 Model Compression Task: Quantising and Distilling TildeOpen},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2006--2015},
  abstract  = {This paper describes Tilde's submission to the unconstrained track of the WMT~2026 Model Compression task in the Czech-German direction. Our baseline is TildeOpen-15B-64k, a model distilled from the TildeOpen-30B-64k European foundation model. We adapt it for CS-DE translation, using supervised fine-tuning followed by GRPO-based reinforcement learning. From this baseline, we explore two axes of compression: post-training quantisation and knowledge distillation. For quantisation, we use LLM Compressor to produce a GPTQ-calibrated model with 4-bit NVFP4 weights (10.1\,GB, $\times$3 smaller than the 30.3\,GB footprint of the 15B baseline) and a data-free RTN model with FP8 weights and activations (16.3\,GB). In parallel, we prune and distil the base 15B model into an 8B student and fine-tune it with the same supervised fine-tuning recipe. Applying the two quantisation schemes to the 8B model yields our smallest models at 8.9\,GB and 5.7\,GB. The latter is over $\times$5 more compact and $\times$1.8 faster than the 15B baseline. Evaluation on four test sets across a variety of metrics shows that quantisation and distillation have minimal impact on translation quality. The NVFP4-quantised 15B model offers the best quality/size/speed trade-off and is our primary submission for this competition. All six models are publicly released.},
  url       = {https://aclanthology.org/2026.wmt-1.137}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0001-6832-5600
@InProceedings{martin-bandarkar-peng:2026:wmt,
  author    = {Martin, Liu O.  and  Bandarkar, Lucas  and  Peng, Nanyun},
  title     = {ESTS at WMT26: Routing-Informed Expert Pruning for Model Compression},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2016--2023},
  abstract  = {We describe six submissions under the team name ESTS to the unconstrained WMT26 Model Compression Shared Task for English--Simplified Chinese and English--Egyptian Arabic. We submit three compression operating points per translation direction, all derived from GPT-OSS-20B. We use task-specific routing mass to rank experts and cross-lingual routing divergence to allocate retained capacity across layers, then physically remove low-importance experts. The resulting specialists are recovery-tuned on GPT-5.1-generated synthetic translation data and further compressed by applying MXFP4 quantization to the retained expert projection weights. We additionally implement a robust inference system for the instruction-conditioned WMT26 setting, including category inference, output validation, retries, segmented fallback, and source-owned JSON reconstruction. Across our six submissions, parameter counts range from 4.186B to 7.770B and packed artifact sizes from 4.55 to 6.33~GiB. Internal xCOMET-XL evaluation using GPT-5.1 pseudo-references provides an internal comparison across the submitted compression operating points.},
  url       = {https://aclanthology.org/2026.wmt-1.138}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{ono:2026:wmt,
  author    = {Ono, Nobutaka},
  title     = {TMU-onono at WMT26: DiBA-Based Model Compression for Czech-to-German Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2024--2031},
  abstract  = {We describe TMU-onono's Czech-to-German systems submitted to the constrained track of the WMT26 Model Compression shared task. We apply DiBA (Diagonal and Binary Matrix Approximation) (Ono, 2026), which approximates a dense matrix $A$ as $D_1B_1D_2B_2D_3$, where $D_1$, $D_2$, and $D_3$ are real diagonal matrices and $B_1$ and $B_2$ are 0/1 binary matrices. We replaced 337 Gemma 3 12B weight matrices, including the tied embedding/lm-head, and retuned only diagonal entries on translations generated by the original model. Each 2.51 GB system package excludes base-model files used during setup and provides 9.72× compression relative to the original 24.37 GB BF16 weights. The primary diba-triton\_direct directly applies bitpacked factors for low-memory inference; the contrastive diba-cached\_unpacked uses unpacked caches for higher throughput. Both were slower than the original. To avoid out-of-memory errors, we used relatively short sequences for retuning. The submitted systems also limited inputs to 384 tokens including the prompt and outputs to 128 new tokens. Primary achieved 17.73 BLEU and 48.44 chrF on 128 held-out short segments, versus 8.25 and 32.16 on 256 WMT25 development paragraphs. Relaxing inference limits alone did not consistently improve quality, suggesting that short-sequence retuning may have limited performance on longer translation units.},
  url       = {https://aclanthology.org/2026.wmt-1.139}
}

Author{1}{Orcid}:
@InProceedings{hettich-okabe-fraser:2026:wmt,
  author    = {Hettich, Niklas  and  Okabe, Shu  and  Fraser, Alexander},
  title     = {SNAPP: Segment-Level Neural Alignment Post-Processing for Low-Resource Parallel Sentence Mining},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {246--260},
  abstract  = {Machine Translation (MT) for low-resource languages has been challenging so far, mainly due to the small size or lack of parallel sentences. Parallel sentence mining and filtering are two automatic approaches which aim to identify and curate such resources. Yet, they usually rely on multilingual representation as a backend, which is known to be of poorer quality for low-resource language pairs, leading to the introduction of noisy pairs. We devise SNAPP, a post-processing pipeline which combines a segment-level filtering technique with an unsupervised neural aligner to remove highly similar but non-parallel sentence pairs. Modular components further enable more specific adaptation to the language pair under consideration. We evaluate filtering performance on three language pairs of varying typological distance. Finally, we show improvement on the downstream MT performance with our post-processing pipeline for the two harder language pairs.},
  url       = {https://aclanthology.org/2026.wmt-1.14}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0009-0003-0169-3689
Author{3}{Orcid}:
@InProceedings{palomino:2026:wmt,
  author    = {Palomino, Alonso},
  title     = {Layer-Aware Native Quantization for Compact Machine Translation: A WMT26 English-to-Chinese Submission},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2032--2036},
  abstract  = {This paper describes a submission to the constrained English-to-Simplified-Chinese track of the WMT26 Model Compression shared task. Starting from Gemma~3 12B, the system applies native BitsAndBytes NF4 with double quantization to the 144 gate, up, and down projections in the model's 48 text-transformer MLP blocks. Attention projections, embeddings, normalization layers, the tied language-model head, and vision modules remain in BF16. The method requires no calibration or fine-tuning data and produces a mixed checkpoint whose quantized weights remain low-bit at inference. On the official 945-segment blind set, the system scores 0.7669 CometKiwi-XXL and 2.7836 MetricX-24-XXL, slightly improving on the organizer global-q4 baseline's 0.7665 and 2.8147, respectively. Its 11.8~GB artifact is 48.4\% of BF16 size, and H100 throughput is essentially tied with global q4 (866.9 versus 867.9 source characters/s). On 332 reference-backed WMT25 segments, its chrF and BLEU gains over global q4 are significant under paired bootstrap resampling. A transferred selected-layer q3+Zstd codec reaches statistically indistinguishable local quality but expands to BF16-like memory after decoding.},
  url       = {https://aclanthology.org/2026.wmt-1.140}
}

Author{1}{Orcid}:
@InProceedings{rayarios-EtAl:2026:wmt,
  author    = {Raya-Rios, Vania  and  Klimchuk, Aleksandr  and  Cuadrado Avila, Nicolas M.  and  Gollini Navarrete, Ivo  and  Horvath, Samuel  and  Takác, Martin},
  title     = {Tiny Titans at WMT26: Task-Calibrated Quantization and Activation-Aware Low-Rank Compression of Gemma 3 12B},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2037--2044},
  abstract  = {We present the TinyTitans submissions for the WMT26 Model Compression shared task, which focuses on Czech-to-German translation. Starting with the instruction-tuned Gemma 3 12B model, we compare two approaches: 4-bit GPTQ, which uses task-specific calibration examples, and a more aggressive low-rank factorization method that replaces attention and feed-forward projections with truncated Singular Value Decomposition (SVD) factors. On a 300-sentence development set, the calibrated GPTQ retains 99.7\% of the dense model's COMET score while reducing GPU memory usage from 26.6 GiB to 17.6 GiB. The recovered SVD model reduces the number of parameters by 46.6\% and reaches 0.7804 COMET in a separate evaluation using 1,997 sentence pairs. In an external evaluation using 457 WMT25 paragraphs, the GPTQ, SVD, and SVD-plus-quantization submissions received COMET scores of 0.5751, 0.4263, and 0.3354, respectively. These findings suggest that while post-training quantization effectively reduces memory usage, combining low-rank truncation with low-bit quantization may increase approximation errors without clear efficiency gains.},
  url       = {https://aclanthology.org/2026.wmt-1.141}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{ponce-etchegoyhen:2026:wmt,
  author    = {Ponce, David  and  Etchegoyhen, Thierry},
  title     = {Vicomtech@WMT 2026: Mask-based Pruning for Model Compression},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2045--2057},
  abstract  = {We describe Vicomtech's participation in the constrained track of the WMT 2026 Shared Task on Model Compression. We addressed all three translation directions of the task, namely Czech to German, English to Simplified Chinese, and English to Egyptian Arabic, using Gemma 3 12B IT as our base model. Our approach performs task-aware structured post-training pruning, which learns differentiable masks to prune model components using machine translation calibration data and an objective combining next-token cross-entropy, knowledge distillation, and intermediate representation matching. We systematically evaluated pruning at two target compression ratios (0.25 and 0.50) and explored different structured pruning configurations, combined with supervised fine-tuning to recover performance after pruning. Additionally, we explored language-specific vocabulary pruning and 4-bit AWQ quantization to achieve further reductions in model size and memory requirements. Our experimental results demonstrate that the combination of structured pruning, targeted post-training, vocabulary reduction, and quantization can achieve substantial model compression while maintaining competitive translation quality across all language directions.},
  url       = {https://aclanthology.org/2026.wmt-1.142}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{sofianopoulos-prokopidis:2026:wmt,
  author    = {Sofianopoulos, Sokratis  and  Prokopidis, Prokopis},
  title     = {The ILSP/ARC submission to the WMT26 Model Compression Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2058--2067},
  abstract  = {This paper describes our submission to the constrained track of the WMT26 Model Compression task, filed under the team name arc/ilsp. The task requires the compression of google/gemma-3-12b-it for translation in cs→de, en→zh, and en→ar in the Egyptian register. Our systems combine the removal of the vision tower, the pruning of the vocabulary by Unicode script, and GPTQ W4A16 quantization. The two checkpoints we file as primary models occupy 6.64 and 7.08 GiB, i.e. 29\% and 31\% respectively of the 22.7 GiB disk footprint of the bf16 original. No interval separates the pruned vocabulary from the full-vocabulary INT4 checkpoint on any direction, and at the paragraph granularity on which the task scores, our vocab-INT4 primary costs 0.0048 COMET against bf16 on cs→de and nothing measurable on the other two directions. We also submit two contrastive systems: an FP8 system that is faster than the bf16 original at lower memory, and a self-MBR decoder that adds roughly 0.009 COMET on en→ar without any additional parameters. We do not submit a depthpruned system; a default generation cap had clipped the distillation targets and taught the student to end a turn mid-paragraph, a defect whose repair recovers a quarter of the deficit against full depth.},
  url       = {https://aclanthology.org/2026.wmt-1.143}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-7305-5289
Author{2}{Orcid}:https://orcid.org/0000-0003-3963-3000
@InProceedings{suman-bentivogli:2026:wmt,
  author    = {Suman, Dhairya  and  Bentivogli, Luisa},
  title     = {FBK's Submission to WMT26 Model Compression Task: Improving Inference Speed for Quantized Models},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2068--2073},
  abstract  = {This paper presents FBK's submission to the WMT26 Shared Task on Model Compression. The task requires compressing the Gemma 3 12B model for machine translation on the Czech-German language pair under the task's constrained track. Post-training quantization (PTQ) methods such as GPTQ have become an industry standard for model compression, but in their standard, weight-only form they reduce only the on-disk size of the model; leaving the activations are in full-precision. As part of this submission, we quantize both weights and activations to 8-bits (W8A8), aiming to also improve inference speed in addition to memory savings, while preserving translation quality. We submit two systems: both quantizing activations d using round-to-nearest quantization, one of which additionally applies SmoothQuant's channel-wise smoothing before quantization. Our experiments demonstrate that both the W8A8 systems increase throughput by more than 20\% while only slightly reducing in quality relative to the full precision model.},
  url       = {https://aclanthology.org/2026.wmt-1.144}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{a-EtAl:2026:wmt,
  author    = {A, VISHNURAJ K.  and  Bodana, Yuvrajsinh D.  and  Hingrajiya, Heli Hitesh bhai  and  Dasari, Priyanka  and  Krishnamurthy, Parameswari},
  title     = {LTRC\_Plural: Better Data and Better Rewards for Low-Resource Indic MT},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2074--2084},
  abstract  = {We present our submission to the WMT 2026 Shared Task on Low-Resource Indic Language Translation, covering four language pairs (English-Assamese, English-Mizo, English-Khasi, and English-Tagin in both directions) spanning a wide range of resource levels, from moderately-resourced Assamese down to Tagin, which barely appears in existing multilingual MT systems at all. We compare three models (NLLB-200-1.3B, Sarvam-Translate, and TranslateGemma) across three setups: fine-tuning on the official shared-task data alone, fine-tuning with added external parallel data, and a reinforcement-learning stage (RLOO) that optimizes a combined BLEU/chrF++ reward on top of the fine-tuned model. Adding external parallel data helps most for Khasi, improving English-to-Khasi by 18 BLEU, and for Mizo, where our system places first among contrastive submissions in both directions; the same augmentation leaves Assamese essentially unchanged despite contributing twice as many sentence pairs, suggesting that data provenance matters more than volume at these scales. The RLOO system degrades performance on Assamese and Mizo, where supervised fine-tuning had already produced competitive systems, but improves over the primary system on both Khasi directions and on every metric for English-to-Tagin, where it obtains the highest METEOR score of any contrastive submission. Reward-based training therefore appears most useful precisely where supervised fine-tuning has least to work with.},
  url       = {https://aclanthology.org/2026.wmt-1.145}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{aditya-ekbal:2026:wmt,
  author    = {Aditya, Aditya  and  Ekbal, Asif},
  title     = {ADI-IITP: Reward-Guided Preference Optimization for Low-Resource English-Assamese Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2085--2092},
  abstract  = {This paper describes our submission to the WMT 2026 shared task on Low-Resource Indic Language Translation for the English-Assamese pair. Our systems build on the compact, distilled 200M-parameter IndicTrans2 model, adapted with parameter-efficient Low-Rank Adaptation (LoRA). Beyond standard supervised fine-tuning (SFT), we built a preference-optimization pipeline that generates candidate translations, scores them with a composite reward combining a GEMBA-style LLM-as-judge score, a reference-free CometKiwi quality-estimation signal, and an xCOMET score, and then trains with Direct Preference Optimization (DPO) on the resulting preference pairs. We used this pipeline to guide our system choices: for Assamese-to-English the SFT-only model scored best on our development set, so we submitted it as primary and the SFT-then-DPO model as a contrastive system; for English-to-Assamese we went with SFT-then-DPO as primary. On the official WMT 2026 blind test set, the Assamese-to-English primary system scored 24.08 BLEU (25.11 for the contrastive system), and the English-to-Assamese primary system scored 15.57 BLEU. We also report METEOR, TER, chrF++, BERTScore, and COMET.},
  url       = {https://aclanthology.org/2026.wmt-1.146}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0003-3612-8834
@InProceedings{arora-EtAl:2026:wmt1,
  author    = {Arora, Palak  and  Sharma, Adhikarimayum Meerajita  and  Indoria, Mrityunjaya  and  Lhoungu, Keneiwenuo  and  Lalthafamkimi, Lalthafamkimi  and  Nathani, Bharti  and  Joshi, Nisheeth},
  title     = {BVSLP: Enhancing Machine Translation through NMT Fine-Tuning, Back-Translation, and LLM-Assisted Synthetic Data Generation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2093--2100},
  abstract  = {This paper presents a two-stage neural machine translation framework for English and selected low-resource Northeastern Indian languages. The approach incorporated transfer learning, bidirectional fine-tuning, back-translation, and LLM-assisted synthetic data augmentation. In Stage 1, the NMT model was fine-tuned on cleaned gold-standard parallel data. In Stage 2, additional synthetic sentence pairs were generated using back-translation and Qwen2.5-Instruct, followed by filtering based on language identification, length ratio, duplicate removal, named-entity preservation, semantic similarity, and round-trip consistency. The framework was evaluated on English–Assamese, English–Mizo, English–Manipuri, English–Bodo, English–Khasi, and Nagamese–English translation directions. Results show stronger performance for several Northeastern-language-to-English directions, while the contrastive system notably improved Manipuri (Bengali)–English translation from 26.56 to 30.59 BLEU. The findings indicate that controlled synthetic augmentation can improve low-resource MT, although its effectiveness remains language- and direction-dependent.},
  url       = {https://aclanthology.org/2026.wmt-1.147}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
@InProceedings{devi-lee:2026:wmt,
  author    = {Devi, Laishram Thoibisana  and  Lee, Grace},
  title     = {EROL: Transformer-Based Multilingual Translation for Northeastern Indian Language Pairs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2101--2107},
  abstract  = {We present EROL, a multilingual neural machine translation system for four English–Indic language pairs: English↔Assamese, English↔Bodo, English↔Manipuri (Bengali script), and English↔Manipuri (Meitei Mayek). Our systems are built on the Transformer-Base architecture with 6 encoder and 6 decoder layers, a 512-dimensional model size, 2048-dimensional feed-forward networks, and 8 attention heads. We employ shared source and target embeddings, GELU activation, layer normalization, dropout regularization, and byte-pair encoding (BPE) tokenization. The models are trained using the standard sequence-to-sequence cross-entropy objective, providing a simple yet effective baseline for multilingual translation in low-resource settings. Experimental results demonstrate competitive performance across all translation directions, achieving BLEU scores of 25.10 for Assamese→English, 23.50 for Bodo→English, 22.47 for Manipuri (Bengali)→English, 24.71 for Manipuri (Meitei Mayek)→English, 15.31 for English→Assamese, 14.25 for English→Bodo, 10.21 for English→Manipuri (Bengali), and 3.96 for English→Manipuri (Meitei Mayek).},
  url       = {https://aclanthology.org/2026.wmt-1.148}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{divyasree-gupta:2026:wmt,
  author    = {Divya Sree, Kuruva  and  Gupta, Deepa},
  title     = {NLP-MT\_Amrita: A RL-Inspired GRPO based Agentic RAG Framework for Manipuri and Bodo},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2108--2117},
  abstract  = {Low-resource Indic machine translation (MT) remains challenging due to limited parallel corpora, linguistic diversity, and contextual ambiguities that often result in inaccurate translations and hallucinations. This study proposes AgenticRAG-GRPO, an RL-inspired retrieval-augmented translation framework that combines semantic retrieval with GRPO-inspired reward-guided candidate selection for context-aware English↔Indic MT. The framework dynamically retrieves relevant translation contexts and ranks candidate translations using a reward function based on SacreBLEU and BERT score. Experiments were conducted on the WMT 2026 Low-Resource IndicMT Shared Task for English↔Manipuri (Mni) and English↔Bodo (Brx) using IndicTrans2, NLLB-200 Distilled 600M, and Sarvam-Translate. The proposed study achieved s performance, obtaining BLEU scores up to 32.36, METEOR up to 67.01, BERTScore up to 95.04, and COMET up to 80.51 across the languages, while improving semantic faithfulness and reducing hallucinations through retrieval-guided generation. The framework also maintained inference efficiency without requiring task-specific model fine-tuning. These results demonstrate that AgenticRAG-GRPO provides an effective and scalable solution for low-resource English-Indic MT.},
  url       = {https://aclanthology.org/2026.wmt-1.149}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{hoang-axelrod-post:2026:wmt,
  author    = {Hoang, Hieu  and  Axelrod, Amittai  and  Post, Matt},
  title     = {Dynamic Lagging using Stable-Prefix Training for Simultaneous Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {261--275},
  abstract  = {In streaming simultaneous speech translation, the speech translation system is trained to learn a read-write policy that alternates between consuming source words and generating target ones. In a cascaded setting, the output from the speech recognizer is passed to a separate machine translation component, making it more difficult to learn such a policy. Approximations such as fixed wait-k strategies or target-suffix deletion can be employed, but these approaches do not provide the model with a streaming system's flexibility to make contextual read-write decisions. This paper presents a training strategy for a cascaded machine translation system that enables it to dynamically decide how much of the growing source prefix to translate. We achieve this by fine-tuning a large language model (Qwen3-8B) on stable prefixes of the training data, which are produced by pairing every source sentence prefix in the training data with the longest translation of that prefix that is shared with the full source sentence translation. We fine-tune variants of the model on different subsets of the prefixes and compare against wait-k and target-suffix deletion. We also investigate the effect of fine-tuning the target-token generation confidence. Our experiments show that stable prefixes improve the quality-latency tradeoff when translating from English into German, Japanese, and Chinese across a range of test sets.},
  url       = {https://aclanthology.org/2026.wmt-1.15}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-1297-6794
@InProceedings{dhawan-EtAl:2026:wmt,
  author    = {Dhawan, Aashish  and  Driggers-Ellis, Christopher William  and  Kasinets, Dzmitry  and  Grant, Christan  and  Wang, Zhe},
  title     = {gators: BM25-Augmented Many-Shot Translation for Low-Resource North-Eastern Indian Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2118--2130},
  abstract  = {This paper describes the University of Florida gators submission to the WMT26 Low-Resource Indic Language Translation shared task. We adapt the retrieval-augmented many-shot translation pipeline from our AmericasNLP 2026 system to translate between English and eleven North-Eastern Indian languages in both directions. At inference time, BM25 retrieves the most similar parallel examples from a language-specific training bank, and Gemini 2.5 Flash translates the input conditioned on these examples. No model fine-tuning is involved. Training banks combine official WMT26 data with publicly available corpora such as Samanantar and prior WMT shared task releases. A grid search over retrieval count r and development exemplar count d across all 22 language-direction pairs selects the best configuration for each submission.},
  url       = {https://aclanthology.org/2026.wmt-1.150}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-6684-3620
Author{5}{Orcid}:
@InProceedings{kachi-akiba-tsukada:2026:wmt,
  author    = {Kachi, Takumi  and  Akiba, Tomoyosi  and  Tsukada, Hajime},
  title     = {AkibaNLP-TUT: Improving Low-Resource Machine Translation via Language-Specific and Length-Adaptive Noise Injection},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2131--2141},
  abstract  = {We present a language-specific word-level noise injection method for low-resource machine translation that eliminates the need for an external monolingual corpus by constructing the low-resource language vocabulary directly from the source-side text of the target dataset. We also investigate two important hyperparameters of the original method: the maximum edit distance used for candidate word selection and the target noise ratio. Specifically, we propose a dynamic edit-distance constraint based on word length and evaluate multiple noise ratios from 5\% to 30\%. Experiments on Assamese-English translation show that the proposed dynamic constraint consistently outperforms the fixed edit-distance setting across all evaluated noise ratios, while a moderate noise ratio achieves the best translation performance. We further report our official results for the WMT 2026 Low-Resource Indic Language Translation Shared Task and discuss the effectiveness and limitations of the proposed approach under different resource conditions.},
  url       = {https://aclanthology.org/2026.wmt-1.151}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-7917-1843
@InProceedings{hb-EtAl:2026:wmt,
  author    = {HB, Barathi Ganesh  and  Ptaszynski, Michal  and  Sharma, Meenakshi  and  R, Jairam},
  title     = {TIP-RBG-AI: Overcoming Orthographic Discrepancies via Algorithmic Translinear Pipelines and Phylogenetic Script Mapping},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2142--2152},
  abstract  = {This paper details the submission of team TIP- RBG-AI for the WMT26 Indic-MT shared task. Addressing severe data sparsity, unstandardized orthographies, and script-level discrepancies across ten Indic languages, we introduce a robust, resource-agnostic cross-lingual framework. Rather than relying on standard downstream parameter fine-tuning, our methodology optimizes zero-shot inference topologies across three massively multilingual machine translation architectures: NLLB-200, MADLAD-400, and IndicTrans2. To systematically resolve vocabulary mismatch and tokenizer fragmentation, we implement a three-tiered preprocessing pipeline comprising baseline untuned direct inference for native scripts, translinear normalization via algorithmic back-transliteration for Romanized textual representations, and phylogenetic script mapping for undocumented, scriptless vernaculars. Empirical evaluations demonstrate that structurally aligning orthographic decoding environments and exploiting cross-lingual family networks substantially enriches semantic fidelity. Without executing a single downstream parameter update, our purely zero-shot framework secured multiple podium placements, including second-place finishes in the Bodo and Manipuri (Meitei Mayek) to English tracks, validating the potential of frontend representation alignment for moderate low-resource translation, while also exposing clear limits at the most extreme end of data scarcity. The code used for reproducing the experiments is publicly available at https: //github.com/rbg-research/EMNLP-2026.},
  url       = {https://aclanthology.org/2026.wmt-1.152}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-1150-2773
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{hede-EtAl:2026:wmt1,
  author    = {Hede, Vedarth  and  Gupta, Neha  and  Bapat, Harish  and  Ekbote, Harsh},
  title     = {MTG-LIRA: Scaling Neural Machine Translation for Low-Resource Indian Languages through Multilingual Learning and Synthetic Data},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2153--2160},
  abstract  = {MTG-LIRA's submission for the WMT 2026 Low-Resource Indic Language Translation Shared Task presents a unified, scalable multilingual Transformer architecture developed using OpenNMT-py to address severe data scarcity across several low-resource Indian languages. The framework partitions languages into family-based groupings—such as Indo-Aryan and Tibeto-Burman—to optimize cross-lingual knowledge sharing and vocabulary overlap using Byte Pair Encoding (BPE). Data preparation integrates official WMT datasets, the BPCC multilingual corpus, and ~5 million synthetic English–Mizo sentence pairs generated via Meta's NLLB model, all cleaned using a rigorous preprocessing pipeline. Experimental results across task directions demonstrate that combining multilingual base models with targeted, language-specific adaptation strategies yields significant gains: fine-tuning on domain-specific data improved BLEU scores for English→Assamese (12.17 to 14.44) and English→Mizo (16.2 to 22.72); checkpoint averaging of the 5 best models enhanced stability for English↔Manipuri (Meitei Mayek); a weighted 3:1 checkpoint averaging strategy improved performance for English→Bodo; and zero-shot style direct transfer proved effective for Assamese→English and English→Manipuri (Bengali script) due to shared script features.},
  url       = {https://aclanthology.org/2026.wmt-1.153}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{huidrom-EtAl:2026:wmt2,
  author    = {Huidrom, Rudali  and  Kumar, Vikas  and  Pangsatabam, Hoomexsun  and  Das, Pinaki  and  Khanganba, K. Kabi  and  Zeno, Nongmaithem  and  KHUMUKCHAM, NICHOLAS  and  Okram, Mangalton  and  Konjengbam, Justice  and  Jamalpoor, Sai Varun  and  Khangembam, Alex D. Nelson  and  Konjengbam, Anand  and  Goyal, Vikram},
  title     = {PANINI: Improving Low-Resource English-Manipuri Translation in Two Scripts via rsLoRA and Three-Stream Self-Augmentation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2161--2169},
  abstract  = {We describe team PANINI's submission to the WMT 2026 Shared Task on Low-Resource Indic Language Translation. We submit primary systems for English↔Manipuri (Meiteilon) in both the Bengali (mni Beng) and Meitei Mayek (mni Mtei) scripts, covering all four directions. Rather than one shared multilingual adapter, we train four independent rank-stabilized LoRA (rsLoRA) adapters over IndicTrans2-1B, one per direction. Each adapter is trained on the authentic bitext plus three filtered synthetic streams the model generates for itself: back-translation, forward translation, and iterative pseudo-labelling, with subword regularisation and a script-aware processing pipeline. Our system obtains a COMET score of 92.31 and a BERTScore of 98.85 on English↔Manipuri (Meitei Mayek), the best in the direction. On the into-English directions, it reaches a COMET of 82.90 for mni Beng↔en and 80.20 for mniMtei↔en. These results show that per-direction adapters combined with self-generated synthetic data are effective even when authentic bitext is limited to a few thousand sentence pairs, and that a script-aware processing pipeline is necessary to produce well-formed output in both Manipuri scripts. The code and resources are released on GitHub.},
  url       = {https://aclanthology.org/2026.wmt-1.154}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
Author{10}{Orcid}:
Author{11}{Orcid}:
Author{12}{Orcid}:
Author{13}{Orcid}:
@InProceedings{j-EtAl:2026:wmt,
  author    = {J, Siva Bhavani  and  Kankanwadi, Daneshwari  and  Gugulothu, Abhinav  and  Paul, Biswajit},
  title     = {ANVITA : Machine Translation System for Low-resource Indian Languages - Khasi, Nagamese and Tagin},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2170--2179},
  abstract  = {In this paper, we present ANVITA machine translation system submitted to WMT 2026 shared task on Low-Resource Indic Language Translation, where the team participated for six translation directions: {Khasi, Nagamese, and Tagin} ↔ English. Two core ideas shaped the design of ANVITA system. Firstly, transfer learning by symmetric fine-tuning of T5-base model and enhanced cross-lingual transfer from related languages (Assamese to Nagamese) by mitigating script mismatch through transliteration. Secondly, to alleviate data scarcity, primary submissions are trained on datasets augmented with related language corpora and paraphrased pairs (English sentences), utilizing only the organizers provided datasets. For the contrastive submissions, training corpora are further expanded by compiling synthetic parallel data and related language data followed by corpora distillation with suitable selection strategies. ANVITA submissions are evaluated on the official test sets using multiple metrics which include BLEU, METEOR, TER, CHRF++, BERT score, and COMET. In the primary submissions, ANVITA achieved first rank for English→Tagin translation. For the contrastive submissions, ANVITA secured first rank for English → Khasi and Khasi → English with BLEU scores of 32.89 and 26.48 respectively. Furthermore, the contrastive submissions for Nagamese ↔ English and Tagin → English ranked second, underscoring efficacy of the data augmentation strategies applied.},
  url       = {https://aclanthology.org/2026.wmt-1.155}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{maiti-muley-sahoo:2026:wmt1,
  author    = {Maiti, Agniva  and  Muley, Aarsh  and  Sahoo, Sovan Kumar},
  title     = {SCE-KIIT: Fine-Tuning NLLB-200 for English-Nagamese Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2180--2189},
  abstract  = {Nagamese is a critically under-resourced Assamese-based creole language spoken in Nagaland, India, serving as the primary lingua franca for over two million speakers across 16 Naga tribal communities. Written in the Latin script, it is entirely absent from all modern massively multilingual machine translation (MT) systems. We describe the SCE-KIIT submission to the WMT 2026 Shared Task on Low-Resource Indic Language Translation (Category 2: English-Nagamese), for which we submitted a primary EN-NAG run. Our system builds on Facebook's NLLB-200 distilled with 600M parameters using a two-phase surrogate-token strategy: we fine-tune with Assamese (asm\_Beng) as the target-side language prefix, then force asm\_Latn at inference time. This token is absent from NLLB's vocabulary and resolves to the unknown token (UNK); we hypothesise that seeding generation with an out-of-vocabulary token leaves the decoder's script preference unconstrained, allowing the Romanised output learned during fine-tuning to surface. On our 228-sentence held-out test split, the system achieves SacreBLEU 40.02, chrF 53.61, and COMET 0.7212 (wmt22-comet-da), against a zero-shot baseline of 0.40 BLEU and 0.4487 COMET: an absolute gain of 39.62 BLEU. On the official shared-task test set, which is out-of-domain news text, the same system scores 12.51 BLEU and 43.81 chrF++; we analyse this domain gap in detail. The model and corpus are publicly released at https://huggingface.co/agnivamaiti/nllb-200-en-nagamese.},
  url       = {https://aclanthology.org/2026.wmt-1.156}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{maiti-muley-sahoo:2026:wmt2,
  author    = {Maiti, Agniva  and  Muley, Aarsh  and  Sahoo, Sovan Kumar},
  title     = {SCE-KIIT: KokLLaMA: Cross-Task Adaptation for Low-Resource English--Kokborok Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2190--2199},
  abstract  = {Kokborok is a critically under-resourced Tibeto-Burman language spoken by over one million people primarily in Tripura, India, and is absent from modern massively multilingual machine translation (MT) systems. We describe the SCE-KIIT submission to the WMT 2026 Shared Task on Low-Resource Indic Language Translation (Category 2: English-Kokborok). We ask whether a large language model can be adapted to translate a language using no parallel supervision at all. Our system, KokLLaMA-3.2-3B-Instruct, fine-tunes Llama-3.2-3B-Instruct via QLoRA purely on Kokborok conversational instruction data, and recovers translation behaviour at inference time through structured prompting and a rule-based post-processing pipeline. On the official WMT 2026 test set it obtains 5.11 BLEU / 27.09 chrF++ (EN->TRP) and 3.58 BLEU / 26.14 chrF++ (TRP->EN), placing third of four and third of five submitted primary systems, and coming within 1.9 BLEU of the best WMT 2025 system for this pair despite using no parallel data. The approach is substantially more effective at generating Kokborok than at generating English, and we analyse why. We further show that our in-domain validation split under-estimated official performance by more than an order of magnitude, a cautionary result for participants who tune on held-out slices of provided training corpora, and we characterise a pathological Translation Edit Rate (TER) failure mode in which an untrained baseline scores well by silently under-generating.},
  url       = {https://aclanthology.org/2026.wmt-1.157}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{nyalang-tekcham-debbarma:2026:wmt,
  author    = {Nyalang, Badal  and  Tekcham, Riya  and  Debbarma, Biman},
  title     = {KrenTransV0.1: Multi-Model Routing and Reranking for Low-Resource Northeast Indian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2200--2205},
  abstract  = {KrenTransV0-1 is MWire Labs' submission to the WMT 2026 Low-Resource Indic MT Shared Task. We achieve \#1 primary rank on both English→Khasi and English→Kokborok. The system covers all 11 Northeast Indian language pairs using a hybrid architecture combining fine-tuned IndicTrans2 for Indic-script languages, NLLB-200 1.3B mono fine-tuned models for Latin-script languages, and M2M-100 418M fine-tuned models for three severely data-starved languages that exhibited complete output collapse under NLLB. All submissions are primary, covering the English-to-target direction only. We additionally report a negative reranking result: neither BharatGen PARAM-1 (2.9B) perplexity reranking nor NE-BERT fluency reranking improved over a beam-search baseline, and all final submissions use the baseline decoding configuration.},
  url       = {https://aclanthology.org/2026.wmt-1.158}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{oinam-saharia:2026:wmt,
  author    = {Oinam, Dingku Singh  and  Saharia, Navanath},
  title     = {DELAB-IIITM WMT26: Full Fine-Tuning and LoRA-Based Adaptation for English-to-Low-Resource Indic Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2206--2213},
  abstract  = {This paper describes DELAB-IIITM's submission system for the WMT26 machine translation shared task. We participated in the English-to-low-resource Indic translation direction, covering four language pairs: English to Assamese (Bengali), English to Manipuri (Bengali), English to Manipuri (Meitei Mayek) and English to Bodo (Devanagari). Our fine-tuning process leverages two pre-trained multilingual models: NLLB-200-Distilled-600M (Meta AI) and IndicTrans2-1.1B (AI4Bharat). For NLLB-200, we perform full fine-tuning on English-to-Assamese and English-to-Manipuri (Bengali) directions.nFor IndicTrans2, we employ Low-Rank Adaptation (LoRA) on English-to-Manipuri (Meitei Mayek) and English-to-Bodo directions, reducing trainable parameters from 1.1B to approximately 5.9M. All models are fine-tuned exclusively on the WMT26-provided training data. Our submissions achieved competitive results across all four language pairs. The contrastive systems consistently outperformed the primary systems for Assamese, Manipuri (Bengali) and Bodo, achieving improvements of up to +2.89 BLEU for Bodo and +3.14 BLEU for Manipuri (Bengali). However, an interesting anomaly was observed for Manipuri (Meitei Mayek), where the primary system outperformed the contrastive system despite the latter being subjectively better. We analyze this discrepancy and attribute it to BLEU's sensitivity to surface-level n-gram matching rather than semantic quality.},
  url       = {https://aclanthology.org/2026.wmt-1.159}
}

Author{1}{Orcid}:https://orcid.org/0009-0001-4554-1885
Author{2}{Orcid}:
@InProceedings{huang-tsai:2026:wmt,
  author    = {Huang, Zheng-Yu  and  Tsai, Richard Tzong-Han},
  title     = {Evaluating Pragmatic Appropriateness in English–Vietnamese Machine Translation with Dyadic Profiles},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {276--287},
  abstract  = {English workplace requests often omit relationship information needed for pragmatically appropriate Vietnamese person reference and politeness. We evaluate translations generated with and without explicit sender--recipient profiles for 30 requests using two large language models (LLMs), Gemini 3.5 Flash and SEA-LION, with judgments from two native speakers of Vietnamese. Two native speakers of Vietnamese judged whether translations were correct and natural and whether they suited the displayed relationship. Adding profiles produced no statistically detectable overall improvement in appropriateness; translations generated without profiles already received high ratings. In a prespecified role-reversal evaluation, the same translations were judged first under their intended hierarchical relationships and later under reversed relationships. All 20 Gemini translations generated with profiles were judged appropriate under the intended relationship, compared with four after reversal; SEA-LION's result was inconclusive. A post-hoc side-by-side comparison also favored Gemini's translations generated with profiles but found no detectable SEA-LION preference. These results suggest that when most translations receive positive yes/no ratings, role reversal and direct comparison can reveal differences those ratings alone miss. Fixed evaluation order and reuse of the annotators in the follow-up limit causal interpretation.},
  url       = {https://aclanthology.org/2026.wmt-1.16}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{rao-EtAl:2026:wmt,
  author    = {Rao, Soujanya  and  Taneja, Sakhil  and  Chaitanya, Krishna  and  Mamidi, Radhika},
  title     = {Decepticons-IIITH : Morphology-Aware Lexicon Filtering of Back-Translated Data for Low-Resource English-to-Indic Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2214--2219},
  abstract  = {This study tackles the WMT 2026 Low-Resource Indic Language Translation task, covering English → Assamese, Bodo, Mizo and Nagamese. Low-resource languages lack parallel corpora of sufficient size and quality to train machine translation models. To address this challenge, we propose Morphology-Aware Lexicon Filtering of Back-Translated data to generate good-quality parallel corpora. Initially, monolingual text from publicly available sources is back-translated to English with an NLLB-200-3.3B model fine-tuned on the official training data. To filter this synthetic data, we introduce MALM (Morphology-Aware Lexicon Matching), a filter that checks whether words from a bilingual lexicon are preserved in a translation, using Morfessor stems to handle rich morphology and edit-distance matching to handle borrowed words. For our primary method, training occurs in two stages: fine-tuning on filtered synthetic (back-translated) data followed by fine-tuning on the official parallel data. Whereas for the contrastive method, we are only fine-tuning on the official parallel data. We fine-tune the NLLB-200-3.3B model for Assamese, Mizo, Nagamese and the IndicTrans2-1B model for Bodo. Our primary systems ranked first among primary submissions on two Category-2 pairs viz. English → Bodo and English → Nagamese. On Nagamese, where only 2,000 official sentence pairs exist, of the two methods we used, the primary system improved by a BLEU score of 8.33.},
  url       = {https://aclanthology.org/2026.wmt-1.160}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0003-0171-0816
@InProceedings{rittikar-ramanna:2026:wmt,
  author    = {Rittikar, Sujay Uday  and  Ramanna, Sheela},
  title     = {Winterpeg: Do Neural Cellular Automata Help Where Pretraining Ends?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2220--2229},
  abstract  = {Machine translation for the low-resource languages of North-East India faces an uneven landscape: some are supported by large multilingual models, while others are absent from every pretrained system and usable subword vocabulary. For the covered pairs, fine-tuning a pretrained language model is a strong baseline, thus, our fine-tuned No Language Left Behind model (NLLB-200) ranks first in the WMT-2026 Indic MT shared task on English to Manipuri (Bengali-Assamese) at 11.95 BLEU and 44.67 chrF++. This paper addresses the remaining languages, for which no pretrained coverage exists. We train a vocabulary-free, UTF-8 byte-level, decoder-only language model from scratch and adapt it into a prompted translator that represents any script without tokenizer engineering. Operating at the byte level removes the vocabulary problem but shifts the cost onto the depth, as the model must compose byte sequences into the units, which a tokenizer would otherwise supply. We therefore introduce a causal Neural Cellular Automaton (NCA) as a front end, with a single local update rule iterated in place providing the required depth at the cost of a single set of shared weights. A front-end ablation at matched parameters attributes the observed gains to the iterated rule rather than to the addition of layers, while an equal-depth stack of independently parameterised layers recovers almost none of the improvement. Our causal NCA model places second on English to Mizo among primary submissions, degrades on Khasi, and reaches its limit on Meitei-Mayek, delineating a clear resource-and-script boundary from scratch byte-level modelling. Our code is available at https://github.com/sujayrittikar/wmt\_2026\_indic\_task.},
  url       = {https://aclanthology.org/2026.wmt-1.161}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ruan-EtAl:2026:wmt,
  author    = {Ruan, Lu  and  jia, shaoying  and  ning, fan  and  wang, wei  and  cai, zhengzhen  and  hu, fei  and  hu, heng  and  wang, chenzi},
  title     = {An LLM-Driven Pipeline for Extremely Low-Resource English-to-Indic Translation --- From Vocabulary Injection to Post-Training and Decoding},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2230--2237},
  abstract  = {This submission addresses WMT26 Category 2 for five extremely low-resource English-to-X translation directions: Bodo, Karbi, Kokborok, Nagamese, and Tagin. We develop an LLM-driven pipeline in which large language models are used both to construct bilingual lexical resources and as translation models. First, an English seed dictionary and qwen3-max are used to build word-level bilingual vocabularies, whose top candidates are injected into translation prompts as a "Helpful Vocabulary" block. On an 80/20 train-held-out split of the Tagin data, vocabulary injection is associated with a 7.09-point BLEU improvement and a 49.35-point TER reduction. Second, we compare LoRA adaptation of Qwen2.5-32B-Instruct with full-parameter fine-tuning of Hunyuan-MT-7B, with optional DPO. Hunyuan performs better in the reported Bodo and Kokborok comparisons, including an 11.87-point ChrF advantage on Bodo. Third, post-hoc repeated-ngram and length truncation, combined with per-sentence selection of the shorter of two decoding outputs, mitigates generation degeneration and improves Bodo BLEU from 19.27 to 27.18 without retraining. We submit two contrastive systems per direction: a take-shorter system with output truncation for Bodo, and augmented and baseline SFT systems for the other directions. Code and configurations are publicly available.},
  url       = {https://aclanthology.org/2026.wmt-1.162}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{sinha-EtAl:2026:wmt,
  author    = {Sinha, Aparajita  and  Agarwal, Monika  and  Bhat, Shreya Narayana  and  Bonal, Shreya  and  Agrahari, Aryan},
  title     = {JH\_NLP\_APAShreya2: LoRA Fine-Tuning of NLLB-200 for English–Manipuri Machine Translation at WMT 2026},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2238--2245},
  abstract  = {This paper describes the submission of team JH\_NLP\_APAShreya2 to the English--Manipuri (Bengali-script) track of the WMT 2026 Low-Resource Indic Machine translation shared task. Our system adapts the multilingual NLLB-200-distilled-600M model using Low-Rank Adaptation (LoRA) on the official English-Manipuri parallel corpus. The corpus contains 23,687 sentence pairs. We used no external parallel or monolingual data, synthetic data, or back-translation. We conducted a progressive sequence of experiments with different data-pool sizes and training durations. The final model was trained for seven epochs and used to translate all 1,000 English source sentences in the official blind test set. The primary submission obtained 8.03 BLEU, 20.15 METEOR, 82.56 TER, 40.93 chrF++, 84.86 BERTScore, and 68.18 COMET. The system ranked within the top three primary systems according to five of the six reported evaluation metrics. These results indicate that parameter-efficient adaptation of multilingual models is a practical approach to low-resource English-Manipuri machine translation.},
  url       = {https://aclanthology.org/2026.wmt-1.163}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{subedi-karki:2026:wmt,
  author    = {Subedi, Bipesh  and  Karki, Nischal},
  title     = {BRS-NK: Bidirectional Finetuning of NLLB-200 for Low-Resource Indic Languages: Assamese-English and Bodo-English Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2246--2253},
  abstract  = {This paper describes our submission to the WMT 2026 Low-Resource Indic Language Translation Shared Task for the English-Assamese and English-Bodo pairs in both translation directions. We fine-tune the pretrained NLLB-200 distilled 600M model for primary as well as contrastive tasks. Since Bodo is not supported by NLLB-200, we use the Hindi (hin\_Deva) language tag as a proxy for Bodo during fine-tuning and inference, based on an empirical evaluation of Devanagari-script language tags. The constrained systems are trained exclusively on the official task data, while the contrastive systems additionally incorporate publicly available parallel corpora, including AI4Bharat BPCC and Google Smol. A rigorous preprocessing pipeline, combined with the extra parallel data, yields consistent gains across all automatic metrics, with the largest improvements observed for the lower-resource English-Bodo pair.},
  url       = {https://aclanthology.org/2026.wmt-1.164}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-6427-7434
Author{2}{Orcid}:
@InProceedings{yadav-chaudhary-kuila:2026:wmt,
  author    = {Yadav, Suyash  and  Chaudhary, Utkarsh  and  Kuila, Alapan},
  title     = {RozarNLP-cr7 at WMT 2026: LoRA-Fine-Tuned NLLB for Low-Resource Machine Translation in Northeastern Indian Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2254--2261},
  abstract  = {We present constrained WMT 2026 submissions for low-resource Northeastern Indian languages using NLLB-200 Distilled 1.3B with multilingual LoRA adapters trained only on the official organizer-provided parallel data. The adapted systems improve over the submitted baselines for both Mizo directions and Bengali-script Manipuri→English, but show substantial variation across languages and directions. We further manually analyze approximately 100 Indic→English outputs and validate the systems on WMT 2025 data, highlighting generation artifacts and the challenges of parameter-efficient adaptation across languages, scripts, and translation directions.},
  url       = {https://aclanthology.org/2026.wmt-1.165}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{aldabbas-EtAl:2026:wmt,
  author    = {Aldabbas, Farizeh  and  Altahan, Zyad  and  Elsafty, Hossam  and  Sifa, Rafet},
  title     = {FARABI at WMT 2026: Post-Hoc Channel Controllers for Low-Resource Arabic–Asian Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2262--2272},
  abstract  = {We describe our submission to the WMT26 Low-Resource Arabic–Asian Language Translation shared task, covering ten translation directions between Arabic and five Asian languages. Our central question is whether a post-hoc controller can improve translation quality beyond a fine-tuned backbone without modifying any model weights. For the five into-Arabic directions, our primary system jointly fine-tunes facebook/nllb-200-3.3B on all source languages and augments it at inference time with per-direction NTK-Mirror controllers, each containing fewer than 6K parameters, that rescale frozen transformer output channels. For the five from-Arabic directions, we use Qwen2.5-7B-Instruct as the backbone, adapted via QLoRA and likewise augmented with NTK-Mirror. In the official evaluation, our system ranked first on Bengali-to-Arabic and Indonesian-to-Arabic, second on Hindi-to-Arabic and Urdu-to-Arabic, and third on English-to-Arabic, placing in the top three across all five into-Arabic directions. Post-submission analysis indicates that the weaker results in the from-Arabic directions stem primarily from a model-selection decision made at submission time.},
  url       = {https://aclanthology.org/2026.wmt-1.166}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{alotaibi:2026:wmt,
  author    = {Alotaibi, Abrar M.},
  title     = {OpenBracket at WMT 2026: COMET-MBR System Combination over Arabic-Native and Multilingual Models for Low-Resource Arabic-English Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2273--2279},
  abstract  = {OpenBracket submits constrained systems to both English-paired directions of the WMT 2026 Low-Resource Arabic-Asian Language Translation task, trained on the official data only (21,000 pairs) from publicly available checkpoints. Per direction we submit a primary COMET-22 MBR system combination over five fine-tuned models (NLLB-3.3B, two X-ALMA variants, ALLaM-7B, OPUS-MT) plus two contrastive systems: an ALLaM-7B fine-tune and sampling-based MBR over NLLB-3.3B. On the official test set, our primary ranks first among all teams in Arabic-to-English (BLEU 33.83, chrF 59.72, TER 57.00, COMET 82.49), and in English-to-Arabic it attains the highest COMET on the primary leaderboard (BLEU 21.81, chrF 54.99, TER 69.29, COMET 85.23).},
  url       = {https://aclanthology.org/2026.wmt-1.167}
}

Author{1}{Orcid}:
@InProceedings{arora-EtAl:2026:wmt2,
  author    = {Arora, Palak  and  Ahmed, Syed Afroz  and  Jangid, Mansi  and  Nathani, Bharti  and  Joshi, Nisheeth},
  title     = {BVSLP at WMT 2026: An Empirical Study of Multilingual Transfer for Low-Resource Arabic-Centric Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2280--2286},
  abstract  = {This paper presents HEMANT framework, the BSVLP submission to the WMT 2026 Low-Resource Arabic–Asian Machine Translation Shared Task. The system addresses ten bidirectional translation directions involving Arabic, English, Hindi, Bangla, Indonesian, and Urdu. HEMANT integrates language-specific Unicode normalization, spelling correction, named-entity recognition, knowledge-base-assisted entity translation, transliteration, multilingual transfer learning, and parameter-efficient adaptation of the NLLB-200 model using LoRA. Two directional multilingual models were trained for many-to-Arabic and Arabic-to-many translations. The official results showed considerable variation across language pairs. Arabic–Hindi produced the strongest relative performance, while English–Arabic remained the most challenging pair. In several directions, COMET scores were more competitive than BLEU and TER, suggesting that semantic adequacy was often better preserved than lexical overlap. The findings highlight both the potential and limitations of entity-aware multilingual adaptation for low-resource Arabic–Asian translation.},
  url       = {https://aclanthology.org/2026.wmt-1.168}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{boudraa:2026:wmt,
  author    = {Boudraa, Hossam},
  title     = {MAGADIR at WMT 2026: Cross-System QE Re-ranking and Entity-Aware Selection for Low-Resource Arabic-English Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2287--2304},
  abstract  = {We present MAGADIR's submissions to the WMT 2026 Low-Resource Arabic–Asian Language Translation Shared Task for English–Arabic translation. Our systems combine a small official parallel corpus, parameter-efficient adaptation of open-weight multilingual models, and segment-level, reference-free quality-estimation re-ranking over heterogeneous cross-system candidate pools. For Arabic-to-English translation, the pool includes supervised LoRA-adapted NLLB and Hunyuan-family models alongside zero-shot Hunyuan/HY-MT and Gemma models. For English-to-Arabic, we select among zero-shot Hunyuan/HY-MT, Gemma, Aya, and NLLB candidates. The Arabic-to-English primary system uses MetricX-24-QE together with a Wikidata-derived entity-recall floor that vetoes candidates omitting recognized proper names before quality estimation determines the final output; the English-to-Arabic primary uses COMETKiwi without an entity constraint. On the official Challenge Test, the Arabic-to-English primary achieves 27.45 BLEU, 56.61 chrF, and 81.94 COMET, while the English-to-Arabic primary obtains 14.59 BLEU, 51.48 chrF, and 83.88 COMET. Relative to our own single-model contrastives, QE-guided cross-system selection improves COMET by up to 1.94 points for Arabic-to-English and 0.81 points for English-to-Arabic, although some contrastive systems retain stronger lexical-overlap scores. This exposes a persistent trade-off between neural adequacy objectives and reference-based surface metrics. A post-hoc matched control separates the Arabic-to-English entity floor from the otherwise confounded change in QE utility. The floor accounts for 83\% of the entity-recall gain while changing only 7\% of outputs and imposing effectively no COMETKiwi cost (+0.0002); the switch from COMETKiwi to MetricX rewrites 74\% of outputs and accounts for the full −0.0085 COMETKiwi difference. Our audit also identifies an important limitation: on approximately 65\% of entity-bearing segments, no candidate preserves every recognized entity in canonical form, leaving the veto with no feasible alternative. These findings motivate routine reference-free entity auditing, more diverse candidate pools, expanded gazetteer coverage, and transliteration-aware entity matching for QE-selected translation systems.},
  url       = {https://aclanthology.org/2026.wmt-1.169}
}

Author{1}{Orcid}:
@InProceedings{huidrom-EtAl:2026:wmt1,
  author    = {Huidrom, Rudali  and  Kumar, Vikas  and  Pangsatabam, Hoomexsun  and  Das, Pinaki  and  Singh, Kshetrimayum Boynao  and  Konjengbam, Anand},
  title     = {Which Metric for Which Script? A Quantified Meta-Evaluation of Automatic Metrics and LLM Judges for Low-Resource Indic Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {288--310},
  abstract  = {Machine translation shared tasks publish several automatic metrics and rank systems on one of them, assuming the choice is inconsequential. We test this assumption on the WMT 2025 and WMT 2026 Low-Resource Indic Language Translation tasks, where pre-trained multilingual encoders are unavailable for several target scripts and languages. As no human gold standard evaluation exists at scale, we adapt the Quantified Reproducibility Assessment (QRA) framework across 20 WMT26 and 14 WMT25 directions, to measure agreement between six automatic metrics and two LLM judges as independent measuring instruments. We show that QRA's Type I measure fails under this repurposing when instruments sit on different scales, and suggest a rank-based repair. Using a script-controlled natural experiment (Manipuri evaluated in both Bengali and Meitei Mayek) and a null-output validity test, we observe that neural metrics inflate scores for text in scripts outside the encoder's coverage and compress it into a narrow discriminative band. Switching to Meitei Mayek raises BERTScore and COMET by 21.6 and 23.3 points, respectively, though language and content are unchanged, and no metric preserves system ranking; on degenerate output the two retain 0.81 and 0.61 points on English-to-Indic directions. The effect reverses in the opposite direction. Metric choice should follow script coverage, not language identity. Lacking human judgments, our claims concern instrument behaviour, not human preference.},
  url       = {https://aclanthology.org/2026.wmt-1.17}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{deb-nayak:2026:wmt,
  author    = {Deb, Satarupa  and  Nayak, Prashanth},
  title     = {NCI-MT at WMT 2026: A Quality-Gated Cascade for Low-Resource Arabic-Asian Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2305--2310},
  abstract  = {This paper describes NCI-MT's submission to the low-resource Arabic-to-Asian language translation task as part of the WMT shared task. Our submission covered six of the ten language pairs. We built two separate systems, the primary and the contrastive. The primary was built using the data provided by WMT. In this approach, we fine-tuned our baseline model (NLLB) with per-direction LoRA adapters. For the contrastive system, we introduced an additional quality-gated refinement stage, which includes two stages: in the first stage, we use a reference-free quality estimator to evaluate the quality of the translations produced by our base model. In the second stage, we iteratively re-translate those translations that are below a certain threshold using LLM-based in-context learning. Our results show that both systems are successful in improving the translation quality in low-resource scenarios.},
  url       = {https://aclanthology.org/2026.wmt-1.170}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{hede-EtAl:2026:wmt2,
  author    = {Hede, Vedarth  and  Gupta, Neha  and  Bapat, Harish  and  Ekbote, Harsh},
  title     = {MTG-LIRA at WMT 2026: Data-Centric Arabic-Hindi Machine Translation through Corpus Augmentation, Synthetic Parallel Data Generation and Transformer Models.},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2311--2317},
  abstract  = {MTG-LIRA's submission for the WMT 2026 Arabic–Asian Machine Translation Shared Task presents a data-centric Neural Machine Translation (NMT) pipeline to address low-resource Arabic–Hindi translation. Built using the OpenNMT-py framework with a 6-layer encoder-decoder Transformer architecture, the system relies on extensive corpus augmentation, combining official WMT data with ~4.05 million parallel sentence pairs from OPUS, alongside ~7 million synthetic Hindi–Arabic sentence pairs generated by translating the English side of the BPCC corpus (English-Hindi) into Arabic using Meta's NLLB model. The data undergoes rigorous preprocessing, including Unicode normalization, quality filtering, deduplication, and alignment verification, before evaluating both Byte Pair Encoding (BPE) and SentencePiece tokenization (32,000 vocabulary size) across two training strategies. Results show that while BPE and SentencePiece achieve comparable performance, the impact of synthetic data depends heavily on the translation direction: augmenting with synthetic data yields substantial performance improvements for Hindi→Arabic (increasing BLEU from 7.2 to 9.5 and COMET from 0.8166 to 0.8262 using BPE), whereas it degrades translation quality for Arabic→Hindi (where the OPUS-only BPE model achieves a higher BLEU of 25.3 compared to 17.7 with synthetic data).},
  url       = {https://aclanthology.org/2026.wmt-1.171}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{jana-EtAl:2026:wmt,
  author    = {Jana, Pramit  and  Adhikari, Arindam  and  Acharya, Priyobroto  and  Das, Dipankar},
  title     = {IndicMT at WMT 2026: Full Fine-Tuning of NLLB-200-600M with Round-Trip Filtering and Forward Self-Training for Arabic-Asian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2318--2324},
  abstract  = {We describe the IndicMT submission to the WMT 2026 Shared Task on Low-Resource Arabic-Asian Language Translation. Our system covers all ten translation directions between Arabic and Bangla, English, Hindi, Indonesian, and Urdu. For each direction, we fully fine-tune a separate NLLB-200-distilled-600M model using two NVIDIA T4 graphics processing units. The training pipeline combines Unicode normalization and length filtering with direction-specific round-trip consistency filtering of the available parallel data. We additionally augment each direction with 500 source-side sentences whose target translations are generated by the pretrained NLLB model, a procedure corresponding to forward self-training rather than conventional back-translation. On the official test set, our strongest relative results are obtained for Arabic$\rightarrow$Indonesian and Indonesian$\rightarrow$Arabic. Across all five language pairs, translation from Arabic obtains higher BLEU scores than translation into Arabic. Our results demonstrate that full adaptation of a compact multilingual translation model, combined with model-based data selection and synthetic-target augmentation, provides a practical baseline under limited computational resources.},
  url       = {https://aclanthology.org/2026.wmt-1.172}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-8110-9344
@InProceedings{karim-EtAl:2026:wmt,
  author    = {Karim, Misbahul  and  Ahmed, Faruk  and  Islam, Mishbahul Al  and  Laskar, Sahinur Rahman  and  Laskar, Rabul Hussain},
  title     = {NERDS-NITS at WMT2026: Pivot-Based and Direct Neural Machine Translation for Arabic–Asian Low-Resource Language Pairs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2325--2330},
  abstract  = {This paper presents NERDS-NITS's sub mission to the WMT 2026 Low-Resource Arabic–Asian Language Translation Shared Task, which covers the following language pairs: Arabic↔English, Arabic↔Hindi, Arabic↔Bengali, Arabic↔Urdu, and Arabic↔Indonesian. For the three Indic language pairs, we propose a pivot-based framework that combines a fine-tuned Arabic↔English NLLB-200 model with pretrained IndicTrans2 checkpoints, motivated by the scarcity of direct Arabic–Indic parallel data. For Arabic↔Indonesian, where Indic Trans2 offers no coverage, we instead fine-tune NLLB-200-1.3B directly using QLoRA under memory-constrained conditions. Our Arabic↔English backbone, built on fine-tuned NLLB-200-600M model, forms the pivot leg for all Indic-language systems. Across all evaluated pairs, fine-tuning yields consistent improvements over zero-shot baselines, and the proposed pivot architecture further outperforms direct multilingual fine-tuning for Arabic–Indic pairs, with gains of up to 6.68 BLEU and 0.02 COMET},
  url       = {https://aclanthology.org/2026.wmt-1.173}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{koirala-bhetwal-choudhury:2026:wmt,
  author    = {Koirala, Nabin  and  Bhetwal, Rhythm  and  Choudhury, Nurul Amin},
  title     = {NITM AI Lab at WMT 2026 Shared Task: Low-Resource Arabic-Asian Language Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2331--2335},
  abstract  = {This paper describes the NITM AI Lab submissions to the WMT26 Low-Resource Arabic-Asian Translation shared task. We participate in two translation directions: English-to-Arabic (Sub-Task 1A) and Arabic-to-Hindi (Sub-Task 2B). To address the data scarcity inherent to these language pairs, we compare three neural architectures: a domain-specialized bilingual model (OPUS-MT-TC-Big), a distilled massively multilingual model (NLLB-200Distilled-600M), and a generalist multilingual sequence-to-sequence transformer (mBART50-Many-to-Many). Our systems contrast full-parameter fine-tuning with parameter-efficient fine-tuning (PEFT) via Low-Rank Adaptation (LoRA), and we measure the effect of rule-based orthographic and numerical normalization together with targeted data augmentation from OPUS. On the official blind test set, the domain-specialized OPUS-MT model is the strongest system for English-to-Arabic, while LoRA-tuned NLLB-200 performs best for Arabic-to-Hindi, with OPUS augmentation giving a small additional gain in the latter direction.},
  url       = {https://aclanthology.org/2026.wmt-1.174}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{sharma-EtAl:2026:wmt,
  author    = {Sharma, Pushkar  and  Ahtasam, Mo  and  Singh, Kshetrimayum Boynao  and  Kumar, Deepak  and  Ekbal, Asif},
  title     = {NLP-IIT Patna at WMT 2026: Corpus-Driven Adaptation of Multilingual MT Models for Low-Resource Arabic–Asian Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2336--2344},
  abstract  = {We investigate low-resource Arabic-Asian machine translation across six Arabic-centric translation directions: Arabic↔English, Arabic↔Hindi, and Arabic↔Urdu. Fine-tuned MADLAD-400 ranks first in four of the six directions on the official primary leaderboard and places within the top three in the remaining two. We fine-tune three pretrained multilingual models, MADLAD-400, NLLB-200, and GemmaX2, and compare them with their zero-shot counterparts using BLEU, chrF2++, COMET-22, and TER. Across all language pairs, supervised fine-tuning consistently outperforms zero-shot inference. Among the evaluated models, fine-tuned MADLAD-400 achieves the strongest overall performance, NLLB-200 delivers competitive results, and GemmaX2-28-9B exhibits the largest relative improvement over its zero-shot baseline. These findings demonstrate the effectiveness of supervised adaptation for low-resource Arabic-Asian machine translation and highlight MADLAD-400 as the most robust model across the evaluated translation directions.},
  url       = {https://aclanthology.org/2026.wmt-1.175}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:0000-0003-3612-8834
@InProceedings{singh-EtAl:2026:wmt,
  author    = {Singh, Sumit  and  Vishwakarma, Anish Kumar  and  Sahu, Brijmohan Lal  and  Kumar, Ashwani},
  title     = {UPES\_NLP at WMT 2026 Low-Resource Arabic--Asian Machine Translation Shared Task: Arabic-Centric Machine Translation with NLLB-200-MOE and Prompt-Constrained Entity-Preserving Translation with GPT-5-mini},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2345--2350},
  abstract  = {This paper presents the UPES\_NLP submission to the WMT 2026 Low-Resource Arabic–Asian Machine Translation Shared Task. We used the NLLB-200 Mixture-of-Experts model for translation inference and GPT-5-mini with entity-aware prompting to preserve named entities. Our system participated in both Primary and Contrastive tracks across multiple language pairs. It ranked 3rd in Arabic→English with a BLEU score of 32.08 and 2nd in Arabic→Indonesian with a BLEU score of 22.77. The results show that the proposed approach performs competitively across low-resource Arabic–Asian translation tasks.},
  url       = {https://aclanthology.org/2026.wmt-1.176}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{yadav-kuila:2026:wmt,
  author    = {Yadav, Raman Kumar  and  Kuila, Alapan},
  title     = {Code \& Corpus at WMT 2026: Checkpoint-Pool Minimum Bayes Risk Decoding for Arabic-Hindi and Arabic-Bangla Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2351--2357},
  abstract  = {We present a constrained submission to the WMT 2026 Low-Resource Arabic-Asian Language Translation shared task for Arabic–Hindi and Arabic–Bangla translation. Our system fine-tunes the NLLB-200 distilled 600M model jointly on all four translation directions using the official task data. At inference, instead of relying on a single best checkpoint, we retain multiple training checkpoints and generate candidate translations from each checkpoint. We then apply checkpoint-pool Minimum Bayes Risk (MBR) decoding with chrF as the utility function to select the final translation. This approach exploits diversity across training checkpoints without additional training runs. On the official evaluation, our primary system ranked 4th of 8 for Hindi→Arabic, 3rd of 6 for Bangla→Arabic, 3rd of 10 for Arabic→Hindi, and 2nd of 7 for Arabic→Bangla.},
  url       = {https://aclanthology.org/2026.wmt-1.177}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{zafar-EtAl:2026:wmt,
  author    = {Zafar, Shomaiza  and  Abdul Rauf, Sadaf  and  Firdous, Sheema  and  Munir, Muhammad Saad  and  Fakhar, Nadeem},
  title     = {SLPG\_FJWU at WMT 2026: Domain Adapted Augmentation for Low-Resource Arabic-Asian Language Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2358--2366},
  abstract  = {This paper describes a submission to the WMT 2026 Low-Resource Arabic-Asian Language Translation shared task, covering three directions: Urdu→Arabic, Arabic→Urdu, and Arabic→English. For each direction we submit a primary system (official WMT data only) and a contrastive system trained on an augmented corpus. For the low resource Urdu-Arabic pair, we build a silver standard corpus by mining CC-100 Urdu monolingual text, applying domain adaptation to filter for relevance against WMT DevTest anchors using LEALLA sentence embeddings, and translating the retained sentences into Arabic — adding 27,172 domain matched pairs to the 20,299 official pairs. For Arabic-English, we augment the WMT data with News Commentary to expand the training pool roughly fivefold. All systems are built through transfer learning, fine-tuning the pretrained multilingual NLLB-200 model across all three directions. Our contrastive systems outperform our own primary systems in every direction (BLEU gains of 1.6–3.2 points), and on the official leaderboard our contrastive systems rank 1st in Arabic→Urdu, 3rd in Urdu→Arabic, and 5th in Arabic→English. Our results show that domain adaptation, rather than raw data volume, is the more effective lever for improving low resource NMT.},
  url       = {https://aclanthology.org/2026.wmt-1.178}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0003-0400-3869
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{haobin:2026:wmt,
  author    = {HAOBIN, WEN},
  title     = {ASEAN-8B: A Prompted Qwen3.5-9B System for Chinese–Southeast Asian Web Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2367--2372},
  abstract  = {We present a training-free system for the WMT26 Chinese–Southeast Asian Multilingual Machine Translation Task. The system uses the publicly available Qwen3.5-9B model with a direction-specific prompt, non-thinking decoding, and single-pass generation. We first compare 17 models on a balanced 210-example development probe covering 14 translation directions between Chinese and seven Southeast Asian languages. We then evaluate the selected system on a separate 1,400-example probe containing 100 examples per direction. Using sentence-level sacreBLEU with the FLORES-200 tokenizer and COMET-22, Qwen3.5-9B achieves 24.86 BLEU, 83.23 COMET, and an equally weighted combined score of 54.04. Analysis reveals substantial variation across directions and highlights the effect of noisy website text and imperfect references on lexical and semantic metrics. The resulting system is compact, reproducible, and suitable for efficiency-aware evaluation under the task's 20-billion-parameter limit.},
  url       = {https://aclanthology.org/2026.wmt-1.179}
}

Author{1}{Orcid}:
@InProceedings{issam-EtAl:2026:wmt,
  author    = {Issam, Abderrahmane  and  Semerci, Yusuf Can  and  Scholtes, Jan  and  Spanakis, Gerasimos},
  title     = {Align, Integrate, and Fire: Efficient Token-Level Alignment for Zero-Shot SpeechLLMs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {311--324},
  abstract  = {While Large Language Models excel in natural language processing, efficiently extending their capabilities to spoken input remains a significant challenge. Existing methods for building SpeechLLMs often rely on computationally expensive full-model fine-tuning, or employ parameter-efficient projectors that suffer from inefficient token sequence lengths and costly full-model supervision. In this paper, we introduce Aligned Continuous Integrate-and-Fire, a highly efficient framework for zero-shot speech processing. Our method dynamically compresses continuous acoustic frames into the exact discrete token length of the target text utilizing explicit Dynamic Time Warping alignments. This allows our initial training stage to establish a robust acoustic-to-semantic bridge using lightweight distance metrics, entirely bypassing the computationally expensive LLM forward pass. For subsequent fine-tuning, we propose a memory-efficient knowledge distillation objective that targets a single LLM layer, performing competitively with full-model cross-entropy training at a fraction of the computational cost. Through extensive evaluations on Automatic Speech Recognition and Speech Translation, we demonstrate that our method achieves superior performance compared to prior parameter-efficient baselines.},
  url       = {https://aclanthology.org/2026.wmt-1.18}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-0799-0241
@InProceedings{su-liu-fang:2026:wmt,
  author    = {Su, Meng  and  Liu, Jian  and  Fang, Youxuan},
  title     = {Jiutian-MT-8B at WMT26: Faithfulness-First Training for Chinese-Southeast Asian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2373--2377},
  abstract  = {This paper describes Jiutian-MT-8B, our submission to the WMT 2026 Chinese-Southeast Asian Multilingual Machine Translation shared task. The system adapts the 8-billion-parameter Jiutian-8B-Instruct checkpoint to fourteen translation directions between Chinese and Thai, Vietnamese, Lao, Burmese, Khmer, Indonesian, and Malay. We formulate translation as a tightly controlled instruction-following task and use a four-stage full-parameter training recipe: (i) supervised multilingual adaptation on released parallel data, (ii) faithfulness first confidence-gated multi-pair preference alignment (F2MTA), (iii) direction-coupled, evidence-filtered self-training from task monolingual data, and (iv) a final clean-data calibration stage. The preference stage uses a partial order: a candidate must gain source adequacy without materially sacrificing target fluency, rather than merely win a scalarized quality score. The recipe directly updates all Jiutian-8B parameters with sharded bf16 training while retaining authentic data anchors to limit synthetic-data drift. The official test set is blind at the time of writing; numerical main results are therefore deliberately unreported. We nevertheless specify the data scope, translation interface, training protocol, and evaluation plan needed to complete the report once official scores are returned.},
  url       = {https://aclanthology.org/2026.wmt-1.180}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{wang-EtAl:2026:wmt3,
  author    = {Wang, Zhenhan  and  Fan, Fengzhao  and  Huang, Yuxin  and  Tan, Kaiwen  and  Yu, Zhengtao  and  Gao, Shengxiang  and  Mao, Cunli  and  Zhang, Siqi  and  Jiang, Shuting},
  title     = {Xiaoyu-MT: Kunming University of Science and Technology at WMT 2026 Chinese-Southeast Asian Multilingual MT Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2378--2387},
  abstract  = {This paper presents Xiaoyu-MT, our submission system for the WMT 2026 Chinese-Southeast Asian Multilingual MT Task, focusing on bidirectional translation between Chinese and seven Southeast Asian languages. To address limited parallel resources and imbalanced multilingual data distributions, we select Gemma-4-12B-it as the base model and construct multilingual training data by combining official WMT resources with additional monolingual and bilingual corpora. Xiaoyu-MT adopts a multi-stage training pipeline consisting of Continual Pre-Training (CPT), Supervised Fine-Tuning (SFT), and Direct Preference Optimization (DPO). During SFT, we introduce a language-aware Prompt encoding mechanism to explicitly model source language, target language, and their directed translation relationships through structured Prompt representations and Prompt compression. DPO further optimizes translation preferences using preference data constructed from reference and model-generated translations. Official evaluation results on WMT 2026 show that Xiaoyu-MT achieves an overall BLEU score of 34.73, a COMET score of 81.87, and a Final Result score of 61.80, demonstrating its effectiveness for multilingual and low-resource machine translation scenarios.},
  url       = {https://aclanthology.org/2026.wmt-1.181}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:0000-0002-2980-8420
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
@InProceedings{yan-hu:2026:wmt,
  author    = {Yan, Xiaojing  and  HU, Rile},
  title     = {A Direction-Specialized System for WMT26 Chinese-Southeast Asian Multilingual Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2388--2395},
  abstract  = {We describe a lightweight system for the WMT26 Chinese-Southeast Asian Multilingual Translation Task, covering 14 directions between Chinese and seven Southeast Asian languages. It uses two independently trained Transformer Base models: a one-to-many (OTM) model for translation from Chinese and a many-to-one (MTO) model for translation into Chinese. Each model has 68,727,808 unique parameters. Both use 134,645 real parallel sentence pairs, augmented with 154,452 back-translated pairs for OTM and a 96,000-pair subset with synthetic Chinese targets for MTO. On the 6,859-pair internal test set, the final CTranslate2-based system achieves direction-family macro-average BLEU scores of 43.70 and 49.03; on the FLORES-200 crossdomain test set, they are 17.52 and 26.03, respectively. A CT2-only microbenchmark on a single NVIDIA GeForce RTX 5090 achieves 237.02 and 241.96 sentences per second on presegmented, source-length-stratified development samples, using the median of three steadystate measurements.},
  url       = {https://aclanthology.org/2026.wmt-1.182}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ziwei:2026:wmt,
  author    = {Ziwei, Ma},
  title     = {Hy-MT2-1.8B for WMT26: Staged Multilingual Adaptation for Chinese–Southeast Asian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2396--2403},
  abstract  = {We present a Hy-MT2-1.8B system for the WMT26 Chinese–Southeast Asian translation task, covering fourteen translation directions between Chinese and seven Southeast Asian languages. The system uses a staged adaptation pipeline consisting of monolingual continued pretraining followed by bilingual supervised fine-tuning. We compare parameter-efficient LoRA SFT with full-parameter SFT from the same continued-pretraining checkpoint under matched data and evaluation conditions. On 26,474 validation examples, full-parameter SFT achieves higher macro-average sacreBLEU and COMET scores than LoRA SFT, although the gains vary across language directions. The implementation separates data construction, model preparation, quality evaluation, and serving so that the submitted system can be reproduced and evaluated under the shared-task conditions.},
  url       = {https://aclanthology.org/2026.wmt-1.183}
}

Author{1}{Orcid}:
@InProceedings{ayasi:2026:wmt,
  author    = {Ayasi, Ananya},
  title     = {HT WMT 2026 CreoleMT System Description: Learning Hierarchies for Low-Resource Creole Language Identification},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2404--2412},
  abstract  = {This work is a submission to the Creole MT Creole Language Identification Shared Task conducted as a part of the Eleventh Conference on Machine Translation (WMT '26) colocated with EMNLP 2026, investigating whether hierarchical classification can improve language identification for closely related French-lexifier Creoles. We compare expert-designed linguistic hierarchies with data-driven hierarchies learned from model confusions and representation similarity, together with both hard and soft routing strategies. All models are built on a shared character-level TF--IDF representation with linear classifiers and are evaluated against a strong flat LinearSVC baseline and the multilingual GlotLID system. The best-performing approach is a data-driven combined hierarchy with soft routing, achieving a Macro-F1 of 87.54 while maintaining one of the lowest Macro False Positive Rates (0.00485). Analysis shows that training data availability remains the strongest predictor of language-level performance, with recall significantly correlated with training set size. Manual inspection further reveals that most remaining errors arise from named entities, lexical borrowing, noisy text, and confusions between closely related Creole varieties. Overall, our results demonstrate that learned hierarchical organization provides a practical and effective alternative to expert-designed taxonomies for low-resource language identification while offering improved balance between classification performance and false-positive control.},
  url       = {https://aclanthology.org/2026.wmt-1.184}
}

Author{1}{Orcid}:0009-0008-9086-7666
@InProceedings{bellune-le-sadat:2026:wmt,
  author    = {Bellune, Tabitha Megane  and  Le, Ngoc Tan  and  Sadat, Fatiha},
  title     = {Enhancing Cultural Awareness for Haitian Creole Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2413--2419},
  abstract  = {Natural Language Processing for low-resource languages faces a major challenge: the lack of structured and culturally grounded data. This article presents our contribution to the machine translation task from Haitian Creole to English/French. Our hybrid approach relies on the creation of an original parallel corpus of 19,148 sentence pairs, derived primarily from 55 hours of transcriptions from the Atlas Linguistique d'Haïti, combined with the synthetic generation of code-switching data. By comparing a Transformer model trained from scratch with the fine-tuning of the pre-trained NLLB- 200 multilingual model, we demonstrate that cultural adaptation is crucial. Our best system records a dramatic increase in its BLEU score of 53.61\% on the cultural domain and achieves an average CSI-Match score of 76.99\% in the direction of hat-eng, and 30.69\% BLEU and and 73.56\% in terms of average CSI-Match in the direction of hat-fra respectively, proving its capacity to accurately translate complex idiomatic expressions.},
  url       = {https://aclanthology.org/2026.wmt-1.185}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0000-0002-9292-722X
@InProceedings{jourdain-pradelles-semmar:2026:wmt,
  author    = {Jourdain, Louis  and  Pradelles, Aurélie  and  Semmar, Nasredine},
  title     = {ChapsVision (CHV) at WMT 2026 CreoleMT: Shifting Creole MT from Train-Time to Test-Time Compute for Haitian Creole},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2420--2446},
  abstract  = {We describe a training-free system for the WMT26 Creole MT shared task, targeting Haitian Creole in four directions (eng↔hat, fra↔hat). Rather than fine-tune, we wrap a sin- gle fixed 8B instruction model (qwen3-8b) in a language-agnostic harness that retrieves lexi- cal, parallel-data and grammatical evidence and injects it directly into the translation prompt. On FLORES+ devtest the harness lifts the base model by up to +14 chrF++ and clears the shared task's primary baseline in all four di- rections, though a purpose-fine-tuned system (kreyòl-MT) and frontier LLM still lead the mean. On the task's blind test set that ordering reverses: we finish below both baselines, while the harness gain itself holds, staying large into Creole and small out of it. Two findings frame the paper. First, injected knowledge behaves as a reasoning equalizer: the same evidence that lifts a weak open model leaves a strong frontier model flat, so grounding substitutes for a capa- bility the model lacks rather than amplifying one it has. Second, how evidence is delivered dominates: forcing it into the prompt beats let- ting the model self-serve the same tools in an agentic (ReAct) loop, which costs about 6× the tokens.},
  url       = {https://aclanthology.org/2026.wmt-1.186}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{meng-anastasopoulos:2026:wmt,
  author    = {Meng, Chutong  and  Anastasopoulos, Antonios},
  title     = {GMU WMT 2026 CreoleMT System Description},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2447--2453},
  abstract  = {We present GMU's submissions to the WMT26 Creole Language Translation Shared Task, exploring LLM fine-tuning and few-shot prompting for English-Creole language translation. Our experiments show that LLM-based systems outperform NLLB-based systems, while few-shot prompting is particularly effective for the lowest-resource language pairs. We also demonstrate that extracting incidental parallel sentences from archival grammar books improves translation performance.},
  url       = {https://aclanthology.org/2026.wmt-1.187}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{merx-thieberger-vylomova:2026:wmt,
  author    = {Merx, Raphael  and  Thieberger, Nick  and  Vylomova, Ekaterina},
  title     = {The University of Melbourne WMT 2026 CreoleMT Submission: A Domain-Balanced Approach to Low-Resource Pacific Creole Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2454--2461},
  abstract  = {For our submission to the WMT26 Creole Language Translation Shared Task, we focus on MT models for Pacific creoles: Tok Pisin, Bislama, and Solomon Pijin, with particular attention to broad domain performance. After pre-training on a large mix of domain-imbalanced data, we continue fine-tuning on a diverse mix of domain-balanced data. We rely on a number of data collection and preparation techniques, including LLM-assisted respelling and alignment, back-translation, and distillation from Gemini for domains originally not present in training data. Evaluated on Bouquet and a novel test set made of spoken transcripts, our models beat open model baselines by 3+ chrF++ points in all directions with human-original references. Looking ahead, we plan to develop human-translated test sets for Solomon Pijin and Bislama, and to distil our best models into much smaller ones that retain broad domain coverage.},
  url       = {https://aclanthology.org/2026.wmt-1.188}
}

Author{1}{Orcid}:https://orcid.org/0009-0007-3242-2311
Author{2}{Orcid}:0000-0001-8797-1018
Author{3}{Orcid}:
@InProceedings{ztop:2026:wmt2,
  author    = {Öztop, Yusuf},
  title     = {KYX WMT 2026 CreoleMT System Description: A Contamination- and Orthography-Aware Evaluation for Réunion Creole to English},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2462--2469},
  abstract  = {This paper presents the KYX submission to the WMT 2026 CreoleMT shared task for Réunion Creole to English (rcf-eng). It is mostly about how such a system should be evaluated. The pair is an extreme case: 191 training pairs, a 34-sentence development set, and no standard orthography. On a benchmark this small, a single chrF++ score is easy to over-read, so we pair our system with a contamination- and orthography-aware evaluation. The system continue-trains the official baseline adapter with QLoRA and raises the raw dev score by +13.3 chrF++. But a near-duplicate audit finds that six of the 34 dev sources are paraphrases of training sentences, which the exact-match check WMT26 mandates reports as 0\% overlap. After decontamination the gain is +10.0. Our official test score then comes in 0.8 chrF++ below the decontaminated estimate and 5.1 below the raw one, but our margin over the baseline falls to +3.9, so even the adjusted gain was optimistic. A data-scaling ablation traces the gain to the provided real pairs, not to synthetic data. A source-spelling stress test then shows the advantage shrinks when the input is re-spelled into equally valid variants, though at this sample size the effect is not stable. We conclude that exact-match contamination checks are not enough on tiny benchmarks, and that scores like ours should be read with care. The same caution applies to any low-resource pair with only a few dozen dev sentences. We release the system, the audit, and all data.},
  url       = {https://aclanthology.org/2026.wmt-1.189}
}

Author{1}{Orcid}:
@InProceedings{kanojia-EtAl:2026:wmt1,
  author    = {Kanojia, Diptesh  and  Sindhujan, Archchana  and  Deoghare, Sourabh Dattatray  and  Sokova, Daria  and  Qian, Shenbin  and  Koushik, Girish  and  Ranasinghe, Tharindu  and  Orasan, Constantin  and  Zerva, Chrysoula  and  Rei, Ricardo  and  Blain, Frederic  and  Martins, André  and  Turchi, Marco  and  Negri, Matteo  and  Kunchukuttan, Anoop  and  Khapra, Mitesh M.  and  Bhattacharyya, Pushpak},
  title     = {IndicQE-APE: A Consolidated Benchmark for Quality Estimation and Automatic Post-Editing over Indic Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {325--351},
  abstract  = {Indic quality estimation (QE) and automatic post-editing (APE) data is spread across separate releases, so no single resource supports training and evaluation across tasks and language pairs on one footing. We consolidate the WMT 2020-2024 shared-task lineage with an extended English-Malayalam resource into IndicQE-APE: 126,754 instances over nine directional pairs, with up to four label types aligned on the same segment, a direct assessment, a human post-edit, word-level tags and an error explanation, and a test set stratified over four difficulty axes. We benchmark six prompted LLMs and three COMET metrics on segment-level QE, and three systems on APE. Two of the axes are defined partly on direct assessment and select a compressed slice of it. Segments whose segment-level and token-level signals disagree are ranked below equally scored segments of the same language. Four-shot prompting costs every model at or below 3.4B both correlation and output-format compliance. Unedited MT beats every APE system we run on three of the four pairs. The benchmark and code are released.},
  url       = {https://aclanthology.org/2026.wmt-1.19}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-8814-0080
Author{2}{Orcid}:https://orcid.org/0000-0002-6467-6873
Author{3}{Orcid}:https://orcid.org/0000-0003-3566-888X
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:0000-0003-3207-3821
Author{8}{Orcid}:0000-0003-2067-8890
Author{9}{Orcid}:https://orcid.org/0000-0002-4031-9492
Author{10}{Orcid}:https://orcid.org/0000-0001-8265-1939
Author{11}{Orcid}:https://orcid.org/0000-0003-3017-3722
Author{12}{Orcid}:
Author{13}{Orcid}:https://orcid.org/0000-0002-5899-4496
Author{14}{Orcid}:https://orcid.org/0000-0002-8811-4330
Author{15}{Orcid}:
Author{16}{Orcid}:
Author{17}{Orcid}:https://orchid.org/0000-5319-5508
@InProceedings{strickland:2026:wmt,
  author    = {Strickland, Emmett},
  title     = {Gwo-K Kreyol Classifier: Rock Pidgins (RP) Submission to the WMT 2026 CreoleMT Language Identification Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2470--2474},
  abstract  = {This paper describes ongoing efforts to build an n-gram text classifier for several major French-lexifier creoles. Our model is based on a simple text processing pipeline, limited training data, and a deliberately lightweight architecture. Despite this, it achieves substantially stronger performance than existing baselines on our target languages. These results suggest that long-established machine learning approaches remain effective for new language identification needs.},
  url       = {https://aclanthology.org/2026.wmt-1.190}
}

Author{1}{Orcid}:
@InProceedings{ahmed-chakma:2026:wmt,
  author    = {Ahmed, Firoz  and  Chakma, Dinalo},
  title     = {A Human-Translated FLORES+ Dataset for Chakma Language},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2475--2480},
  abstract  = {We present a human-translated, native-script Chakma (ccp\_Cakm) extension to FLORES+. The corpus comprises the 997-sentence dev and 1,012-sentence devtest English splits, for 2,009 records in total. We recruited three native Chakma speakers through a screening task, conducted six virtual training sessions, and required direct translation without machine-translation or language-model output. A native reviewer checked meaning, grammar, script, terminology, named entities, numbers, and punctuation, returning problematic items for revision. We describe the Bangladeshi Chakma variety used and our orthographic and Unicode decisions. We then report a completed release audit: FLORES+ identifiers were recovered from the canonical English order and renumbered to the 1-based scheme, all 2,009 records were verified one-to-one against the English source, text was normalised to Unicode NFC, a character inventory was produced, and all twenty-seven residual non-Chakma characters were located and resolved. The package passes every check with no open defects, is fixed by SHA-256 checksums, and has been submitted to FLORES+ through the Open Language Data Initiative under CC BY-SA 4.0. The work provides a native-script evaluation resource for a language whose existing computational resources frequently rely on Bengali-script transliteration.},
  url       = {https://aclanthology.org/2026.wmt-1.191}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ali-mateus-mangue:2026:wmt,
  author    = {Ali, Felermino Dario Mario  and  Mateus, Delfina Lázaro  and  Mangue, Manuel Valente},
  title     = {Measuring the Cost of Variety Conflation in Multilingual MT Evaluation: Adding Mozambican Xichangana, Nyanja and Sena to FLORES+},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2481--2491},
  abstract  = {In this paper, we extend FLORES+ with Portuguese-source evaluation sets for three Mozambican Bantu varieties: Xichangana, Mozambican Nyanja, and Sena. We compare Xichangana with the existing Tsonga reference and Mozambican Nyanja with Chichewa, and evaluate NLLB-200, Google Translate, GPT, and a variant-aware NLLB model. Holding system output fixed reveals substantial reference sensitivity. On \textit{devtest}, changing only the reference from Tsonga to Xichangana reduces spBLEU by 13.10 points for NLLB-200 and 15.30 for Google. On matched Nyanja subsets, replacing Chichewa with Mozambican Nyanja produces smaller but consistent reductions of 3.03 and 6.10 spBLEU, respectively. Variant-aware fine-tuning reverses this pattern on the intended targets: relative to NLLB-200, it improves Xichangana by 7.04 spBLEU and Mozambican Nyanja by 5.33 on \textit{devtest}, while losing performance on the sibling references. GPT is competitive on Tsonga and Chichewa but substantially weaker on the Mozambican varieties. For Sena, the finetuned model reaches 12.64 spBLEU and 36.21 chrF++ on \textit{devtest}. These findings motivate variety-aware language identifiers, references, and reporting for cross-border languages or language dialects/variants. The data is publicly available on Hugging Face at \url{https://huggingface.co/datasets/MOZNLP/FLORES_MOZ}.},
  url       = {https://aclanthology.org/2026.wmt-1.192}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{frontull-EtAl:2026:wmt2,
  author    = {Frontull, Samuel  and  Castlunger, Lara  and  Oberhollenzer, Maximilian  and  Comploj, Karin  and  Videsott, Ruth  and  Sama, Robert  and  Perathoner, Gabriel  and  Zoli, Carlo  and  Videsott, Paul},
  title     = {Adding ciofs and ciüfs to BOUQuET: Benchmark Extension to Ladin},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2492--2503},
  abstract  = {Benchmarks do more than report scores: they provide a shared basis for evaluating methods, comparing systems, and measuring progress. As language technologies are now widely integrated into everyday life, efforts should also be made to include minority languages in the resources that underpin language-technology development. If languages are absent from relevant benchmarks, they remain harder to evaluate, compare, and incorporate into future language technologies. Expanding benchmark coverage is therefore essential for ensuring fair evaluation, increasing visibility, and enabling continued development of language technologies for these communities. To support MT development for Ladin, we contribute a Ladin extension of BOUQuET to the WMT 2026 OLDI shared task, covering the Val Badia and Gherdëina varieties. We present a systematic reference-creation workflow combining human translation, expert revision, cross-variant alignment, and blind ranking to produce high-quality references. The dataset is released under the CC BY-SA 4.0 license.},
  url       = {https://aclanthology.org/2026.wmt-1.193}
}

Author{1}{Orcid}:0009-0004-1230-4666
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
@InProceedings{grigorian-umishov:2026:wmt,
  author    = {Grigorian, Vladislav  and  Umishov, Abu-Viskhan},
  title     = {Expanding Chechen Machine Translation Resources: SMOL, BOUQuET, and WMT24++ Extensions},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2504--2508},
  abstract  = {We introduce a new fork of the SMOL training dataset along with forks of the BOUQuET and WMT24++ translation benchmarks for the Chechen language. We train a model on SMOL and calculate BLEU and chrF++ scores on the introduced benchmarks to evaluate how training on such a limited amount of data may contribute to the model's Chechen-Russian translation capability. The introduced resources provide a unified evaluation framework for Chechen machine translation and allow measuring the transfer of improvements across different benchmark datasets. Our experiments establish baseline results for future research on Chechen translation and demonstrate the potential of utilizing small-scale parallel corpora for improving low-resource language models.},
  url       = {https://aclanthology.org/2026.wmt-1.194}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ibrahim:2026:wmt,
  author    = {Ibrahim, Beshir},
  title     = {Closing the Resource Gap for Tigre: A Diaspora-Sourced English–Tigre Parallel Dataset for SMOL},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2509--2515},
  abstract  = {Tigre, an Afro-Semitic language maintaining strong linguistic continuity with Classical Ge'ez, holds a significant presence in Eritrea as well as across native and intergenerational refugee communities in eastern Sudan. It remains absent from major machine translation benchmarks and resources such as FLORES-200, despite an established orthography and active use in education. As a submission to the WMT26 Open Language Data shared task, we present a community-driven contribution of 11,998 English–Tigre pairs (word-, phrase-, and sentence-level) to the SMOL parallel corpus, written in the standardized Ge'ez-script form of Tigre used in Eritrean education and media. The data were produced through a structured, multi-stage post-editing and crossvalidation pipeline involving two diaspora native-speaker translator groups. We document the collection methodology and translator workflow, and validate the data's usefulness by fine-tuning NLLB-200-3.3B on the contributed pairs, observing about a 4.4× improvement in chrF++ over an untrained baseline for English-to-Tigre translation (4.92 to 21.63), with gains in both translation directions. We release the dataset under a Creative Commons Attribution (CC-BY 4.0) license to support future NLP research on this severely underresourced language},
  url       = {https://aclanthology.org/2026.wmt-1.195}
}

Author{1}{Orcid}:
@InProceedings{marmonier-bawden-sagot:2026:wmt,
  author    = {Marmonier, Malik  and  Bawden, Rachel  and  Sagot, Benoît},
  title     = {A French Version of the SmolSent Corpus},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2516--2525},
  abstract  = {We present a French version of the SmolSent corpus, a contribution to the WMT 2026 Open Language Data Initiative (OLDI) shared task. Originally curated to maximize unique-word coverage across 863 English sentences, SmolSent constitutes a valuable, token-efficient training set for the machine translation of low-resource languages. We describe our translation workflow, which relied on a custom-built interface to select and refine translation hypotheses generated by a traditional encoder-decoder model, alongside two state-of-the-art reasoning-enabled large language models (LLMs). Experimental validation based on the MetricX-24-XXL model for quality estimation, supported by bootstrap resampling significance tests, indicates that human post-editing results in significant quality improvements over all raw model outputs, even as reasoning models like Gemma-4-31B-it achieve a competitive 39.51\% zero-edit rate. This French partition is not an end in itself, but a pivot resource meant to facilitate ongoing data collection efforts for under-resourced regional languages of France, such as Savoyard and Gallo.},
  url       = {https://aclanthology.org/2026.wmt-1.196}
}

Author{1}{Orcid}:https://orcid.org/0009-0001-7398-7067
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-0107-8526
@InProceedings{marmonier-EtAl:2026:wmt,
  author    = {Marmonier, Malik  and  Favre, Alain  and  Sagot, Benoît  and  Bawden, Rachel},
  title     = {A Savoyard Version of the FLORES+ Corpus},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2526--2545},
  abstract  = {We present a Savoyard version of the FLORES+ corpus, a contribution to the Open Language Data Initiative (OLDI) shared task at WMT 2026. Savoyard, an endangered Arpitan dialect of the Gallo-Romance language family, is currently undergoing active revitalization. We give an overview of its linguistic history, describe our data collection process, and report baseline scores for a range of systems on the resulting French-Savoyard evaluation set as a point of comparison for future work. We hope that this dataset will stimulate further data collection efforts for Savoyard and the broader Arpitan dialect continuum, supporting the development of NLP resources for these endangered languages and contributing to the preservation of humanity's linguistic heritage.},
  url       = {https://aclanthology.org/2026.wmt-1.197}
}

Author{1}{Orcid}:https://orcid.org/0009-0001-7398-7067
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-0107-8526
Author{4}{Orcid}:
@InProceedings{signoroni-rychly:2026:wmt,
  author    = {Signoroni, Edoardo  and  Rychly, Pavel},
  title     = {FIÙR: A Benchmark Dataset for Eastern Lombard Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2546--2559},
  abstract  = {While recent initiatives have expanded machine translation benchmarks to include low-resource and minoritized languages, regional dialect continuums remain severely underrepresented. Current datasets for the Lombard language predominantly feature Western Lombard, with Eastern Lombard varieties almost absent from the digital landscape. In this paper, we introduce, a new benchmark dataset providing an Eastern Lombard translation of the FLORES+ dev and devtest splits. We detail the methodology of translating a largely oral, unstandardized language, including orthographic regularization and the handling of Italian loanwords. Finally, we establish zero-shot baselines by prompting current LLMs and MT models, demonstrating that existing systems struggle with Lombard varieties overall, and exhibit a strong Western Lombard bias, struggling even more to accurately generate Eastern Lombard text.},
  url       = {https://aclanthology.org/2026.wmt-1.198}
}

Author{1}{Orcid}:0000-0002-1029-8299
Author{2}{Orcid}:https://orcid.org/0000-0001-5097-4610
@InProceedings{singh-ekbal-pakray:2026:wmt,
  author    = {Singh, Kshetrimayum Boynao  and  Ekbal, Asif  and  Pakray, Partha},
  title     = {Script Matters: Benchmarking Open-Source Machine Translation for Manipuri in Meetei-Mayek and Bengali Script},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2560--2569},
  abstract  = {Manipuri (Meiteilon) officially recognised the Meetei-Mayek script alongside the Bengali script in 2021, with both scripts permitted for concurrent use during a 10-year transition period. However, the FLORES-200 devtest release still encodes Manipuri only as Bengali script, leaving the language's current official script without a standard benchmark. We address this gap by contributing the Meetei-Mayek script (mni\_Mtei) layer, aligned with the existing Bengali-script (mni\_Beng) data, to form a four-way parallel resource comprising English, Hindi, Meetei-Mayek Manipuri, and Bengali-script Manipuri, with 1,012 sentences per language. This resource enables separate evaluation of open-source MT and LLM systems on the two Manipuri scripts, rather than treating them as a single "Manipuri" target. Across three IndicTrans2 checkpoint families, Sarvam-Translate, and five general-purpose LLMs, evaluated using six automatic metrics, we find that legacy Bengali-script output often scores higher than Meetei-Mayek despite the latter being the official script; IndicTrans2's family degrades sharply on Bengali-script Manipuri; Sarvam-Translate cannot produce Bengali script at all; and none of the five general-purpose LLMs we test can reliably read or write Meetei-Mayek. Script identity is therefore a first-class evaluation variable for Manipuri MT: a single "Manipuri" score can mask large, direction-dependent quality gaps.},
  url       = {https://aclanthology.org/2026.wmt-1.199}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0003-3612-8834
Author{3}{Orcid}:https://orcid.org/0000-0003-3834-5154
@InProceedings{bathala-EtAl:2026:wmt,
  author    = {Bathala, Prasanth  and  Shrimal, Anubhav  and  Singh Kharbhanda, Sukhdeep  and  Lanka, Pradyumna  and  Dhaipule, Rohit},
  title     = {TACTICS: Taxonomy-Aware Intelligent Corpus Sampling for Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {14--32},
  abstract  = {Large-scale machine-translation (MT) systems are typically evaluated on random samples from a corpus whose distributional composition is an artifact of how it was assembled. Such a sample inherits the phenomena the collection happens to contain rather than the full space a system must handle, spanning rule-governed conventions (terminology, punctuation, currency formatting) and context-dependent phenomena (tone, honorifics, document-level coherence), and thus provides no coverage guarantee for assessing robustness. We propose TACTICS (Taxonomy-Aware Coverage-opTimized Intelligent Corpus Sampling), which recasts coverage as an explicit objective. TACTICS induces a hierarchical taxonomy from a locale style guide, classifies segments against it, and selects a fixed-budget subset jointly optimizing coverage of rare categories, document-level coherence, and distributional fidelity to the full corpus. Applied to MT evaluation across four translation directions, TACTICS improves coverage of rare categories over lexical and embedding-based selection. By targeting the phenomena that separate systems, TACTICS makes a fixed evaluation budget go further, recovering the true system ranking from far fewer segments than random sampling wherever a real quality gap exists and never signaling a difference where none exists.},
  url       = {https://aclanthology.org/2026.wmt-1.2}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{kimura-suzuki:2026:wmt,
  author    = {Kimura, Subaru  and  Suzuki, Jun},
  title     = {Text-Only vs. Image-Aware VLM Judges for Manga Translation Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {352--374},
  abstract  = {In manga translation, visual context outside speech-bubble text, including speaker identities, referents, and scene tone, is depicted in page images, making it natural to refer to images during translation evaluation. However, most conventional manga translation evaluations rely solely on text-only automatic metrics, leaving the impact of image input insufficiently explored. We propose a practical validation framework that employs the same vision-language model (VLM) as both a text-only judge (TOJ) and an image-aware judge (IAJ) to investigate evaluation divergences associated with image-aware evaluation procedures. We use a Japanese-English manga dataset of official translations aligned at the page and speech-bubble levels. First, adding page images to TOJ enabled the primary VLM to detect more degraded translations. Second, we analyzed evaluation divergences between TOJ and IAJ, finding different error counts and error-category distributions. Human verification did not reveal a consistent preference between TOJ and IAJ.},
  url       = {https://aclanthology.org/2026.wmt-1.20}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0003-2108-1340
@InProceedings{zundanovi-EtAl:2026:wmt,
  author    = {Zundanović, Dragana  and  Leventić, Hrvoje  and  Božić Lenard, Dragana  and  Romić, Krešimir},
  title     = {The Croatian Dataset Seed Submission to the WMT26 Open Language Data Initiative Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2570--2582},
  abstract  = {We present a new English–Croatian parallel corpus for the WMT26 Open Language Data Initiative (OLDI) Seed shared task, consisting of 6,193 sentence pairs. To ensure high quality, the resource was produced via machine translation followed by a two-stage verification process (post-editing and independent revision) conducted by professional translators. We validate the corpus by fine-tuning NLLB-200, TranslateGemma, and Gemma-4-31B. The dataset improves every dedicated translation model in both directions and the strongest open LLMs in the Croatian→English direction. Through a training-data ablation, we demonstrate that the benefit of human verification increases with model capacity and is visible only to neural metrics, not to surfaceoverlap metrics such as chrF++. Finally, we analyze the quality ceiling for the strongest models, identifying a trade-off between naturalness and surface overlap in the English→Croatian direction.},
  url       = {https://aclanthology.org/2026.wmt-1.200}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{arora-EtAl:2026:wmt3,
  author    = {Arora, Palak  and  Jangid, Mansi  and  Nathani, Bharti  and  Joshi, Nisheeth},
  title     = {QwenSub-MT Framework for WMT 2026 Video Subtitle Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2583--2590},
  abstract  = {The paper introduces QwenSub-MT, a system that is context-aware and constraint-controlled and has been developed for the WMT 2026 Video Subtitle Translation shared task. It carried out the translation of Simplified Chinese subtitles into 6 different languages. QwenSub-MT used Qwen3.5-Instruct together with a multistage 4-bit QLoRA adaptation and treated adjacent subtitle cues as contextual blocks. In order to enhance consistency and disambiguation, the system incorporated video synopsis information, previous translations, terminology memory and selective visual grounding. A revision stage was carried out to correct semantic, terminology and formatting errors, while duration-aware length control, line-break optimization and deterministic SRT validation were used to make sure that the outputs were readable and structurally valid. 500 subtitle files were submitted in five language directions and obtained a macro-average score of 58.459, placing sixth in the overall ranking. The results show that contextual modelling and structural control are useful, but candidate diversity, independent quality estimation and stronger adaptation to the target language still needs improvement.},
  url       = {https://aclanthology.org/2026.wmt-1.201}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{ghimire-mahato:2026:wmt,
  author    = {Ghimire, Prajwal  and  Mahato, Aashish},
  title     = {SubVision@WMT26: QLoRA Fine-Tuning of Hy-MT2-1.8B for Chinese to English Video Subtitle Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2591--2606},
  abstract  = {We describe \textbf{SubVision}, our submission to the WMT26 Video Subtitle Translation shared task. The system fine-tunes the 1.8B-parameter multilingual translation model Hy-MT2-1.8B \citep{zheng2026hymt2familyfastefficient} using 4-bit QLoRA \citep{dettmers2023qloraefficientfinetuningquantized} on 60,000 sentence pairs sampled from the TVsub Chinese-to-English subtitle corpus \citep{DBLP:journals/corr/abs-1801-03257}. The training pipeline uses a single instruction style prompt template, LoRA adapters (rank 16, $\alpha=32$) on all attention and MLP projections, and early stopping based on dev set sacreBLEU. We fix the decoding with beam search with beam size 4, repetition penalty 1.15, and no-repeat 3-gram constraints. The fine-tuned model on a 200 pair TVsub test split with multiple references achieves sacreBLEU 37.63, chrF 52.76, and TER 53.93, compared with zero-shot NF4 baselines of sacreBLEU 11.60 for Hy-MT2-1.8B and 15.79 for Hy-MT2-7B on the same test pairs, prompt, and decoding configuration. The same configuration is used for official inference, translating subtitle lines independently while preserving SRT timing. We report our full training and inference configuration, describe a data processing detail in parsing the corpus's multi-document, multi-reference SGM files, and compare our final configuration against nine additional trained configurations that differ in training pair count, sequence-length budget, LoRA rank, and decoding generation length. The paired bootstrap analysis over this comparison shows that given the 200 segment test set size, the nominal ranking of the submitted configuration is not statistically distinguishable from several of the alternatives. We report this analysis along with the point estimates rather than presenting the ranking as an established result. We note that evaluation uses an in-domain TVsub split. The code is available at: \url{https://github.com/praaajg/wmt26-hymt2}.},
  url       = {https://aclanthology.org/2026.wmt-1.202}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{guttmann-nowakowski:2026:wmt,
  author    = {Guttmann, Kamil  and  Nowakowski, Artur},
  title     = {Laniqo at WMT26 Video Subtitle Translation Shared Task: Multi-Objective Fusion of Pareto-Optimal Candidates},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2607--2614},
  abstract  = {This paper describes Laniqo's submission to the WMT26 Video Subtitle Translation Shared Task, translating subtitles from Simplified Chinese into English, Thai, Indonesian, Malay, and Traditional Chinese under a 20B-parameter, open-license model constraint. Our system optimizes translation quality and subtitle compliance entirely at inference time, without fine-tuning, by combining hedged multi-prompt candidate generation, language-identification and compliance pruning, multi-objective Pareto reranking, reasoning-based candidate fusion, and a fallback/compression step. We screened four open-weight models and compared the two strongest, Qwen3.5-9B and Gemma-4-12B-it, across the full pipeline on the complete test corpus. Gemma-4-12B-it reached significantly higher compliance rates in every target language, thus it was chosen as our final system.},
  url       = {https://aclanthology.org/2026.wmt-1.203}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0003-4473-0008
@InProceedings{saud-dawadi-regmi:2026:wmt,
  author    = {Saud, Shiv Ram  and  Dawadi, Sundeep  and  Regmi, Sunil},
  title     = {paramanandaAI@WMT26 Video Subtitle Translation: QLoRA Fine-Tuning with Test-Time Metadata Prompting},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2615--2621},
  abstract  = {Subtitle translation poses challenges dis tinct from general-purpose machine trans lation model trained on web corpus. The outputs must fit strict display constraints as the text is predominantly spoken dia logue, local slangs, proper names, cultural nuances. In the "WMT26 Chinese → En glish task" all test videos belong to the historical costume drama genre, where pe riod vocabulary cannot be resolved from a single line in isolation. We address these challenges with a two-part system. We fine-tune Hy-MT2-1.8B (Zheng et al., 2026) on 500,000 Chinese-English subtitle pairs from the TVsub corpus (Wang et al., 2018) using 4-bit QLoRA (Dettmers et al., 2023), and at test time we inject show-level metadata (title, episode name, summary) into a structured system prompt to ground character names and period terms. On an in-domain 200-pair development split, QLoRA fine-tuning raises BLEU from 6.93 to 31.29 (+24.4 points) which demon strates the value of domain adaptation over the base multilingual model. Across four prompt configurations evaluated via LLM-as-judge on 20 sampled test trans lations the metadata-augmented, previous context-free variant (structured\_noctx) wins 14 of 20 judged dialogues and is sent for submission. The fine-tuned model weights are released under Apache 2.0 at ShivRamSaud/hy-mt2-1.8b-wmt26.},
  url       = {https://aclanthology.org/2026.wmt-1.204}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{yudhistira-EtAl:2026:wmt,
  author    = {Yudhistira, Pieter Christy Yan  and  Azzahra, Sayyidah Fatimah  and  Malik, Dzaki Rafif  and  Fatyanosa, Tirana Noor},
  title     = {~rupiah at WMT26: Training-Free Context-Aware Subtitle Translation with QE Selection},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2622--2629},
  abstract  = {We present the training-free subtitle translation system developed by Team ~rupiah for the WMT26 Video Subtitle Translation shared task. The system translates Chinese subtitles into English and Indonesian using Hy-MT2-7B while preserving the source cue indices and timestamps. To support discourse and terminology consistency, each cue is conditioned on episode metadata, preceding source cues, translation history, and an automatically constructed episode glossary. At inference time, the system generates candidates at six temperatures, removes exact duplicates, filters candidates containing Han characters, and selects a hypothesis using reference-free CometKiwi quality estimation. A deterministic lexical-recovery step then replaces glossary terms that remain untranslated in the selected hypothesis. In the preliminary official evaluation over 100 videos per direction, ~rupiah ranks fifth of eight systems for Chinese-to-English and third of four for Chinese-to-Indonesian. Rubric-level results indicate that the largest performance gap relative to the top-ranked system is in accuracy and fidelity, whereas the gaps in fluency and subtitle conventions are smaller. Reference-based scores over 89 videos per direction provide additional diagnostics of the complete pipeline. These aggregate evaluations do not isolate the contribution of individual components. Source code available and can be accessed here: https://github.com/Pieter414/-rupiah-training-free-video-translation-subtitle},
  url       = {https://aclanthology.org/2026.wmt-1.205}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-2801-5947
@InProceedings{wu:2026:wmt,
  author    = {Wu, jianfeng},
  title     = {Full-Video Context and Anonymous Candidate Selection for WMT26 Video Subtitle Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2630--2635},
  abstract  = {We present ReopenAI's constrained-track system for the WMT26 Video Subtitle Translation Shared Task. The system translates Simplified Chinese subtitles into English, Thai, Indonesian, Malay, and Traditional Chinese for Taiwan. It combines Hy-MT2-7B and Gemma-4-12B-it without fine-tuning, using approximately 19B unique parameters. Both models receive the complete Chinese subtitle sequence and video titles as context and generate two candidates for each subtitle cue. A Gemma-based selector anonymously compares four candidates in a stable order and its reverse. A selection is accepted only when both judgments agree; otherwise, a deterministic Gemma candidate is used as fallback. In the official preliminary evaluation, ReopenAI ranked first overall and in every target direction, achieving a macro-average score of 89.357 across 100 videos. On an internal three-video diagnostic covering 705 cues per target, the fusion system achieved a 93.8\% pooled acceptable-translation rate. A separate 122-cue ablation showed that full-video context improved Hy-MT2 in all five target languages, with gains ranging from 4.1 to 18.0 points. The system uses no task-specific parallel data, audio, or video frames.},
  url       = {https://aclanthology.org/2026.wmt-1.206}
}

Author{1}{Orcid}:
@InProceedings{yekkisetty:2026:wmt,
  author    = {Yekkisetty, Aaryan},
  title     = {A Translator–Critic System for Chinese-to-English Subtitle Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2636--2650},
  abstract  = {This paper describes our constrained-track Chinese-to-English submission to the WMT 2026 video subtitle translation shared task, entered under the team name PeloTranslate. We introduce a text-only system that translates Chinese video subtitles into English using two open-weight models under a 20B parameter budget. A 7B translation-specialized model generates English candidates, while a quantized 12B instruction-tuned model selects candidates and audits the selected line against eight subtitle-specific accuracy checks, one check per model call. A skeptical second pass re-verifies every flag, and a confirmed error is turned into a short correction instruction and sent back to the translation model for guided retranslation. Deterministic code then enforces character-name consistency and line layout and rebuilds each SRT file with its original indices and timestamps.},
  url       = {https://aclanthology.org/2026.wmt-1.207}
}

Author{1}{Orcid}:
@InProceedings{diskin:2026:wmt,
  author    = {Diskin, Michael},
  title     = {One Model, Five Tasks, Two RTX 3090 GPUs: The HSE System for the Sorbian Track of WMT26 Multitask LLMs with Limited Resources},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2651--2660},
  abstract  = {We describe HSE's submission to the Sorbian track of WMT26 Multitask LLMs with Lim- ited Resources, where a single 2B-parameter model must translate between German, Up- per Sorbian, and Lower Sorbian and also an- swer multiple-choice questions, check spelling and grammar, and solve maths problems in both Sorbian languages. Our system is one LoRA adapter on Qwen3.5-2B, trained on the official parallel data, synthetic spelling and grammar examples, a question-answering proxy, and English GSM8K in about nine GPU- hours on two consumer RTX 3090 GPUs. Ex- tended translation-only fine-tuning cost about ten points of question-answering accuracy, re- producing a WMT25 finding, and nearly erased the output conventions of the other tasks; the multitask mixture recovered them at almost no translation cost. The system ranked fourth of four teams. It far exceeded the baseline on translation (21.7 → 61.5 chrF++) and spell checking (6.7 → 63.6), but on grammar check- ing and maths reasoning it learned the answer format without the skill: the grammar score is almost exactly the share of error-free sentences, and maths accuracy did not exceed the baseline. Two negative results may transfer to other low- resource settings. Synthetic holdouts did not predict performance on the official tasks in ei- ther direction, and self-translated training prob- lems passed every automatic check yet changed meaning in at least 6 of 20 inspected cases. The model is publicly available.},
  url       = {https://aclanthology.org/2026.wmt-1.208}
}

Author{1}{Orcid}:
@InProceedings{khasykov:2026:wmt,
  author    = {Khasykov, Mikhail},
  title     = {HeyBusan at WMT26: A System for the Sorbian Track},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2661--2666},
  abstract  = {We describe our submission (team HeyBusan) to the Sorbian track of the WMT26 Multitask LLMs with Limited Resources shared task. A single Qwen3.5-2B model handles machine translation, question answering, spell checking, grammar checking, and mathematical reasoning in Upper Sorbian, Lower Sorbian, and German. We fine-tuned all model parameters on public data (the official task data plus external corpora) for approximately 55 GPU-hours on one 24 GB consumer GPU. Task-specific decoding then improved the results further without changing the model's weights. Team HeyBusan placed first in machine translation, spell checking, and grammar checking, and tied with team LT3 for the overall track win.},
  url       = {https://aclanthology.org/2026.wmt-1.209}
}

Author{1}{Orcid}:
@InProceedings{koo:2026:wmt,
  author    = {Koo, Zi Chen},
  title     = {From Error Detection to Credit Assignment: Do xCOMET Error Spans Provide Useful Policy Credit for Chinese–Malay Translation?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {375--386},
  abstract  = {Sequence-level rewards assign the same group-relative advantage to every generated token, even when only a short translation span is erroneous. We ask whether xCOMET error spans provide useful token-level policy credit for Group Relative Policy Optimization (GRPO) in Chinese–Malay translation. Signed Lexical-Residual GRPO (SLR) adds a bounded, zero-mean local residual to the sequence advantage. Across three matched seeds, SLR improves reward-aligned COMET and MetricX-24, a learned metric not used in training, while chrF++ and BLEU differences remain unresolved. A blinded 300-item evaluation by the bilingual author does not resolve an adequacy or fluency preference and finds a higher omission rate under SLR (9.33\% to 16.67\%). A repeated span audit finds high sentence-level error detection but weak target-word localization; fewer than half of consistently marked error words receive negative residual credit. In this setting, error detection, localization, and policy-credit actionability are distinct: explainable metric spans change optimization behavior without automatically providing human-aligned token supervision.},
  url       = {https://aclanthology.org/2026.wmt-1.21}
}

Author{1}{Orcid}:
@InProceedings{moerman-tezcan:2026:wmt,
  author    = {Moerman, Thomas  and  Tezcan, Arda},
  title     = {Getting More from Small LLMs and Limited Data: Fuzzy-Match Retrieval and Candidate Scoring for Multitask Sorbian NLP},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2667--2680},
  abstract  = {We describe LT3's submission to the WMT26 shared task on Multitask LLMs with Limited Resources for Upper Sorbian (hsb) and Lower Sorbian (dsb): one QLoRA-fine-tuned Qwen3.5-2B model serves machine translation, multiple-choice question answering, spell checking, grammar checking, and maths reasoning. Our approach centres on three components. First, retrieval augmentation: fuzzy-match (FM) exemplars, retrieved by character- and embedding-based similarity from authentic and back-translated sentence pools, are included in prompts at training and inference times. Second, synthetic task data: training data for spell- and grammar-checking is generated from dictionaries and monolingual text. Third, a scoring-based inference system: the model generates free text only for the open-ended tasks–translation and maths reasoning–while, for the other three tasks, it scores candidates from constrained spaces by ranking multiple-choice options in place and selecting spelling and grammar corrections from dictionary-derived candidates. Our primary system was the joint winner of the Sorbian track, ranking first or second across all five tasks and outperforming the next-best system by 7.7 accuracy points on question answering and 1.4 points on mathematical reasoning. These results show that, for low-resource languages, FM augmentation combined with back-translation can improve MT performance for a relatively small LLM, while inference-level strategies can further improve its performance on other tasks without compromising MT performance.},
  url       = {https://aclanthology.org/2026.wmt-1.210}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0002-8707-6176
@InProceedings{temesgen-kurtyigit-fraser:2026:wmt,
  author    = {Temesgen, Tsedeniya Kinfe  and  Kurtyigit, Sinan  and  Fraser, Alexander},
  title     = {TUMHN@WMT2026: LLMs with Limited Resources for Slavic Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2681--2687},
  abstract  = {We describe the TUMHN team's systems for the WMT26 Shared Tasks on LLMs with Lim- ited Resources for Slavic Languages. We partic- ipated in the Ukrainian track, covering machine translation, question answering, spell checking, grammar checking, and math reasoning tasks. We restructured our training dataset into a chat-style format and fine-tuned the Qwen3.5-2B model for the Ukrainian track in a multitask learn- ing setting. Our model outperforms the base- line on three of the five tasks, with significant improvements on Question Answering, Spell Checking, and Grammar Checking. In Machine Translation, the model shows a slight improve- ment for the English–Ukrainian direction. For Spell Checking and Grammar Checking specif- ically, the largest gains were observed in spell or grammatical error detection as opposed to correction. In Math Reasoning, however, our model does not outperform the baseline. The fine-tuned model 1 and training dataset 2 are publicly available on Hugging Face.},
  url       = {https://aclanthology.org/2026.wmt-1.211}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{yang-EtAl:2026:wmt2,
  author    = {Yang, Lingchu  and  Fu, Zitong  and  Kathy, Hämmerl  and  Ito, Masaki},
  title     = {Zolint at WMT 2026: A Multitask LLM Submission for the Low-Resource Ukrainian Track},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2688--2695},
  abstract  = {We present a unified multi-task fine-tuning system submitted to the WMT 2026 Shared Task on Multitask LLMs with Limited Resources (Ukrainian track), covering machine translation, spelling correction, grammar correction, question answering, and mathematical reasoning. Our system is built on Qwen3.5-2B-Instruct and adopts a two-stage training pipeline that gradually introduces heterogeneous tasks while replaying selected data from earlier stages to mitigate catastrophic forgetting. The resulting model achieves a strong balance across all evaluated tasks, consistently outperforming the official baseline and obtaining the best performance on every task except question answering. Our results demonstrate that staged multi-task fine-tuning is an effective strategy for adapting compact multilingual LLMs under limited-resource constraints.},
  url       = {https://aclanthology.org/2026.wmt-1.212}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{zhang-EtAl:2026:wmt2,
  author    = {zhang, cong  and  Wang, Yutong  and  Liu, Xuebo  and  Zhang, Min},
  title     = {HITSZ at WMT 2026: Mixed Continued Pre-training and Supervised Fine-tuning for Low-Resource Sorbian},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2696--2708},
  abstract  = {We describe the HITSZ system submitted to the Sorbian track of the WMT 2026 Shared Task on Multitask LLMs with Limited Resources. The track requires a single Qwen3.5-2B model to jointly perform machine translation (MT), question answering (QA), spell checking (SC), grammar checking (GC), and mathematical reasoning (MR) for Upper and Lower Sorbian. To strengthen the model's limited Sorbian linguistic knowledge while retaining previously acquired capabilities, we first perform mixed continued pre-training (CPT) on Sorbian monolingual and parallel data together with multilingual and task-oriented replay. We then apply full-parameter multitask supervised fine-tuning (SFT) to align the adapted model with the five heterogeneous task formats. Compared with the official Qwen3.5-2B baseline, our primary submission improves MT by 41.44 chrF++ points and improves QA, SC, GC, and MR by 13.94, 66.47, 60.77, and 13.20 points, respectively, ranking third overall in the Sorbian track. Additional analysis shows that removing CPT reduces our five-task development score by 5.24 points, with the largest losses on SC, MT, and GC. We also find that heavily increasing MT supervision does not further improve translation and can degrade other jointly trained tasks.},
  url       = {https://aclanthology.org/2026.wmt-1.213}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:0000-0002-3895-5510
@InProceedings{kunilovskaya:2026:wmt,
  author    = {Kunilovskaya, Maria},
  title     = {Translationese as a Rational Response to Translation Task Difficulty},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {387--404},
  abstract  = {Translated texts exhibit systematic differences from comparable texts originally written in the target language. Explaining this phenomenon, commonly known as translationese, remains an open challenge. Translationese has been attributed to production tendencies (e.g. interference, simplification), socio-cultural variables, and language-pair effects, yet a unified explanatory account is lacking. We investigate the hypothesis that translationese is a response to the cognitive load inherent in the translation task. We test whether observable translationese can be predicted from quantifiable measures of translation task difficulty. Translationese is measured as a segment-level probability of being a translation produced by an automatic classifier (translatedness score). Translation task difficulty includes source-text and cross-lingual transfer components. They are captured by information-theoretic metrics based on LLM surprisal and by established syntactic and semantic alternatives. We use a bidirectional English-German corpus comprising written and spoken subcorpora. Results indicate that translationese can only be explained in part by translation task difficulty, especially in the English-to-German direction. For most experiments, cross-lingual transfer difficulty contributes more than source-text complexity. Information-theoretic indicators match or outperform traditional features in written mode, but offer no advantage in spoken mode. Source-text syntactic complexity and translation-solution entropy emerged as the strongest predictors of translationese across language pairs and modes.},
  url       = {https://aclanthology.org/2026.wmt-1.22}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-1473-4684
@InProceedings{liu-koehn:2026:wmt,
  author    = {Liu, Ruoxi  and  Koehn, Philipp},
  title     = {RT-SFT: Text Style Transfer from Non-Parallel Corpora by Roundtrip Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {405--418},
  abstract  = {Text style transfer (TST) is naturally a supervised task — rewrite a sentence in a target style while preserving its meaning — yet the parallel corpora that supervision requires exist for only a handful of style domains. A common workaround is to normalize an input into a style-agnostic intermediate and then stylize it into the target style, but the normalizer is typically a lightweight, task-specific paraphraser applied only at test time, feeding a correspondingly small stylizer. We observe that a style-stripping normalizer already exists at scale: neural MT systems trained on hundreds of millions of general-domain sentence pairs preserve content while regressing toward generic phrasing, so roundtrip translation through a pivot language strips stylistic signal without any task-specific training. This turns normalization from an inference-time patch into a data-generation tool. Roundtrip-translating a monolingual in-style corpus yields a pseudo-parallel corpus on which we LoRA-finetune an instruction-tuned LLM as the stylizer (RT-SFT); applying the same normalizer to test queries keeps that stylizer in-distribution. We show that across four style domains, RT-SFT outperforms state-of-the-art methods, such as few-shot in-context learning, by considerable margins. We also report on effective retrieval augmentation methods for expert style domains with strict terminology and naming conventions.},
  url       = {https://aclanthology.org/2026.wmt-1.23}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{mazumder-EtAl:2026:wmt,
  author    = {Mazumder, Arnav  and  Zhang, Dengjia  and  Li, Shuyue Stella  and  Tsvetkov, Yulia  and  Bafna, Niyati},
  title     = {Multilingual Reasoning Cascades Need More Context},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {419--468},
  abstract  = {Translation cascades for reasoning translate the query from another language to English, reason in English, and translate the answer back to the original language. This is a competitive approach to multilingual reasoning, but structurally lossy, since each stage discards information later stages may need, including cues for cultural grounding, register, and disambiguation. We examine the benefits of a simple and training-free intervention: a context-aware translation cascade, which additionally provides the original question, the English-translated question, and the reasoning trace to the final cascade stage for target-language generation. We evaluate gains across ten multilingual benchmarks including various task types, three backbone models, and 285 high-, mid-, and low-resource languages, and demonstrate strong gains for open-ended generation across models and resource regimes. We show that the original language question carries most of the beneficial context. Our study emphasizes the need to better design information flow in multilingual reasoning cascades for mitigating error propagation, and provides a simple and actionable default strategy: preserve the original user question until the end of the pipeline.},
  url       = {https://aclanthology.org/2026.wmt-1.24}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0000-0003-1769-6772
Author{4}{Orcid}:0000-0002-4634-7128
Author{5}{Orcid}:
@InProceedings{noh-EtAl:2026:wmt,
  author    = {Noh, Dongwon  and  Koh, Donghyeok  and  Lim, Yeon-Soo  and  Jeong, Seohyeong  and  Kim, Yunsu  and  Kim, Gyuwan  and  Do, SooJong  and  Bak, JinYeong  and  Eo, Sugyeong  and  Park, Cheoneum},
  title     = {Lost in Mimicry: Verified Training-Free Post-Editing for LLM Translationese},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {469--498},
  abstract  = {Translations generated by Large Language Models (LLMs) often exhibit unnatural stylistic traces distinct from human-written translations, which we refer to as LLM translationese. We propose \textit{MELT}, a training-free post-editing framework for mitigating LLM translationese across five target languages (Korean, Chinese, Japanese, German, and French) without weight updates. MELT defines translationese patterns based on faithfulness, voice, register, naturalness, and surface form, and combines a three-stage verification gate with selective multi-agent debate to preserve meaning and mitigate self-bias. Experiments on FLORES-200 show that verification contributes most to quality improvement and that MELT achieves the lowest calibration error with respect to Oracle preference across the five languages. Our analysis further shows that LLM translationese exhibits language-specific realizations while sharing a common mechanism of English-structure mimicry.},
  url       = {https://aclanthology.org/2026.wmt-1.25}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-3292-0691
Author{5}{Orcid}:
Author{6}{Orcid}:0000-0002-5081-3945
Author{7}{Orcid}:
Author{8}{Orcid}:https://orcid.org/0000-0002-3212-5241
Author{9}{Orcid}:https://orcid.org/0000-0002-8008-6160
Author{10}{Orcid}:0000-0001-5386-0483
@InProceedings{nehrdich-keutzer:2026:wmt,
  author    = {Nehrdich, Sebastian  and  Keutzer, Kurt},
  title     = {MITRA-MT: Continued Pretraining, Parallel Corpus Mining, and Multi-Directional Machine Translation for Sanskrit, Tibetan, and Buddhist Chinese},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {499--513},
  abstract  = {Buddhist literature is preserved in a network of classical languages, above all Sanskrit, Pāli, Buddhist Chinese, and Tibetan, for which machine translation support remains limited, and largely restricted to translation into English. In order to address this shortcoming, we present three connected contributions. First, we release MITRA-parallel v2, a corpus of 1.69 million automatically aligned parallel records (2.34 million segment pairs) between Sanskrit, Tibetan, and Buddhist Chinese, mined with an embedding-based span-mining pipeline. Second, we present MITRA-MT, a domain-adapted large language model built by continued pretraining of Qwen3.5-9B on a 22.6-billion-token corpus of classical Asian languages and related modern material, followed by a lightweight translation fine-tune and pref- erence optimization. Third, we introduce a multi-directional evaluation suite covering 17 translation directions at sentence level (ten of them also at paragraph level), including cross-classical directions (Sanskrit↔Tibetan, Sanskrit↔Chinese, Tibetan↔Chinese) and from Tibetan into seven modern languages beyond English. On this suite, MITRA-MT outperforms all open baselines, including instruction-tuned LLMs up to 122B parameters and dedicated MT systems, on all directions, and matches commercial references on the large majority of directions when evaluated via chrF, with the largest margins for translation into Tibetan, Sanskrit, and Classical Chinese. We release the parallel corpus, the evaluation data, and the model weights.},
  url       = {https://aclanthology.org/2026.wmt-1.26}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{nehrdich-EtAl:2026:wmt,
  author    = {Nehrdich, Sebastian  and  Allport, David  and  Sandhan, Jivnesh  and  Jagadeeshan, Manoj Balaji  and  Kumar, Sujeet  and  Sellmer, Sven  and  Goyal, Pawan  and  Keutzer, Kurt},
  title     = {Mitrasamgraha: A Comprehensive Classical Sanskrit Machine Translation Dataset},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {514--526},
  abstract  = {Although machine translation is often considered solved for high-resource languages, it remains challenging for texts involving poetic language, philosophical concepts, and layered metaphors, characteristics that are central to Sanskrit literature. Additionally, Sanskrit's rich morphology, sandhi, and compounding, combined with its multi-millennial and multi-domain textual tradition, highlight the need for large, diverse parallel resources, which are currently scarce. We introduce Mitrasaṃgraha, a large-scale Sanskrit-English machine translation dataset containing 391,548 aligned sentence pairs, over 4 times larger than the previously available Itihāsa corpus (Aralikatte et al., 2021). The dataset spans more than 3 millennia of Sanskrit literature across multiple domains and includes temporally annotated data for studying domain and period effects on translation performance. We release manually corrected development (5,587) and test (5,552) sets. Benchmarks show that fine-tuning models such as NLLB and Gemma substantially improves translation quality, and retrieval-augmented prompting further enhances performance. In a controlled cross-dataset comparison, training on Mitrasaṃgraha outperforms both Itihāsa and web-mined bitext for two model families. Despite these gains, our results reveal persistent challenges in translating complex compounds, philosophical terminology, and metaphorical language in Sanskrit, highlighting the need for further research in this area.},
  url       = {https://aclanthology.org/2026.wmt-1.27}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{oostermeijer-jones:2026:wmt,
  author    = {Oostermeijer, Koen  and  Jones, Teryn},
  title     = {LAND: Learning an Adaptive Number of Draws for Bayesian Best-of-N Selection},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {527--536},
  abstract  = {Best-of-N decoding improves generation quality by sampling multiple candidates and selecting the best, but it allocates equal compute to sources regardless of how much they benefit from additional sampling: the expected benefit is small for sources with low-variance score distributions and potentially much larger for those with high-variance distributions. To address this shortcoming, we propose LAND (Learning an Adaptive Number of Draws), which dynamically allocates samples according to their posterior-predictive marginal value. LAND learns a discrete empirical mixture of source-conditioned score distributions from calibration data, updates its belief about a source as completion scores are observed, and stops sampling when the expected improvement of another draw falls below a shared compute price. Across machine-translation experiments with multiple models and language pairs, LAND achieves better MetricX scores than fixed best-of-N at matched average compute. Conversely, at matched translation quality, LAND requires approximately 20-40\% fewer candidate generations. Adaptive allocation therefore provides a simple way to retain the benefits of best-of-N while substantially improving its compute efficiency.},
  url       = {https://aclanthology.org/2026.wmt-1.28}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{otmar-bojar:2026:wmt,
  author    = {Otmar, Antonín  and  Bojar, Ondřej},
  title     = {How many raters do we need to recognize a bad translation via perplexity?},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {537--547},
  abstract  = {We evaluated candidate translation pairs with various kinds of damage using perplexity assigned to them by LLMs to see how it holds up as a fluency and adequacy metric. We also explored how the discriminating power of perplexity can be improved by ensembling.},
  url       = {https://aclanthology.org/2026.wmt-1.29}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-0606-0050
@InProceedings{bennett:2026:wmt,
  author    = {Bennett, Eric R.},
  title     = {Synthetic Slang Generation for Benchmarking Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {33--46},
  abstract  = {Slang presents a persistent challenge in machine translation (MT) because its informal, socially situated meanings can depend heavily on context, especially when a system has not encountered the form or sense in training. We introduce a language-adaptable pipeline for generating and validating realistic synthetic slang, and use it to create a benchmark of 1,087 items across Chinese, Russian, and Farsi. We evaluate ten large language model (LLM)-based MT systems under three conditions: translation from a single usage, translation with an additional example of usage in context, and translation conditioned on a gold definition. We find additional context consistently improves translation quality, especially on slang types which are most difficult without assistance. The benefit of providing the gold definition generally grows as model size shrinks. We also test the ability for models to recover the synthetic slang definition, rising to 99\% accuracy with 2 examples in context. These results demonstrate how synthetic slang can provide a controlled stress test of meaning inference from context in MT systems.},
  url       = {https://aclanthology.org/2026.wmt-1.3}
}

Author{1}{Orcid}:
@InProceedings{pan-seeber:2026:wmt,
  author    = {PAN, Dongpeng  and  Seeber, Kilian G.},
  title     = {Comparing Machine and Human Simultaneous Interpreting},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {548--560},
  abstract  = {We evaluate four machine simultaneous interpreting (SI) systems, an open-weight and a commercial instance of both the cascaded and the end-to-end speech-native architecture, on simulated conference discourse interpreted into five languages. All outputs are compared with professional interpretations. Reference-free quality metrics are first calibrated against blind expert ratings of the human renditions: their association with expert ratings is strongest for content, weaker for presentation, and weakest for style. Machine systems attain content-similarity scores within or above the professional range and higher median segment coverage at the reported alignment threshold. Coverage can count professional compression as omission and is sensitive to transcription and alignment error, so the surplus does not establish better interpreting. For latency, only the commercial speech-native system approaches professional timing. Estimated content lag thus ranges from under twice to more than ten times the professional median, and each system's delay is associated with an observable operating property: commit granularity, synthesis queuing, or compute limits. Together, the rating-calibrated metrics and professional distribution provide a reference-free procedure for evaluating machine interpreting on unsegmented, full-length conference discourse.},
  url       = {https://aclanthology.org/2026.wmt-1.30}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ploeger-EtAl:2026:wmt,
  author    = {Ploeger, Esther  and  Bjerva, Johannes  and  Nguyen, Dong  and  Östling, Robert},
  title     = {An Analysis of Source-Side Diversity in Machine Translation Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {561--580},
  abstract  = {Benchmarks are central to assessing progress in machine translation (MT), yet the content of widely used MT test sets is increasingly scrutinized. Concerns about data contamination and benchmark difficulty have gained traction, but the role of source‑side diversity remains largely overlooked. Many popular benchmarks contain overlapping source items and even exact duplicates, raising the question of how such repetition affects evaluation quality. We empirically examine how source-side diversity shapes MT evaluation. First we measure the inter-source diversity of nine popular MT datasets. Next, our downstream analysis, based on WMT24, suggests that benchmarks with low diversity may be less discriminative and may provide artificially narrow confidence intervals. Ultimately, we call for greater attention to inter‑source diversity in general MT benchmark design.},
  url       = {https://aclanthology.org/2026.wmt-1.31}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{qian-scherrer:2026:wmt,
  author    = {Qian, Shenbin  and  Scherrer, Yves},
  title     = {TransClean: A Benchmark for Detecting and Extracting Clean Translations from Large Language Model Outputs},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {581--596},
  abstract  = {Large language models (LLMs) are increasingly used for machine translation, yet their outputs often contain additional text beyond the translation itself, such as language labels, explanations or bilingual repetitions, which we term translation noise. Despite its prevalence, this problem lacks dedicated benchmarks and systematic study. We analyze over 790,000 translation outputs from 12 LLMs across 22 language pairs (LPs) and identify 12 recurring noise patterns, which we group into formatting and content noise. Building on the observed patterns, we construct TransClean, a controlled benchmark of 9,900 pairs of noisy and clean translation outputs, comprising 8,800 synthetically generated instances and 1,100 manually curated authentic instances. We evaluate two extraction approaches on the TransClean benchmark: 1) a span-based extraction method leveraging translation quality estimation models for span detection, and 2) an LLM-based extraction method that prompts an LLM to isolate the translation. Our benchmark and analysis provide the first systematic framework to evaluate and improve the cleanliness of LLM translation outputs.},
  url       = {https://aclanthology.org/2026.wmt-1.32}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-5247-5073
@InProceedings{rajaee-EtAl:2026:wmt,
  author    = {Rajaee, Sara  and  Vincent, Sebastian  and  Berard, Alexandre  and  Fadaee, Marzieh  and  Marchisio, Kelly  and  Kocmi, Tom},
  title     = {Unlocking Reasoning Capability on Machine Translation in Large Language Models},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {597--613},
  abstract  = {Reasoning-oriented large language models (RLMs) achieve strong gains on tasks such as mathematics and coding by generating explicit intermediate reasoning. However, their impact on machine translation (MT) remains underexplored. We systematically evaluate several RLMs on the WMT24++ benchmark and find that enabling explicit reasoning consistently degrades translation quality across languages and models. Our structural analysis shows that MT reasoning traces are highly linear, lacking revision, self-correction, and exploration of alternative translations, which limits their usefulness. Controlled reasoning-injection experiments demonstrate that providing high-quality reasoning traces from stronger models does not reliably improve weaker models' performance, indicating the failure lies within the format of MT reasoning. To address this mismatch, we propose a structured reasoning framework tailored to translation, based on multi-step improvements and dynamic iterative revision over difficult segments. Post-training a 111B RLM on such structured reasoning traces yields consistent gains over standard translation fine-tuning and injected generic reasoning baselines. Our findings demonstrate that reasoning must be task-structured to benefit MT.},
  url       = {https://aclanthology.org/2026.wmt-1.33}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-8975-165X
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{salvin-EtAl:2026:wmt,
  author    = {Salvin, G.L. John  and  Ramesan, Amisha  and  Chigwededza, Abigairl Nyasha  and  Budde, Shrikant Tryambak  and  Hingmire, Swapnil},
  title     = {DoDS-IITPKD: LoRA Fine-Tuning and LLM Post-Editing for Low-Resource Indic Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {614--619},
  abstract  = {We describe the DoDS-IITPKD submissions to the WMT 2026 Shared Task on Low-Resource Indic Language Translation. We cover six English-centric pairs in both directions: Assamese, Mizo, Khasi, and Manipuri in Category 1, and Bodo and Kokborok in Category 2. Each system adapts a frozen multilingual backbone with low-rank adapters (LoRA with DoRA and rank-stabilised scaling). NLLB-200-3.3B handles Mizo, Khasi, and Kokborok, and IndicTrans2-1B handles Assamese, Manipuri, and Bodo. For the two languages outside the NLLB tokenizer (Khasi and Kokborok) we use same-script surrogate language tags. We add self back-translation for the Indic-English directions and, for some systems, a line-aligned LLM post-editing pass. On the official test sets, our systems obtain the best BLEU of any submission for Bodo to English (36.10) and English to Kokborok (7.36), and the best on time BLEU for Kokborok to English (20.28). LLM post-editing gives large gains in the Indic-English direction (up to +8.20 BLEU) and almost none in the English-Indic direction.},
  url       = {https://aclanthology.org/2026.wmt-1.34}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{salvin-hingmire:2026:wmt,
  author    = {Salvin, G.L. John  and  Hingmire, Swapnil},
  title     = {Making COMET Comparable Across Scripts: Diagnosis and Correction of Tokeniser-Induced Script Bias in Indic MT Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {620--637},
  abstract  = {COMET reports translation quality as a single number, and that number is routinely compared across target languages written in different scripts. Such a comparison assumes Script Invariance: the score should not depend on the writing system that carries the target. We test it on IndicMT Eval by re-encoding the target into Latin script, which changes orthographic form while holding content and human ratings fixed. Script identity then accounts for 22.9\% of native-script COMET variance, and agreement with annotators falls in all five languages studied. We trace the effect to the tokeniser and measure it with three label-free diagnostics. The bias is two faults, not one. Scores from different scripts occupy incompatible ranges, and within a single script the metric orders translations less accurately. No order-preserving transform of the score can repair the second fault. The first is removed exactly by COMET-QN, which maps the score distribution of each (language, script) pair onto a shared reference. Pooled agreement with annotators rises from 0.300 to 0.399, which is what makes scores from different scripts safe to place on one axis, and every within-language ordering is provably preserved. A regressor over parity features recovers a further 17.1\% of the lost sensitivity. The remainder belongs to the encoder, and no post-processing can reach it. We therefore recommend publishing the normalised score, the three diagnostics, and the identity of the tokeniser they were computed against, so that a reader can tell how much of a score reflects translation quality and how much reflects the writing system.},
  url       = {https://aclanthology.org/2026.wmt-1.35}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{scholz-EtAl:2026:wmt,
  author    = {Scholz, Niklas  and  Thulke, David  and  Nasir, Abdallah  and  Allred, Will  and  Matusov, Evgeny  and  Ney, Hermann},
  title     = {Fine-Tuning LLMs for Translation: General Forgetting Mitigation Does Not Preserve MT-Specific Instruction Following},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {638--660},
  abstract  = {Fine-tuning large language models on parallel data improves translation quality but can cause catastrophic forgetting. Mitigation methods are generally evaluated by retention on general benchmarks. We ask whether these findings transfer to machine translation (MT) fine-tuning and to MT-specific instruction following (MT-IF): instructions that modify a translation, such as formality, grammatical gender, and length control. We compare methods anchored to auxiliary data, to model outputs, and to the base model parameters, first in a screening study with Llama 3.2 1B Instruct, then on Llama 3.1 8B Instruct fine-tuned on bidirectional Arabic-English or Spanish-English data. Elastic Weight Consolidation preserves general capabilities best in both stages; on the 8B Spanish model the average score on general benchmarks drops 1.7 points versus 11.0 for standard fine-tuning, yet its scores for formality and grammatical gender control remain close to standard fine-tuning. Only data mixing with control-task examples preserves these controls, but its gains do not transfer to unseen prompts for the same task.},
  url       = {https://aclanthology.org/2026.wmt-1.36}
}

Author{1}{Orcid}:https://orcid.org/0009-0007-5754-6011
Author{2}{Orcid}:https://orcid.org/0000-0002-9808-7073
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{serdioukova-zhilko:2026:wmt,
  author    = {Serdioukova, Anastasiya  and  Zhilko, Denis},
  title     = {AQI: a composite human-calibrated metric for machine translation quality assessment},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {661--679},
  abstract  = {Manual evaluation of machine translation quality is the gold standard but expensive and noisy. Professional MQM annotation and holistic human scoring both require trained annotators, scale linearly with throughput, and suffer from substantial inter-annotator disagreement, particularly on strong neural systems of recent generations. At production volumes manual evaluation does not scale: automatic evaluation must carry the main load, and humans should remain a corrective signal on a limited subset of segments. The modern automatic-metric inventory includes surface metrics (chrF, TER), embedding-based metrics (BERTScore F1), and learned metrics (COMETKiwi-XXL, MetricX-XXL, xCOMET-XXL). Our audit on the WMT25 General-MT corpus (n=3,450 segment-system ratings, 5 language pairs) shows that no single metric reaches the human-agreement ceiling: the strongest (MetricX-XXL) only approaches it, and combining metrics adds a small but reliable margin. Production stakeholders (PMs, lead translators, clients) at the same time expect a single scalar in the 0–100 range with clear Good/Borderline/Bad thresholds, not a vector of six heterogeneous numbers. We propose the Alconost Quality Index (AQI) — a composite, human-calibrated index of machine translation quality, published in two equal-billing versions differing in the normalisation of the key metric and accompanied by an optional per-LP calibration. AQI is designed as an interpretable, production-friendly instrument: a single score, explicit weights, a transparent path from input automatic metrics to the final number. Beyond purely automatic deployment we derive a simple human-in-the-loop index — a linear blend of one human score into the auto-only AQI — and empirically select the blend weight via honest validation against an independent annotator.},
  url       = {https://aclanthology.org/2026.wmt-1.37}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{shayegh-kazemi:2026:wmt,
  author    = {Shayegh, Behzad  and  Kazemi, Niloofar},
  title     = {Mind Which Bird You Favour: Parameterizing Adequacy-Fluency Balance in Meta-Evaluation of Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {680--704},
  abstract  = {There is a tradeoff in machine translation meta-evaluation between prioritizing alignment with adequacy versus fluency. The balance depends on the combination of translation systems in the meta-evaluation dataset. This system set is a small, filtered sample whose characteristics change heavily across years and language pairs; it does not represent the true system distribution. Consequently, the adequacy-fluency balance is often unrepresentative and subject to change. For sensitive domains, controlling this balance is critical. We expose this balance as a tunable choice. To achieve a target balance, we reweight existing systems while minimizing distortion from uniform weighting, ensuring the evaluated systems remain real and representative. We provide an exact optimization algorithm with theoretical guarantees and pruning mechanisms to compute these weights. To validate meta-evaluation internal consistency, we design a scorer-augmentation framework that establishes a known relative identity for the scorers. Results demonstrate that our reweighting method effectively controls the adequacy-fluency balance and preserves the internal consistency, outperforming prior approaches. Finally, we analyze the performance of popular scorers across a sweep of this parameter.},
  url       = {https://aclanthology.org/2026.wmt-1.38}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{sheng-dimarco-fraser:2026:wmt,
  author    = {Sheng, Qimin  and  Di Marco, Marion  and  Fraser, Alexander},
  title     = {In-Context Learning for Upper Sorbian to German Translation: Comparing Lexical and Morpho-Syntactic Information},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {705--717},
  abstract  = {LLM-based translation remains challenging for low-resource languages, as LLMs often lack linguistic knowledge due to their limited representation in pre-training data. This is particularly problematic for morphologically rich languages where a large vocabulary can lead to further data sparsity. Previous work has shown that providing external guidance on the source sentence in the form of language-specific information can help LLMs with tasks such as machine translation. In this study, we investigate zero-shot context augmentation for Upper Sorbian-to-German translation by enriching prompts with different types of contextual information: translation candidates in the form of bilingual dictionary entries, and morpho-syntactic analysis provided through morphological tagging. We first compare the effectiveness of these types of information in a zero-shot setting and then evaluate whether they generalize across additional experimental settings. We find that context-augmented prompting can improve Upper Sorbian-to-German translation. Contexts only containing translation candidates perform overall best, whereas morpho-syntactic analysis offers limited benefits, both on its own and in combination with lexical cues.},
  url       = {https://aclanthology.org/2026.wmt-1.39}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{carraro-EtAl:2026:wmt,
  author    = {Carraro, Fabrício  and  Zevallos, Rodolfo Joel  and  Gonçalves de Souza, Rodrigo  and  Faustino da Silva, Caio Henrique  and  Ortega, John E.},
  title     = {The Constitution Speaks Nheengatu: An Open MT System and Reproducible Corpus Pipeline for the Amazonian Língua Geral},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {47--63},
  abstract  = {Nheengatu (yrl), the Amazonian Língua Geral, has 6-10 thousand speakers and is co-official in São Gabriel da Cachoeira (Brazil), yet remains poorly served by evaluated Portuguese-Nheengatu MT resources. We present the largest parallel corpus assembled for the language (14.5k Portuguese-Nheengatu pairs), anchored by the 2023 official translation of the Brazilian Federal Constitution; the first MT evaluation set built from transcribed spontaneous speech; and, to our knowledge, the first open-weight MT models dedicated to bidirectional Portuguese-Nheengatu translation. Three findings organize the paper. First, translation into Nheengatu on held-out constitutional articles plateaus on development data regardless of training length or model size, and much of the remaining error sits in conventional legal terms that the written sources do not attest consistently; a terminology success rate over a frozen lexicon shows that glossary mechanisms at training time and lexical constraints at decoding each raise term realization, the latter at a cost in repetitive output that chrF++ barely registers. Second, a corpus expansion that adds transcribed conversation raises speech chrF++ by 24-27 points while leaving written domains unchanged, a register gap that written-only benchmarks cannot see. Third, a larger NLLB model helps translation into Portuguese but not into Nheengatu on legal or speech text. For Portuguese-to-Nheengatu translation, our final systems reach 31.7-32.8 chrF++ on article-disjoint constitutional text, 64.1-64.5 on the near-domain set (62.1 after removing items with target-side content overlap), and 45.2-47.5 on speech.},
  url       = {https://aclanthology.org/2026.wmt-1.4}
}

Author{1}{Orcid}:https://orcid.org/0009-0007-9392-7305
Author{2}{Orcid}:https://orcid.org/0000-0003-0192-7740
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:0000-0002-2328-3205
@InProceedings{sukhareva-enikeeva:2026:wmt,
  author    = {Sukhareva, Maria  and  Enikeeva, Ekaterina},
  title     = {Retrieval-Augmented Terminology Translation for English-Russian: A Multi-Domain Study},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {718--733},
  abstract  = {Neural machine translation reaches near-human quality on general-domain text but systematically fails on specialized terminology, where polysemous and low-frequency terms carry domain-specific meanings absent from general training data. We investigate two complementary interventions for English-Russian specialized translation across ten professional domains: (i) a four-stage retrieval-augmented pipeline of LLM-based term extraction, domain classification, lemma-based knowledge-base retrieval, and glossary-augmented translation; and (ii) a preference-based fine-tuning stage combining supervised fine-tuning (SFT) with Contrastive Preference Optimization (CPO) layered on top of retrieval. We release an open multi-domain English-Russian terminology knowledge base of 50,000 terms with 120,161 domain-tagged translations, a 200-item curated test set selected for terminology difficulty, and a 1,300-item Wikipedia-derived test set. Across four open base models, retrieval-augmented prompting alone delivers the bulk of the terminology gain and already exceeds the two commercial reference systems; the additional preference-tuning stage contributes only a small further gain at a measurable fluency cost. The pattern transfers to the WMT25 Terminology Translation Task, the most recent iteration of the shared task that included English-Russian as a language pair.},
  url       = {https://aclanthology.org/2026.wmt-1.40}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{thavarasa-thevakumar-sukumar:2026:wmt,
  author    = {Thavarasa, Luxshan  and  Thevakumar, Jubeerathan  and  Sukumar, Sivasuthan},
  title     = {Obligatory Slots: Under Reference-Free Evaluation, Dropping a Distinction the Source Never Made Is Free},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {734--753},
  abstract  = {When the target language must mark a distinction the source does not, a translation system must still pick a value. English they came becomes one Tamil verb if those who came are people and another if they are not; English does not say which. We ask what automatic evaluation charges when the choice is wrong, or never made. We release TamilLingBench, an English→Tamil challenge set for three verb-agreement distinctions, checked by a morphological analyser: rationality (திணை tiṇai), gender, and number as a control. We pre-registered a prediction that neural metrics would not notice a single wrong morpheme. It is wrong for the neural reference-based metrics: COMET-22 charges 8.52× the score difference that human raters reliably notice. On the systems' own errors, both reference-free metrics rank the wrong form first more often than not, and a legitimate Tamil form that leaves the distinction out is not charged at all. Under reference-free evaluation, dropping the distinction is free; reference-based metrics do charge for it. xCOMET finds the error's region but not the morpheme. Inside one model, activation patching finds the same asymmetry: a distinction English marks is settled by the middle of the network, one it leaves unmarked only near the output.},
  url       = {https://aclanthology.org/2026.wmt-1.41}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{thompson:2026:wmt,
  author    = {Thompson, Isaac},
  title     = {AmanaMT: Translation Direction Predicts Automatic Metric Reliability},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {754--771},
  abstract  = {When machine translation (MT) metrics fail in low-resource settings, the field typically blames data scarcity. We show this diagnosis is often wrong using AmanaMT, a unified benchmark spanning 730,748 segments, 33 languages, three annotation tiers, and 10 metrics: the dominant predictor of metric unreliability is not resource level but domain mismatch and translation direction. Direction is the single largest predictor of between-language variance in metric reliability (η2dir = 0.309, nearly a third of between-language variance); morphological type is a substantially smaller main effect (η2morph = 0.073), though the joint direction×morphology cells account for η2cell = 0.618, capturing direction-specific morphological variation. Ukrainian's near-zero COMET correlation is fully explained by a domain label (other, not news): a genre confound, not a language problem. Russian's weak aggregate dissolves into clean per-year signals once stratified by WMT year, a textbook Simpson's Paradox from pooling heterogeneous system generations. Neural metrics (MetricX-24, xCOMET, COMET) substantially outperform n-gram baselines at every tier; the gap is largest in morphologically complex and non-news settings where surface overlap is an especially poor proxy for adequacy.},
  url       = {https://aclanthology.org/2026.wmt-1.42}
}

Author{1}{Orcid}:
@InProceedings{uramov-EtAl:2026:wmt,
  author    = {Uramová, Šárka  and  Petrov, Petar Kirilov  and  Jon, Josef  and  Bojar, Ondřej  and  Novák, Michal},
  title     = {ImgMT: Wikipedia-based Dataset For Evaluating Text-in-Image Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {772--788},
  abstract  = {Text-in-image translation aims to translate all textual content embedded in an image. In Text Image Translation (TIT), the output is translated text; In-Image Machine Translation (IIMT) additionally renders the translation back into the image. We introduce ImgMT, a massively multilingual dataset and benchmark mined from localized SVG illustrations on Wikimedia Commons, focusing on diagrams, maps, and infographics. ImgMT contains 35,612 aligned image pairs grouped into 1,855 image sets, spanning 129 languages, 20+ writing systems, and 4,886 language pairs. We propose a multi-level evaluation protocol that measures source-side text-region detection and OCR accuracy alongside target-side translation quality. We benchmark several multimodal large language models and a dedicated TIT system in English-to-X and X-to-English settings, and find that the source language and writing system substantially affect TIT quality. Finally, we show that ImgMT is not well suited for isolating the contribution of visual information in multimodal translation.},
  url       = {https://aclanthology.org/2026.wmt-1.43}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-0606-0050
Author{5}{Orcid}:0000-0002-6052-7459
@InProceedings{wang-EtAl:2026:wmt1,
  author    = {Wang, Xiaotian  and  Lin, Youyuan  and  Shen, Zhan  and  Yanaka, Hitomi},
  title     = {Doc2FRC: Length-Consistent Document-Level Machine Translation via Fixed-Range Chunking},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {789--824},
  abstract  = {Advanced large language models (LLMs) with long context windows can substantially reduce input truncation in document-level machine translation (DocMT). However, direct Doc2Doc translation remains prone to n-gram repetition and progressive quality degradation. A common remedy is to segment the document into finer-grained chunks. Nonetheless, conventional rule-based chunking approaches fail to handle the length distribution mismatch between training and inference. To address this, we introduce Fixed-Range Chunking (FRC), utilizing dynamic programming to partition documents into chunks within a predefined length interval. By consistently applying FRC during training and inference, the input documents of any length are mapped to the same length distribution, substantially reducing train-test length mismatch. Centered on FRC, we propose a lightweight dual-boundary matching algorithm for chunk alignment, alongside four distinct training strategies. Experimental results show that FRC-based fine-tuning substantially improves 7B LLMs over direct Doc2Doc fine-tuning and outperforms existing DocMT methods on IWSLT2017. We further construct GlobVDoc, a 10-language test set independent of mainstream DocMT training sources, and show that FRC improves out-of-distribution document translation.},
  url       = {https://aclanthology.org/2026.wmt-1.44}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{wu-wieting-smith:2026:wmt,
  author    = {Wu, Si  and  Wieting, John  and  Smith, David A.},
  title     = {Learning from Many Voices: Literary MT Using Multi-Reference Human and Synthetic Data},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {825--848},
  abstract  = {Unlike many other texts, literary works are often translated multiple times. We investigate strategies for leveraging these multi-reference datasets to improve literary machine translation. We propose a filtering framework based on semantic similarity to identify source texts whose references display meaningful variation while remaining faithful. We find that fine-tuning with medium to high semantic similarity data substantially outperforms low semantic similarity data. Moreover, using medium and high semantic similarity data achieves comparable or better performance than using the full unfiltered data. Synthetic translations generated by LLMs are economical and convenient alternatives to human expert translations; however, we find fine-tuning on human expert translations outperforms fine-tuning on synthetically augmented data in automatic metrics and human evaluations, demonstrating the indispensable value of human expert translations for fine-tuning literary machine translation models.},
  url       = {https://aclanthology.org/2026.wmt-1.45}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{zhao-EtAl:2026:wmt,
  author    = {Zhao, Jim  and  Maskey, Sohir  and  Oostermeijer, Koen  and  Orr, Douglas  and  Jones, Teryn},
  title     = {Studying quantization trade-offs for efficient inference deployment in machine translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {849--866},
  abstract  = {Deploying large language models in realistic server environments poses challenges, as the system needs to provide high-quality responses with low latency. Quantization is a common approach to reduce the memory footprint and improve inference efficiency, yet its impact on latency and throughput is rarely evaluated under controlled, orchestration-level workloads. In this work we study the quantization trade-offs of EuroLLM \citep{martins2025eurollm} across three model sizes ranging from 1.7B to 22B for efficient deployment on a single A100 or H100 GPU. We demonstrate that combining a document-chunking strategy with W4A8 or W8A8 quantization improves the latency-throughput Pareto-curve under a wide range of workloads. Furthermore, since standard machine translation (MT) benchmarks rely on isolated sentences and fail to capture long-context dynamics, we introduce a document-level evaluation based on DocHPLT \cite{o2025dochplt} to assess how text chunking strategies affect translation quality under quantization. Our results indicate that standard segment-level evaluation can potentially underestimate the interaction between quantization and long-context document translation, for some quantization formats, translation direction and models. Overall, our experiments show that the trade-off between inference efficiency and translation quality depends not only on the quantization format, but also on the choice of text chunking strategy.},
  url       = {https://aclanthology.org/2026.wmt-1.46}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{zouhar-EtAl:2026:wmt,
  author    = {Zouhar, Vilem  and  Grundkiewicz, Roman  and  Rajaee, Sara  and  Riley, Parker  and  Bawden, Rachel  and  Koehn, Philipp  and  Carpuat, Marine  and  Kocmi, Tom},
  title     = {Contrastive ESA: Human Evaluation of Multiple Translations at Once},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {867--879},
  abstract  = {Current human evaluation of machine translation typically assesses single outputs in isolation, a paradigm that suffers from high annotator noise and cost. We introduce Contrastive Error Span Annotation (cESA), a protocol that presents multiple translations of the source input (text, video, audio, image). In cESA, the annotator sees multiple translations of the same document, marks major and minor error spans, and then assigns a score from 0\% to 100\% on absolute scale. By allowing annotators to access the shared context across multiple outputs, cESA facilitates more consistent and efficient judgments. We validate cESA using a large-scale human evaluation of English->Japanese translations of 12 models, demonstrating reductions in annotation time and noise compared to standard pointwise evaluation. Unlike existing contrastive ranking methods, cESA yields absolute quality judgments that enable simple, interpretable non-parametric model rankings without the need for post-hoc corrections.},
  url       = {https://aclanthology.org/2026.wmt-1.47}
}

Author{1}{Orcid}:0000-0001-9874-2069
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:https://orcid.org/ 0000-0003-1693-0782
Author{8}{Orcid}:
@InProceedings{kocmi-EtAl:2026:wmt1,
  author    = {Kocmi, Tom  and  Artemova, Ekaterina  and  Avramidis, Eleftherios  and  Bawden, Rachel  and  Bojar, Ondřej  and  Dukanov, Sergey  and  Dvorkovich, Anton  and  Fishel, Mark  and  Freitag, Markus  and  Frontull, Samuel  and  Gowda, Thamme  and  Grundkiewicz, Roman  and  Haddow, Barry  and  Kharevich, Stan  and  Koehn, Philipp  and  Li, Zheng  and  Maillard, Jean  and  Monz, Christof  and  Murauski, Alexander  and  Murray, Kenton  and  Nagata, Masaaki  and  Perrella, Stefano  and  Popel, Martin  and  Popović, Maja  and  Proietti, Lorenzo  and  Rajaee, Sara  and  Riley, Parker  and  Shmatova, Mariya  and  Steingrímsson, Steinþór  and  Yankovskaya, Lisa  and  Zouhar, Vilem},
  title     = {Findings of the WMT26 General Machine Translation Shared Task: Contrastive Dynamic Human Evaluation at Scale},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {880--930},
  abstract  = {This paper presents the results of the General Machine Translation Task organized under the 2026 Conference on Machine Translation (WMT). Participants build systems for any of the 23 language pairs spanning four to five domains. This year we brought major changes to the human evaluation: (1) new annotation platform Pearmut for better reproducibility, (2) new human protocol "contrastive Error Span Annotation" for higher quality and side-by-side evaluation, and (3) Dynamic human evaluation that allows assessing all submitted models while evaluating higher-performing models more frequently. Beyond human evaluation, we (4) extended difficulty sampling with a human-driven stage, (5) added two new domains, (6) made the test sets fully document-level without requiring segment-level alignment and relying on HTML or JSON structure, (7) prepared some human references by post-editing open-weight model outputs, and (8) introduced contextual instructions governing formality and structural style. We evaluated 46 systems in total: 30 submitted by participants and 16 consisting of translations from large language models (LLMs) and industry translation providers.},
  url       = {https://aclanthology.org/2026.wmt-1.48}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-5671-573X
Author{4}{Orcid}:
Author{5}{Orcid}:https://orcid.org/0000-0002-0606-0050
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:https://orcid.org/0000-0003-1932-2600
Author{9}{Orcid}:
Author{10}{Orcid}:0009-0004-1230-4666
Author{11}{Orcid}:https://orcid.org/0000-0001-5422-8674
Author{12}{Orcid}:
Author{13}{Orcid}:
Author{14}{Orcid}:
Author{15}{Orcid}:
Author{16}{Orcid}:
Author{17}{Orcid}:https://orcid.org/0000-0003-0025-1021
Author{18}{Orcid}:
Author{19}{Orcid}:
Author{20}{Orcid}:
Author{21}{Orcid}:https://orcid.org/0009-0006-6115-4566
Author{22}{Orcid}:https://orcid.org/0000-0001-8900-8049
Author{23}{Orcid}:
Author{24}{Orcid}:
Author{25}{Orcid}:https://orcid.org/0000-0002-2428-5878
Author{26}{Orcid}:
Author{27}{Orcid}:
Author{28}{Orcid}:
Author{29}{Orcid}:
Author{30}{Orcid}:
Author{31}{Orcid}:0000-0001-9874-2069
@InProceedings{lavie-EtAl:2026:wmt,
  author    = {Lavie, Alon  and  Hanneman, Greg  and  Perrella, Stefano  and  Ding, Shuoyang  and  Avramidis, Eleftherios  and  Proietti, Lorenzo  and  Lo, Chi-kiu  and  Shurtz, Ammon  and  Zerva, Chrysoula  and  Sindhujan, Archchana  and  Zouhar, Vilem  and  Kanojia, Diptesh  and  Blain, Frederic  and  Thompson, Brian  and  Filandrianos, Giorgos  and  Menis Mastromichalakis, Orfeas  and  Kocmi, Tom  and  Gupta, Pranav},
  title     = {Findings of the WMT26 Shared Task on Automated Translation Quality Evaluation Systems: Compact Open Models are Competitive Quality Evaluators},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {931--1030},
  abstract  = {We present the findings of the WMT26 Shared Task on Automated Translation Quality Evaluation Systems, continuing last year's unification of the earlier separate WMT Metrics and Quality Estimation shared tasks. This year we evaluated three complementary views of segment-level translation quality on a common test set: fine-grained error-span detection, continuous quality-score prediction, and a new task on identifying error-free translations. Submissions to upstream tasks were also converted automatically to downstream predictions. The evaluation covered 21 translation directions using human cESA judgments from the WMT26 General Machine Translation task, with optional reference translations generated as either native, post-edited, or pseudo-references, depending on the translation direction. Official evaluation data was complemented by five submitted challenge sets. Across all three primary tasks, unsupervised LLM-as-a-judge approaches outperformed all traditional and supervised metrics, while open-weight models such as Gemma 4 were found to be largely competitive with proprietary frontier models. Reference-free LLM judges are highly competitive, and the benefit of adding a reference largely depends on its provenance and on the evaluator model family.},
  url       = {https://aclanthology.org/2026.wmt-1.49}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0001-8900-8049
Author{4}{Orcid}:
Author{5}{Orcid}:https://orcid.org/0000-0002-5671-573X
Author{6}{Orcid}:https://orcid.org/0000-0002-2428-5878
Author{7}{Orcid}:https://orcid.org/0000-0001-8714-7846
Author{8}{Orcid}:
Author{9}{Orcid}:https://orcid.org/0000-0002-4031-9492
Author{10}{Orcid}:https://orcid.org/0000-0002-6467-6873
Author{11}{Orcid}:0000-0001-9874-2069
Author{12}{Orcid}:https://orcid.org/0000-0001-8814-0080
Author{13}{Orcid}:https://orcid.org/0000-0003-3017-3722
Author{14}{Orcid}:
Author{15}{Orcid}:https://orcid.org/ 0000-0002-7015-7746
Author{16}{Orcid}:
Author{17}{Orcid}:
Author{18}{Orcid}:
@InProceedings{chakma-EtAl:2026:wmt,
  author    = {Chakma, Aunabil  and  Chakma, Aditya  and  Hasan, Masum  and  Khisa, Soham  and  Tripura, Chumui  and  Shahriyar, Rifat},
  title     = {ChakmaNMT: Machine Translation for a Low-Resource and Endangered Language via Transliteration},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {64--81},
  abstract  = {We present a systematic study of machine translation for Chakma, an endangered and extremely low-resource Indo-Aryan language. We introduce a large Chakma–Bangla machine translation resource comprising 15,021 parallel translation pairs, 42,783 Chakma monolingual sentences, and a trilingual evaluation benchmark. To address data scarcity and the script mismatch between Chakma and Bangla, we develop a character-level transliteration framework that enables transfer from Bangla and multilingual pretrained models. We evaluate from-scratch machine translation systems, fine-tuned pretrained models, and large language models using in-context learning. Our results show that transliteration is crucial for the tested pretrained models and that performance is strongly direction-dependent: in-context learning performs best for Chakma-to-Bangla translation, while Bangla-to-Chakma remains considerably more challenging, with the strongest approach depending on the evaluation metric.},
  url       = {https://aclanthology.org/2026.wmt-1.5}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:https://orcid.org/0000-0001-9540-315X
@InProceedings{schmidtova-EtAl:2026:wmt,
  author    = {Schmidtova, Patricia  and  Artemova, Ekaterina  and  Aycock, Seth  and  Bafna, Niyati  and  Banga, Shobhit  and  Kaur, Manmeet  and  Kocmi, Tom  and  Koehn, Philipp  and  Liu, Danni  and  Luu, Nam  and  Papi, Sara  and  Savoldi, Beatrice  and  Shmatova, Mariya  and  Sidh, Hanuman  and  Zerminova, Evfrosiniya  and  Zouhar, Vilem  and  Züfle, Maike  and  Chen, Pinzhen},
  title     = {Findings of the WMT26 Multilingual Instruction Shared Task: Small Models Are Not Yet Polyglots},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1031--1052},
  abstract  = {We present the findings of the WMT26 Multilingual Instruction Shared Task (MIST), evaluating small (<=10B parameter) open-weight language models across 24 languages on three sub-tasks: context-based QA, cross-lingual summarization, and open-ended generation. Submissions from eight teams explore fine-tuning, knowledge distillation, and prompt engineering. Our findings demonstrate that small models are not yet polyglots: while leading systems perform well on high-resource languages, performance drops sharply on low-resource varieties, and comprehension degrades significantly when transferring across non-English language pairs. In open-ended generation, human evaluation and automatic rule-based verification exhibit strong overall rank correlation (0.79) but diverge locally at the top: human evaluators favor distilled pipelines with superior fluency and cross-lingual equity, whereas automatic verifiers reward concise prompt-engineered systems. Furthermore, multi-stage pipelines suffer from prompt-language inertia on cross-lingual instructions, and summarization exhibits severe English leakage unless penalized by language gating. Finally, while baseline fluency is largely achieved by modern small models, multi-constraint instruction following and unanswerable question refusal remain the primary performance bottlenecks. We release all test sets, system outputs, human judgments, and evaluation code under a permissive license.},
  url       = {https://aclanthology.org/2026.wmt-1.50}
}

Author{1}{Orcid}:0009-0008-5516-798X
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:https://orcid.org/0000-0001-5419-1963
Author{10}{Orcid}:
Author{11}{Orcid}:https://orcid.org/0000-0002-4494-8886
Author{12}{Orcid}:
Author{13}{Orcid}:
Author{14}{Orcid}:
Author{15}{Orcid}:
Author{16}{Orcid}:0000-0001-9874-2069
Author{17}{Orcid}:https://orcid.org/0009-0001-7238-7705
Author{18}{Orcid}:https://orcid.org/0000-0003-0089-5118
@InProceedings{charkiewicz-EtAl:2026:wmt,
  author    = {Charkiewicz, Adrian  and  Chen, Pinzhen  and  Etchegoyhen, Thierry  and  Gete, Harritxu  and  Guttmann, Kamil  and  Huang, Xu  and  Ponce, David  and  Nowakowski, Artur  and  Odermatt, Frederic  and  Oncevay, Arturo  and  Zhu, Dawei  and  Zouhar, Vilem  and  Semenov, Kirill},
  title     = {Findings of the WMT26 Terminology Translation Task: The Hard Part is Finding the Terms, Not Using Them},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1053--1086},
  abstract  = {The WMT26 Terminology Translation Task aims to evaluate machine translation in high-stakes, term-heavy domains (technology, finance, medicine). This year, we focus solely on document-level translation and run two tasks: (1) MT with explicit document-level dictionaries, (2) MT with bitext samples that contain the specific terms. Participants are presented with the texts in three translation directions, two of which feature mid-to-low-resourced morphologically rich languages: Spanish${\rightarrow}$Basque, English${\rightarrow}$Polish, and Traditional Chinese${\rightarrow}$English. This year, the main metrics were multiple variants of overall translation quality and terminology success rate; in line with previous shared tasks, we also compared systems with no terminology, proper dictionaries, and random dictionaries to causally analyze terminology utility. 17 teams participated in our task, submitting 21 systems to Track 1 and 18 systems to Track 2. The results show that document-level translation with explicit dictionaries is close to saturation, and the best systems nearly reach the reference texts. In contrast, for translation with sample bitexts, the spread of the systems is bigger, and the best scores are lower, highlighting the need to concentrate on terminology extraction rather than its use from an explicit source. We also evaluate the grammaticality of the generated texts in Basque and Polish and observe slight trends toward using dictionary forms of the terms and assigning the most frequent grammemes.},
  url       = {https://aclanthology.org/2026.wmt-1.51}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0003-0089-5118
Author{3}{Orcid}:
Author{4}{Orcid}:0009-0005-3955-5059
Author{5}{Orcid}:
Author{6}{Orcid}:https://orcid.org/0009-0006-0385-4054
Author{7}{Orcid}:
Author{8}{Orcid}:https://orcid.org/0000-0003-4473-0008
Author{9}{Orcid}:
Author{10}{Orcid}:
Author{11}{Orcid}:
Author{12}{Orcid}:0000-0001-9874-2069
Author{13}{Orcid}:
@InProceedings{gowda-EtAl:2026:wmt,
  author    = {Gowda, Thamme  and  Gaido, Marco  and  Grundkiewicz, Roman  and  Negri, Matteo},
  title     = {Findings of the WMT 2026 Shared Task on Model Compression: No Free Lunch at Extreme Compression},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1087--1097},
  abstract  = {We present the second edition of the WMT 2026 Model Compression (WMT26MC) shared task, which studies how large language models can be made efficient for machine translation under model-size and latency constraints. WMT26MC offers a constrained track based on Gemma~3~12B and an unconstrained track for compressing models below 20B. Both tracks cover cs-de, en-zh, and en-ar. Participants submit complete runnable systems, which we run in a standardized environment on single H100 GPU. Evaluation jointly considers translation quality, model size on disk, and decoding speed, analyzing the resulting quality--size and quality--speed trade-offs. We received 41 submissions from 13 teams. The proposed systems span weight and activation quantization, structured and expert pruning, low-rank factorization, knowledge distillation, vocabulary and vision-component removal, mixed precision, and inference-time reranking. Evaluating with reference-free quality-estimation metrics on the blind test sets, we find that post-training quantization is the reliable lever: INT4 and FP8 systems shrink the constrained base to roughly a quarter of its on-disk size and run up to an order of magnitude faster while staying within quality-estimation noise of the uncompressed model, whereas aggressive low-rank and structural pruning collapse quality.},
  url       = {https://aclanthology.org/2026.wmt-1.52}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-5422-8674
Author{2}{Orcid}:0000-0003-4217-1396
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-8811-4330
@InProceedings{pakray-EtAl:2026:wmt,
  author    = {Pakray, Partha  and  Pal, Santanu  and  Vetagiri, Advaitha  and  Singh, Kshetrimayum Boynao  and  Dash, Sandeep Kumar  and  Maji, Arnab Kumar  and  Lyngdoh, Saralin A.  and  Laitonjam, Lenin  and  Jamatia, Anupam  and  Das, Ajit  and  Warjri, Sunita  and  Sharma, Uzzal  and  Sambyo, Koj  and  Manna, Riyanka},
  title     = {Findings of WMT 2026 shared task on Low-resource Indic Languages Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1098--1123},
  abstract  = {This paper presents the findings of the Lowresource Indic Languages Translation Shared Task organized at the Eleventh Conference on Machine Translation (WMT 2026). We evaluated machine translation systems across ten English–Indic language pairs spanning three major language families of Northeast India, grouped into moderately resourced and extremely low-resource categories. Built upon the expanded INDICNE-CORP 2.0 dataset, this edition introduced a rigorous, bidirectional multidomain benchmark covering the Healthcare, Political, Travel, Sports, and Entertainment domains to assess real-world generalizability. The task attracted 24 participating teams who deployed diverse methodologies, including parameter-efficient fine-tuning (PEFT) of multilingual foundations, retrieval-augmented generation (RAG) using frontier large language models (LLMs), and hybrid neuro-symbolic decoding constraints. The systems were evaluated using a comprehensive suite of lexical metrics (BLEU, METEOR, TER, ChrF++) and semantic metrics (BERTScore, COMET). Our analysis reveals a persistent performance cliff under extreme data scarcity and a pronounced directional asymmetry, demonstrating that while current architectures readily decode Indic representations into English, they struggle to generate morphologically complex Indic target text. By publicly releasing these datasets, evaluation protocols, and competitive baselines, this shared task establishes a foundation for advancing research in translation technologies for underrepresented and indigenous languages.},
  url       = {https://aclanthology.org/2026.wmt-1.53}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-3834-5154
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
Author{10}{Orcid}:
Author{11}{Orcid}:
Author{12}{Orcid}:
Author{13}{Orcid}:
Author{14}{Orcid}:
@InProceedings{laskar-EtAl:2026:wmt,
  author    = {Laskar, Sahinur Rahman  and  Alam, Firoj  and  Paul, Bishwaraj  and  Ahmad, Irfan  and  Lydia, Maya Silvi  and  Dadure, Pankaj},
  title     = {Findings of the WMT 2026 Shared Task on Low-Resource Arabic-Asian Language Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1124--1137},
  abstract  = {We present the findings of the WMT 2026, which evaluated bidirectional translation between Arabic and English, Hindi, Bangla, Indonesian, and Urdu. We release AraAsian1.0, a news-domain parallel corpus with 20.1K-21.0K training pairs per language pair, and evaluate submissions on ten translation directions using BLEU, ChrF2, TER, COMET, and BERT-based metrics. Of 23 registered teams, 13 submitted systems, producing 70 primary and 65 contrastive runs. Fine-tuned multilingual MT models remained the strongest overall: MADLAD-based systems won four primary directions, while target-aware pivoting, joint multilingual adaptation, and MBR-based ensembling led the remaining tracks. The best contrastive run exceeded the best primary run in only three of ten directions, and each gain was below 0.2 BLEU, indicating that additional data, prompting, or multi-pass refinement did not provide consistent improvements. Bangla had the lowest winning BLEU in both directions, despite comparable training-set size. BLEU and COMET selected the same winner in nine of ten primary results but only six of ten contrastive results, with the largest disagreements occurring for LLM-based and reranked outputs. These results emphasize domain-matched adaptation, target-aware transfer, and multi-metric reporting in low-resource Arabic-Asian translation. We made the AraAsian1.0 dataset available for the community.},
  url       = {https://aclanthology.org/2026.wmt-1.54}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-8413-2718
Author{2}{Orcid}:https://orcid.org/0000-0001-7172-1997
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{dong-EtAl:2026:wmt,
  author    = {Dong, Tianyu  and  Chen, Ziyan  and  Liu, Jingsong  and  Zhu, Shaolin},
  title     = {Findings of the WMT26 Chinese–Southeast Asian Multilingual MT Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1138--1144},
  abstract  = {We present the findings of the WMT26 Chinese–Southeast Asian Multilingual Machine Translation shared task. The task evaluates fourteen translation directions between Chinese and Thai, Vietnamese, Lao, Burmese, Khmer, Indonesian, and Malay. Organizers released manually aligned parallel data, domainmatched monolingual data, validation data, and a blind test set. The published leaderboard contains eleven systems and combines translation quality, measured by the equal-weight average of sacreBLEU and COMET, with a throughput bonus measured on a shared four-GPU platform. Jiutian-MT ranks first overall, while AiBabelMT achieves the highest throughput. We describe the task design, data resources, evaluation protocol, and published aggregate results, then examine how quality and throughput shape the final ranking. The results also motivate more detailed reporting by language direction and domain},
  url       = {https://aclanthology.org/2026.wmt-1.55}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{robinson-EtAl:2026:wmt,
  author    = {Robinson, Nathaniel R.  and  Armstrong, Ruth-Ann Hazel  and  Bizon Monroc, Claire  and  Dent, Rasul  and  Gupta, Pranav  and  Dabre, Raj  and  Coy, Andre  and  Murray, Kenton},
  title     = {Findings of the Second Shared Task for Creole Language Machine Translation, at WMT 2026},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1145--1165},
  abstract  = {We present the second annual Shared Task for Creole Language Machine Translation, the only published shared task specializing in machine translation (MT) methods for Creole languages and their communities, presented at the Conference of Machine Translation (WMT) 2026. We accepted system submissions for two tasks: MT and language identification (LID). We received submissions from eight teams: 13 MT systems from six teams across 88 language directions; and four LID systems from two teams, identifying 11 languages each. Our analysis of MT systems indicates fine-tuning with newly curated datasets and/or synthetic data as an effective strategy for producing state-of-the-art MT. For the LID task, we found that smaller, resource-efficient systems achieved state-of-the-art scores on some languages, especially when paired with data augmentation.},
  url       = {https://aclanthology.org/2026.wmt-1.56}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-8012-1387
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{burchell-EtAl:2026:wmt,
  author    = {Burchell, Laurie  and  Dale, David  and  Maillard, Jean  and  Abdulmumin, Idris  and  Anastasopoulos, Antonios  and  Caswell, Isaac  and  Koehn, Philipp},
  title     = {Findings of the WMT 2026 Shared Task of the Open Language Data Initiative},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1166--1175},
  abstract  = {We present the results of the WMT 2026 shared task of the Open Language Data Initiative (OLDI). Participants were invited to contribute to the massively multilingual open datasets supported by OLDI, FLORES+ and OLDI Seed, or to create new resources in line with OLDI's ethos. We accepted ten submissions: five extending FLORES+, and five extending other open and massively-parallel datasets (BOUQuET, OLDI Seed, SMOL, and WMT24++). These contributions advance the coverage and quality of multilingual datasets, especially for under-served language varieties. All contributions are released under permissive open-source licenses.},
  url       = {https://aclanthology.org/2026.wmt-1.57}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-0724-350X
Author{2}{Orcid}:https://orcid.org/0000-0003-2045-6833
Author{3}{Orcid}:https://orcid.org/0000-0003-0025-1021
Author{4}{Orcid}:https://orcid.org/0000-0002-3795-8381
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
@InProceedings{li-zheng:2026:wmt,
  author    = {Li, Zheng  and  Zheng, Mao},
  title     = {Findings of the WMT26 Video Subtitle Translation Shared Task: Context and Candidate Selection Matter},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1176--1186},
  abstract  = {We present the setup, systems, evaluation, and findings of the WMT26 Video Subtitle Translation shared task. The task covers translation from Simplified Chinese into English, Thai, Indonesian, Malay, and Traditional Chinese for Taiwan, China. Eight teams submitted 2,200 SRT files across 22 team-language combinations. We evaluated 2,198 unique team-language-video units with repeated LLM-based language quality assessment followed by human checking and correction. Each video was assessed three times, with valid positive scores aggregated by the median. ReopenAI led all five directions and achieved a macro-average of 89.357; laniqo ranked second with 78.265. Corrected scores were extractable from 93.36\% of the evaluation units, and 15.30\% of those scores differed from the automatic scores. Stronger results were associated with contextual prompting, diverse candidate generation, constrained selection, translator-critic specialization, terminology control, and deterministic subtitle validation. These cross-system associations are not controlled ablation evidence. Persistent challenges include cross-cue reference resolution, cultural and register adaptation, multilingual judge calibration, and balancing brevity with semantic completeness.},
  url       = {https://aclanthology.org/2026.wmt-1.58}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{okabe-EtAl:2026:wmt,
  author    = {Okabe, Shu  and  Di Marco, Marion  and  Haemmerl, Kathy  and  Dementieva, Daryna  and  Edman, Lukas  and  Měškank, Marko  and  Hendrichowa, Anita  and  Fraser, Alexander},
  title     = {Findings of the WMT 2026 Shared Task: Multitask LLMs with Limited Resources},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1187--1209},
  abstract  = {We present the Findings of the WMT 2026 Shared Task on Multitask LLMs with Limited Resources. This year, we extended the Shared Task to jointly model five diverse NLP tasks, Machine Translation (MT) for five language pairs (eight directions), Question Answering (QA), Spell Checking (SC), Grammar Checking (GC), and Mathematical Reasoning (MR), for two language tracks: Ukrainian and Sorbian (grouping Upper and Lower Sorbian). As the Shared Task is focused on low-resource conditions, the model is restricted to Qwen3.5-2B to remain below the 3B-parameter threshold. In total, six teams participated across the two language tracks: two in the Ukrainian and four in the Sorbian track. No team participated in both language tracks. All systems have been uploaded to HuggingFace by the participants. Despite our strict restrictions on model and dataset availability, systems were diverse in the resources they used as well as their training and inference strategies. The submitted systems show that a single small model can handle all five tasks, with clear difficulty shown for the challenging MR only.},
  url       = {https://aclanthology.org/2026.wmt-1.59}
}

Author{1}{Orcid}:https://orcid.org/0009-0003-0169-3689
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0003-0929-4140
Author{5}{Orcid}:https://orcid.org/0000-0001-8215-3614
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{dahan-yvon-bawden:2026:wmt,
  author    = {Dahan, Nicolas  and  Yvon, François  and  Bawden, Rachel},
  title     = {TermJudge: A Document-Level Metric Judging, Not Counting, Terminology in Machine Translation Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {82--122},
  abstract  = {Existing automatic metrics for evaluating terminological use in machine translation (MT) penalise any divergence from a fixed reference, conflating translation errors with the valid terminological variation that human translators routinely produce. We introduce TermJudge, a document-level terminology metric that assigns an interpretable verdict to every term occurrence: glossary-conforming occurrences are settled deterministically, while divergences are assessed under a two-step LLM-as-judge procedure using the full document context: the first detects and labels terminology errors; the second sorts valid document-level variations from inconsistencies. Validated against expert error annotations and document-level human MQM scores, TermJudge ranks first in both system- and segment-level meta-evaluation, ahead of glossary-conformity and quality-estimation baselines. When applied to eight systems translating academic documents, under two prompting conditions, we observe that glossary injection improves terminology translation in all paired comparisons, by removing genuine errors rather than valid variation. TermJudge is released as open-source code.},
  url       = {https://aclanthology.org/2026.wmt-1.6}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-7972-7442
Author{3}{Orcid}:
@InProceedings{bueno-EtAl:2026:wmt,
  author    = {Bueno, Mirelle Candida  and  Domingues, Lucas  and  Frontull, Samuel  and  Maillard, Jean  and  Garg, Sushil},
  title     = {LiLa at WMT 2026 General MT Task: Exploring Beyond NMT and Learning the Limits of LLM Adaptation for Ligurian and Ladin},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1210--1219},
  abstract  = {We describe LiLa, our submission to the constrained track of the WMT 2026 General Machine Translation Task. Motivated by recent advances in large language models (LLMs) and their increasing availability as open models, we investigate the adaptation of LLMs for Ligurian and Ladin, two low-resource languages included in the shared task. We evaluate several adaptation strategies for Gemma-4, including Instruction Tuning (IT-only), continuous pre-training followed by Instruction Tuning (CPT+IT), and Low-Rank Adaptation (LoRA), and compare them with a dedicated encoder-decoder pipeline based on fine-tuned No-Language Left Behind (NLLB) models. Our experiments show that, under the evaluated conditions, fine-tuned NLLB models achieve higher automatic translation scores than the investigated LLM adaptation strategies. Based on these results, our final submission adopts a fine-tuned NLLB-based translation pipeline. We discuss the main observations from the different approaches explored during development. We release the code and models produced in these experiments through a publicly accessible GitHub repository https://github.com/schtailmuel/wmt26-lila/},
  url       = {https://aclanthology.org/2026.wmt-1.60}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0009-0004-1230-4666
Author{4}{Orcid}:https://orcid.org/0000-0003-0025-1021
Author{5}{Orcid}:
@InProceedings{cao-EtAl:2026:wmt,
  author    = {Cao, Xiayu  and  Li, Baohang  and  Yuan, Zekun  and  Ye, Zekai  and  Feng, Xiaocheng},
  title     = {SCIR-TG-MT's Submission for the WMT 2026 General Machine Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1220--1227},
  abstract  = {This paper presents SCIR-TG-MT, our primary submission (ID 44) to the unconstrained track of the WMT 2026 General Machine Translation shared task. The system covers ten translation directions using DeepSeek-V4-Pro and STRIDE (Structured Translation with Risk-Informed Directional Execution), an OpenCode translation skill. STRIDE converts source-side evidence and task instructions into a translation contract, an execution route, generation constraints, and selected checks. On five WMT25 development directions, STRIDE raises average SacreBLEU from 31.24 to 31.80 and COMET-22 from 83.61 to 84.17. A cumulative ablation on three directions separates the effects of routing, controls, and validation.},
  url       = {https://aclanthology.org/2026.wmt-1.61}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{cozzini-EtAl:2026:wmt,
  author    = {cozzini, marta  and  Oliver, Antoni  and  López-Sánchez, Gonzalo  and  Vàzquez, Mercè  and  Morales-Hurtado, Patricia  and  Alvarez-Vidal, Sergi},
  title     = {GRIAL-TA participation in the WMT26 General Translation Shared Task: SalamandraTA7bFFT},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1228--1237},
  abstract  = {This work describes the Grial-TA team submission to the constrained track of WMT2026 General Machine Translation Task for the following language pairs: English to-Icelandic, English-to-Ladin and English-to-Ligurian. Our submission leverages full fine-tuning of theSalamandraTA-7b-instruct model. Two separated models were fine-tuned: one for Icelandic and one for Ligurian and Ladin.},
  url       = {https://aclanthology.org/2026.wmt-1.62}
}

Author{1}{Orcid}:0009-0004-1992-9132
Author{2}{Orcid}:0000-0001-8399-3770
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-7983-4029
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{dukanov-tsyurupa-dvorkovich:2026:wmt,
  author    = {Dukanov, Sergey  and  Tsyurupa, Maria  and  Dvorkovich, Anton},
  title     = {Dubformer at WMT26 General MT: Translating the Video, Not the Transcript},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1238--1241},
  abstract  = {We describe the system Dubformer entered in the WMT26 General MT shared task, an un- constrained system built on our video-dubbing pipeline. It rests on two commitments. A doc- ument is never translated as raw text: it is first decomposed into a structure chosen for its do- main — scenes carrying speaker and audio- visual context for spoken dialogue, markup- masked segments for web documents, value- only segments for localization resources — so that non-linguistic material is protected by construction and everything needed for consis- tency is in front of the model at once. And no translation is taken on trust: every segment is checked by a second model from a different family for hallucination and for drift from the meaning of the source, and a segment it rejects is re-translated with the reason attached rather than filled in from a fallback.},
  url       = {https://aclanthology.org/2026.wmt-1.63}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{dobrowolski-EtAl:2026:wmt,
  author    = {Dobrowolski, Adam  and  Przybysz, Paweł  and  Szymański, Marcin  and  Przewłocki, Paweł  and  Janicki, Artur},
  title     = {Relay-MT: Multi-agent LLM Ensembling for Machine Translation – Samsung R\&D Poland Submission for WMT2026},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1242--1254},
  abstract  = {We propose a method to improve machine translation quality by ensembling multiple large language models within a multi-agent framework. The method addresses the problem of ensembling models with different tokenizations by abstracting away from token-level representations and operating at the level of generated text. The approach integrates several established concepts in MT, including ensembling, quality estimation, and reranking into one translation system. The log-likelihood reranking applied in this paper significantly improves results, even without ensembling. Experimental results show improvements comparable to COMET-MBR.},
  url       = {https://aclanthology.org/2026.wmt-1.64}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{frontull-EtAl:2026:wmt1,
  author    = {Frontull, Samuel  and  Maillard, Jean  and  Haberland, Christopher R.  and  Videsott, Ruth  and  Lusito, Stefano},
  title     = {Grammatist at WMT 2026 General MT Task: An Agentic LLM Framework for Machine Translation with Explicit Linguistic Knowledge},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1255--1264},
  abstract  = {We describe Grammatist, our submission to the unconstrained track of the WMT 2026 General Machine Translation Task, developed in direct collaboration with community members of two regional languages of Italy, Ladin and Ligurian. Ladin and Ligurian represent challenging scenarios for machine translation (MT), as they have limited resources for conventional data-driven approaches. However, this scarcity of large-scale parallel data does not reflect a lack of linguistic documentation: both languages are extensively documented in dictionaries and grammar books, which contain valuable knowledge but are not readily exploitable by conventional MT approaches. Grammatist addresses this gap by making this information available to a large language model (LLM) at inference time. The framework is both agentic and model-agnostic: it dynamically consults dictionaries and grammar books during translation to retrieve the information most relevant to the input. Any LLM can serve as the underlying base model, enabling Grammatist to benefit from advances in increasingly capable models. For the WMT 2026 task, we apply Grammatist to structured document-level translation from English into Ligurian and Ladin.},
  url       = {https://aclanthology.org/2026.wmt-1.65}
}

Author{1}{Orcid}:0009-0004-1230-4666
Author{2}{Orcid}:https://orcid.org/0000-0003-0025-1021
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{he-EtAl:2026:wmt,
  author    = {He, Yu  and  Lan, Xiaoqing  and  Wei, Daimeng  and  GUO, Jiaxin  and  Luo, Yuanchang  and  Shang, Hengchao  and  Li, Zongyao  and  Yang, Jinlong  and  Wu, Zhanglin  and  Huang, Boqi},
  title     = {HW-TSC's Submission to the WMT 2026 General Machine Translation Track},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1265--1270},
  abstract  = {Machine translation based on large language models (LLMs) has achieved remarkable progress in recent years. However, relying on a single translation model and fixed translation instructions often limits translation quality, especially for multilingual and multi-domain scenarios. To address this issue, we propose HW-TSC-Agent, a multi-stage translation agent framework that integrates translation model fine-tuning, dual-model candidate generation, and instruction-aware translation refinement into a unified pipeline. First, we perform Supervised Fine-Tuning (SFT) and Contrastive Preference Optimization (CPO) on openPangu-Embedded-7B using high-quality bilingual data to enhance its multilingual translation capabilities. The fine-tuned openPangu-Embedded-7B and HY-MT2-7B are then used to generate candidate translations. Finally, source-language analysis, candidate translations, and task-specific instructions are incorporated into a dynamic prompt, which guides DeepSeek-V4-Flash to compare, refine, and fuse the candidate translations to produce the final output. Our system is submitted to the WMT2026 General Machine Translation Shared Task, covering 13 translation directions. On the WMT2025 English-to-Chinese development test set, enriching the dynamic prompt generally improves translation quality over an instruction-only DeepSeek-V4-Flash baseline.},
  url       = {https://aclanthology.org/2026.wmt-1.66}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:https://orcid.org/0000-0002-5211-3221
Author{7}{Orcid}:https://orcid.org/my-orcid?orcid=0000-0001-7196-2782
Author{8}{Orcid}:
Author{9}{Orcid}:https://orcid.org/0000-0002-2920-0773
Author{10}{Orcid}:
@InProceedings{hrabal-jon-bojar:2026:wmt,
  author    = {Hrabal, Miroslav  and  Jon, Josef  and  Bojar, Ondřej},
  title     = {CUNI-MH and CUNI-EdUKate at the WMT26 General MT Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1271--1277},
  abstract  = {We describe our constrained CUNI-MH-v3 (based on Qwen3-14B) and unconstrained CUNI-EdUKate (based on Granite 4.1 8B) submissions for the WMT26 General MT shared task. Both submissions use a three-stage training procedure: QLoRA supervised fine-tuning on synthetic sentence pairs, continued supervised fine-tuning on a smaller mixture of paragraph-level and instruction-oriented examples, and iterative preference optimization with automatically generated and scored preference pairs. We document the data construction, training setup, checkpoint selection, and evaluation. We also share the training data and model checkpoints for our constrained system.},
  url       = {https://aclanthology.org/2026.wmt-1.67}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0002-0606-0050
@InProceedings{jon-EtAl:2026:wmt,
  author    = {Jon, Josef  and  Bondok, Rawan  and  Hrabal, Miroslav  and  Bojar, Ondřej},
  title     = {CUNI-AR Team at WMT26 General Translation Task for Egyptian Arabic},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1278--1290},
  abstract  = {We describe the Charles University CUNI-AR submission to the WMT26 General Translation shared task for English into Egyptian Arabic. Our pipeline is built around GEMBA (LLM-as-judge quality) estimator that scores translations independently for accuracy and dialect authenticity. We use it throughout our pipeline: to filter training data, to construct preference pairs for DPO, to select best checkpoints, and to pick the best hypothesis per segment at inference time. Our primary translation model is Gemma-4-12B-it finetuned via QLoRA, we also finetune Jais-2-8B-Chat and Aya-expanse-8B. Gemma-4-12B-it provides a high-quality Egyptian Arabic translation, so SFT yields only marginal quality gains in automatic metrics; DPO is able to improve the metrics it optimizes for, but the real gain in translation quality is to be assessed by human evaluation. The final submission selects the best-scoring translation per segment from 31 candidate checkpoints, with the unmodified base model contributing 31\% of the output.},
  url       = {https://aclanthology.org/2026.wmt-1.68}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-0606-0050
@InProceedings{kanojia-EtAl:2026:wmt2,
  author    = {Kanojia, Diptesh  and  Lo, Chi-kiu  and  Sindhujan, Archchana  and  Larkin, Samuel  and  Hanneman, Greg  and  Lavie, Alon},
  title     = {In the Blind: Building Pseudo-References for MT Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1291--1311},
  abstract  = {The WMT26 General MT task evaluates systems on 10 language pairs that have no human references (neither translated from scratch nor post-edited from MT output by humans). We describe how we built the pseudo-references for these pairs and six other language pairs (in which some forms of human references are available): seven models translate the 3,277 official documents under up to five prompt conditions, giving a total of 26 system--prompt combinations; then three reference-free quality estimation (QE) models score every candidate; and a per-document selector picks one translation, which GPT-5.5 post-edits where needed. Working without references exposed a failure mode of QE-guided selection: the metrics rank fluent output in the wrong language above correct translations. Adding a confidence-scaled language identification penalty to the score fusion drives the wrong-language count to zero, and the resulting selector still scores better on MetricX than the rank-fusion baseline it replaces. Since no references were available for these pairs while we were building them, we calibrate every selection decision on last year's WMT25 human judgments. The human evaluation, released after construction, shows the cost of getting selection wrong: our references stand with the strongest participating systems when the selector kept a frontier-model candidate, and fall up to 17 ESA points below them when it did not. We release the selection method and the provenance of every reference.},
  url       = {https://aclanthology.org/2026.wmt-1.69}
}

Author{1}{Orcid}:https://orcid.org/0000-0001-8814-0080
Author{2}{Orcid}:https://orcid.org/0000-0001-8714-7846
Author{3}{Orcid}:https://orcid.org/0000-0002-6467-6873
Author{4}{Orcid}:0009-0000-6147-9631
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{dasari-EtAl:2026:wmt,
  author    = {Dasari, Priyanka  and  Bodana, Yuvrajsinh D.  and  Mujadia, Vandan Vasantlal  and  Ahsan, Arafat  and  Sharma, Dipti Misra  and  Krishnamurthy, Parameswari},
  title     = {BaatCheet: A Multilingual Corpus for Dialogue Translation in Indian Languages},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {123--152},
  abstract  = {Existing translation models are typically trained on sentence-level and formal text, limiting their ability to capture everyday conversational dialogue phenomena such as informality, speaker interaction, and discourse coherence. Most existing Indic translation resources and evaluation benchmarks focus on sentence-level or formal text, making it difficult to assess translation quality of dialogue phenomena. In this work, we introduce BaatCheet, a multilingual dialogue corpus named after the Hindi term for conversation or chitchat, containing approximately 49,000 dialogues for dialogue translation across five translation directions. We fine-tune five open-source LLMs across seven training data configurations and find that fine-tuning yields substantial gains over zero- and few-shot baselines. To comprehensively evaluate dialogue translation quality, we employ multiple evaluation strategies, including automatic metrics, LLM-as-judge, and human assessments using an SQM-guided Direct Assessment (DA) Protocol.},
  url       = {https://aclanthology.org/2026.wmt-1.7}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0009-0009-9709-3832
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{kocmi-EtAl:2026:wmt2,
  author    = {Kocmi, Tom  and  Berard, Alexandre  and  Blunsom, Phil  and  Cahyawijaya, Samuel  and  Cassini, Shaun Rafael  and  de Gibert, Ona  and  Gomez, Aidan  and  Govindarajan, Nithya  and  Kiyono, Shun  and  Lasche, Olivia  and  Rogers, Lawrence  and  Marchisio, Kelly  and  Moghe, Nikita  and  More, Yash  and  Moran-Hidalgo, Camila  and  Nan, Yiyang  and  Sachs, Michael  and  Starostina, Trisha  and  van Stigt, Daan  and  Rarrick, Spencer Taylor  and  Vincent, Sebastian  and  Zhang, Ivan  and  Frosst, Nicholas},
  title     = {North Small Translate: Advanced Cost-Effective Translation (Cohere CAT+)},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1312--1324},
  abstract  = {We present North Small Translate, an open-weight, LLM-based machine translation (MT) model with instruction-following capabilities built on the same foundation as Cohere's Command A Plus, a mixture-of-experts architecture with 25 billion active parameters out of 218 billion total parameters. North Small Translate is trained using difficulty sampling to obtain challenging documents and a five-step training protocol combining supervised fine-tuning, direct preference optimization, and online reinforcement learning. We prioritized throughput through a non-reasoning base model and supplemented with optional agentic capabilities to unlock translation quality gains. North Small Translate is trained to perform MT-related tasks, including post-editing and quality estimation, as well as related tasks such as general instruction following. The model achieves top MT performance across 50 languages in the class of models under 1T parameters, with no need to run expensive reasoning at inference time.},
  url       = {https://aclanthology.org/2026.wmt-1.70}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-9891-1608
Author{5}{Orcid}:0009-0006-3821-7688
Author{6}{Orcid}:0000-0002-7163-4807
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
Author{10}{Orcid}:
Author{11}{Orcid}:
Author{12}{Orcid}:
Author{13}{Orcid}:
Author{14}{Orcid}:
Author{15}{Orcid}:
Author{16}{Orcid}:
Author{17}{Orcid}:
Author{18}{Orcid}:
Author{19}{Orcid}:
Author{20}{Orcid}:
Author{21}{Orcid}:https://orcid.org/0000-0001-8975-165X
Author{22}{Orcid}:
Author{23}{Orcid}:
@InProceedings{lee:2026:wmt1,
  author    = {Lee, Soyoung},
  title     = {AdaptiveMT at WMT26: A Controller–Executor–Reducer Agent Harness for Document-Level Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1325--1333},
  abstract  = {We describe AdaptiveMT, our submission to the English–Korean condition of the WMT26 General Machine Translation task in the unconstrained track. AdaptiveMT is a language-general adaptive translation harness with target-language-specific profiles; this run uses the Korean profile. The system uses a controller–executor–reducer agent harness. For each document, the harness analyzes the source, generates independent candidate translations from two heterogeneous models, compares them with token-level and semantic diffs, and synthesizes a final translation with a judge model. Mandatory QA hooks and deterministic Korean-specific guards run before finalization; failures trigger bounded local repair. Multimodal documents use structured visual context extracted from images or sampled video frames before translation. We processed all 2,078 distributed blindset records under two Korean target tags; the subsequently released General MT evaluation set contains 198 English-to-Korean documents tagged kor\_Hang. On this official subset, the harness made 1,441 successful LLM calls at an estimated list-price inference cost of \$12.03, and all 198 documents finished with passing hooks. In blindset-wide diagnostics, synthesis eliminated observed candidate-level mechanical errors, and a GPT-5 judge preferred the final output over a bare GPT-4.1 baseline on 600 sampled documents (44.5\% vs. 30.0\%, 25.5\% ties).},
  url       = {https://aclanthology.org/2026.wmt-1.71}
}

Author{1}{Orcid}:
@InProceedings{mash-EtAl:2026:wmt,
  author    = {Mash, Audrey  and  Ayebakuro, Jonathan Orama  and  Bohman, Ella Paulina  and  Liao, Xixian  and  De Luca Fornaciari, Francesca  and  Melero, Maite},
  title     = {BSC Submission for WMT26 General Machine Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1334--1350},
  abstract  = {We present the BSC submission to the WMT26 General Machine Translation shared task: a 7B decoder-only model covering 23 directions, adapted from the SALAMANDRA (Gonzalez-Agirre et al., 2025) family through vocabulary replacement, two rounds of continued pre-training (monolingual, followed by instruction-formatted parallel data), supervised fine-tuning on a translation-only instruction mixture, and preference optimisation applied selectively to six weak directions via CPO-SimPO (Xu et al., 2024; Meng et al., 2024) with LoRA adapters (Hu et al., 2021). Submissions are decoded with Minimum Bayes Risk selection over a candidate pool filtered for truncation, degeneracy and off-target script. As official human evaluation was unavailable at the time of writing, we report a stage-wise internal evaluation on FLORES+ and WMT24++, together with automatic scores for the submitted system on the 12 test-set directions for which references have been released. Instruction tuning accounts for almost all measurable gain, but its size differs by an order of magnitude between the two evaluation sets; preference optimisation moves automatic metrics by less than the drift among untreated directions. The model is released under a Research-Only Licence and is available on request.},
  url       = {https://aclanthology.org/2026.wmt-1.72}
}

Author{1}{Orcid}:0009-0006-6625-5841
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:0009-0002-7821-1433
Author{6}{Orcid}:0000-0001-9933-3224
@InProceedings{matsuda-EtAl:2026:wmt,
  author    = {Matsuda, Ryosuke  and  Kudo, Keito  and  Fujii, Ryo  and  Ito, Takumi  and  Morishita, Makoto  and  Suzuki, Jun},
  title     = {RFMT at WMT 2026 General Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1351--1378},
  abstract  = {We participated in the constrained (open-weight) track of the WMT 2026 General MT Task for the English-to-Japanese and Simplified Chinese-to-Japanese directions. Our system builds on Marco-MT-Algharb, a strong open-weight translation model from the previous WMT General MT task. We further fine-tuned this model to generate more natural Japanese translations. To this end, we constructed a Context-Aware JApanese Linguistic Acceptability dataset (CAJALA). During annotation, human annotators were shown multiple paraphrased variants of a segment within its document-level context and asked to select the most natural one. We then back-translated the CAJALA dataset to create parallel data for Direct Preference Optimization (DPO). We also trained the model to perform fill-in-the-middle (FIM) translation, in which the model reconstructs target-side segments from their surrounding document. At inference, we use Minimum Bayes Risk (MBR) decoding, followed by selective FIM post-editing of low-quality segments.},
  url       = {https://aclanthology.org/2026.wmt-1.73}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-7380-2587
Author{5}{Orcid}:
Author{6}{Orcid}:https://orcid.org/0000-0003-2108-1340
@InProceedings{nishihama:2026:wmt,
  author    = {Nishihama, Chris},
  title     = {MakotoAI at WMT26: Per-Pair Model Selection with Deterministic Output Normalization for the General Machine Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1379--1384},
  abstract  = {We describe MakotoAI, an unconstrained submission to the WMT26 General Machine Translation shared task covering all 23 language pairs (4,897 documents). The system performs no training or fine-tuning: it is a lightweight orchestration layer over six off-the-shelf commercial LLMs, in which each language pair is routed to the single model that won a blinded pre-competition shootout for that pair. Translation is document-level with the full task instruction provided in-prompt. Structured output (JSON and HTML domains) is repaired by a small deterministic normalization pass rather than by model-based retry: an ablation on real test data found that a post-hoc validate-and-retry instruction layer was inert on clean documents and net-negative on JSON documents, while deterministic fence and preamble stripping achieved the intended effect at zero inference cost. Quality assurance combined an exhaustive deterministic instruction-adherence scan, a two-judge LLM evaluation whose self-grading bias we measured to be directional by model family, and the organizers' official alignment checker, against which the final submission scores 1,914 of 1,914 structurally checkable documents aligned. Four defective documents (three truncations, one repetition-loop degeneration) were detected and regenerated, one requiring escalation to a different model; all four are disclosed. The entire system runs on commodity hardware; total API cost to produce and verify the submission was just under \$25. We argue that for API-orchestrated MT, verification discipline and deterministic post-processing are higher-leverage than added inference-time machinery.},
  url       = {https://aclanthology.org/2026.wmt-1.74}
}

Author{1}{Orcid}:
@InProceedings{pilinka-EtAl:2026:wmt,
  author    = {Pilinka, Mikita  and  Kliuyeu, Aliaksandr  and  Samuel, David  and  Scherrer, Yves},
  title     = {ReMova: Fine-tuning LLMs for English to Belarusian translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1385--1397},
  abstract  = {This paper presents a Belarusian-specific data-cleaning pipeline and fine-tuning for English-Belarusian machine translation. Our cleaning pipeline distinguishes itself from others by employing a correction tool that addresses the issue of the two orthographies of the Belarusian language, noise in the training data, interference from other languages, and other misspelling issues common in Belarusian on the internet. A matched ablation on unfiltered training data shows substantial benefits from filtering for all fine-tuned models, with the LLM-based models gaining roughly twice as much from filtering as the dedicated encoder-decoder MT system, supporting the view that for Belarusian MT one of the primary bottlenecks is data quality.},
  url       = {https://aclanthology.org/2026.wmt-1.75}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0000-0003-2866-1022
Author{4}{Orcid}:https://orcid.org/0000-0001-5247-5073
@InProceedings{pirinen:2026:wmt,
  author    = {Pirinen, Flammie A.},
  title     = {Apertium-sme-eng for WMT2026 shared task---creating new pair by crossing dictionaries and translating rules},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1398--1403},
  abstract  = {In this article I describe an experiment on creation of a new rule-based machine translation model build on two existing models. I built Apertium system between North Sámi and English based on existing translation models between North Sámi and Finnish, and Finnish and English. There were two expert-driven processes involved in building of the system: firstly crossing and validating the bilingual dictionaries and secondly translating and modifying the grammatical and structural rules. I breifly explain why there are no pre-existing systems for this language pair and what limitations and ethical issues are related to it. I will also evaluate some of the training and development corpora provided in the shared task to elaborate on the problem that automatic low quality machine translation can cause to vulnerable minority languages.},
  url       = {https://aclanthology.org/2026.wmt-1.76}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-1207-5395
@InProceedings{popov-EtAl:2026:wmt,
  author    = {Popov, Dmitrii  and  Kozlova, Elizaveta  and  Taracheva, Ekaterina  and  Makhotina, Elizaveta  and  Mekhraliev, Artem  and  Baratelia, Miron  and  Enikeeva, Ekaterina  and  Karpachev, Nikolay},
  title     = {Yandex at WMT26: Translation Adaptation at 235B and Hard-Label Distillation to 8B},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1404--1411},
  abstract  = {We describe Yandex and Yandex-8B, our unconstrained and constrained submissions to the WMT26 General Machine Translation task. Our previous WMT25 submission, a 7B translation specialist, shows strong Fluency but comparatively weaker Accuracy. To combine its translation behavior with the capacity of a much larger model without replaying the full 7B training pipeline at 235B scale, we add two forms of supervision to the instruction-tuning data of an in-house 235B pretrained model: direct translation targets generated by the 7B specialist and in-house translation-related tasks. The resulting Yandex model improves both Accuracy and Fluency and reduces the LLM-based error score relative to the same 235B model tuned only on general instructions. Relative to the 7B specialist, it improves Accuracy and reduces the error score, while their Fluency point estimates remain similar. Yandex then provides hard targets for Yandex-8B, which is trained for English-to-Russian, English-to-Belarusian, English-to-Kazakh, and English-to-Armenian using direct and instruction-conditioned translation data. Across four directions, Yandex-8B obtains the highest paragraph-level ChrF++ and the highest Accuracy and Fluency under our LLM-based evaluation among five public baselines.},
  url       = {https://aclanthology.org/2026.wmt-1.77}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{shao-EtAl:2026:wmt,
  author    = {Shao, Liangying  and  Wu, Xinwei  and  Huang, Yichong  and  Wang, Jifang  and  Xu, Ruoxi  and  Shi, Ling  and  Yang, Baosong  and  Xu, Linlong},
  title     = {Wayfinder: An Agentic Ensemble-and-Postcheck System for WMT26 GenMT},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1412--1420},
  abstract  = {This paper presents Wayfinder, our submission system for the WMT26 General Machine Translation (GenMT) shared task. GenMT evaluates document-level translation under natural-language instructions, where structural preservation, formatting constraints, and output cleanliness are part of translation quality. Rather than training a new translation model, Wayfinder treats four API-based system outputs as a document-level candidate pool and produces one final hypothesis through a four-stage agentic pipeline. It aggregates source, instruction, metadata, candidates, and a DeepSeek-V4-Pro ranking-then-scoring signal; applies an anonymized pick-or-rewrite agent; and runs a conservative postcheck for severe objective defects. During development, a judge-guided loop compares outputs with a GPT-5.5 baseline, analyzes loss cases, updates prompts and repair constraints, and selectively reprocesses difficult documents. In automatic pairwise evaluation against GPT-5.5, Wayfinder obtains a pooled document-level weighted win rate of 0.6537 and remains above parity in every evaluated direction. The results suggest that candidate-pool translation, constrained agentic selection, conservative repair, and selective judge-guided iteration are practical components for instruction-conditioned shared-task MT.},
  url       = {https://aclanthology.org/2026.wmt-1.78}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{steingrmsson-rarson-daason:2026:wmt,
  author    = {Steingrímsson, Steinþór  and  Þórðarson, Sveinbjörn  and  Daðason, Jón Friðrik},
  title     = {What a DRAG (it is being small): The AMI Submission to the WMT 2026 General Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1421--1431},
  abstract  = {We describe the AMI team's submission to the WMT 2026 General Translation shared task, participating in the English->Icelandic translation direction. Building on our 2025 submission, which relied on a 3B-parameter model and substantial rule-based post-processing, this year we aim to reduce that dependency and rely more on the translation model itself. We continually pre-train a Qwen3-4B checkpoint on a mixture of Icelandic, English, code, math, and structured data, then fine-tune it with LoRA on a retrieval-augmented, domain-conditioned instruction-tuning dataset covering five domains. At both training and inference time, source sentences are enriched with dictionary entries and back-translated example translations similar to the sentence being translated, retrieved by a dedicated RAG server, and served through a self-healing inference client that retries and validates generations. We report COMET and chrF++ scores comparing this system against last year's submission, a larger Llama-3.1-8B model adapted with the same recipe, and an instruction-tuned Qwen3-4B baseline, alongside ablations that isolate the contribution of each part of the retrieval-augmented prompt.},
  url       = {https://aclanthology.org/2026.wmt-1.79}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-9776-9507
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0009-0009-5817-674X
@InProceedings{dhaipule-EtAl:2026:wmt,
  author    = {Dhaipule, Rohit  and  Kharbanda, Sukhdeep Singh  and  Bathala, Prasanth  and  Lanka, Pradyumna  and  Shrimal, Anubhav},
  title     = {StalePO: Anchored Token-Level Preference Optimization using Legacy Post-Edits in Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {153--163},
  abstract  = {Machine translation systems are periodically upgraded to stronger models, but the available preference signal is human post-edits of an older system's outputs, which the newer model may already surpass. Moreover, collecting fresh post-edits for every new model is prohibitively expensive. We call this the Stale Preference problem. Standard DPO can fail in this setting: it may increase the likelihood of inferior post-edits, erode the model's existing quality, and fail to provide the per-token control needed to correct localized errors. We introduce StalePO, an objective derived from three requirements this regime imposes. Likelihood movement must be downward on both responses, the policy must be anchored to its own base response, and the KL constraint must apply at the token level. These requirements are jointly necessary. In ablations, each mechanism in isolation leaves the model's performance indistinguishable from the base model, and only their combination converts stale feedback into gains. On English-to-Hindi and English-to-Turkish localization data, StalePO improves the fraction of segments passing all LLM-as-judge MQM quality checks by 14.9 and 4.6 percentage points, respectively, with gains concentrated on style and fluency. A human evaluation under the same framework confirms these gains on English-to-Hindi, raising the fraction of segments passing all seven human checks by 13.8 percentage points.},
  url       = {https://aclanthology.org/2026.wmt-1.8}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{volchek-poritski:2026:wmt,
  author    = {Volchek, Oksana  and  Poritski, Vladislav},
  title     = {Krosny at WMT26 General Translation Task: Dictionary Augmentation and Iterative Refinement for English–Belarusian Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1432--1436},
  abstract  = {This paper describes our submission to the WMT26 General Machine Translation shared task for the English → Belarusian direction. In lower-resourced morphologically rich languages like Belarusian, open-weight large language models (LLMs) frequently struggle with orthographic fidelity, hallucinated vocabulary, and negative transfer from related high-resource languages. To address these issues, we implemented Krosny, an iterative, resource-augmented pipeline built on top of TranslateGemma 12B. The system translates each segment in three stages: (1) dictionary-augmented generation, (2) refinement using a reference grammatical database, and (3) rule-based post-processing followed by an LLM-as-a-judge candidate selection. We find a small but consistent improvement over the baseline TranslateGemma 12B. In comparison with other constrained-track systems of WMT26, the performance of Krosny is mid-tier. We release the source code of our implementation (https://github.com/volchek/Krosny-WMT26).},
  url       = {https://aclanthology.org/2026.wmt-1.80}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{wang-EtAl:2026:wmt2,
  author    = {Wang, Hao  and  Liu, Yangyang  and  Zhao, Xiaohu  and  Liu, Heng  and  Shang, Zifu  and  Li, Tianhao  and  Gao, Ruize  and  Tang, Jialong  and  Shao, Shiao  and  Wei, Haoran  and  Yang, Baosong  and  Xu, Linlong  and  Wang, Longyue  and  Luo, Weihua},
  title     = {Lumen at WMT2026: Instruction-Specialized Translation via Quality-Aware Training and On-Policy Distillation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1437--1446},
  abstract  = {We present Lumen, our submission to the WMT2026 General Machine Translation Shared Task. We focus on instruction-specialized translation, where systems must produce accurate translations while satisfying fine-grained requirements on style, terminology, and structured output. Lumen follows a four-stage pipeline. First, continued pre-training and supervised fine-tuning jointly strengthen multilingual translation competence and the ability to follow translation-specific instructions, such as terminology, style, and format constraints. Second, we extend the originally offline M\^{}{2}PO preference-optimization framework into an online preference-optimization stage: the current policy samples candidate translations, an LLM judge assigns continuous direct-assessment scores, and the resulting multi-pair preferences are used to update the model. Third, on-policy distillation transfers stronger translation-instruction-following behavior from a larger teacher model to the Qwen3-14B student, improving translation-instruction adherence while retaining a 14B primary translation backbone. Finally, structure-aware inference and postprocessing preserve required formats, while selective repair addresses clear structural or decoding failures before submission. We participate in 23 translation directions. Under an absolute 1--5 LLM-judge rubric that evaluates translation-instruction following and translation quality separately, Lumen reaches an overall mean score of 4.60. Despite using a 14B translation backbone, it achieves performance comparable to leading proprietary systems in both translation quality and translation-instruction following.},
  url       = {https://aclanthology.org/2026.wmt-1.81}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
Author{9}{Orcid}:
Author{10}{Orcid}:
Author{11}{Orcid}:
Author{12}{Orcid}:
Author{13}{Orcid}:
Author{14}{Orcid}:
@InProceedings{wu-EtAl:2026:wmt,
  author    = {Wu, Di  and  Troshin, Sergey  and  Mohammed, Wafaa  and  Aycock, Seth  and  Tokarchuk, Evgeniia  and  Niculae, Vlad  and  Monz, Christof},
  title     = {UvA-MT's Participation in the WMT26 General Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1447--1453},
  abstract  = {This paper presents UvA-MT's submission to the WMT 2026 General Machine Translation shared task, competing in the unconstrained track across all 23 translation directions. This year, we fully leverage the test-time methods of Large Language Models (LLMs) for machine translation. Specifically, (1) we use sequential sampling with a refinement prompt to generate a pool of translation candidates for each source sentence, a strategy shown to outperform traditional parallel sampling; and (2) we employ LLMs as pairwise judges to select the best candidates via a round-robin voting mechanism, with offline evaluation on the WMT25 benchmark showing a very clear improvement over point-wise best-of-N selection. Lastly, we simply apply these two strategies for the submission of WMT26.},
  url       = {https://aclanthology.org/2026.wmt-1.82}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
@InProceedings{xia-EtAl:2026:wmt,
  author    = {Xia, Tian  and  Chen, Chao  and  Yang, Mengpeng  and  Yang, Jingxu  and  Sun, Yabo  and  Liu, Qiang},
  title     = {QINGQIU-MT-9B: An Instruction-Following Multilingual Translation Model},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1454--1469},
  abstract  = {We present Qingqiu-MT-9B, an instruction-following multilingual translation model developed for the constrained track of the WMT26 General Machine Translation shared task. Our model supports translation among 20 languages and covers 22 of the 23 official language pairs. It uses a source-grounded synthesis pipeline with separate LLM-based steps for generating instance-specific translation instructions and their corresponding translations. The resulting corpus covers diverse translation requirements, task complexity, instruction formulations, domains, document lengths, and glossary constraints, reducing reliance on fixed prompt templates and supporting generalization to unseen translation instructions. Our model is based on Qwen3.5-9B and trained through two-stage full-parameter supervised fine-tuning (SFT) followed by reinforcement learning (RL). The first SFT stage performs broad adaptation on a large-scale, medium-quality corpus, while the second refines the model on a smaller, higher-quality set. The RL stage further optimizes the model with Group Relative Policy Optimization (GRPO). During development, we use an LLM-based, rubric-guided contrastive method that scores candidates by dimension and aggregates an overall score to assess translation quality. Experiments show that Qingqiu-MT-9B demonstrates strong generalization and competitiveness among similarly sized models.We release the model at https://huggingface.co/WPS-Qingqiu/Qingqiu-MT-9B.},
  url       = {https://aclanthology.org/2026.wmt-1.83}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{yang-EtAl:2026:wmt1,
  author    = {Yang, Mengpeng  and  Xia, Tian  and  Yang, Jingxu  and  Chen, Chao  and  Sun, Yabo  and  Liu, Qiang},
  title     = {TRIVE: A Tri-Fusion Verified Translation Agent},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1470--1483},
  abstract  = {For the WMT26 General Machine Translation Task (Unconstrained Track), we propose TRIVE (Tri-Fusion Verified Translation Agent), a model-agnostic multi-agent translation framework. TRIVE organizes translation as a closed loop of Generation--Fusion--Verification--Correction: it first generates candidates in parallel under three complementary strategies---Faithful, Fluent, and Free---then synthesizes a draft that balances semantic fidelity and natural expression, and finally iteratively refines the output via rule-based validation and large language model evaluation. Because each module is implemented primarily via prompting, TRIVE can be directly overlaid onto diverse backbone models without task-specific fine-tuning. Experiments show that overlaying TRIVE on multiple backbones consistently improves instruction-following translation quality. Finally, we submitted Gemini-3.5-Flash + TRIVE as our primary system to WMT26.},
  url       = {https://aclanthology.org/2026.wmt-1.84}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{yu-EtAl:2026:wmt,
  author    = {Yu, Hao  and  Chen, Ziyan  and  Zhao, Hongyu  and  Zhang, Yulong  and  Wang, Xiaogang  and  Huang, Kaiyu  and  Huang, Degen},
  title     = {TransCore: The NewTranx Submission to the Constrained Track of the WMT26 General Machine Translation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1484--1490},
  abstract  = {This paper describes TransCore, the NewTranx submission to the WMT26 General Machine Translation Shared Task constrained track. Our system is built upon the Qwen3-8B large language model and adopts a two-stage supervised fine-tuning framework trained exclusively on publicly available data. Our approach focuses on high-quality multilingual data construction, including parallel corpus construction, knowledge-distilled corpus construction, and instruction-following data construction. For the parallel corpus, we construct a high-quality multilingual training corpus through multilingual data collection and automatic quality filtering. For knowledge distillation, we construct a KD corpus from publicly available monolingual resources using a large language model, followed by automatic quality filtering. To support the instruction-following evaluation introduced in WMT26, we further construct instruction data covering immutable entity preservation, document-level translation, and structured translation, which are incorporated into the first-stage supervised fine-tuning. Our system participates in four language directions of the WMT26 General Machine Translation Shared Task: English$\rightarrow$Indonesian, English$\rightarrow$Kazakh, English$\rightarrow$Thai, and Chinese$\rightarrow$Japanese.},
  url       = {https://aclanthology.org/2026.wmt-1.85}
}

Author{1}{Orcid}:https://orcid.org/0009-0004-2060-6816
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:https://orcid.org/0000-0002-8860-7805
@InProceedings{choi-EtAl:2026:wmt,
  author    = {Choi, Minjoo  and  Jung, Jaejun  and  Han, Gunwoo  and  Kim, Junhyeop  and  Yoon, Yiji  and  Kim, Hyemi  and  Cho, Yeonseo  and  Jang, Jinwoo},
  title     = {KUEST: Korean game Usage Evaluation Suite for LLM Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1491--1512},
  abstract  = {Despite the growing demand for multilingual game localization, systematic evaluation frameworks for game translation constraints remain largely unexplored. As part of the WMT26 Test Suite Track, we propose KUEST, a test suite comprising 1,796 English-to-Korean segments across four core categories, and evaluate 27 submitted systems. Experimental results demonstrate that standard metrics obscure true model capabilities (e.g., the overall 1st-place system dropping to 15th in creative translation). Furthermore, we reveal residual capability fragmentation—category-specific variance that survives the shared general-competence factor and inverts top-tier rankings—and observe a "prompt gap'' where context metadata degrades performance. As such, KUEST serves not only as a test suite but also as an evaluation framework that multi-dimensionally dissects domain-specific translation blind spots hidden behind single composite scores. The dataset is available at https://huggingface.co/datasets/JudyChoi/KUEST.},
  url       = {https://aclanthology.org/2026.wmt-1.86}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
Author{8}{Orcid}:
@InProceedings{dinatale-EtAl:2026:wmt,
  author    = {Di Natale, Paolo  and  Chiocchetti, Elena  and  Ralli, Natascia  and  Alber, Marlies  and  Stemle, Egon W.},
  title     = {Terminology Resources as (MT) Benchmarks: Evaluation of Legal Terminology Challenges in German Language Varieties by EURAC},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1513--1539},
  abstract  = {Language varieties and dialects remain an open challenge for machine translation (MT). Plus, existing benchmarks rarely combine the evaluation of specific varieties with domain-specific knowledge. We introduce a termbase-derived MT benchmark for legal terminology across four German varieties: Germany, Austria, Switzerland, and South Tyrol (a low-resource German variety spoken in Italy). In the inter-variety setting, models translate Italian source sentences by selecting the correct South Tyrolean German term among semantically equivalent terms from other German varieties. In the intra-variety settings, we test sense disambiguation and ontological relation capabilities using homographic and ontologically related distractors for all four language varieties. We find that distinguishing language varieties remains difficult. High performance on low-resourced South Tyrolean terminology is partly driven by indirect exposure to terms shared with better-resourced varieties; when this overlap is removed, even the best-performing models collapse toward small models. We also find mild evidence of dominant-variety interference, although stronger systems more often err toward the geographically and culturally closer Austrian variety. Generally, locale-oriented adaptation appears more challenging than linguistic and semantic reasoning, where the ontological relation task is largely saturated and no generalized bias is shown against a specific variety. We conclude that model scaling mainly reinforces dominant language varieties and that LLM-based MT shifts part of the performance bottleneck from individual term rarity to the underrepresentation of domain-variety intersections across training and alignment stages.},
  url       = {https://aclanthology.org/2026.wmt-1.87}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{han-EtAl:2026:wmt,
  author    = {Han, Lifeng  and  Liang, Jiahui  and  Latusek, Anna  and  El Haff, Karim  and  Haddad Haddad, Amal  and  Höfgen, Josua  and  Evang, Kilian  and  Ma, Min  and  Zhyrko, Maryia},
  title     = {Mind the Gap: Exposing LLM Translation Blind Spots Using the AlphaMWE Multilingual Parallel Corpus},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1540--1561},
  abstract  = {LLMs' performance on machine translation (MT) tasks is often dependent on the data availability in the specific domains and language pairs that they are trained upon. To examine if multiword expressions (MWEs) still present a bottleneck for LLMs regarding language understanding and translation, we report the performance of systems from the WMT 2026 Test Suites shared task, using portions of the publicly available multilingual parallel corpus AlphaMWE as test suites. We received 31 MT systems' outputs covering English to Chinese (zh), Polish (pl), German (de), and Arabic (ar) including Modern Standard Arabic (MSA) and two dialectal ones (Egyptian and Tunisian Arabic). We carried out automatic evaluations using BLEU, ChrF, and BERT-score to select the top 3 systems per language pair, followed up with human evaluations on the selected systems. Our findings show that figurative and MWE-related phenomena remain challenging for contemporary MT systems, automatic metrics sometimes disagree in system ranking, and human evaluation uncovers language-specific errors that remain hidden by aggregate scores. Inter-annotator analysis further reveals challenges in consistently identifying and calibrating linguistically subtle translation errors, highlighting the need for explicit evaluation guidelines and careful annotator calibration.},
  url       = {https://aclanthology.org/2026.wmt-1.88}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-3221-2185
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0009-0001-5439-6149
Author{4}{Orcid}:https://orcid.org/0009-0000-0519-6418
Author{5}{Orcid}:https://orcid.org/0000-0002-7949-612X
Author{6}{Orcid}:
Author{7}{Orcid}:0000-0003-0895-9841
Author{8}{Orcid}:
Author{9}{Orcid}:
@InProceedings{lee-kim-seo:2026:wmt,
  author    = {Lee, Deun Sol  and  Kim, Dae Woong  and  Seo, Yun Ha},
  title     = {Lost in Morphology at WMT26: A Korean-English Test Suite for LLM Translation Blind Spots},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1562--1574},
  abstract  = {We present lost-in-morphology, a 3,536-item Korean to English challenge suite spanning nine phenomena and evaluated on 23 systems submitted to WMT26. The suite targets ellipsis punctuation, stacked particles, wordplay, topic/subject marking, evidentiality, scrambling, auxiliary chains, connective endings and ideophones. We match each phenomenon to its observable evidence: deterministic rules for visible or lexically constrained contrasts, structured LLM observations for semantic features, joint scoring for minimal pairs, distributional metrics for multi-variant frames, and a strategy-aware panel for wordplay. Learned judges and classifiers must pass subset-specific admission tests before their outputs contribute to a system score. Results reveal failures hidden by aggregate MT quality. Ellipsis is usually preserved, but one model family deletes the marked utterance in most of its failures. Particle fidelity falls from 0.95 at stack depth 1 to 0.70 at depth 4, while distinct scalar particles collapse into bare English even. Evidential and auxiliary meanings are often neutralised although the core proposition survives, and wordplay remains the hardest subset: the best system scores 0.385 against 0.546 for the curated references. Judge validation is itself diagnostic: the only conflict-free wordplay judge fails through ceiling saturation, whereas three judges that are also evaluated systems pass without detectable family preference. The suite therefore measures both translation failures and the reliability limits of the instruments used to identify them.},
  url       = {https://aclanthology.org/2026.wmt-1.89}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{dhankhar-EtAl:2026:wmt,
  author    = {Dhankhar, Harshit  and  Gain, Baban  and  Ekbal, Asif  and  Tripathi, Yogesh Mani},
  title     = {Balancing Global Quality and Pronoun-Specific Feedback for Context-Aware Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {164--174},
  abstract  = {Context-aware machine translation can expose the evidence needed for pronoun choice, but standard fine-tuning does not explicitly prioritize these sparse discourse-sensitive decisions. We study ProNMT, a reward-guided iterative self-training method that combines sentence-level quality estimation with a signed confidence signal at generated pronoun positions. For each current sentence and its preceding source context, ProNMT samples candidate translations, scores them using reference-free quality estimation together with a reference-derived pronoun label, and fine-tunes on the highest-scoring candidate. On filtered English--German Europarl and English--French News Commentary data, ProNMT improves over context-aware supervised fine-tuning on BLEU and COMET. Ablations show that pronoun-only feedback can severely degrade sentence-level translation quality on these pronoun-focused data, while hard binary feedback underperforms confidence-weighted feedback. These results indicate that targeted linguistic feedback is most useful when combined with both a global quality signal and the context relevant to the targeted decision. We make the code publicly available at https://github.com/Harshit2807161/ProNMT.},
  url       = {https://aclanthology.org/2026.wmt-1.9}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0001-8673-7078
Author{3}{Orcid}:https://orcid.org/0000-0003-3612-8834
Author{4}{Orcid}:
@InProceedings{ztop:2026:wmt1,
  author    = {Öztop, Yusuf},
  title     = {Evaluating Occupational Gender Bias and Male Default in Turkish-to-English Machine Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1575--1583},
  abstract  = {Turkish has no grammatical gender: its third-person pronoun o refers to a man, a woman, or an inanimate object alike, so a system translating into English must supply a gender the source never states. We introduce tr\_gender\_bias, a reference-free Turkish-to-English test suite of 200 templated items, each built around a single human referent, and use it to measure how 24 machine translation systems and large language models assign gender when the source withholds it. The suite has three parts: ambiguous items whose only signal is an occupational stereotype grounded in labor-force statistics, items whose gender is fixed by a kinship term or a name against that stereotype, and neutral controls. Every output is scored under a six-way, abstention-aware taxonomy that treats neutralization, hedging, and malformed output as distinct outcomes. Under ambiguity, stereotype-driven gender assignment is universal, significant for all 22 testable systems after correction, and uniformly male-skewed, with no system ever favoring the feminine. The effect is sharply asymmetric: every system renders male-typed occupations as he almost without exception, while agreement on female-typed occupations is barely above chance, so what separates systems is how strong a feminine stereotype must be to displace the masculine default. Gender that the context states, by contrast, is now resolved almost perfectly, a marked change from earlier evaluations in which stereotype overrode context. The systems split into committers and neutralizers, and neutralizing with singular they removes both stereotype bias and male skew at no cost in accuracy. Gender bias in current translation is therefore less a failure to use context than an unsettled policy for what to do when the source leaves gender genuinely open.},
  url       = {https://aclanthology.org/2026.wmt-1.90}
}

Author{1}{Orcid}:
@InProceedings{sigursson-EtAl:2026:wmt,
  author    = {Sigurðsson, Einar Freyr  and  Steingrímsson, Steinþór  and  Ármannsson, Bjarki  and  Jasonarson, Atli  and  Sigurdardottir, Helga Svala  and  Markússon, Jón Símon},
  title     = {Translating Syntactic Structures: An English-Icelandic Test Suite for Syntactic Divergence},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1584--1600},
  abstract  = {We present a test suite intended to study how well different machine translation systems do when translating syntactic constructions found in English that do not have a direct equivalent in Icelandic. We manually evaluate the translations and compare it to an automatic evaluation. Whereas the results demonstrate that the best-performing models achieve high accuracy in various categories, they still struggle with others.},
  url       = {https://aclanthology.org/2026.wmt-1.91}
}

Author{1}{Orcid}:https://orcid.org/0009-0005-2783-9388
Author{2}{Orcid}:
Author{3}{Orcid}:0009-0008-3279-8439
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{vasileva:2026:wmt,
  author    = {Vasileva, Lisa},
  title     = {"I'm Sorry, I Can't Translate That": A Challenge Set for Refusal to Translate, Overgeneration and Hallucination in LLM Translation Systems},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1601--1608},
  abstract  = {Large language models (LLMs) have significantly raised the bar for translation quality in the recent years. However, they are still prone to novel failure modes, including hallucinations and overgenerations, where the generated translation is partially or fully detached from the source text and contains information that is not present in the input. To facilitate the analysis of these behaviors and assess the robustness of LLM-based translation systems, we introduce a challenge set specifically designed to probe scenarios that we have observed to be particularly prone to overgeneration, hallucination, and other forms of semantic detachment from the source. We describe the construction of the challenge set and present the evaluation of the systems submitted to the WMT 2026 General Machine Translation shared task.},
  url       = {https://aclanthology.org/2026.wmt-1.92}
}

Author{1}{Orcid}:
@InProceedings{yadav-mukherjee-shrivastava:2026:wmt,
  author    = {Yadav, Saumitra  and  Mukherjee, Ananya  and  Shrivastava, Manish},
  title     = {Finding the Gaps in LLM Translation with CoST},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1609--1621},
  abstract  = {Large language models (LLMs) have advanced machine translation substantially, but generic benchmarks often fail to expose where these systems struggle with complex linguistic structures and stylistic nuance. We use CoST (Complex Structures Test), a challenge suite of 1,947 English-Hindi sentences spanning eight genres, autobiography, conversation, legal, mixed, narration, play, poetry, and technical writing, previously introduced for WMT24, to evaluate 22 machine translation systems submitted to the WMT26 General Translation Shared Task for English-Hindi with reference-free, reference-based, and manual evaluation. Our results reveal a consistent genre-level pattern: systems perform well on narrative text such as autobiography and play, but degrade sharply on poetry, legal, and conversational text. We further identify a distinct failure mode in poetry translation, where systems reproduce memorized canonical Hindi verse instead of translating the given English text, and manual analysis surfaces recurring issues in named-entity handling, wrong-language output, and lexical choice, several of which persist from our prior CoST evaluation on WMT24 submissions. Our test suite is available at https://github.com/deciphyre/CoST-WMT-26-Test-Suite-Task.},
  url       = {https://aclanthology.org/2026.wmt-1.93}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0009-0001-2254-0548
Author{3}{Orcid}:https://orcid.org/0000-0001-8705-6637
@InProceedings{eckhardt-glembek-stewart:2026:wmt,
  author    = {Eckhardt, Alan  and  Glembek, Ondrej  and  Stewart, Craig},
  title     = {Phrase at WMT26 Automated MT Evaluation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1622--1630},
  abstract  = {We describe the Phrase submission to WMT26's shared task on automated MT quality evaluation, covering all three of its tasks: segment-level error detection and span annotation (T1), segment-level quality score prediction (T2), and detection of error-free segments (T3). Treating independent LLMs as a panel of ESA (Error Span Annotation)-style error-span judges, we measure a ceiling: a perfect filter over the panel's pooled spans would raise MPP-F (our span-matching F-score) by roughly two thirds over the strongest single judge. An exhaustive search for how to realize that headroom (diversity, rank-weighted aggregation, a per-judge reliability prior, and boundary refinement) never beats the strongest single judge on MPP-F (T1), because the judges' shared recall is too low for any gold-free span selection to close the gap. The two tasks derived from the same spans behave differently: a static reliability prior wins the segment score (T2), and simple agreement wins the binary error-free decision (T3). Our primary submission, Phrase01, follows this split; a secondary submission, Phrase02, forces in a third judge (UFAL's cat-v4) across all tasks as an explicit ablation, despite evidence it does not help. The lever for span-level detection is annotation coverage through generation, not aggregation; for the derived tasks, it is matching the aggregation mechanism to the metric.},
  url       = {https://aclanthology.org/2026.wmt-1.94}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0003-4665-6550
Author{3}{Orcid}:
@InProceedings{ding-duan:2026:wmt,
  author    = {Ding, Jie  and  Duan, Xiangyu},
  title     = {FECE: A Challenge Set for Evaluating Fine-Grained Translation Error Correction},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1631--1644},
  abstract  = {Existing machine translation evaluation metrics assess translation quality only at the segment level and therefore cannot assess fine-grained translation quality. For translation error correction systems, this means that the quality of fine-grained corrections cannot be evaluated using existing metrics. To highlight this evaluation challenge, we propose the FECE dataset for evaluating fine-grained correction quality. Experiments on FECE show that existing metrics do not always reflect fine-grained correction performance. To enable the evaluation of fine-grained correction performance, we further propose COMET\_Fine, a fine-grained evaluation metric. COMET\_Fine can effectively assess fine-grained correction quality while also revealing inconsistencies between existing metrics and COMET\_Fine.},
  url       = {https://aclanthology.org/2026.wmt-1.95}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{garland-berger:2026:wmt,
  author    = {Garland, Evelyn Y.  and  Berger, Carola F.},
  title     = {Pterodactyl: Pushing the Limits of Efficient Automated Translation Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1645--1650},
  abstract  = {In this paper, we present 1020 Group's submission to the WMT26 shared task on automated quality evaluation. Our system, Pterodactyl, was engineered specifically for subtasks 2 and 3, that is, segment-level quality score prediction and the detection of error-free segments. Pterodactyl is a highly customizable, ultra-low-cost system optimized for binary classification in translation triage and data curation. It is based on an LLM-as-a-judge approach, with prompt engineering and model selection guided by ROC analysis. We also developed several variants of Pterodactyl that were specifically adapted for subtasks 3.},
  url       = {https://aclanthology.org/2026.wmt-1.96}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0009-0005-5745-9351
@InProceedings{griesbeck-geierhos:2026:wmt,
  author    = {Griesbeck, Lena  and  Geierhos, Michaela},
  title     = {AGGREE: An Aggregation-Grounded Challenge Set for Segment-Level MT Metric Evaluation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1651--1658},
  abstract  = {This paper introduces AGGREE, a WMT26 Metrics shared task challenge set, which is designed for the diagnostic evaluation of automatic machine translation quality evaluation systems. It is motivated by the observation that metric agreement can be inflated by aggregation as metrics may correlate well at a system level, but disagree at an individual segment level. AGGREE therefore focuses on naturally occurring system outputs where automatic metrics disagree locally or diverge from available WMT human judgments. The challenge set contains 350 single-hypothesis examples across six language pairs drawn from WMT23 and WMT24 system outputs. The submitted set was mined using diagnostics for metric disagreement and metric-human mismatch, then filtered using strict automated quality control. This WMT26 analysis serves only to describe the behavior of the metrics in AGGREE and does not provide an assessment of the performance degradation compared to standard test data.},
  url       = {https://aclanthology.org/2026.wmt-1.97}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-8180-5606
@InProceedings{hrabal-bojar:2026:wmt,
  author    = {Hrabal, Miroslav  and  Bojar, Ondřej},
  title     = {CUNI at the WMT26 Automated MT Evaluation Shared Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1659--1664},
  abstract  = {We describe the CUNI submissions to the WMT26 Automated MT Evaluation Shared Task. We take part in error-span prediction (Task~1) and segment-level quality score prediction (Task~2). Our systems use a single open-weight Gemma 4 31B instruction-tuned model as a backbone. For Task~1, we decompose error prediction into semantic error detection, self-review, and a separate span-alignment step that maps free-form error descriptions to source and target character spans. We submit two variants of this approach, one additionally predicting MQM-style error categories. For Task~2, we explore two complementary approaches: regression over features extracted from the Task~1 predictions and LLM-based direct assessment on a 0--100 scale.},
  url       = {https://aclanthology.org/2026.wmt-1.98}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-0606-0050
@InProceedings{jang-cho-choi:2026:wmt,
  author    = {Jang, Jisoo  and  Cho, Haewon  and  Choi, Seungtaek},
  title     = {HUFS-DILAB at WMT 2026 Automated Translation Quality Evaluation Task 1: Language-Pair Prompt Routing for Error Span Annotation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1665--1673},
  abstract  = {This paper describes HUFS-DILAB's submission to Task 1, segment-level error detection and span annotation, of the WMT26 Automated Translation Quality Evaluation task. We investigate how far an off-the-shelf LLM can be taken as a judge without task-specific training: Gemma-4-31B-it is prompted to annotate error spans, omissions, and severities as constrained JSON given the source, the reference, and the translation, with each language pair routed to a prompt variant specialized for that pair. This design is motivated by the observation that, even with the same prompt, the judge exhibits different severity and error-density distributions across language pairs. These pair-specific patterns recur in at least 24 of the 26 MT systems, suggesting systematic tendencies of the judge rather than effects specific to individual MT systems. To address these tendencies, we calibrate each prompt variant against human annotations where available. We also correct a rule-based wrong-language filter whose errors on closely related languages were silently overriding the judge's annotations. Our submission achieves a score of 0.5854 on the official metric, averaged over the eight prioritized language pairs.},
  url       = {https://aclanthology.org/2026.wmt-1.99}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0000-0003-3570-0907
