@InProceedings{ztop:2026:wmt2,
  author    = {Öztop, Yusuf},
  title     = {KYX WMT 2026 CreoleMT System Description: A Contamination- and Orthography-Aware Evaluation for Réunion Creole to English},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2462--2469},
  abstract  = {This paper presents the KYX submission to the WMT 2026 CreoleMT shared task for Réunion Creole to English (rcf-eng). It is mostly about how such a system should be evaluated. The pair is an extreme case: 191 training pairs, a 34-sentence development set, and no standard orthography. On a benchmark this small, a single chrF++ score is easy to over-read, so we pair our system with a contamination- and orthography-aware evaluation. The system continue-trains the official baseline adapter with QLoRA and raises the raw dev score by +13.3 chrF++. But a near-duplicate audit finds that six of the 34 dev sources are paraphrases of training sentences, which the exact-match check WMT26 mandates reports as 0\% overlap. After decontamination the gain is +10.0. Our official test score then comes in 0.8 chrF++ below the decontaminated estimate and 5.1 below the raw one, but our margin over the baseline falls to +3.9, so even the adjusted gain was optimistic. A data-scaling ablation traces the gain to the provided real pairs, not to synthetic data. A source-spelling stress test then shows the advantage shrinks when the input is re-spelled into equally valid variants, though at this sample size the effect is not stable. We conclude that exact-match contamination checks are not enough on tiny benchmarks, and that scores like ours should be read with care. The same caution applies to any low-resource pair with only a few dozen dev sentences. We release the system, the audit, and all data.},
  url       = {https://aclanthology.org/2026.wmt-1.189}
}

