@InProceedings{ahmed-chakma:2026:wmt,
  author    = {Ahmed, Firoz  and  Chakma, Dinalo},
  title     = {A Human-Translated FLORES+ Dataset for Chakma Language},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2475--2480},
  abstract  = {We present a human-translated, native-script Chakma (ccp\_Cakm) extension to FLORES+. The corpus comprises the 997-sentence dev and 1,012-sentence devtest English splits, for 2,009 records in total. We recruited three native Chakma speakers through a screening task, conducted six virtual training sessions, and required direct translation without machine-translation or language-model output. A native reviewer checked meaning, grammar, script, terminology, named entities, numbers, and punctuation, returning problematic items for revision. We describe the Bangladeshi Chakma variety used and our orthographic and Unicode decisions. We then report a completed release audit: FLORES+ identifiers were recovered from the canonical English order and renumbered to the 1-based scheme, all 2,009 records were verified one-to-one against the English source, text was normalised to Unicode NFC, a character inventory was produced, and all twenty-seven residual non-Chakma characters were located and resolved. The package passes every check with no open defects, is fixed by SHA-256 checksums, and has been submitted to FLORES+ through the Open Language Data Initiative under CC BY-SA 4.0. The work provides a native-script evaluation resource for a language whose existing computational resources frequently rely on Bengali-script transliteration.},
  url       = {https://aclanthology.org/2026.wmt-1.191}
}

