@InProceedings{choi-EtAl:2026:wmt,
  author    = {Choi, Minjoo  and  Jung, Jaejun  and  Han, Gunwoo  and  Kim, Junhyeop  and  Yoon, Yiji  and  Kim, Hyemi  and  Cho, Yeonseo  and  Jang, Jinwoo},
  title     = {KUEST: Korean game Usage Evaluation Suite for LLM Translation},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1491--1512},
  abstract  = {Despite the growing demand for multilingual game localization, systematic evaluation frameworks for game translation constraints remain largely unexplored. As part of the WMT26 Test Suite Track, we propose KUEST, a test suite comprising 1,796 English-to-Korean segments across four core categories, and evaluate 27 submitted systems. Experimental results demonstrate that standard metrics obscure true model capabilities (e.g., the overall 1st-place system dropping to 15th in creative translation). Furthermore, we reveal residual capability fragmentation—category-specific variance that survives the shared general-competence factor and inverts top-tier rankings—and observe a "prompt gap'' where context metadata degrades performance. As such, KUEST serves not only as a test suite but also as an evaluation framework that multi-dimensionally dissects domain-specific translation blind spots hidden behind single composite scores. The dataset is available at https://huggingface.co/datasets/JudyChoi/KUEST.},
  url       = {https://aclanthology.org/2026.wmt-1.86}
}

