@InProceedings{yadav-mukherjee-shrivastava:2026:wmt,
  author    = {Yadav, Saumitra  and  Mukherjee, Ananya  and  Shrivastava, Manish},
  title     = {Finding the Gaps in LLM Translation with CoST},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1609--1621},
  abstract  = {Large language models (LLMs) have advanced machine translation substantially, but generic benchmarks often fail to expose where these systems struggle with complex linguistic structures and stylistic nuance. We use CoST (Complex Structures Test), a challenge suite of 1,947 English-Hindi sentences spanning eight genres, autobiography, conversation, legal, mixed, narration, play, poetry, and technical writing, previously introduced for WMT24, to evaluate 22 machine translation systems submitted to the WMT26 General Translation Shared Task for English-Hindi with reference-free, reference-based, and manual evaluation. Our results reveal a consistent genre-level pattern: systems perform well on narrative text such as autobiography and play, but degrade sharply on poetry, legal, and conversational text. We further identify a distinct failure mode in poetry translation, where systems reproduce memorized canonical Hindi verse instead of translating the given English text, and manual analysis surfaces recurring issues in named-entity handling, wrong-language output, and lexical choice, several of which persist from our prior CoST evaluation on WMT24 submissions. Our test suite is available at https://github.com/deciphyre/CoST-WMT-26-Test-Suite-Task.},
  url       = {https://aclanthology.org/2026.wmt-1.93}
}

