@InProceedings{marmonier-bawden-sagot:2026:wmt,
  author    = {Marmonier, Malik  and  Bawden, Rachel  and  Sagot, Benoît},
  title     = {A French Version of the SmolSent Corpus},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {2516--2525},
  abstract  = {We present a French version of the SmolSent corpus, a contribution to the WMT 2026 Open Language Data Initiative (OLDI) shared task. Originally curated to maximize unique-word coverage across 863 English sentences, SmolSent constitutes a valuable, token-efficient training set for the machine translation of low-resource languages. We describe our translation workflow, which relied on a custom-built interface to select and refine translation hypotheses generated by a traditional encoder-decoder model, alongside two state-of-the-art reasoning-enabled large language models (LLMs). Experimental validation based on the MetricX-24-XXL model for quality estimation, supported by bootstrap resampling significance tests, indicates that human post-editing results in significant quality improvements over all raw model outputs, even as reasoning models like Gemma-4-31B-it achieve a competitive 39.51\% zero-edit rate. This French partition is not an end in itself, but a pivot resource meant to facilitate ongoing data collection efforts for under-resourced regional languages of France, such as Savoyard and Gallo.},
  url       = {https://aclanthology.org/2026.wmt-1.196}
}

