@InProceedings{dinatale-EtAl:2026:wmt,
  author    = {Di Natale, Paolo  and  Chiocchetti, Elena  and  Ralli, Natascia  and  Alber, Marlies  and  Stemle, Egon W.},
  title     = {Terminology Resources as (MT) Benchmarks: Evaluation of Legal Terminology Challenges in German Language Varieties by EURAC},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1513--1539},
  abstract  = {Language varieties and dialects remain an open challenge for machine translation (MT). Plus, existing benchmarks rarely combine the evaluation of specific varieties with domain-specific knowledge. We introduce a termbase-derived MT benchmark for legal terminology across four German varieties: Germany, Austria, Switzerland, and South Tyrol (a low-resource German variety spoken in Italy). In the inter-variety setting, models translate Italian source sentences by selecting the correct South Tyrolean German term among semantically equivalent terms from other German varieties. In the intra-variety settings, we test sense disambiguation and ontological relation capabilities using homographic and ontologically related distractors for all four language varieties. We find that distinguishing language varieties remains difficult. High performance on low-resourced South Tyrolean terminology is partly driven by indirect exposure to terms shared with better-resourced varieties; when this overlap is removed, even the best-performing models collapse toward small models. We also find mild evidence of dominant-variety interference, although stronger systems more often err toward the geographically and culturally closer Austrian variety. Generally, locale-oriented adaptation appears more challenging than linguistic and semantic reasoning, where the ontological relation task is largely saturated and no generalized bias is shown against a specific variety. We conclude that model scaling mainly reinforces dominant language varieties and that LLM-based MT shifts part of the performance bottleneck from individual term rarity to the underrepresentation of domain-variety intersections across training and alignment stages.},
  url       = {https://aclanthology.org/2026.wmt-1.87}
}

