@InProceedings{pan-seeber:2026:wmt,
  author    = {PAN, Dongpeng  and  Seeber, Kilian G.},
  title     = {Comparing Machine and Human Simultaneous Interpreting},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {548--560},
  abstract  = {We evaluate four machine simultaneous interpreting (SI) systems, an open-weight and a commercial instance of both the cascaded and the end-to-end speech-native architecture, on simulated conference discourse interpreted into five languages. All outputs are compared with professional interpretations. Reference-free quality metrics are first calibrated against blind expert ratings of the human renditions: their association with expert ratings is strongest for content, weaker for presentation, and weakest for style. Machine systems attain content-similarity scores within or above the professional range and higher median segment coverage at the reported alignment threshold. Coverage can count professional compression as omission and is sensitive to transcription and alignment error, so the surplus does not establish better interpreting. For latency, only the commercial speech-native system approaches professional timing. Estimated content lag thus ranges from under twice to more than ten times the professional median, and each system's delay is associated with an observable operating property: commit granularity, synthesis queuing, or compute limits. Together, the rating-calibrated metrics and professional distribution provide a reference-free procedure for evaluating machine interpreting on unsegmented, full-length conference discourse.},
  url       = {https://aclanthology.org/2026.wmt-1.30}
}

