@InProceedings{siani:2026:wmt,
  author    = {Siani, Assaf},
  title     = {Reference-Free Consensus Evaluation of Machine Translation: Ensembling LLM Error-Span Judgments with Neutral Cross-Verification and STAPLE Reliability Fusion},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1746--1751},
  abstract  = {We present a reference-free system for the WMT 2026 automated translation-quality-evaluation shared task, covering all three subtasks: character-level error-span detection with Minor/Major severity, segment level quality scoring on a 0–100 scale, and binary error-free classification. Rather than treating a single large language model (LLM) as an oracle, the system treats LLMs as fallible annotators. Three heterogeneous judges independently propose target-side error spans; a neutral cross-verification round converts these open-ended proposals into a complete, anonymized candidate-by-judge vote matrix; a character-level adaptation of STAPLE estimates latent error truth and per-judge reliability without references; and a two-state HMM imposes minimal span structure. Omissions are handled by a dedicated source-coverage pass rather than by scanning the target. Calibrating the single free prior against the official metric with human gold yields several findings: plain majority voting matches or beats STAPLE; the error rate prior barely matters once outputs are schema-conformant; precision—judge over-flagging at roughly 3× the human rate—is the binding constraint; and LLM-derived silver labels cannot calibrate the prior without circularity. The system was deployed across six language pairs with partial-failure-tolerant, resumable execution.},
  url       = {https://aclanthology.org/2026.wmt-1.108}
}

