@InProceedings{serdioukova-zhilko:2026:wmt,
  author    = {Serdioukova, Anastasiya  and  Zhilko, Denis},
  title     = {AQI: a composite human-calibrated metric for machine translation quality assessment},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {661--679},
  abstract  = {Manual evaluation of machine translation quality is the gold standard but expensive and noisy. Professional MQM annotation and holistic human scoring both require trained annotators, scale linearly with throughput, and suffer from substantial inter-annotator disagreement, particularly on strong neural systems of recent generations. At production volumes manual evaluation does not scale: automatic evaluation must carry the main load, and humans should remain a corrective signal on a limited subset of segments. The modern automatic-metric inventory includes surface metrics (chrF, TER), embedding-based metrics (BERTScore F1), and learned metrics (COMETKiwi-XXL, MetricX-XXL, xCOMET-XXL). Our audit on the WMT25 General-MT corpus (n=3,450 segment-system ratings, 5 language pairs) shows that no single metric reaches the human-agreement ceiling: the strongest (MetricX-XXL) only approaches it, and combining metrics adds a small but reliable margin. Production stakeholders (PMs, lead translators, clients) at the same time expect a single scalar in the 0–100 range with clear Good/Borderline/Bad thresholds, not a vector of six heterogeneous numbers. We propose the Alconost Quality Index (AQI) — a composite, human-calibrated index of machine translation quality, published in two equal-billing versions differing in the normalisation of the key metric and accompanied by an optional per-LP calibration. AQI is designed as an interpretable, production-friendly instrument: a single score, explicit weights, a transparent path from input automatic metrics to the final number. Beyond purely automatic deployment we derive a simple human-in-the-loop index — a linear blend of one human score into the auto-only AQI — and empirically select the blend weight via honest validation against an independent annotator.},
  url       = {https://aclanthology.org/2026.wmt-1.37}
}

