@InProceedings{nishihama:2026:wmt,
  author    = {Nishihama, Chris},
  title     = {MakotoAI at WMT26: Per-Pair Model Selection with Deterministic Output Normalization for the General Machine Translation Task},
  booktitle      = {Proceedings of the Eleventh Conference on Machine Translation},
  month          = {October},
  year           = {2026},
  address        = {Budapest, Hungary},
  publisher      = {Association for Computational Linguistics},
  pages     = {1379--1384},
  abstract  = {We describe MakotoAI, an unconstrained submission to the WMT26 General Machine Translation shared task covering all 23 language pairs (4,897 documents). The system performs no training or fine-tuning: it is a lightweight orchestration layer over six off-the-shelf commercial LLMs, in which each language pair is routed to the single model that won a blinded pre-competition shootout for that pair. Translation is document-level with the full task instruction provided in-prompt. Structured output (JSON and HTML domains) is repaired by a small deterministic normalization pass rather than by model-based retry: an ablation on real test data found that a post-hoc validate-and-retry instruction layer was inert on clean documents and net-negative on JSON documents, while deterministic fence and preamble stripping achieved the intended effect at zero inference cost. Quality assurance combined an exhaustive deterministic instruction-adherence scan, a two-judge LLM evaluation whose self-grading bias we measured to be directional by model family, and the organizers' official alignment checker, against which the final submission scores 1,914 of 1,914 structurally checkable documents aligned. Four defective documents (three truncations, one repetition-loop degeneration) were detected and regenerated, one requiring escalation to a different model; all four are disclosed. The entire system runs on commodity hardware; total API cost to produce and verify the submission was just under \$25. We argue that for API-orchestrated MT, verification discipline and deterministic post-processing are higher-leverage than added inference-time machinery.},
  url       = {https://aclanthology.org/2026.wmt-1.74}
}

