UniMaia: Steering Chess Policies with Language for Human-like Play. Siu, S. & Istead, L. 2026.
Paper abstract bibtex Recent advances in large language models have enabled natural language to serve as a flexible interface for controlling complex systems, but often at the cost of large-scale multimodal training or weakened domain-specific inductive biases. In structured decision-making domains such as chess, specialized policy networks achieve strong performance but lack semantic controllability, while prompt-conditioned language models are more flexible yet typically exhibit weaker domain grounding. We propose UniMaia, a framework for prompt-conditioned policy modulation that adapts a frozen Lc0-based chess policy network using a parameter-efficient text encoder and a ControlNet-style conditioning mechanism. UniMaia enables semantic control over gameplay, including opening selection and player strength, while preserving the pretrained policy representations. We further introduce UniMaia-Aux, which incorporates auxiliary temporal conditioning and behavioral prediction objectives. To support this work, we construct a large-scale metadata-augmented Lichess dataset, develop a semi-automated prompt-generation pipeline, and introduce benchmarks spanning both prompt-conditioned and metadata-conditioned settings. UniMaia achieves state-of-the-art expected accuracy on several prompt-conditioned benchmarks and competitive top-move accuracy on general instruction-following tasks, while remaining competitive with dedicated metadata-conditioned approaches on human move prediction benchmarks. UniMaia-Aux further improves expected accuracy and behavioral modeling across several evaluation settings, with modest trade-offs in top-move accuracy. Overall, our results demonstrate that prompt-conditioned control of domain-specific policy networks is feasible without end-to-end multimodal training, while highlighting trade-offs between controllability and predictive performance.
@misc{siu:2026:unimaia-steering-chess-policies-with-language-for-human-like-play,
title = {UniMaia: Steering Chess Policies with Language for Human-like Play},
author = {Sherman Siu and Lesley Istead},
year = {2026},
eprint = {2605.27767},
url = {https://arxiv.org/abs/2605.27767},
abstract = {Recent advances in large language models have enabled natural language to serve as a flexible interface for controlling complex systems, but often at the cost of large-scale multimodal training or weakened domain-specific inductive biases. In structured decision-making domains such as chess, specialized policy networks achieve strong performance but lack semantic controllability, while prompt-conditioned language models are more flexible yet typically exhibit weaker domain grounding. We propose UniMaia, a framework for prompt-conditioned policy modulation that adapts a frozen Lc0-based chess policy network using a parameter-efficient text encoder and a ControlNet-style conditioning mechanism. UniMaia enables semantic control over gameplay, including opening selection and player strength, while preserving the pretrained policy representations. We further introduce UniMaia-Aux, which incorporates auxiliary temporal conditioning and behavioral prediction objectives. To support this work, we construct a large-scale metadata-augmented Lichess dataset, develop a semi-automated prompt-generation pipeline, and introduce benchmarks spanning both prompt-conditioned and metadata-conditioned settings. UniMaia achieves state-of-the-art expected accuracy on several prompt-conditioned benchmarks and competitive top-move accuracy on general instruction-following tasks, while remaining competitive with dedicated metadata-conditioned approaches on human move prediction benchmarks. UniMaia-Aux further improves expected accuracy and behavioral modeling across several evaluation settings, with modest trade-offs in top-move accuracy. Overall, our results demonstrate that prompt-conditioned control of domain-specific policy networks is feasible without end-to-end multimodal training, while highlighting trade-offs between controllability and predictive performance.},
affiliation = {University of Waterloo, Carleton University},
archiveprefix = {arXiv},
primaryclass = {cs.CL},
}
Downloads: 0
{"_id":"M4WfFREZqQs8tZmjn","bibbaseid":"siu-istead-unimaiasteeringchesspolicieswithlanguageforhumanlikeplay-2026","author_short":["Siu, S.","Istead, L."],"bibdata":{"bibtype":"misc","type":"misc","title":"UniMaia: Steering Chess Policies with Language for Human-like Play","author":[{"firstnames":["Sherman"],"propositions":[],"lastnames":["Siu"],"suffixes":[]},{"firstnames":["Lesley"],"propositions":[],"lastnames":["Istead"],"suffixes":[]}],"year":"2026","eprint":"2605.27767","url":"https://arxiv.org/abs/2605.27767","abstract":"Recent advances in large language models have enabled natural language to serve as a flexible interface for controlling complex systems, but often at the cost of large-scale multimodal training or weakened domain-specific inductive biases. In structured decision-making domains such as chess, specialized policy networks achieve strong performance but lack semantic controllability, while prompt-conditioned language models are more flexible yet typically exhibit weaker domain grounding. We propose UniMaia, a framework for prompt-conditioned policy modulation that adapts a frozen Lc0-based chess policy network using a parameter-efficient text encoder and a ControlNet-style conditioning mechanism. UniMaia enables semantic control over gameplay, including opening selection and player strength, while preserving the pretrained policy representations. We further introduce UniMaia-Aux, which incorporates auxiliary temporal conditioning and behavioral prediction objectives. To support this work, we construct a large-scale metadata-augmented Lichess dataset, develop a semi-automated prompt-generation pipeline, and introduce benchmarks spanning both prompt-conditioned and metadata-conditioned settings. UniMaia achieves state-of-the-art expected accuracy on several prompt-conditioned benchmarks and competitive top-move accuracy on general instruction-following tasks, while remaining competitive with dedicated metadata-conditioned approaches on human move prediction benchmarks. UniMaia-Aux further improves expected accuracy and behavioral modeling across several evaluation settings, with modest trade-offs in top-move accuracy. Overall, our results demonstrate that prompt-conditioned control of domain-specific policy networks is feasible without end-to-end multimodal training, while highlighting trade-offs between controllability and predictive performance.","affiliation":"University of Waterloo, Carleton University","archiveprefix":"arXiv","primaryclass":"cs.CL","bibtex":"@misc{siu:2026:unimaia-steering-chess-policies-with-language-for-human-like-play,\n title = {UniMaia: Steering Chess Policies with Language for Human-like Play},\n author = {Sherman Siu and Lesley Istead},\n year = {2026},\n eprint = {2605.27767},\n url = {https://arxiv.org/abs/2605.27767},\n abstract = {Recent advances in large language models have enabled natural language to serve as a flexible interface for controlling complex systems, but often at the cost of large-scale multimodal training or weakened domain-specific inductive biases. In structured decision-making domains such as chess, specialized policy networks achieve strong performance but lack semantic controllability, while prompt-conditioned language models are more flexible yet typically exhibit weaker domain grounding. We propose UniMaia, a framework for prompt-conditioned policy modulation that adapts a frozen Lc0-based chess policy network using a parameter-efficient text encoder and a ControlNet-style conditioning mechanism. UniMaia enables semantic control over gameplay, including opening selection and player strength, while preserving the pretrained policy representations. We further introduce UniMaia-Aux, which incorporates auxiliary temporal conditioning and behavioral prediction objectives. To support this work, we construct a large-scale metadata-augmented Lichess dataset, develop a semi-automated prompt-generation pipeline, and introduce benchmarks spanning both prompt-conditioned and metadata-conditioned settings. UniMaia achieves state-of-the-art expected accuracy on several prompt-conditioned benchmarks and competitive top-move accuracy on general instruction-following tasks, while remaining competitive with dedicated metadata-conditioned approaches on human move prediction benchmarks. UniMaia-Aux further improves expected accuracy and behavioral modeling across several evaluation settings, with modest trade-offs in top-move accuracy. Overall, our results demonstrate that prompt-conditioned control of domain-specific policy networks is feasible without end-to-end multimodal training, while highlighting trade-offs between controllability and predictive performance.},\n affiliation = {University of Waterloo, Carleton University},\n archiveprefix = {arXiv},\n primaryclass = {cs.CL},\n}\n\n","author_short":["Siu, S.","Istead, L."],"key":"siu:2026:unimaia-steering-chess-policies-with-language-for-human-like-play","id":"siu:2026:unimaia-steering-chess-policies-with-language-for-human-like-play","bibbaseid":"siu-istead-unimaiasteeringchesspolicieswithlanguageforhumanlikeplay-2026","role":"author","urls":{"Paper":"https://arxiv.org/abs/2605.27767"},"metadata":{"authorlinks":{}}},"bibtype":"misc","biburl":"https://raw.githubusercontent.com/lichess-org/papers/main/lichess.bib","dataSources":["shB3Z7oaBcgBmet5C"],"keywords":[],"search_terms":["unimaia","steering","chess","policies","language","human","play","siu","istead"],"title":"UniMaia: Steering Chess Policies with Language for Human-like Play","year":2026}