
GenARM
An autoregressive reward model guides generation token by token, turning alignment into a controllable test-time procedure.
Spend computation where it changes the answer. We treat decoding as a control problem: use rewards, value estimates, lightweight interventions, and multiple models to guide generation without retraining the base model.
Research question
We treat decoding as a control problem: use rewards, value estimates, lightweight interventions, and multiple models to guide generation without retraining the base model.
Representative projects
Each project connects its central idea with papers, code, datasets, demonstrations, and public explanations.

An autoregressive reward model guides generation token by token, turning alignment into a controllable test-time procedure.



Controlled decoding combines a mixture of agents so complementary model strengths can guide a single aligned response.
Selected publications
Browse the full publication database for the broader body of work in reasoning control.
The Thirteenth International Conference on Learning Representations (ICLR), 2025
@inproceedings{xu2025genarmceb1,
title = {GenARM: Reward Guided Generation with Autoregressive Reward Model for Test-Time Alignment},
author = {Yuancheng Xu and Udari Madhushani Sehwag and Alec Koppel and Sicheng Zhu and Bang An and Furong Huang and Sumitra Ganesh},
booktitle = {The Thirteenth International Conference on Learning Representations (ICLR), 2025},
year = {2025},
eprint = {2410.08193},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2410.08193},
}Forty-third International Conference on Machine Learning (ICML), 2026
@inproceedings{ghosal2026safetyd87b,
title = {Safety Recovery in Reasoning Models Is Only a Few Early Steering Steps Away},
author = {Soumya Suvra Ghosal and Souradip Chakraborty and Vaibhav Singh and Furong Huang and Dinesh Manocha and Amrit Singh Bedi},
booktitle = {Forty-third International Conference on Machine Learning (ICML), 2026},
year = {2026},
eprint = {2602.11096},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2602.11096},
}The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS), 2024
@inproceedings{chakraborty2024transferf3e1,
title = {Transfer Q-star: Principled Decoding for LLM Alignment},
author = {Souradip Chakraborty and Soumya Suvra Ghosal and Ming Yin and Dinesh Manocha and Mengdi Wang and Amrit Bedi and Furong Huang},
booktitle = {The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS), 2024},
year = {2024},
eprint = {2405.20495},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2405.20495},
}The Thirteenth International Conference on Learning Representations (ICLR), 2025
@inproceedings{chakraborty2025collabbdbe,
title = {Collab: Controlled Decoding using Mixture of Agents for LLM Alignment},
author = {Souradip Chakraborty and Sujay Bhatt and Udari Madhushani Sehwag and Soumya Suvra Ghosal and Jiahao Qiu and Mengdi Wang and Dinesh Manocha and Furong Huang and Alec Koppel and Sumitra Ganesh},
booktitle = {The Thirteenth International Conference on Learning Representations (ICLR), 2025},
year = {2025},
eprint = {2503.21720},
archivePrefix = {arXiv},
url = {https://arxiv.org/abs/2503.21720},
}