
@misc{nabail2026ubp2,
archiveprefix = {arXiv},
author = {Mohamed Nabail and Leo Cheng and Jingmin Wang and Nicholas Rhinehart},
eprint = {2606.19328},
primaryclass = {cs.LG},
title = {UBP2: Uncertainty-Balanced Preference Planning for Efficient Preference-based Reinforcement Learning},
url = {https://arxiv.org/abs/2606.19328},
year = {2026}
}

@article{ruan2026qpilots,
title = {QPILOTS: Efficient Test-Time Q-Steering for Flow Policies},
author = {Ruan, Yifan and Cao, Chenyang and Burger, Andreas and Pesaranghader, Ali and Kamali, Kaveh and Kim, Jaehong and Vijaykumar, Nandita and Aspuru-Guzik, Alan and Gilitschenski, Igor and Rhinehart, Nicholas},
journal = {arXiv preprint arXiv:2606.14801},
year = {2026},
doi = {10.48550/arXiv.2606.14801},
url = {https://arxiv.org/abs/2606.14801}
}
@misc{sahak2026oscar,
archiveprefix = {arXiv},
author = {Hshmat Sahak and Aoran Jiao and Nicholas Rhinehart and Tim Barfoot},
eprint = {2606.00990},
primaryclass = {cs.RO},
title = {OSCAR: Obstacle Survival Curves for Adaptive Robot Navigation},
url = {https://arxiv.org/abs/2606.00990},
year = {2026}
}
@article{pourkeshavarz2026autoworld,
title = {AutoWorld: Scaling Multi-Agent Traffic Simulation with Self-Supervised World Models},
author = {Pourkeshavarz, Mozhgan and Liu, Tianran and Rhinehart, Nicholas},
journal = {arXiv preprint arXiv:2603.28963},
year = {2026},
doi = {10.48550/arXiv.2603.28963},
url = {https://arxiv.org/abs/2603.28963}
}
@article{liu2026occsim,
title = {OccSim: Multi-kilometer Simulation with Long-horizon Occupancy World Models},
author = {Liu, Tianran and Zhao, Shengwen and Pourkeshavarz, Mozhgan and Li, Weican and Rhinehart, Nicholas},
journal = {arXiv preprint arXiv:2603.28887},
year = {2026},
doi = {10.48550/arXiv.2603.28887},
url = {https://arxiv.org/abs/2603.28887}
}
@misc{han2025ratatouilleimitationlearningingredients,
title={Ratatouille: Imitation Learning Ingredients for Real-world Social Robot Navigation},
author={James R. Han and Mithun Vanniasinghe and Hshmat Sahak and Nicholas Rhinehart and Timothy D. Barfoot},
year={2025},
eprint={2509.17204},
archivePrefix={arXiv},
primaryClass={cs.RO},
url={https://arxiv.org/abs/2509.17204},
}
@article{cao2025residualrewardmodelspreferencebased,
title={Residual Reward Models for Preference-based Reinforcement Learning},
author={Chenyang Cao and Miguel Rogel-García and Mohamed Nabail and Xueqian Wang and Nicholas Rhinehart},
year={2025},
eprint={2507.00611},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2507.00611},
}
@article{liu2025foundationallidarworldmodels,
title={Towards foundational LiDAR world models with efficient latent flow matching},
author={Tianran Liu and Shengwen Zhao and Nicholas Rhinehart},
year={2025},
eprint={2506.23434},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2506.23434},
}@article{han2024dr,
title={DR-MPC: Deep Residual Model Predictive Control for Real-world Social Navigation},
author={Han, James R and Thomas, Hugues and Zhang, Jian and Rhinehart, Nicholas and Barfoot, Timothy D},
journal={IEEE Robotics and Automation Letters (RA-L)},
year={2025}
}

@article{yang2024carff,
author = {Yang, Jiezhi and Desai, Khushi and Packer, Charles and Bhatia, Harshil and Rhinehart, Nicholas and McAllister, Rowan and Gonzalez, Joseph},
journal = {arXiv preprint arXiv:2401.18075},
title = {CARFF: Conditional Auto-encoded Radiance Field for 3D Scene Forecasting},
year = {2024}
}
@inproceedings{packer2023anyone,
author = {Packer, Charles and Rhinehart, Nicholas and McAllister, Rowan Thomas and Wright, Matthew A and Wang, Xin and He, Jeff and Levine, Sergey and Gonzalez, Joseph E},
booktitle = {Conference on Robot Learning},
organization = {PMLR},
pages = {1607--1617},
title = {Is anyone there? learning a planner contingent on perceptual uncertainty},
year = {2023}
}

@article{montali2024waymo,
author = {Montali, Nico and Lambert, John and Mougin, Paul and Kuefler, Alex and Rhinehart, Nicholas and Li, Michelle and Gulino, Cole and Emrich, Tristan and Yang, Zoey and Whiteson, Shimon and others},
journal = {Advances in Neural Information Processing Systems},
title = {The waymo open sim agents challenge},
volume = {36},
year = {2024}
}
@inproceedings{dashora2022hybrid,
author = {Dashora, Nitish and Shin, Daniel and Shah, Dhruv and Leopold, Henry and Fan, David and Agha-Mohammadi, Ali and Rhinehart, Nicholas and Levine, Sergey},
booktitle = {2022 International Conference on Robotics and Automation (ICRA)},
organization = {IEEE},
pages = {4452--4458},
title = {Hybrid imitative planning with geometric and predictive costs in off-road environments},
year = {2022}
}
@article{shah2022offline,
author = {Shah, Dhruv and Bhorkar, Arjun and Leen, Hrish and Kostrikov, Ilya and Rhinehart, Nick and Levine, Sergey},
journal = {arXiv preprint arXiv:2212.08244},
title = {Offline reinforcement learning for visual navigation},
year = {2022}
}

@inproceedings{weng2022s2net,
author = {Weng, Xinshuo and Nan, Junyu and Lee, Kuan-Hui and McAllister, Rowan and Gaidon, Adrien and Rhinehart, Nicholas and Kitani, Kris},
booktitle = {European Conference on Computer Vision (ECCV)},
title = {S2Net: Stochastic Sequential Pointcloud Forecasting},
year = {2022}
}

@misc{vernaza2021traffic,
author = {Vernaza, Paul and Rhinehart, Nicholas},
month = {November~30},
note = {US Patent 11,189,171},
title = {Traffic prediction with reparameterized pushforward policy for autonomous vehicles},
year = {2021}
}
@inproceedings{rhinehart2021contingencies,
author = {Rhinehart, Nicholas and He, Jeff and Packer, Charles and Wright, Matthew A and McAllister, Rowan and Gonzalez, Joseph E and Levine, Sergey},
booktitle = {2021 IEEE International Conference on Robotics and Automation (ICRA)},
organization = {IEEE},
pages = {13663--13669},
title = {Contingencies from observations: Tractable contingency planning with learned behavior models},
year = {2021}
}

@article{fickinger2021explore,
author = {Fickinger, Arnaud and Jaques, Natasha and Parajuli, Samyak and Chang, Michael and Rhinehart, Nicholas and Berseth, Glen and Russell, Stuart and Levine, Sergey},
journal = {arXiv preprint arXiv:2107.07394},
title = {Explore and control with adversarial surprise},
year = {2021}
}

@article{rhinehart2021information,
author = {Rhinehart, Nicholas and Wang, Jenny and Berseth, Glen and Co-Reyes, John and Hafner, Danijar and Finn, Chelsea and Levine, Sergey},
journal = {Advances in Neural Information Processing Systems},
pages = {10745--10758},
title = {Information is power: Intrinsic control via information capture},
volume = {34},
year = {2021}
}

@inproceedings{weng2021inverting,
author = {Weng, Xinshuo and Wang, Jianren and Levine, Sergey and Kitani, Kris and Rhinehart, Nicholas},
booktitle = {Conference on robot learning},
organization = {PMLR},
pages = {11--20},
title = {Inverting the pose forecasting pipeline with SPF2: Sequential pointcloud forecasting for sequential pose forecasting},
year = {2021}
}

@article{shah2021rapid,
author = {Shah, Dhruv and Eysenbach, Benjamin and Kahn, Gregory and Rhinehart, Nicholas and Levine, Sergey},
journal = {arXiv preprint arXiv:2104.05859},
title = {Rapid exploration for open-world navigation with latent goal models},
year = {2021}
}

@inproceedings{shah2021ving,
author = {Shah, Dhruv and Eysenbach, Benjamin and Kahn, Gregory and Rhinehart, Nicholas and Levine, Sergey},
booktitle = {2021 IEEE International Conference on Robotics and Automation (ICRA)},
organization = {IEEE},
pages = {13215--13222},
title = {Ving: Learning open-world navigation with visual goals},
year = {2021}
}

@misc{vernaza2020generative,
author = {Vernaza, Paul and Choi, Wongun and Rhinehart, Nicholas},
month = {July~7},
note = {US Patent 10,705,531},
title = {Generative adversarial inverse trajectory optimization for probabilistic vehicle forecasting},
year = {2020}
}

@inproceedings{filos2020can,
author = {Filos, Angelos and Tigkas, Panagiotis and McAllister, Rowan and Rhinehart, Nicholas and Levine, Sergey and Gal, Yarin},
booktitle = {International Conference on Machine Learning},
organization = {PMLR},
pages = {3145--3153},
title = {Can autonomous vehicles identify, recover from, and adapt to distribution shifts?},
year = {2020}
}

@article{bharadhwaj2020conservative,
author = {Bharadhwaj, Homanga and Kumar, Aviral and Rhinehart, Nicholas and Levine, Sergey and Shkurti, Florian and Garg, Animesh},
journal = {arXiv preprint arXiv:2010.14497},
title = {Conservative safety critics for exploration},
year = {2020}
}
@inproceedings{rhinehart2020deep,
author = {Rhinehart, Nicholas and McAllister, Rowan and Levine, Sergey},
booktitle = {International Conference on Learning Representations (ICLR)},
title = {Deep Imitative Models for Flexible Inference, Planning, and Control},
year = {2020}
}

@article{singh2020parrot,
author = {Singh, Avi and Liu, Huihan and Zhou, Gaoyue and Yu, Albert and Rhinehart, Nicholas and Levine, Sergey},
journal = {arXiv preprint arXiv:2011.10024},
title = {Parrot: Data-driven behavioral priors for reinforcement learning},
year = {2020}
}
@inproceedings{sharma2019directed,
author = {Sharma, Arjun and Sharma, Mohit and Rhinehart, Nicholas and Kitani, Kris M},
booktitle = {International Conference on Learning Representations (ICLR)},
title = {Directed-Info GAIL: Learning Hierarchical Policies from Unsegmented Demonstrations using Directed Information},
year = {2019}
}
@article{guan2019generative,
author = {Guan, Jiaqi and Yuan, Ye and Kitani, Kris M and Rhinehart, Nicholas},
journal = {arXiv preprint arXiv:1904.06250},
title = {Generative Hybrid Representations for Activity Forecasting with No-Regret Learning},
year = {2019}
}

@phdthesis{rhinehart2019jointly,
author = {Rhinehart, Nicholas},
school = {Carnegie Mellon University},
title = {Jointly Forecasting and Controlling Behavior by Learning from High-Dimensional Data},
year = {2019}
}
@inproceedings{rhinehart2019precog,
author = {Rhinehart, Nicholas and McAllister, Rowan and Kitani, Kris and Levine, Sergey},
booktitle = {Proceedings of the IEEE International Conference on Computer Vision},
title = {PRECOG: PREdiction Conditioned On Goals in Visual Multi-Agent Settings},
year = {2019}
}
@inproceedings{berseth2019smirl,
author = {Berseth, Glen and Geng, Daniel and Devin, Coline and Rhinehart, Nicholas and Finn, Chelsea and Jayaraman, Dinesh and Levine, Sergey},
booktitle = {arXiv preprint arXiv:1912.05510},
title = {SMiRL: Surprise Minimizing RL in Dynamic Environments},
year = {2019}
}
@article{rhinehart2018first,
author = {Rhinehart, Nicholas and Kitani, Kris},
journal = {IEEE Transactions on Pattern Analysis and Machine Intelligence},
publisher = {IEEE},
title = {First-Person Activity Forecasting from Video with Online Inverse Reinforcement Learning},
year = {2018}
}

@inproceedings{pan2018human,
author = {Pan, Xinlei and Ohn-Bar, Eshed and Rhinehart, Nicholas and Xu, Yan and Shen, Yilin and Kitani, Kris M.},
booktitle = {Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems},
organization = {International Foundation for Autonomous Agents and Multiagent Systems},
pages = {1380--1387},
title = {Human-Interactive Subgoal Supervision for Efficient Inverse Reinforcement Learning},
year = {2018}
}

@inproceedings{shankar2018learning,
author = {Shankar, Tanmay and Rhinehart, Nicholas and Muelling, Katharina and Kitani, Kris M.},
booktitle = {arXiv:1806.07822},
title = {Learning Neural Parsers with Deterministic Differentiable Imitation Learning},
year = {2018}
}

teacher' network as input and outputs a compressed student’ network derived from the teacher' network. In the first stage of our method, a recurrent policy network aggressively removes layers from the large teacher’ model. In the second stage, another recurrent policy network carefully reduces the size of each remaining layer. The resulting network is then evaluated to obtain a reward – a score based on the accuracy and compression of the network. Our approach uses this reward signal with policy gradients to train the policies to find a locally optimal student network. Our experiments show that we can achieve compression rates of more than 10x for models such as ResNet-34 while maintaining similar performance to the input teacher' network. We also present a valuable transfer learning result which shows that policies which are pre-trained on smaller teacher’ networks can be used to rapidly speed up training on larger `teacher’ networks.@inproceedings{ashok2018n2n,
author = {Ashok, Anubhav and Rhinehart, Nicholas and Beainy, Fares and Kitani, Kris M.},
booktitle = {International Conference on Learning Representations (ICLR)},
title = {N2N learning: Network to Network Compression via Policy Gradient Reinforcement Learning},
year = {2018}
}
@inproceedings{rhinehart2018r2p2,
author = {Rhinehart, Nicholas and Kitani, Kris M. and Vernaza, Paul},
booktitle = {Proceedings of the European Conference on Computer Vision (ECCV)},
pages = {772--788},
title = {R2P2: A Reparameterized Pushforward Policy for Diverse, Precise Generative Path Forecasting},
year = {2018}
}
@inproceedings{rhinehart2017first,
author = {Rhinehart, Nicholas and Kitani, Kris M.},
booktitle = {The IEEE International Conference on Computer Vision (ICCV)},
pages = {3716--3725},
title = {First-Person Activity Forecasting with Online Inverse Reinforcement Learning},
year = {2017}
}

@inproceedings{venkatraman2017predictive,
author = {Venkatraman, Arun and Rhinehart, Nicholas and Sun, Wen and Pinto, Lerrel and Hebert, Martial and Boots, Byron and Kitani, Kris M. and Bagnell, J. A.},
booktitle = {Advances in Neural Information Processing Systems},
pages = {1172--1183},
title = {Predictive-state decoders: Encoding the future into recurrent networks},
year = {2017}
}
@inproceedings{rhinehart2016learning,
author = {Rhinehart, Nicholas and Kitani, Kris M.},
booktitle = {The IEEE Conference on Computer Vision and Pattern Recognition (CVPR)},
title = {Learning Action Maps of Large Environments Via First-Person Vision},
year = {2016}
}

@inproceedings{rhinehart2015visual,
author = {Rhinehart, Nicholas and Zhou, Jiaji and Hebert, Martial and Bagnell, J Andrew},
booktitle = {2015 IEEE International Conference on Robotics and Automation (ICRA)},
organization = {IEEE},
pages = {5448--5454},
title = {Visual chunking: A list prediction framework for region-based object detection},
year = {2015}
}