{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:54:37Z","timestamp":1773536077096,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T00:00:00Z","timestamp":1773619200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Science Foundation","award":["2433429"],"award-info":[{"award-number":["2433429"]}]},{"name":"Office of Naval Research","award":["N00014-24-1-2784"],"award-info":[{"award-number":["N00014-24-1-2784"]}]},{"name":"Office of Naval Research","award":["N00014-24-1-2603"],"award-info":[{"award-number":["N00014-24-1-2603"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,16]]},"DOI":"10.1145\/3757279.3785585","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:27:38Z","timestamp":1773102458000},"page":"227-236","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LEGS-POMDP: Language and Gesture-Guided Object Search in Partially Observable Environments"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0141-1915","authenticated-orcid":false,"given":"Ivy Xiao","family":"He","sequence":"first","affiliation":[{"name":"Brown University, Providence, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2905-4075","authenticated-orcid":false,"given":"Stefanie","family":"Tellex","sequence":"additional","affiliation":[{"name":"Brown University, Providence, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7732-3666","authenticated-orcid":false,"given":"Jason Xinyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Brown University, Providence, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,16]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Kevin Black Noah Brown Danny Driess Adnan Esmail Michael Equi Chelsea Finn Niccolo Fusai Lachy Groom Karol Hausman Brian Ichter et al. 2024. \u03c0 0: A vision-language-action flow model for general robot control. CoRR abs\/2410.24164 2024. doi: 10.48550. arXiv preprint ARXIV.2410.24164."},{"key":"e_1_3_2_1_2_1","volume-title":"Robotics: Science and Systems.","author":"Brohan Anthony","year":"2023","unstructured":"Anthony Brohan, Noah Brown, et al. 2023. RT-1: Robotics Transformer for Real-World Control at Scale. In Robotics: Science and Systems."},{"key":"e_1_3_2_1_3_1","volume-title":"Conference on Robot Learning.","author":"Brohan Anthony","year":"2023","unstructured":"Anthony Brohan, Noah Brown, et al. 2023. RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control. In Conference on Robot Learning."},{"key":"e_1_3_2_1_4_1","unstructured":"Devendra Singh Chaplot Dhiraj Gandhi Abhinav Gupta and Ruslan Salakhutdinov. 2020. Object Goal Navigation using Goal-Oriented Semantic Exploration. arxiv:2007.00643. arxiv:2007.00643"},{"key":"e_1_3_2_1_5_1","volume-title":"Song-Chun Zhu, Tao Gao, Yixin Zhu, and Siyuan Huang.","author":"Chen Yixin","year":"2021","unstructured":"Yixin Chen, Qing Li, Deqian Kong, Yik Lun Kei, Song-Chun Zhu, Tao Gao, Yixin Zhu, and Siyuan Huang. 2021. YouRefIt: Embodied Reference Understanding with Language and Gesture. arxiv:2109.03413. arxiv:2109.03413"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.026"},{"key":"e_1_3_2_1_7_1","volume-title":"International Joint Conference on Artificial Intelligence (IJCAI).","author":"Cohen Vanya","year":"2024","unstructured":"Vanya Cohen, Jason Xinyu Liu, Raymond Mooney, Stefanie Tellex, and David Watkins. 2024. A Survey of Robotic Language Grounding: Tradeoffs between Symbols and Embeddings. In International Joint Conference on Artificial Intelligence (IJCAI)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.3390\/electronics14122346"},{"key":"e_1_3_2_1_9_1","unstructured":"Zipeng Fu Tony Z. Zhao and Chelsea Finn. 2024. Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation. arxiv:2401.02117. arxiv:2401.02117"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Sourav Garg Krishan Rana Mehdi Hosseinzadeh Lachlan Mares Niko S\u00fcnderhauf Feras Dayoub and Ian Reid. 2024. RoboHop: Segment-based Topological Map Representation for Open-World Visual Navigation. arxiv:2405.05792. arxiv:2405.05792","DOI":"10.1109\/ICRA57147.2024.10610234"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Haoran Geng Helin Xu Chengyang Zhao Chao Xu Li Yi Siyuan Huang and He Wang. 2024. GAPartNet: Cross-Category Domain-Generalizable Object Perception and Manipulation via Generalizable and Actionable Parts. arXiv preprint arXiv:2303.04137v5.","DOI":"10.1109\/CVPR52729.2023.00684"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.adf6991"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812169"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1177\/02783649241229725"},{"key":"e_1_3_2_1_15_1","volume-title":"CAESAR: An Embodied Simulator for Generating Multimodal Referring Expression Datasets. Neural Information Processing Systems.","author":"Mofijul Islam Md.","year":"2022","unstructured":"Md. Mofijul Islam, Reza Mirzaiee, Alexi Gladstone, Haley N. Green, and Tariq Iqbal. 2022. CAESAR: An Embodied Simulator for Generating Multimodal Referring Expression Datasets. Neural Information Processing Systems."},{"key":"e_1_3_2_1_16_1","volume-title":"Openvla: An open-source vision-language-action model. arXiv preprint arXiv:2406.09246.","author":"Kim Moo Jin","year":"2024","unstructured":"Moo Jin Kim, Karl Pertsch, Siddharth Karamcheti, Ted Xiao, Ashwin Balakrishna, Suraj Nair, Rafael Rafailov, Ethan Foster, Grace Lam, Pannag Sanketi, et al. 2024. Openvla: An open-source vision-language-action model. arXiv preprint arXiv:2406.09246."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","unstructured":"Oleg Kobzarev Artem Lykov and Dzmitry Tsetserukou. 2025. GestLLM: Advanced Hand Gesture Interpretation via Large Language Models for Human-Robot Interaction. https:\/\/doi.org\/10.48550\/arXiv.2501.07295 arXiv:2501.07295 [cs] 10.48550\/arXiv.2501.07295","DOI":"10.48550\/arXiv.2501.07295"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/11678816_34"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.3390\/electronics14122346"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","unstructured":"Feng Li Hao Zhang Peize Sun Xueyan Zou Shilong Liu Jianwei Yang Chunyuan Li Lei Zhang and Jianfeng Gao. 2023. Semantic-SAM: Segment and Recognize Anything at Any Granularity. https:\/\/doi.org\/10.48550\/arXiv.2307.04767 arXiv:2307.04767 [cs] 10.48550\/arXiv.2307.04767","DOI":"10.48550\/arXiv.2307.04767"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"e_1_3_2_1_22_1","volume-title":"End-to-End-Learning in High-Performance Tactile Robotic Manipulation. In Conference on Robot Learning (CoRL). 516","author":"Lin Hsiu-Chin","year":"2020","unstructured":"Hsiu-Chin Lin, Soshi Iba, and Fabio Ramos. 2020. Multi-Level Structure vs. End-to-End-Learning in High-Performance Tactile Robotic Manipulation. In Conference on Robot Learning (CoRL). 516."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","unstructured":"Li-Heng Lin Yuchen Cui Yilun Hao Fei Xia and Dorsa Sadigh. 2023. Gesture-Informed Robot Assistance via Foundation Models. https:\/\/doi.org\/10.48550\/arXiv.2309.02721 arXiv:2309.02721 [cs] 10.48550\/arXiv.2309.02721","DOI":"10.48550\/arXiv.2309.02721"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10802696"},{"key":"e_1_3_2_1_25_1","volume-title":"Grounding Complex Natural Language Commands for Temporal Tasks in Unseen Environments. In Conference on Robot Learning (CoRL).","author":"Liu Jason Xinyu","year":"2023","unstructured":"Jason Xinyu Liu, Ziyi Yang, Ifrah Idrees, Sam Liang, Benjamin Schornstein, Stefanie Tellex, and Ankit Shah. 2023. Grounding Complex Natural Language Commands for Temporal Tasks in Unseen Environments. In Conference on Robot Learning (CoRL)."},{"key":"e_1_3_2_1_26_1","volume-title":"European conference on computer vision. 38\u201355","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, et al. 2024. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European conference on computer vision. 38\u201355."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1906.08172"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Atharv Mahesh Mane Dulanga Weerakoon Vigneshwaran Subbaraju Sougata Sen Sanjay E. Sarma and Archan Misra. 2025. Ges3ViG: Incorporating Pointing Gestures into Language-Based 3D Visual Grounding for Embodied Reference Understanding. arxiv:2504.09623. arxiv:2504.09623","DOI":"10.1109\/CVPR52734.2025.00843"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","unstructured":"Arjun Mani Nobline Yoo Will Hinthorn and Olga Russakovsky. 2022. Point and Ask: Incorporating Pointing into Visual Question Answering. https:\/\/doi.org\/10.48550\/arXiv.2011.13681 arXiv:2011.13681 [cs] 10.48550\/arXiv.2011.13681","DOI":"10.48550\/arXiv.2011.13681"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.9"},{"key":"e_1_3_2_1_31_1","unstructured":"Shu Nakamura Yasutomo Kawanishi Shohei Nobuhara and Ko Nishino. 2023. DeePoint: Visual Pointing Recognition and Direction Estimation. arxiv:2304.06977. arxiv:2304.06977"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10341492"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/958432.958460"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2005.12.020"},{"key":"e_1_3_2_1_35_1","volume-title":"Dorsa Sadigh, Chelsea Finn, and Sergey Levine.","author":"Team Octo Model","year":"2024","unstructured":"Octo Model Team, Dibya Ghosh, Homer Walke, Karl Pertsch, Kevin Black, Oier Mees, Sudeep Dasari, Joey Hejna, Charles Xu, Jianlan Luo, Tobias Kreiman, You Liang Tan, Dorsa Sadigh, Chelsea Finn, and Sergey Levine. 2024. Octo: An Open-Source Generalist Robot Policy. In Robotics: Science and Systems."},{"key":"e_1_3_2_1_36_1","volume-title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models. In IEEE International Conference on Robotics and Automation.","author":"O\u2019Neill Abby","year":"2024","unstructured":"Abby O\u2019Neill, Abdul Rehman, et al. 2024. Open X-Embodiment: Robotic Learning Datasets and RT-X Models. In IEEE International Conference on Robotics and Automation."},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society, 46","author":"Pelgrim Madeline Helmer","year":"2024","unstructured":"Madeline Helmer Pelgrim, Ivy Xiao He, Kyle Lee, Falak Pabari, Stefanie Tellex, Thao Nguyen, and Daphna Buchsbaum. 2024. Find it like a dog: Using Gesture to Improve Object Search. Proceedings of the Annual Meeting of the Cognitive Science Society, 46, 0 (2024), https:\/\/escholarship.org\/uc\/item\/0nk6w9fd"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","unstructured":"Leah Perlmutter Eric Kernfeld and Maya Cakmak. 2016. Situated Language Understanding with Human-like and Visualization-Based Transparency. In Robotics: Science and Systems XII. Robotics: Science and Systems Foundation. isbn:978-0-9923747-2-3 https:\/\/doi.org\/10.15607\/RSS.2016.XII.040 10.15607\/RSS.2016.XII.040","DOI":"10.15607\/RSS.2016.XII.040"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","unstructured":"Mohit Shridhar and David Hsu. 2018. Interactive Visual Grounding of Referring Expressions for Human-Robot Interaction. https:\/\/doi.org\/10.48550\/arXiv.1806.03831 arXiv:1806.03831 [cs] 10.48550\/arXiv.1806.03831","DOI":"10.48550\/arXiv.1806.03831"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 24th International Conference on Neural Information Processing Systems -","volume":"2","author":"Silver David","year":"2010","unstructured":"David Silver and Joel Veness. 2010. Monte-Carlo planning in large POMDPs. In Proceedings of the 24th International Conference on Neural Information Processing Systems - Volume 2 (NIPS\u201910). Curran Associates Inc., Red Hook, NY, USA. 2164\u20132172."},{"key":"e_1_3_2_1_41_1","volume-title":"First Workshop on Vision-Language Models for Navigation and Manipulation at ICRA","author":"Tanada Kosei","year":"2024","unstructured":"Kosei Tanada, Shigemichi Matsuzaki, Kazuhito Tanaka, Shintaro Nakaoka, Yuki Kondo, and Yuto Mori. 2024. Pointing Gesture Understanding via Visual Prompting and Visual Question Answering for Interactive Robot Navigation. In First Workshop on Vision-Language Models for Navigation and Manipulation at ICRA 2024. https:\/\/openreview.net\/forum?id=sJjwtGvK5D"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-101119-071628"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793888"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2407.05530"},{"key":"e_1_3_2_1_45_1","unstructured":"Zihan Wang Xiangyu Yang Jiahao Liang Jing Xu Yuehu Luo Zhiming Yang Haojian Zhang Xiaoyu Hu and Yandong Wang. 2024. Sim-to-Real Transfer via 3D Feature Fields for Vision-and-Language Navigation. arXiv preprint arXiv:2406.09798."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989121"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","unstructured":"Jianwei Yang Hao Zhang Feng Li Xueyan Zou Chunyuan Li and Jianfeng Gao. 2023. Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V. https:\/\/doi.org\/10.48550\/arXiv.2310.11441 arXiv:2310.11441 [cs] 10.48550\/arXiv.2310.11441","DOI":"10.48550\/arXiv.2310.11441"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","unstructured":"Yang Yang Xibai Lou and Changhyun Choi. 2022. Interactive Robotic Grasping with Attribute-Guided Disambiguation. https:\/\/doi.org\/10.48550\/arXiv.2203.08037 arXiv:2203.08037 [cs] 10.48550\/arXiv.2203.08037","DOI":"10.48550\/arXiv.2203.08037"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610712"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2108.11092"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","unstructured":"Kaiyu Zheng Deniz Bayazit Rebecca Mathew Ellie Pavlick and Stefanie Tellex. 2021. Spatial Language Understanding for Object Search in Partially Observed City-scale Environments. https:\/\/doi.org\/10.48550\/arXiv.2012.02705 arXiv:2012.02705 [cs] 10.48550\/arXiv.2012.02705","DOI":"10.48550\/arXiv.2012.02705"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","unstructured":"Kaiyu Zheng Anirudha Paul and Stefanie Tellex. 2023. A System for Generalized 3D Multi-Object Search. https:\/\/doi.org\/10.48550\/arXiv.2303.03178 arXiv:2303.03178 [cs] 10.48550\/arXiv.2303.03178","DOI":"10.48550\/arXiv.2303.03178"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","unstructured":"Xueyan Zou Jianwei Yang Hao Zhang Feng Li Linjie Li Jianfeng Wang Lijuan Wang Jianfeng Gao and Yong Jae Lee. 2023. Segment Everything Everywhere All at Once. https:\/\/doi.org\/10.48550\/arXiv.2304.06718 arXiv:2304.06718 [cs] 10.48550\/arXiv.2304.06718","DOI":"10.48550\/arXiv.2304.06718"}],"event":{"name":"HRI '26: 21st ACM\/IEEE International Conference on Human-Robot Interaction","location":"Edinburgh Scotland UK","acronym":"HRI '26","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction","IEEE RAS"]},"container-title":["Proceedings of the 21st ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3757279.3785585","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:31:27Z","timestamp":1773534687000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757279.3785585"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,16]]},"references-count":53,"alternative-id":["10.1145\/3757279.3785585","10.1145\/3757279"],"URL":"https:\/\/doi.org\/10.1145\/3757279.3785585","relation":{},"subject":[],"published":{"date-parts":[[2026,3,16]]},"assertion":[{"value":"2026-03-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}