{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,16]],"date-time":"2025-01-16T05:39:07Z","timestamp":1737005947103,"version":"3.33.0"},"reference-count":34,"publisher":"Elsevier BV","issue":"8","license":[{"start":{"date-parts":[[2007,8,1]],"date-time":"2007-08-01T00:00:00Z","timestamp":1185926400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Robotics and Autonomous Systems"],"published-print":{"date-parts":[[2007,8]]},"DOI":"10.1016\/j.robot.2007.03.003","type":"journal-article","created":{"date-parts":[[2007,4,23]],"date-time":"2007-04-23T11:04:07Z","timestamp":1177326247000},"page":"628-642","source":"Crossref","is-referenced-by-count":12,"title":["Application of SONQL for real-time learning of robot behaviors"],"prefix":"10.1016","volume":"55","author":[{"given":"Marc","family":"Carreras","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junku","family":"Yuh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joan","family":"Batlle","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pere","family":"Ridao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"year":"1998","series-title":"Behavior-Based Robotics","author":"Arkin","key":"10.1016\/j.robot.2007.03.003_b1"},{"issue":"1","key":"10.1016\/j.robot.2007.03.003_b2","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1109\/JRA.1986.1087032","article-title":"A robust layered control system for a mobile robot","volume":"RA-2","author":"Brooks","year":"1986","journal-title":"IEEE Journal of Robotics and Automation"},{"year":"1998","series-title":"Reinforcement Learning, an Introduction","author":"Sutton","key":"10.1016\/j.robot.2007.03.003_b3"},{"key":"10.1016\/j.robot.2007.03.003_b4","doi-asserted-by":"crossref","first-page":"311","DOI":"10.1016\/0004-3702(92)90058-6","article-title":"Automatic programming of behavior-based robots using reinforcement learning","volume":"55","author":"Mahadevan","year":"1992","journal-title":"Artificial Intelligence"},{"key":"10.1016\/j.robot.2007.03.003_b5","unstructured":"M. Ryan, M. Pendrith, RL-TOPs: An architecture for modularity and re-use in reinforcement learning, in: Fifteenth International Conference on Machine Learning, Madison, Wisconsin, 1998"},{"issue":"3\u20134","key":"10.1016\/j.robot.2007.03.003_b6","doi-asserted-by":"crossref","first-page":"365","DOI":"10.1177\/105971239700500307","article-title":"Measuring the effectiveness of reinforcement learning for behavior-based robotics","volume":"5","author":"Shackleton","year":"1997","journal-title":"Adaptive Behavior"},{"key":"10.1016\/j.robot.2007.03.003_b7","unstructured":"Y. Takahashi, M. Asada, Vision-guided behavior acquisition of a mobile robot by multilayered reinforcement learning, in: IEEE\/RSJ International Conference on Intelligent Robots and Systems, 2000"},{"key":"10.1016\/j.robot.2007.03.003_b8","doi-asserted-by":"crossref","first-page":"251","DOI":"10.1016\/S0921-8890(97)00042-0","article-title":"Neural reinforcement learning for behavior synthesis","volume":"22","author":"Touzet","year":"1997","journal-title":"Robotics and Autonomous Systems"},{"key":"10.1016\/j.robot.2007.03.003_b9","doi-asserted-by":"crossref","unstructured":"D. Gachet, M. Salichs, L. Moreno, J. Pimental, Learning emergent tasks for an autonomous mobile robot, in: Proceedings of the International Conference on Intelligent Robots and Systems, vol. 1, IROS\u201994, Munich, Germany, 1994, pp. 290\u2013297","DOI":"10.1109\/IROS.1994.407378"},{"key":"10.1016\/j.robot.2007.03.003_b10","doi-asserted-by":"crossref","unstructured":"Z. Kalmar, C. Szepesvari, A. Lorincz, Module-based reinforcement learning: Experiments with a real robot, in: Proceedings of the 6th European Workshop on Learning Robots, 1997","DOI":"10.1007\/3-540-49240-2_3"},{"key":"10.1016\/j.robot.2007.03.003_b11","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1016\/S0921-8890(05)80028-4","article-title":"Situated agents can have goals","volume":"6","author":"Maes","year":"1990","journal-title":"Robotics and Automation Systems"},{"key":"10.1016\/j.robot.2007.03.003_b12","doi-asserted-by":"crossref","unstructured":"E. Martinson, A. Stoytchev, R. Arkin, Robot behavioral selection using Q-learning, in: IEEE\/RSJ International Conference on Intelligent Robots and Systems, Lausanne, Switzerland, 2002","DOI":"10.21236\/ADA640010"},{"key":"10.1016\/j.robot.2007.03.003_b13","doi-asserted-by":"crossref","unstructured":"B. Lee, R. Arkin, Adaptive multi-robot behavior via learning momentum, in: IEEE\/RSJ International Conference on Intelligent Robots and Systems, Las Vegas, USA, 2003","DOI":"10.21236\/ADA443160"},{"key":"10.1016\/j.robot.2007.03.003_b14","unstructured":"W. Uther, M. Veloso, Tree based discretization for continuous state space reinforcement learning, in: Proceedings of the Fifteenth National Conference on Artificial Intelligence, 1998"},{"key":"10.1016\/j.robot.2007.03.003_b15","doi-asserted-by":"crossref","first-page":"163","DOI":"10.1177\/105971239700600201","article-title":"Experiments with reinforcement learning in problems with continuous state and action spaces","volume":"6","author":"Santamaria","year":"1998","journal-title":"Adaptive Behavior"},{"key":"10.1016\/j.robot.2007.03.003_b16","unstructured":"W. Smart, Making reinforcement learning work on real robots, Ph.D. Thesis, Department of Computer Science at Brown University, Rhode Island, May 2002"},{"key":"10.1016\/j.robot.2007.03.003_b17","unstructured":"C. Gaskett, Q-learning for robot control, Ph.D. Thesis, Australian National University, 2002"},{"key":"10.1016\/j.robot.2007.03.003_b18","doi-asserted-by":"crossref","first-page":"279","DOI":"10.1007\/BF00992698","article-title":"Q-learning","volume":"8","author":"Watkins","year":"1992","journal-title":"Machine Learning"},{"key":"10.1016\/j.robot.2007.03.003_b19","doi-asserted-by":"crossref","unstructured":"L. Baird, Residual algorithms: Reinforcement learning with function approximation, in: Machine Learning: Twelfth International Conference, San Francisco, USA, 1995","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"issue":"3\u20134","key":"10.1016\/j.robot.2007.03.003_b20","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1007\/BF00992697","article-title":"Practical issues in temporal difference learning","volume":"8","author":"Tesauro","year":"1992","journal-title":"Machine Learning"},{"key":"10.1016\/j.robot.2007.03.003_b21","doi-asserted-by":"crossref","unstructured":"S. Weaver, L. Baird, M. Polycarpou, An analytical framework for local feedforward networks, IEEE Transactions on Neural Networks 9 (3)","DOI":"10.1109\/72.668889"},{"issue":"3\u20134","key":"10.1016\/j.robot.2007.03.003_b22","doi-asserted-by":"crossref","first-page":"293","DOI":"10.1007\/BF00992699","article-title":"Self-improving reactive agents based on reinforcement learning, planning and teaching","volume":"8","author":"Lin","year":"1992","journal-title":"Machine Learning"},{"key":"10.1016\/j.robot.2007.03.003_b23","doi-asserted-by":"crossref","unstructured":"C. Gaskett, D. Wettergreen, A. Zelinsky, Q-learning in continuous state and action spaces, in: Proceedings of the 12th Australian Joint Conference on Artificial Intelligence, Sydney, Australia, 1999","DOI":"10.1007\/3-540-46695-9_35"},{"key":"10.1016\/j.robot.2007.03.003_b24","doi-asserted-by":"crossref","unstructured":"R. Sutton, Integrated architectures for learning, planning, and reacting based on approximating dynamic programming, in: M. Kaufmann (Ed.), Proceedings of the Seventh International Conference on Machine Learning, 1990","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"10.1016\/j.robot.2007.03.003_b25","doi-asserted-by":"crossref","unstructured":"A. Moore, Variable resolution dynamic programming: Efficiently learning action maps on multivariate real-value state-spaces, in: Proceedings of the Eighth International Conference on Machine Learning, 1991","DOI":"10.1016\/B978-1-55860-200-7.50069-6"},{"key":"10.1016\/j.robot.2007.03.003_b26","unstructured":"M. Carreras, J. Batlle, P. Ridao, Hybrid coordination of reinforcement learning-based behaviors for auv control, in: IEEE\/RSJ International Conference on Intelligent Robots and Systems, Hawaii, USA, 2001"},{"year":"1999","series-title":"Neural Networks, a Comprehensive Foundation","author":"Haykin","key":"10.1016\/j.robot.2007.03.003_b27"},{"key":"10.1016\/j.robot.2007.03.003_b28","unstructured":"L. Pyeatt, A. Howe, Decision tree function approximation in reinforcement learning, Tech. Rep. TR CS-98-112, Colorado State University, 1998"},{"key":"10.1016\/j.robot.2007.03.003_b29","doi-asserted-by":"crossref","first-page":"123","DOI":"10.1007\/BF00114726","article-title":"Reinforcement learning with replacing eligibility traces","volume":"22","author":"Singh","year":"1996","journal-title":"Machine Learning"},{"key":"10.1016\/j.robot.2007.03.003_b30","unstructured":"H. Vollbrecht, kd-Q-learning with hierarchic generalization in state space, Tech. Rep. SFB 527, Department of Neural Information Processing, University of Ulm, Germany, 1999"},{"key":"10.1016\/j.robot.2007.03.003_b31","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1023\/A:1017992615625","article-title":"Variable resolution discretization in optimal control","volume":"49","author":"Munos","year":"2002","journal-title":"Machine Learning"},{"key":"10.1016\/j.robot.2007.03.003_b32","first-page":"1038","article-title":"Generalization in reinforcement learning: Successful examples using sparce coarse coding","volume":"9","author":"Sutton","year":"1996","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.robot.2007.03.003_b33","doi-asserted-by":"crossref","unstructured":"R. Kretchmar, C. Anderson, Comparison of cmacs and radial basis functions for local function approximators in reinforcement learning, in: Proceedings of the IEEE International Conference on Neural Networks, Houston, TX, 1997, pp. 834\u2013837","DOI":"10.1109\/ICNN.1997.616132"},{"key":"10.1016\/j.robot.2007.03.003_b34","unstructured":"J. Boyan, A. Moore, Generalization in reinforcement learning: Safely approximating the value function, in: NIPS-7, San Mateo, CA, USA, 1995"}],"container-title":["Robotics and Autonomous Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0921889007000498?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0921889007000498?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,1,15]],"date-time":"2025-01-15T21:46:08Z","timestamp":1736977568000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0921889007000498"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2007,8]]},"references-count":34,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2007,8]]}},"alternative-id":["S0921889007000498"],"URL":"https:\/\/doi.org\/10.1016\/j.robot.2007.03.003","relation":{},"ISSN":["0921-8890"],"issn-type":[{"type":"print","value":"0921-8890"}],"subject":[],"published":{"date-parts":[[2007,8]]}}}