{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,23]],"date-time":"2026-02-23T20:57:42Z","timestamp":1771880262145,"version":"3.50.1"},"reference-count":32,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.neunet.2025.108453","type":"journal-article","created":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T03:00:51Z","timestamp":1765422051000},"page":"108453","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Towards more effective skill discovery in reinforcement learning by incorporating state reachability"],"prefix":"10.1016","volume":"197","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5971-1040","authenticated-orcid":false,"given":"Yang","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0905-0816","authenticated-orcid":false,"given":"Jingchen","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1287-8391","authenticated-orcid":false,"given":"Huarui","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0928-8332","authenticated-orcid":false,"given":"Ying","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6314-7661","authenticated-orcid":false,"given":"Chunjiang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2025.108453_bib0001","series-title":"Proceedings of the 31st international conference on neural information processing systems","first-page":"5055","article-title":"Hindsight experience replay","author":"Andrychowicz","year":"2017"},{"issue":"4","key":"10.1016\/j.neunet.2025.108453_bib0002","doi-asserted-by":"crossref","first-page":"1247","DOI":"10.1007\/s40815-023-01664-1","article-title":"Improved event-triggered-based output tracking for a class of delayed networked t-s fuzzy systems","volume":"26","author":"Aslam","year":"2024","journal-title":"International Journal of Fuzzy Systems"},{"key":"10.1016\/j.neunet.2025.108453_bib0003","series-title":"Proceedings of the 37th international conference on machine learning","first-page":"1317","article-title":"Explore, discover and learn: Unsupervised discovery of state-covering skills","author":"\u0107tor Campos","year":"2020"},{"issue":"4","key":"10.1016\/j.neunet.2025.108453_bib0004","doi-asserted-by":"crossref","first-page":"373","DOI":"10.2478\/jaiscr-2024-0020","article-title":"Exponential state estimation for delayed competitive neural network via stochastic sampled-data control with markov jump parameters under actuator failure","volume":"14","author":"Cao","year":"2024","journal-title":"Journal of Artificial Intelligence and Soft Computing Research"},{"issue":"3","key":"10.1016\/j.neunet.2025.108453_bib0005","doi-asserted-by":"crossref","first-page":"7455","DOI":"10.1109\/LRA.2022.3171915","article-title":"Unsupervised reinforcement learning for transferable manipulation skill discovery","volume":"7","author":"Cho","year":"2022","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10.1016\/j.neunet.2025.108453_bib0006","series-title":"Proceedings of the genetic and evolutionary computation conference","first-page":"81","article-title":"Autonomous skill discovery with quality-diversity and unsupervised descriptors","author":"Cully","year":"2019"},{"key":"10.1016\/j.neunet.2025.108453_bib0007","unstructured":"Ernst, D., Feuerriegel, A. L., Hartmann, S., Janiesch, J., Zschech, C., & P (2024). Introduction to reinforcement learning."},{"key":"10.1016\/j.neunet.2025.108453_bib0008","series-title":"7th international conference on learning representations","article-title":"Diversity is all you need: Learning skills without a reward function","author":"Eysenbach","year":"2019"},{"key":"10.1016\/j.neunet.2025.108453_bib0009","series-title":"2020 28th European signal processing conference (EUSIPCO","first-page":"1517","article-title":"Neural discrete abstraction of high-dimensional spaces: A case study in reinforcement learning","author":"Giannakopoulos","year":"2021"},{"issue":"6","key":"10.1016\/j.neunet.2025.108453_bib0010","doi-asserted-by":"crossref","first-page":"2451","DOI":"10.1214\/aos\/1030741081","article-title":"Mutual information, metric entropy and cumulative relative entropy risk","volume":"25","author":"Haussler","year":"1997","journal-title":"The Annals of Statistics"},{"key":"10.1016\/j.neunet.2025.108453_bib0011","doi-asserted-by":"crossref","unstructured":"He, Z., Song, C., Li, J., & Shi, H. (2024). PDRL: Towards deeper states and further behaviors in unsupervised skill discovery by progressive diversity.","DOI":"10.1109\/TCDS.2024.3471645"},{"key":"10.1016\/j.neunet.2025.108453_bib0012","doi-asserted-by":"crossref","unstructured":"Hu, J., Wang, Z., Stone, P., & \u0144 Mart\u0131 \u0144, R. M. (2024). Disentangled unsupervised skill discovery for efficient hierarchical reinforcement learning. In Conference on Neural Information Processing Systems (NeurIPS). 37, (pp. 76529\u201376552).","DOI":"10.52202\/079017-2437"},{"key":"10.1016\/j.neunet.2025.108453_bib0013","doi-asserted-by":"crossref","first-page":"301","DOI":"10.1609\/icaps.v34i1.31488","article-title":"Rethinking mutual information for language conditioned skill discovery on imitation learning","volume":"34","author":"Ju","year":"2024","journal-title":"Proceedings of the International Conference on Automated Planning and Scheduling"},{"key":"10.1016\/j.neunet.2025.108453_bib0014","series-title":"Proceedings of the 37th international conference on neural information processing systems","first-page":"28226","article-title":"Learning to discover skills through guidance","author":"Kim","year":"2023"},{"issue":"1","key":"10.1016\/j.neunet.2025.108453_bib0015","doi-asserted-by":"crossref","first-page":"77","DOI":"10.1109\/TCYB.2014.2319733","article-title":"Stochastic abstract policies: Generalizing knowledge to improve reinforcement learning","volume":"45","author":"Koga","year":"2014","journal-title":"IEEE Transactions on Cybernetics"},{"key":"10.1016\/j.neunet.2025.108453_bib0016","series-title":"Proceedings of the 22nd international conference on neural information processing systems","first-page":"1015","article-title":"Skill discovery in continuous reinforcement learning domains using skill chaining","author":"Konidaris","year":"2009"},{"key":"10.1016\/j.neunet.2025.108453_bib0017","first-page":"34478","article-title":"Xue bin peng, denis yarats, aravind rajeswaran, and pieter abbeel. unsupervised reinforcement learning with contrastive intrinsic control","volume":"35","author":"Laskin","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2025.108453_bib0018","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1613\/jair.3175","article-title":"Non-deterministic policies in markovian decision processes","volume":"40","author":"Milani","year":"2011","journal-title":"Journal of Artificial Intelligence Research"},{"key":"10.1016\/j.neunet.2025.108453_bib0019","series-title":"2019 Joint IEEE 9th international conference on development and learning and epigenetic robotics (ICDL-EPIROB)","first-page":"1","article-title":"Hindsight experience replay with experience ranking","author":"Nguyen","year":"2019"},{"issue":"3","key":"10.1016\/j.neunet.2025.108453_bib0020","doi-asserted-by":"crossref","first-page":"929","DOI":"10.1007\/s40745-024-00548-x","article-title":"A new kernel density estimation-based entropic isometric feature mapping for unsupervised metric learning","volume":"12","author":"Neto","year":"2024","journal-title":"Annals of Data Science"},{"key":"10.1016\/j.neunet.2025.108453_bib0021","unstructured":"Sharma, A., Gu, S., Levine, S., Kumar, V., & Hausman, K. (2019). Dynamics-aware unsupervised discovery of skills. Technical Report arXiv preprint."},{"key":"10.1016\/j.neunet.2025.108453_bib0022","series-title":"2014 IEEE\/RSJ international conference on intelligent robots and systems","first-page":"1408","article-title":"Simultaneous on-line discovery and improvement of robotic skill options","author":"Stulp","year":"2014"},{"key":"10.1016\/j.neunet.2025.108453_bib0023","series-title":"2012 IEEE\/RSJ international conference on intelligent robots and systems","first-page":"5026","article-title":"Mujoco: A physics engine for model-based control","author":"Todorov","year":"2012"},{"issue":"4","key":"10.1016\/j.neunet.2025.108453_bib0024","doi-asserted-by":"crossref","first-page":"5064","DOI":"10.1109\/TNNLS.2022.3207346","article-title":"Deep reinforcement learning: A survey","volume":"35","author":"Wang","year":"2022","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2025.108453_bib0025","series-title":"Proceedings of the 33rd international conference on neural information processing systems","first-page":"9668","article-title":"Towards optimal off-policy evaluation for reinforcement learning with marginalized importance sampling","author":"Xie","year":"2019"},{"key":"10.1016\/j.neunet.2025.108453_bib0026","series-title":"Conference on robot learning","first-page":"3536","article-title":"XSkill: Cross embodiment skill discovery","author":"Xu","year":"2023"},{"key":"10.1016\/j.neunet.2025.108453_bib0027","series-title":"International conference on machine learning","first-page":"39183","article-title":"Behavior contrastive learning for unsupervised skill discovery","author":"Yang","year":"2023"},{"key":"10.1016\/j.neunet.2025.108453_bib0028","series-title":"Proceedings of the 34th international conference on neural information processing systems","first-page":"13903","article-title":"On function approximation in reinforcement learning: optimism in the face of large state spaces","author":"Yang","year":"2020"},{"key":"10.1016\/j.neunet.2025.108453_bib0029","unstructured":"Yu, X., Dunion, M., Li, X., & Albrecht, S. V. (2024). Skill-aware mutual information optimisation for generalisation in reinforcement learning. Technical Report arXiv preprint."},{"key":"10.1016\/j.neunet.2025.108453_bib0030","series-title":"Proceedings of the 40th international conference on machine learning","first-page":"40531","article-title":"Automatic intrinsic reward shaping for exploration in deep reinforcement learning","author":"Yuan","year":"2023"},{"key":"10.1016\/j.neunet.2025.108453_bib0031","doi-asserted-by":"crossref","first-page":"10887","DOI":"10.1609\/aaai.v35i12.17300","article-title":"Sample efficient reinforcement learning with reinforce","volume":"35","author":"Zhang","year":"2021","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"10.1016\/j.neunet.2025.108453_bib0032","first-page":"26519","article-title":"On the effectiveness of fine-tuning versus meta-reinforcement learning","volume":"35","author":"Zhao","year":"2022","journal-title":"Advances in Neural Information Processing Systems"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608025013346?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608025013346?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,2,23]],"date-time":"2026-02-23T20:00:23Z","timestamp":1771876823000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608025013346"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":32,"alternative-id":["S0893608025013346"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2025.108453","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Towards more effective skill discovery in reinforcement learning by incorporating state reachability","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2025.108453","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108453"}}