Speeding up hierarchical reinforcement learning using state-independent temporal skills | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article Speeding up hierarchical reinforcement learning using state-independent temporal skills Leila Azadkhah, Omid Davoodi, Mohammad Ghazanfari, Nasser Mozayani This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-4678044/v1 This work is licensed under a CC BY 4.0 License Status: Under Review Version 1 posted 3 You are reading this latest preprint version Abstract Hierarchical reinforcement learning has the potential to expedite long-term decision-making by abstracting policies into multiple levels. Encouraging outcomes have been observed in challenging reward environments through the use of skills, defined as sequences of basic actions. While existing methods based on offline data have shown promise, the resulting lower-level policies may suffer from unreliability due to limited demonstration coverage or shifts in distribution. To address this limitation, we propose a novel approach to autonomously identify state-independent temporal skills (SITS) by extracting the most repetitive action sequences from a trained agent. These skills, acquired in a simpler source task, can then be transferred to a more complex target task to enhance the agent’s training efficiency, particularly during the exploration phase. Our method is independent of other hierarchical reinforcement learning techniques and can be used in conjunction with them. Experimental results demonstrate the efficacy of incorporating SITS in addressing complex RL challenges. Reinforcement Learning Hierarchical Reinforcement Learning Skill Discovery State-independent Skills Transfer Learning Full Text Additional Declarations No competing interests reported. Cite Share Download PDF Status: Under Review Version 1 posted Editor assigned by journal 04 Jul, 2024 Submission checks completed at journal 04 Jul, 2024 First submitted to journal 03 Jul, 2024 You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-4678044","acceptedTermsAndConditions":true,"allowDirectSubmit":false,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":326156413,"identity":"e094244d-b4de-49e2-bd67-476e7b75504c","order_by":0,"name":"Leila Azadkhah","email":"","orcid":"","institution":"Iran University of Science and Technology","correspondingAuthor":false,"prefix":"","firstName":"Leila","middleName":"","lastName":"Azadkhah","suffix":""},{"id":326156414,"identity":"5e885a5c-abc1-495e-957c-c2451c1ae369","order_by":1,"name":"Omid Davoodi","email":"","orcid":"","institution":"Carleton University","correspondingAuthor":false,"prefix":"","firstName":"Omid","middleName":"","lastName":"Davoodi","suffix":""},{"id":326156415,"identity":"e1b115bc-3de8-4474-9cad-6e750e1bd1f7","order_by":2,"name":"Mohammad Ghazanfari","email":"","orcid":"","institution":"Iran University of Science and Technology","correspondingAuthor":false,"prefix":"","firstName":"Mohammad","middleName":"","lastName":"Ghazanfari","suffix":""},{"id":326156416,"identity":"e3584c2f-3e6b-4c18-aa5e-2406c3d0a62d","order_by":3,"name":"Nasser Mozayani","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAAA10lEQVRIiWNgGAWjYBADHn5mGJMZnzpkLZLNQPIAKVoYDA7AtBACuu3tDz/z1NyRMT7OY/z5A4OdPAM77wO8WszOHEiW5jn2jMfsMI+ZxAGGZMMGZnYD/FpuJByQ5mE7DNYCdBhzAgMzG36Hmd1IbP7N8+8wj3Ezj/GHAwz1xGhJZpPmbTvMY8DMYwB02GEitJw5xmY5t+8wj8RhtjKJMwbHDdsIajne/vjGm2+H7fn7D2/+UFFRLc/Pfwy/FhBg4oEzgWFFwA4IYPxBjKpRMApGwSgYuQAAK7M7rrV24LAAAAAASUVORK5CYII=","orcid":"","institution":"Iran University of Science and Technology","correspondingAuthor":true,"prefix":"","firstName":"Nasser","middleName":"","lastName":"Mozayani","suffix":""}],"badges":[],"createdAt":"2024-07-03 06:26:06","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-4678044/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-4678044/v1","draftVersion":[],"editorialEvents":[],"editorialNote":"","failedWorkflow":false,"files":[{"id":61338160,"identity":"bdd0a701-5436-4165-b969-60001f5c39b2","added_by":"auto","created_at":"2024-07-29 16:04:11","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":699491,"visible":true,"origin":"","legend":"","description":"","filename":"SITSIntelligentManufaturing.pdf","url":"https://assets-eu.researchsquare.com/files/rs-4678044/v1_covered_8450d6d5-8b7d-4a88-b156-71d8d33faf26.pdf"}],"financialInterests":"No competing interests reported.","formattedTitle":"Speeding up hierarchical reinforcement learning using state-independent temporal skills","fulltext":[],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":false,"hideJournal":false,"highlight":"","institution":"","isAcceptedByJournal":false,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":true,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"the-journal-of-supercomputing","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"","sideBox":"Learn more about [The Journal of Supercomputing](https://www.springer.com/journal/11227)","snPcode":"11227","submissionUrl":"https://submission.nature.com/new-submission/11227/3","title":"The Journal of Supercomputing","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"stoa","reportingPortfolio":"Springer Hybrid","inReviewEnabled":true,"inReviewRevisionsEnabled":false},"keywords":"Reinforcement Learning, Hierarchical Reinforcement Learning, Skill Discovery, State-independent Skills, Transfer Learning","lastPublishedDoi":"10.21203/rs.3.rs-4678044/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-4678044/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"Hierarchical reinforcement learning has the potential to expedite long-term decision-making by abstracting policies into multiple levels. Encouraging outcomes have been observed in challenging reward environments through the use of skills, defined as sequences of basic actions. While existing methods based on offline data have shown promise, the resulting lower-level policies may suffer from unreliability due to limited demonstration coverage or shifts in distribution. To address this limitation, we propose a novel approach to autonomously identify state-independent temporal skills (SITS) by extracting the most repetitive action sequences from a trained agent. These skills, acquired in a simpler source task, can then be transferred to a more complex target task to enhance the agent’s training efficiency, particularly during the exploration phase. Our method is independent of other hierarchical reinforcement learning techniques and can be used in conjunction with them. Experimental results demonstrate the efficacy of incorporating SITS in addressing complex RL challenges.","manuscriptTitle":"Speeding up hierarchical reinforcement learning using state-independent temporal skills","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2024-07-29 15:56:04","doi":"10.21203/rs.3.rs-4678044/v1","editorialEvents":[{"type":"communityComments","content":0},{"type":"editorAssigned","content":"","date":"2024-07-04T12:41:25+00:00","index":"","fulltext":""},{"type":"checksComplete","content":"","date":"2024-07-04T12:40:02+00:00","index":"","fulltext":""},{"type":"submitted","content":"The Journal of Supercomputing","date":"2024-07-03T06:24:40+00:00","index":"","fulltext":""}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"the-journal-of-supercomputing","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":false,"externalIdentity":"","sideBox":"Learn more about [The Journal of Supercomputing](https://www.springer.com/journal/11227)","snPcode":"11227","submissionUrl":"https://submission.nature.com/new-submission/11227/3","title":"The Journal of Supercomputing","twitterHandle":"","acdcEnabled":true,"dfaEnabled":true,"editorialSystem":"stoa","reportingPortfolio":"Springer Hybrid","inReviewEnabled":true,"inReviewRevisionsEnabled":false}}],"origin":"","ownerIdentity":"7a078c62-3563-4821-ac46-6719c0531346","owner":[],"postedDate":"July 29th, 2024","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"under-review","subjectAreas":[],"tags":[],"updatedAt":"2024-07-29T15:56:04+00:00","versionOfRecord":[],"versionCreatedAt":"2024-07-29 15:56:04","video":"","vorDoi":"","vorDoiUrl":"","workflowStages":[]},"version":"v1","identity":"rs-4678044","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-4678044","identity":"rs-4678044","version":["v1"]},"buildId":"qtupq5eGEP_6zYnWcrvyt","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.