Investigation and Imitation of Human Captains’ Maneuver Using Inverse Reinforcement Learning | Research Square window.SnipcartSettings = { analytics: { enabled: false } }; (function() { var accessVector = localStorage.getItem('access_vector') || ''; window.dataLayer = window.dataLayer || []; if (accessVector) { window.dataLayer.push({ user: { profile: { profileInfo: { snid: accessVector } } } }); } })(); (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src='https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-K279D39R'); Browse Preprints In Review Journals COVID-19 Preprints AJE Video Bytes Research Tools Research Promotion AJE Professional Editing AJE Rubriq About Preprint Platform In Review Editorial Policies Our Team Advisory Board Help Center Sign In Submit a Preprint Cite Share Download PDF Research Article Investigation and Imitation of Human Captains’ Maneuver Using Inverse Reinforcement Learning Takefumi Higaki, Hirotada Hashimoto, Hitoshi Yoshioka This is a preprint; it has not been peer reviewed by a journal. https://doi.org/ 10.21203/rs.3.rs-1844861/v1 This work is licensed under a CC BY 4.0 License Status: Published Journal Publication published 30 Jan, 2023 Read the published version in Journal of the Japan Society of Naval Architects and Ocean Engineers → Version 1 posted You are reading this latest preprint version Abstract Automatic collision avoidance is of significant importance to prevent maritime collisions. Although many studies have been conducted in recent years, autonomous system has not completely replaced human captains since it is still difficult to imitate their complicated decisions. Thus, the present paper tries to investigate and imitate experienced captains’ maneuver using maximum entropy inverse reinforcement learning (MaxEnt IRL). We firstly verify that MaxEnt IRL can reproduce appropriate reward function from demonstrative trajectories. Afterwards, we conduct an experiment on a simulator where well-experienced captains maneuver in congested sea and estimate reward from the trajectories. Searching the route which maximizes the obtained reward, finally, we demonstrate the optimized route can avoid collision against multiple ships in compliance with the International Regulations for Preventing Collisions at Sea (COLREGs). Collision Avoidance Imitation Learning Inverse Reinforcement Learning Dangerous Area of Collision COLREGs Figures Figure 1 Figure 2 Figure 3 Figure 4 Figure 5 Figure 6 Figure 7 Figure 8 Figure 9 Figure 10 Figure 11 Figure 12 Figure 13 Full Text Declarations Competing interests: The authors declare no competing interests. Cite Share Download PDF Status: Published Journal Publication published 30 Jan, 2023 Read the published version in Journal of the Japan Society of Naval Architects and Ocean Engineers → Version 1 posted You are reading this latest preprint version Research Square lets you share your work early, gain feedback from the community, and start making changes to your manuscript prior to peer review in a journal. As a division of Research Square Company, we’re committed to making research communication faster, fairer, and more useful. We do this by developing innovative software and high quality services for the global research community. Our growing team is made up of researchers and industry professionals working together to solve the most critical problems facing scientific publishing. Also discoverable on Platform About Our Team In Review Editorial Policies Advisory Board Help Center Resources Author Services Accessibility API Access RSS feed Manage Cookie Preferences © Research Square 2026 | ISSN 2693-5015 (online) Privacy Policy Terms of Service Do Not Sell My Personal Information {"props":{"pageProps":{"initialData":{"identity":"rs-1844861","acceptedTermsAndConditions":true,"allowDirectSubmit":true,"archivedVersions":[],"articleType":"Research Article","associatedPublications":[],"authors":[{"id":138756933,"identity":"4e6ff4aa-c4f3-4279-9f56-16eb5fe8bf1a","order_by":0,"name":"Takefumi Higaki","email":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAZAAAAAyAQMAAABI0h/eAAAABlBMVEX///8AAABVwtN+AAAACXBIWXMAAA7EAAAOxAGVKw4bAAABHklEQVRIie2RvWrDMBRGrxG4ixOv8lK/gozAXUyeJUbQbqXQJZAhNgFNgawNlD5DoC8gI1AWt1m9Jd47eK+HykkK/VEaupWiswndw6fvCsBi+YvIIhMNwef+4UwAifc7YTRAsby+GyU0yNBBcYenlKsp9crLlIjvipm+cjjucTmk67WAph1Qn3sEeTAIwXkyxgSdEnB5HVcMnAVnMVY7hUUZPBvzyCbnOOLyNq4QoF6GEtjMLl49QHq8JEalS0m5TB/nEpDXTpJwnzL5SZkSoesvgWnFlTHZK/KoorvkdaaXjCtGigVf0Ui5N849WUX8SJe+OtvKtvvKeVFvm3YcPSi0hJfROPSxeWOf+DChn+Ti8qTxFX/2a8VisVj+JW/41mJi+gduqAAAAABJRU5ErkJggg==","orcid":"https://orcid.org/0000-0003-4838-4253","institution":"Osaka Prefecture University: Osaka Koritsu Daigaku - Nakamozu Campus","correspondingAuthor":true,"submittingAuthor":false,"prefix":"","firstName":"Takefumi","middleName":"","lastName":"Higaki","suffix":""},{"id":138756934,"identity":"2176a398-437e-47b0-83f4-4bc298fb8506","order_by":1,"name":"Hirotada Hashimoto","email":"","orcid":"","institution":"Osaka Metropolitan University: Osaka Koritsu Daigaku","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Hirotada","middleName":"","lastName":"Hashimoto","suffix":""},{"id":138756935,"identity":"257c2be7-ea36-43b1-9767-c4ac2002261e","order_by":2,"name":"Hitoshi Yoshioka","email":"","orcid":"","institution":"Osaka Prefecture University: Osaka Koritsu Daigaku - Nakamozu Campus","correspondingAuthor":false,"submittingAuthor":false,"prefix":"","firstName":"Hitoshi","middleName":"","lastName":"Yoshioka","suffix":""}],"badges":[],"createdAt":"2022-07-11 02:22:40","currentVersionCode":1,"declarations":"","doi":"10.21203/rs.3.rs-1844861/v1","doiUrl":"https://doi.org/10.21203/rs.3.rs-1844861/v1","draftVersion":[],"editorialEvents":[{"content":"https://doi.org/10.2534/jjasnaoe.36.137","type":"published","date":"2023-01-31T00:00:00+00:00"}],"editorialNote":"","failedWorkflow":false,"files":[{"id":26830142,"identity":"7ff0000c-cc41-44b8-930d-08b8308c316e","added_by":"auto","created_at":"2022-09-22 15:17:33","extension":"png","order_by":1,"title":"Figure 1","display":"","copyAsset":false,"role":"figure","size":202599,"visible":true,"origin":"","legend":"\u003cp\u003eOverview of fundamental rules in each encounter situation: a) overtaking, b) head-on, c) crossing. A give-way vessel has responsibility to keep enough distance, while a stand-on vessel must keep its course and speed.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.11.53PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/b621bfe85374f0f30f95f9fb.png"},{"id":26830423,"identity":"f764afd5-458a-4105-8323-5e23c23ee617","added_by":"auto","created_at":"2022-09-22 15:22:34","extension":"png","order_by":2,"title":"Figure 2","display":"","copyAsset":false,"role":"figure","size":670634,"visible":true,"origin":"","legend":"\u003cp\u003eSee image above for figure legend.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.12.29PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/b1a02007c2168f625098041e.png"},{"id":26830147,"identity":"ae958507-1907-472a-bad7-a1a527bb9226","added_by":"auto","created_at":"2022-09-22 15:17:34","extension":"png","order_by":3,"title":"Figure 3","display":"","copyAsset":false,"role":"figure","size":413825,"visible":true,"origin":"","legend":"\u003cp\u003eSchematic of how to generate a sample trajectory (the dashed black line in this figure). Combining lines which goes straight or turns in a certain angle, we select the shortest path so as not to overlap the DAC.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.12.48PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/11f252963e7bf80682e35a9b.png"},{"id":26830143,"identity":"af3c8e41-d2f4-4913-b4af-1ca7afc99e24","added_by":"auto","created_at":"2022-09-22 15:17:34","extension":"png","order_by":4,"title":"Figure 4","display":"","copyAsset":false,"role":"figure","size":605061,"visible":true,"origin":"","legend":"\u003cp\u003eSee image above for figure legend.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.13.11PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/f4e709b9a88c999412d3ed19.png"},{"id":26830146,"identity":"9a4df729-c06d-4c4f-b247-38a7d9554666","added_by":"auto","created_at":"2022-09-22 15:17:34","extension":"png","order_by":5,"title":"Figure 5","display":"","copyAsset":false,"role":"figure","size":673896,"visible":true,"origin":"","legend":"\u003cp\u003eAreas divided by concentric circles around the own ship.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.13.24PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/587f88fc17ebbc03780c186a.png"},{"id":26830427,"identity":"5772ca90-973f-404a-9f04-1231ec98c4f2","added_by":"auto","created_at":"2022-09-22 15:22:34","extension":"png","order_by":6,"title":"Figure 6","display":"","copyAsset":false,"role":"figure","size":395902,"visible":true,"origin":"","legend":"\u003cp\u003eSchematic of Imazu problem. The light-gray and the dark-gray triangles represent the own ship and the other ships, respectively. The other ships basically have the same speed as the own ship except for the overtaking cases such as scenario No.3.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.13.34PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/af8c4cdfa9eeea8bb867b716.png"},{"id":26831091,"identity":"e75e1359-a3dc-4d7c-8fa0-e901403661c5","added_by":"auto","created_at":"2022-09-22 15:32:34","extension":"jpeg","order_by":7,"title":"Figure 7","display":"","copyAsset":false,"role":"figure","size":84146,"visible":true,"origin":"","legend":"\u003cp\u003eDangerous area of collision (DAC) at 22 scenarios of Imazu problem. The pictures at the top row show the scenario No. 1~4, the second row No. 5~8, …, and the last row No. 21~22. The light-gray and the dark-gray triangles represent the own ship and the other ships, respectively. Axes unit: [km].\u003c/p\u003e","description":"","filename":"groupimage1.jpeg","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/59f925fc0e2be5786ed436f4.jpeg"},{"id":26830806,"identity":"a1904c47-dc49-4eda-9dc0-8d4a3a539d6d","added_by":"auto","created_at":"2022-09-22 15:27:34","extension":"jpeg","order_by":8,"title":"Figure 8","display":"","copyAsset":false,"role":"figure","size":149707,"visible":true,"origin":"","legend":"\u003cp\u003eReward map at 22 scenarios of Imazu problem. The pictures at the top row show the scenario No. 1~4, the second row No. 5~8, …, and the last row No. 21~22. The light-gray and the dark-gray triangles represent the own ship and the other ships, respectively. The dashed lines denote the future routes of the other ships, assuming they does not change their course or speed. Reward at all states was calculated using MaxEnt IRL and then normalized within [-1, 1]. Axes unit: [km].\u003c/p\u003e","description":"","filename":"groupimage2.jpeg","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/e4eda1cdc2f274df7535fc4f.jpeg"},{"id":26830421,"identity":"6176c90e-b1aa-439d-8e81-c089034b52f9","added_by":"auto","created_at":"2022-09-22 15:22:34","extension":"jpeg","order_by":9,"title":"Figure 9","display":"","copyAsset":false,"role":"figure","size":38608,"visible":true,"origin":"","legend":"\u003cp\u003eTrajectories in the three experiment cases on the maneuvering simulator. The red lines show the own ship’s trajectories, and the black lines the other ships’ trajectories. Axes unit: [km].\u003c/p\u003e","description":"","filename":"groupimage3.jpeg","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/ef56fbb4cf080df3f51009e3.jpeg"},{"id":26830424,"identity":"be7bd866-24b6-4df5-8e94-8dbd295e7138","added_by":"auto","created_at":"2022-09-22 15:22:34","extension":"png","order_by":10,"title":"Figure 10","display":"","copyAsset":false,"role":"figure","size":939830,"visible":true,"origin":"","legend":"\u003cp\u003eSee image above for figure legend.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.14.31PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/c0e5b967ce6af60b4be07e97.png"},{"id":26830148,"identity":"fd57fed0-94d2-4b1b-ad82-fb419acfa352","added_by":"auto","created_at":"2022-09-22 15:17:34","extension":"jpeg","order_by":11,"title":"Figure 11","display":"","copyAsset":false,"role":"figure","size":139666,"visible":true,"origin":"","legend":"\u003cp\u003eReward map and optimized routes at 22 scenarios of Imazu problem using the pseudo-expert data based on DAC. The pictures at the top row show the scenario No. 1~4, the second row No. 5~8, …, and the last row No. 21~22. The light-gray and the dark-gray triangles represent the initial positions of the own ship and the other ships, respectively. The black straight lines denote the future trajectories of the other ships; the pink lines denote the optimized routes which maximize the obtained reward. Those lines are marked by ticks at equal time intervals. Axes unit: [km].\u003c/p\u003e","description":"","filename":"groupimage4.jpeg","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/75a763d0170bceb8d3fdab56.jpeg"},{"id":26830807,"identity":"72377db3-0186-4ba7-af4b-2dc063f01f85","added_by":"auto","created_at":"2022-09-22 15:27:34","extension":"jpeg","order_by":12,"title":"Figure 12","display":"","copyAsset":false,"role":"figure","size":142001,"visible":true,"origin":"","legend":"\u003cp\u003eReward map and optimized routes at 22 scenarios of Imazu problem using the expert data. The pictures at the top row show the scenario No. 1~4, the second row No. 5~8, …, and the last row No. 21~22. The light-gray and the dark-gray triangles represent the initial positions of the own ship and the other ships, respectively. The black straight lines denote the future trajectories of the other ships; the pink lines denote the optimized routes which maximize the obtained reward. Those lines are marked by ticks at equal time intervals. Axes unit: [km].\u003c/p\u003e","description":"","filename":"groupimage5.jpeg","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/d0edcd8552ed86536ce76291.jpeg"},{"id":26830152,"identity":"528cf2a6-261b-46ca-9363-79fe31643357","added_by":"auto","created_at":"2022-09-22 15:17:34","extension":"png","order_by":13,"title":"Figure 13","display":"","copyAsset":false,"role":"figure","size":645162,"visible":true,"origin":"","legend":"\u003cp\u003eSee image above for figure legend.\u003c/p\u003e","description":"","filename":"ScreenShot20220921at10.15.27PM.png","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1/e2ed4e6030f6bd0bba2c5194.png"},{"id":26886440,"identity":"1072c46a-7e9a-4797-9f24-514861efb1f8","added_by":"auto","created_at":"2022-09-23 15:16:52","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1086163,"visible":true,"origin":"","legend":"","description":"","filename":"20220711jmstscripthigaki.pdf","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1_covered.pdf"},{"id":26831136,"identity":"e294c261-1ad2-4876-96af-2fe1b52d16d5","added_by":"auto","created_at":"2022-09-22 15:32:49","extension":"pdf","order_by":1,"title":"","display":"","copyAsset":false,"role":"manuscript-pdf","size":1086414,"visible":true,"origin":"","legend":"","description":"","filename":"20220711jmstscripthigaki.pdf","url":"https://assets-eu.researchsquare.com/files/rs-1844861/v1_covered.pdf"}],"financialInterests":"","formattedTitle":"Investigation and Imitation of Human Captains’ Maneuver Using Inverse Reinforcement Learning","fulltext":[{"header":"Full Text","content":"This preprint is available for \u003ca href='/article/rs-1844861/latest.pdf' target='_blank'\u003edownload as a PDF\u003c/a\u003e."},{"header":"Declarations","content":"\u003cp\u003eCompeting interests: The authors declare no competing interests.\u003c/p\u003e"}],"fulltextSource":"","fullText":"","funders":[],"hasAdminPriorityOnWorkflow":false,"hasManuscriptDocX":false,"hasOptedInToPreprint":true,"hasPassedJournalQc":"","hasAnyPriority":true,"hideJournal":true,"highlight":"","institution":"","isAcceptedByJournal":true,"isAuthorSuppliedPdf":true,"isDeskRejected":"","isHiddenFromSearch":false,"isInQc":false,"isInWorkflow":false,"isPdf":false,"isPdfUpToDate":true,"isWithdrawnOrRetracted":false,"journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true},"keywords":"Collision Avoidance, Imitation Learning, Inverse Reinforcement Learning, Dangerous Area of Collision, COLREGs","lastPublishedDoi":"10.21203/rs.3.rs-1844861/v1","lastPublishedDoiUrl":"https://doi.org/10.21203/rs.3.rs-1844861/v1","license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"manuscriptAbstract":"Automatic collision avoidance is of significant importance to prevent maritime collisions. Although many studies have been conducted in recent years, autonomous system has not completely replaced human captains since it is still difficult to imitate their complicated decisions. Thus, the present paper tries to investigate and imitate experienced captains’ maneuver using maximum entropy inverse reinforcement learning (MaxEnt IRL). We firstly verify that MaxEnt IRL can reproduce appropriate reward function from demonstrative trajectories. Afterwards, we conduct an experiment on a simulator where well-experienced captains maneuver in congested sea and estimate reward from the trajectories. Searching the route which maximizes the obtained reward, finally, we demonstrate the optimized route can avoid collision against multiple ships in compliance with the International Regulations for Preventing Collisions at Sea (COLREGs).","manuscriptTitle":"Investigation and Imitation of Human Captains’ Maneuver Using Inverse Reinforcement Learning","msid":"","msnumber":"","nonDraftVersions":[{"code":1,"date":"2022-09-22 15:17:32","doi":"10.21203/rs.3.rs-1844861/v1","editorialEvents":[{"type":"communityComments","content":0}],"status":"published","journal":{"display":true,"email":"
[email protected]","identity":"researchsquare","isNatureJournal":false,"hasQc":true,"allowDirectSubmit":true,"externalIdentity":"","sideBox":"","snPcode":"","submissionUrl":"/submission","title":"Research Square","twitterHandle":"researchsquare","acdcEnabled":true,"dfaEnabled":false,"editorialSystem":"","reportingPortfolio":"","inReviewEnabled":false,"inReviewRevisionsEnabled":true}}],"origin":"","ownerIdentity":"f53d5c92-a44e-4b65-a40d-1c72962ede66","owner":[],"postedDate":"September 22nd, 2022","published":true,"recentEditorialEvents":[],"rejectedJournal":[],"revision":"","amendment":"","status":"posted","subjectAreas":[],"tags":[],"updatedAt":"2023-02-02T19:46:15+00:00","versionOfRecord":{"articleIdentity":"rs-1844861","link":"https://doi.org/10.2534/jjasnaoe.36.137","journal":{"identity":"journal-of-the-japan-society-of-naval-architects-and-ocean-engineers","isVorOnly":true,"title":"Journal of the Japan Society of Naval Architects and Ocean Engineers"},"publishedOn":"2023-01-31 00:00:00","publishedOnDateReadable":"January 31st, 2023"},"versionCreatedAt":"2022-09-22 15:17:32","video":"","vorDoi":"10.2534/jjasnaoe.36.137","vorDoiUrl":"https://doi.org/10.2534/jjasnaoe.36.137","workflowStages":[]},"version":"v1","identity":"rs-1844861","journalConfig":"researchsquare"},"__N_SSP":true},"page":"/article/[identity]/[[...version]]","query":{"redirect":"/article/rs-1844861","identity":"rs-1844861","version":["v1"]},"buildId":"rHA-KDH7Qsr4HCuvH75dn","isFallback":false,"isExperimentalCompile":false,"dynamicIds":[84888],"gssp":true,"scriptLoader":[]}
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.