{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T17:55:01Z","timestamp":1786038901713,"version":"3.56.0"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2020,10,1]],"date-time":"2020-10-01T00:00:00Z","timestamp":1601510400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,10,1]],"date-time":"2020-10-01T00:00:00Z","timestamp":1601510400000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,10,1]],"date-time":"2020-10-01T00:00:00Z","timestamp":1601510400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,10,1]],"date-time":"2020-10-01T00:00:00Z","timestamp":1601510400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1713209"],"award-info":[{"award-number":["U1713209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61520106009"],"award-info":[{"award-number":["61520106009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["ECCS 1526835"],"award-info":[{"award-number":["ECCS 1526835"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Syst. Man Cybern, Syst."],"published-print":{"date-parts":[[2020,10]]},"DOI":"10.1109\/tsmc.2018.2884725","type":"journal-article","created":{"date-parts":[[2019,1,3]],"date-time":"2019-01-03T19:44:25Z","timestamp":1546544665000},"page":"3713-3725","source":"Crossref","is-referenced-by-count":166,"title":["Deterministic Policy Gradient With Integral Compensator for Robust Quadrotor Control"],"prefix":"10.1109","volume":"50","author":[{"given":"Yuanda","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jia","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0002-3103-4452","authenticated-orcid":false,"given":"Haibo","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0001-9269-334X","authenticated-orcid":false,"given":"Changyin","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1","article-title":"Linear off-policy actor-critic","author":"degris","year":"2012","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553501"},{"key":"ref33","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","author":"sutton","year":"2000","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref32","first-page":"1038","article-title":"Generalization in reinforcement learning: Successful examples using sparse coarse coding","author":"sutton","year":"1996","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref30","first-page":"1531","article-title":"A natural policy gradient","author":"kakade","year":"2002","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1080\/00207177208932211"},{"key":"ref36","first-page":"2094","article-title":"Deep reinforcement learning with double Q-learning","volume":"16","author":"van hasselt","year":"2016","journal-title":"Proc AAAI"},{"key":"ref35","first-page":"2613","article-title":"Double Q-learning","author":"hasselt","year":"2010","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref34","article-title":"Reinforcement learning for robots using neural networks","author":"lin","year":"1993"},{"key":"ref28","first-page":"1","article-title":"A deep reinforcement learning strategy for UAV autonomous landing on a moving platform","author":"rodriguez-ramos","year":"2018","journal-title":"J Intell Robot Syst"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487175"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2017.2720851"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.compag.2013.09.008"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CYBER.2016.7574835"},{"key":"ref20","first-page":"1","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"2014","journal-title":"Proc ICML"},{"key":"ref22","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref21","first-page":"1","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2016","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2017.2785794"},{"key":"ref23","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017","journal-title":"arXiv 1707 06347v2 [cs LG]"},{"key":"ref26","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","author":"duan","year":"2016","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2018.2849942"},{"key":"ref10","doi-asserted-by":"crossref","first-page":"1924","DOI":"10.1109\/TCST.2012.2209887","article-title":"Robust adaptive attitude tracking on SO(3) with an application to a quadrotor UAV","volume":"21","author":"lee","year":"2013","journal-title":"IEEE Trans Control Syst Technol"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1049\/iet-cta.2011.0348"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2281663"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2012.2200104"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2018.2790929"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s10846-013-9909-4"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2014.2364982"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2018.2836922"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2005.1545025"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of Go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6385917"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2012.2206473"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.23919\/ECC.2007.7068316"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2004.1389776"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2006.377588"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/s12555-009-0311-8"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICCEREC.2017.8226676"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2017.2698473"},{"key":"ref46","first-page":"265","article-title":"TensorFlow: A system for large-scale machine learning","volume":"16","author":"abadi","year":"2016","journal-title":"Proc OSDI"},{"key":"ref45","year":"2018","journal-title":"Flying Evaluation"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2017.7510679"},{"key":"ref47","first-page":"1","article-title":"ADAM: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2016.2542923"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TMECH.2017.2675913"},{"key":"ref43","year":"2017","journal-title":"Qball 2 for QUARC Set Up and Configuration (User Manual)"}],"container-title":["IEEE Transactions on Systems, Man, and Cybernetics: Systems"],"original-title":[],"link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/ielam\/6221021\/9198254\/8600717-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/http\/xplorestaging.ieee.org\/ielx7\/6221021\/9198254\/08600717.pdf?arnumber=8600717","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T17:20:34Z","timestamp":1651080034000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/document\/8600717\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10]]},"references-count":49,"journal-issue":{"issue":"10"},"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1109\/tsmc.2018.2884725","relation":{},"ISSN":["2168-2216","2168-2232"],"issn-type":[{"value":"2168-2216","type":"print"},{"value":"2168-2232","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,10]]}}}