{"id":"https://openalex.org/W2998069546","doi":"https://doi.org/10.1109/tcyb.2019.2939174","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Toward Continuous Control in Computationally Complex Environments","display_name":"Asynchronous Episodic Deep Deterministic Policy Gradient: Toward Continuous Control in Computationally Complex Environments","publication_year":2019,"publication_date":"2019-12-31","ids":{"openalex":"https://openalex.org/W2998069546","doi":"https://doi.org/10.1109/tcyb.2019.2939174","mag":"2998069546"},"language":"en","primary_location":{"id":"doi:10.1109/tcyb.2019.2939174","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tcyb.2019.2939174","pdf_url":null,"source":{"id":"https://openalex.org/S4210191041","display_name":"IEEE Transactions on Cybernetics","issn_l":"2168-2267","issn":["2168-2267","2168-2275"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Cybernetics","raw_type":"journal-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5101734732","display_name":"Zhizheng Zhang","orcid":"https://orcid.org/0000-0002-5360-7565"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":true,"raw_author_name":"Zhizheng Zhang","raw_affiliation_strings":["CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China"],"affiliations":[{"raw_affiliation_string":"CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China","institution_ids":["https://openalex.org/I126520041"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5100696903","display_name":"Jiale Chen","orcid":"https://orcid.org/0009-0007-4658-5365"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Jiale Chen","raw_affiliation_strings":["CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China"],"affiliations":[{"raw_affiliation_string":"CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China","institution_ids":["https://openalex.org/I126520041"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5079572598","display_name":"Zhibo Chen","orcid":"https://orcid.org/0000-0002-8525-5066"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Zhibo Chen","raw_affiliation_strings":["CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China"],"affiliations":[{"raw_affiliation_string":"CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China","institution_ids":["https://openalex.org/I126520041"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5100415568","display_name":"Weiping Li","orcid":"https://orcid.org/0000-0003-3750-1653"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Weiping Li","raw_affiliation_strings":["CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China"],"affiliations":[{"raw_affiliation_string":"CAS Key Laboratory of Technology in Geo-Spatial Information Processing and Application System, University of Science and Technology of China, Hefei, China","institution_ids":["https://openalex.org/I126520041"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":4,"corresponding_author_ids":["https://openalex.org/A5101734732"],"corresponding_institution_ids":["https://openalex.org/I126520041"],"apc_list":null,"apc_paid":null,"fwci":4.0471,"has_fulltext":false,"cited_by_count":69,"citation_normalized_percentile":{"value":0.95199629,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":97,"max":100},"biblio":{"volume":"51","issue":"2","first_page":"604","last_page":"613"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11689","display_name":"Adversarial Robustness in Machine Learning","score":0.9914000034332275,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12794","display_name":"Adaptive Dynamic Programming Control","score":0.9775999784469604,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.8430414199829102},{"id":"https://openalex.org/keywords/asynchronous-communication","display_name":"Asynchronous communication","score":0.8327853083610535},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7306802272796631},{"id":"https://openalex.org/keywords/inefficiency","display_name":"Inefficiency","score":0.4879119098186493},{"id":"https://openalex.org/keywords/task","display_name":"Task (project management)","score":0.4723949134349823},{"id":"https://openalex.org/keywords/stability","display_name":"Stability (learning theory)","score":0.46518343687057495},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.34582316875457764},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.18207022547721863},{"id":"https://openalex.org/keywords/engineering","display_name":"Engineering","score":0.08028894662857056}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.8430414199829102},{"id":"https://openalex.org/C151319957","wikidata":"https://www.wikidata.org/wiki/Q752739","display_name":"Asynchronous communication","level":2,"score":0.8327853083610535},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7306802272796631},{"id":"https://openalex.org/C2778869765","wikidata":"https://www.wikidata.org/wiki/Q6028363","display_name":"Inefficiency","level":2,"score":0.4879119098186493},{"id":"https://openalex.org/C2780451532","wikidata":"https://www.wikidata.org/wiki/Q759676","display_name":"Task (project management)","level":2,"score":0.4723949134349823},{"id":"https://openalex.org/C112972136","wikidata":"https://www.wikidata.org/wiki/Q7595718","display_name":"Stability (learning theory)","level":2,"score":0.46518343687057495},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.34582316875457764},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.18207022547721863},{"id":"https://openalex.org/C127413603","wikidata":"https://www.wikidata.org/wiki/Q11023","display_name":"Engineering","level":0,"score":0.08028894662857056},{"id":"https://openalex.org/C162324750","wikidata":"https://www.wikidata.org/wiki/Q8134","display_name":"Economics","level":0,"score":0.0},{"id":"https://openalex.org/C31258907","wikidata":"https://www.wikidata.org/wiki/Q1301371","display_name":"Computer network","level":1,"score":0.0},{"id":"https://openalex.org/C175444787","wikidata":"https://www.wikidata.org/wiki/Q39072","display_name":"Microeconomics","level":1,"score":0.0},{"id":"https://openalex.org/C201995342","wikidata":"https://www.wikidata.org/wiki/Q682496","display_name":"Systems engineering","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/tcyb.2019.2939174","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tcyb.2019.2939174","pdf_url":null,"source":{"id":"https://openalex.org/S4210191041","display_name":"IEEE Transactions on Cybernetics","issn_l":"2168-2267","issn":["2168-2267","2168-2275"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Cybernetics","raw_type":"journal-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[{"id":"https://openalex.org/G3231013917","display_name":null,"funder_award_id":"61571413","funder_id":"https://openalex.org/F4320321001","funder_display_name":"National Natural Science Foundation of China"},{"id":"https://openalex.org/G3634038634","display_name":null,"funder_award_id":"61632001","funder_id":"https://openalex.org/F4320321001","funder_display_name":"National Natural Science Foundation of China"}],"funders":[{"id":"https://openalex.org/F4320321001","display_name":"National Natural Science Foundation of China","ror":"https://ror.org/01h0zpd94"}],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":62,"referenced_works":["https://openalex.org/W1522301498","https://openalex.org/W1727219132","https://openalex.org/W1771410628","https://openalex.org/W2017957151","https://openalex.org/W2046376809","https://openalex.org/W2057055688","https://openalex.org/W2097082695","https://openalex.org/W2103836155","https://openalex.org/W2111873046","https://openalex.org/W2112707476","https://openalex.org/W2130281033","https://openalex.org/W2141559645","https://openalex.org/W2145339207","https://openalex.org/W2150513221","https://openalex.org/W2158782408","https://openalex.org/W2165150801","https://openalex.org/W2173248099","https://openalex.org/W2201581102","https://openalex.org/W2436711315","https://openalex.org/W2594466397","https://openalex.org/W2623491082","https://openalex.org/W2736601468","https://openalex.org/W2739678353","https://openalex.org/W2754517384","https://openalex.org/W2774354230","https://openalex.org/W2786928559","https://openalex.org/W2793035934","https://openalex.org/W2795776994","https://openalex.org/W2798705390","https://openalex.org/W2963024489","https://openalex.org/W2963120839","https://openalex.org/W2963296584","https://openalex.org/W2963428623","https://openalex.org/W2963477884","https://openalex.org/W2963864421","https://openalex.org/W2963985863","https://openalex.org/W2964001908","https://openalex.org/W2964043796","https://openalex.org/W2964082094","https://openalex.org/W2964121744","https://openalex.org/W2964340928","https://openalex.org/W4297732320","https://openalex.org/W4300799055","https://openalex.org/W4302570325","https://openalex.org/W6631190155","https://openalex.org/W6637433312","https://openalex.org/W6638018090","https://openalex.org/W6677145610","https://openalex.org/W6684205842","https://openalex.org/W6684921986","https://openalex.org/W6687681856","https://openalex.org/W6692846177","https://openalex.org/W6718359804","https://openalex.org/W6734330393","https://openalex.org/W6739193204","https://openalex.org/W6740801417","https://openalex.org/W6741002519","https://openalex.org/W6744123322","https://openalex.org/W6746581380","https://openalex.org/W6748554570","https://openalex.org/W6749032995","https://openalex.org/W6780559895"],"related_works":["https://openalex.org/W2264067234","https://openalex.org/W3124243301","https://openalex.org/W1571502335","https://openalex.org/W1589409554","https://openalex.org/W2759038785","https://openalex.org/W2172232600","https://openalex.org/W3123876860","https://openalex.org/W3124172198","https://openalex.org/W2046181650","https://openalex.org/W2142633247"],"abstract_inverted_index":{"Deep":[0],"deterministic":[1],"policy":[2],"gradient":[3],"(DDPG)":[4],"has":[5,169],"been":[6],"proved":[7],"to":[8,140,165,176,192],"be":[9],"a":[10,63,132,170],"successful":[11],"reinforcement":[12],"learning":[13,54,164],"(RL)":[14],"algorithm":[15],"for":[16,66,74],"continuous":[17],"control":[18,116,178],"tasks.":[19],"However,":[20],"DDPG":[21,42,205],"still":[22],"suffers":[23],"from":[24,94],"data":[25,67,99,102],"insufficiency":[26],"and":[27,101,153,190],"training":[28,57,81],"inefficiency,":[29],"especially,":[30],"in":[31,69,137,163,180,195,206],"computationally":[32,171,182],"complex":[33,172,183],"environments.":[34,208],"In":[35,104,127],"this":[36,92],"article,":[37],"we":[38,61,106,129,210],"propose":[39],"asynchronous":[40,71,75],"episodic":[41,115],"(AE-DDPG),":[43],"as":[44,84],"an":[45,70],"expansion":[46],"of":[47,87,97,114,135,204,214],"DDPG,":[48],"which":[49,168],"can":[50,121],"achieve":[51],"more":[52],"effective":[53],"with":[55,201],"less":[56,155],"time":[58,156],"required.":[59],"First,":[60],"design":[62],"modified":[64],"scheme":[65],"collection":[68],"fashion.":[72],"Generally,":[73],"RL":[76,161],"algorithms,":[77],"sample":[78,196],"efficiency":[79,197],"or/and":[80],"stability":[82],"diminish":[83],"the":[85,95,112,119,142,177,181,212],"degree":[86],"parallelism":[88],"increases.":[89],"We":[90],"consider":[91],"problem":[93],"perspectives":[96],"both":[98],"generation":[100],"utilization.":[103],"detail,":[105],"redesign":[107],"experience":[108],"replay":[109],"by":[110],"introducing":[111],"idea":[113],"so":[117],"that":[118,147],"agent":[120],"latch":[122],"on":[123,198],"good":[124],"trajectories":[125],"rapidly.":[126],"addition,":[128],"also":[130,186],"inject":[131],"new":[133],"type":[134],"noise":[136],"action":[138],"space":[139],"enrich":[141],"exploration":[143],"behaviors.":[144],"Experiments":[145],"demonstrate":[146],"our":[148],"AE-DDPG":[149,185],"achieves":[150,187],"higher":[151,188],"rewards":[152,189],"requires":[154],"consumption":[157],"than":[158],"most":[159],"popular":[160],"algorithms":[162],"run":[166],"task":[167],"environment.":[173],"Not":[174],"limited":[175],"tasks":[179],"environments,":[184],"two-fold":[191],"four-fold":[193],"improvement":[194],"average":[199],"compared":[200],"other":[202],"variants":[203],"MuJoCo":[207],"Furthermore,":[209],"verify":[211],"effectiveness":[213],"each":[215],"proposed":[216],"technique":[217],"component":[218],"through":[219],"abundant":[220],"ablation":[221],"study.":[222]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":15},{"year":2024,"cited_by_count":12},{"year":2023,"cited_by_count":13},{"year":2022,"cited_by_count":13},{"year":2021,"cited_by_count":11},{"year":2020,"cited_by_count":4}],"updated_date":"2026-03-12T08:34:05.389933","created_date":"2020-01-10T00:00:00"}
