{"id":"https://openalex.org/W3120441334","doi":"https://doi.org/10.1109/tnnls.2020.3045087","title":"Reinforcement Learning and Adaptive Optimal Control for Continuous-Time Nonlinear Systems: A Value Iteration Approach","display_name":"Reinforcement Learning and Adaptive Optimal Control for Continuous-Time Nonlinear Systems: A Value Iteration Approach","publication_year":2021,"publication_date":"2021-01-09","ids":{"openalex":"https://openalex.org/W3120441334","doi":"https://doi.org/10.1109/tnnls.2020.3045087","mag":"3120441334","pmid":"https://pubmed.ncbi.nlm.nih.gov/33417569"},"language":"en","primary_location":{"id":"doi:10.1109/tnnls.2020.3045087","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tnnls.2020.3045087","pdf_url":null,"source":{"id":"https://openalex.org/S4210175523","display_name":"IEEE Transactions on Neural Networks and Learning Systems","issn_l":"2162-237X","issn":["2162-237X","2162-2388"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Neural Networks and Learning Systems","raw_type":"journal-article"},"type":"article","indexed_in":["crossref","pubmed"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5028849529","display_name":"Tao Bian","orcid":"https://orcid.org/0000-0003-1661-2012"},"institutions":[{"id":"https://openalex.org/I57206974","display_name":"New York University","ror":"https://ror.org/0190ak572","country_code":"US","type":"education","lineage":["https://openalex.org/I57206974"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Tao Bian","raw_affiliation_strings":["Department of Electrical and Computer Engineering, Control and Networks Lab, Tandon School of Engineering, New York University, Brooklyn, NY, USA"],"raw_orcid":"https://orcid.org/0000-0003-1661-2012","affiliations":[{"raw_affiliation_string":"Department of Electrical and Computer Engineering, Control and Networks Lab, Tandon School of Engineering, New York University, Brooklyn, NY, USA","institution_ids":["https://openalex.org/I57206974"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5067046312","display_name":"Zhong\u2010Ping Jiang","orcid":"https://orcid.org/0000-0002-4868-9359"},"institutions":[{"id":"https://openalex.org/I57206974","display_name":"New York University","ror":"https://ror.org/0190ak572","country_code":"US","type":"education","lineage":["https://openalex.org/I57206974"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Zhong-Ping Jiang","raw_affiliation_strings":["Department of Electrical and Computer Engineering, Control and Networks Lab, Tandon School of Engineering, New York University, Brooklyn, NY, USA"],"raw_orcid":"https://orcid.org/0000-0002-4868-9359","affiliations":[{"raw_affiliation_string":"Department of Electrical and Computer Engineering, Control and Networks Lab, Tandon School of Engineering, New York University, Brooklyn, NY, USA","institution_ids":["https://openalex.org/I57206974"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":[],"corresponding_institution_ids":["https://openalex.org/I57206974"],"apc_list":null,"apc_paid":null,"fwci":12.3049,"has_fulltext":false,"cited_by_count":126,"citation_normalized_percentile":{"value":0.98943349,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":98,"max":100},"biblio":{"volume":"33","issue":"7","first_page":"2781","last_page":"2790"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T12794","display_name":"Adaptive Dynamic Programming Control","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T12794","display_name":"Adaptive Dynamic Programming Control","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9850999712944031,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10545","display_name":"Optimization and Variational Analysis","score":0.9699000120162964,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/markov-decision-process","display_name":"Markov decision process","score":0.7308194637298584},{"id":"https://openalex.org/keywords/optimal-control","display_name":"Optimal control","score":0.6607626080513},{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.6602153778076172},{"id":"https://openalex.org/keywords/dynamic-programming","display_name":"Dynamic programming","score":0.6086215376853943},{"id":"https://openalex.org/keywords/nonlinear-system","display_name":"Nonlinear system","score":0.5956427454948425},{"id":"https://openalex.org/keywords/bellman-equation","display_name":"Bellman equation","score":0.584396481513977},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.5629795789718628},{"id":"https://openalex.org/keywords/mathematical-optimization","display_name":"Mathematical optimization","score":0.5547378063201904},{"id":"https://openalex.org/keywords/adaptive-control","display_name":"Adaptive control","score":0.5278881788253784},{"id":"https://openalex.org/keywords/discrete-time-and-continuous-time","display_name":"Discrete time and continuous time","score":0.5277500748634338},{"id":"https://openalex.org/keywords/dynamical-systems-theory","display_name":"Dynamical systems theory","score":0.4593304991722107},{"id":"https://openalex.org/keywords/control-theory","display_name":"Control theory (sociology)","score":0.414242684841156},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.34253159165382385},{"id":"https://openalex.org/keywords/markov-process","display_name":"Markov process","score":0.3425000309944153},{"id":"https://openalex.org/keywords/control","display_name":"Control (management)","score":0.32300400733947754},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.1715671420097351}],"concepts":[{"id":"https://openalex.org/C106189395","wikidata":"https://www.wikidata.org/wiki/Q176789","display_name":"Markov decision process","level":3,"score":0.7308194637298584},{"id":"https://openalex.org/C91575142","wikidata":"https://www.wikidata.org/wiki/Q1971426","display_name":"Optimal control","level":2,"score":0.6607626080513},{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.6602153778076172},{"id":"https://openalex.org/C37404715","wikidata":"https://www.wikidata.org/wiki/Q380679","display_name":"Dynamic programming","level":2,"score":0.6086215376853943},{"id":"https://openalex.org/C158622935","wikidata":"https://www.wikidata.org/wiki/Q660848","display_name":"Nonlinear system","level":2,"score":0.5956427454948425},{"id":"https://openalex.org/C14646407","wikidata":"https://www.wikidata.org/wiki/Q1430750","display_name":"Bellman equation","level":2,"score":0.584396481513977},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.5629795789718628},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.5547378063201904},{"id":"https://openalex.org/C107464732","wikidata":"https://www.wikidata.org/wiki/Q235781","display_name":"Adaptive control","level":3,"score":0.5278881788253784},{"id":"https://openalex.org/C55689738","wikidata":"https://www.wikidata.org/wiki/Q15963867","display_name":"Discrete time and continuous time","level":2,"score":0.5277500748634338},{"id":"https://openalex.org/C79379906","wikidata":"https://www.wikidata.org/wiki/Q3174497","display_name":"Dynamical systems theory","level":2,"score":0.4593304991722107},{"id":"https://openalex.org/C47446073","wikidata":"https://www.wikidata.org/wiki/Q5165890","display_name":"Control theory (sociology)","level":3,"score":0.414242684841156},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.34253159165382385},{"id":"https://openalex.org/C159886148","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov process","level":2,"score":0.3425000309944153},{"id":"https://openalex.org/C2775924081","wikidata":"https://www.wikidata.org/wiki/Q55608371","display_name":"Control (management)","level":2,"score":0.32300400733947754},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.1715671420097351},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.0}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1109/tnnls.2020.3045087","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tnnls.2020.3045087","pdf_url":null,"source":{"id":"https://openalex.org/S4210175523","display_name":"IEEE Transactions on Neural Networks and Learning Systems","issn_l":"2162-237X","issn":["2162-237X","2162-2388"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Neural Networks and Learning Systems","raw_type":"journal-article"},{"id":"pmid:33417569","is_oa":false,"landing_page_url":"https://pubmed.ncbi.nlm.nih.gov/33417569","pdf_url":null,"source":{"id":"https://openalex.org/S4306525036","display_name":"PubMed","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I1299303238","host_organization_name":"National Institutes of Health","host_organization_lineage":["https://openalex.org/I1299303238"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE transactions on neural networks and learning systems","raw_type":null}],"best_oa_location":null,"sustainable_development_goals":[{"score":0.7599999904632568,"id":"https://metadata.un.org/sdg/16","display_name":"Peace, Justice and strong institutions"}],"awards":[{"id":"https://openalex.org/G1335624659","display_name":null,"funder_award_id":"ECCS-1501044","funder_id":"https://openalex.org/F4320306076","funder_display_name":"National Science Foundation"},{"id":"https://openalex.org/G7241937758","display_name":null,"funder_award_id":"EPCN-1903781","funder_id":"https://openalex.org/F4320306076","funder_display_name":"National Science Foundation"}],"funders":[{"id":"https://openalex.org/F4320306076","display_name":"National Science Foundation","ror":"https://ror.org/021nxhr62"}],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":41,"referenced_works":["https://openalex.org/W173093232","https://openalex.org/W1499021337","https://openalex.org/W1564089118","https://openalex.org/W1680219837","https://openalex.org/W1854776945","https://openalex.org/W1996670387","https://openalex.org/W1998361253","https://openalex.org/W2002357820","https://openalex.org/W2005229381","https://openalex.org/W2007700211","https://openalex.org/W2028145673","https://openalex.org/W2037025184","https://openalex.org/W2058406815","https://openalex.org/W2064527819","https://openalex.org/W2073990179","https://openalex.org/W2098432798","https://openalex.org/W2124699533","https://openalex.org/W2140531311","https://openalex.org/W2211436427","https://openalex.org/W2400458653","https://openalex.org/W2467518411","https://openalex.org/W2498677259","https://openalex.org/W2570727720","https://openalex.org/W2727279496","https://openalex.org/W2796251577","https://openalex.org/W2822752092","https://openalex.org/W2884970059","https://openalex.org/W2907944110","https://openalex.org/W2962940927","https://openalex.org/W2995552585","https://openalex.org/W3012130149","https://openalex.org/W4213251304","https://openalex.org/W4214717370","https://openalex.org/W4237171445","https://openalex.org/W4239216443","https://openalex.org/W4301747304","https://openalex.org/W4301886962","https://openalex.org/W4302608015","https://openalex.org/W4302609762","https://openalex.org/W4388319978","https://openalex.org/W6778111457"],"related_works":["https://openalex.org/W4255265352","https://openalex.org/W4239477580","https://openalex.org/W4389475841","https://openalex.org/W2152670157","https://openalex.org/W2903299703","https://openalex.org/W4281791088","https://openalex.org/W2117282672","https://openalex.org/W4385342861","https://openalex.org/W1574958246","https://openalex.org/W2602009922"],"abstract_inverted_index":{"This":[0],"article":[1],"studies":[2],"the":[3,22,51,67,82,97,100,105,115,128,173,176],"adaptive":[4,83,136],"optimal":[5,86,137,160],"control":[6,87,121,150],"problem":[7],"for":[8,89,141],"continuous-time":[9,73,90,101],"nonlinear":[10,142],"systems":[11,91,143],"described":[12,92],"by":[13,29,69,93],"differential":[14,94],"equations.":[15,95],"A":[16,148],"key":[17],"strategy":[18],"is":[19,110,139,152],"to":[20,37,50,64,80,113,154,157,171],"exploit":[21],"value":[23],"iteration":[24],"(VI)":[25],"method":[26,75],"proposed":[27,129,153,177],"initially":[28],"Bellman":[30],"in":[31],"1957":[32],"as":[33],"a":[34,71,124,132],"fundamental":[35],"tool":[36],"solve":[38],"dynamic":[39],"programming":[40],"problems.":[41],"However,":[42],"previous":[43],"VI":[44,74,102,130],"methods":[45],"are":[46,169],"all":[47],"exclusively":[48],"devoted":[49],"Markov":[52],"decision":[53],"processes":[54],"and":[55],"discrete-time":[56],"dynamical":[57],"systems.":[58],"In":[59],"this":[60],"article,":[61],"we":[62],"aim":[63],"fill":[65],"up":[66],"gap":[68],"developing":[70],"new":[72,133],"that":[76,108],"will":[77],"be":[78],"applied":[79],"address":[81],"or":[84],"nonadaptive":[85],"problems":[88],"Like":[96],"traditional":[98],"VI,":[99],"algorithm":[103,151],"retains":[104],"nice":[106],"feature":[107],"there":[109],"no":[111],"need":[112],"assume":[114],"knowledge":[116],"of":[117,127,135,175],"an":[118],"initial":[119],"admissible":[120],"policy.":[122],"As":[123],"direct":[125],"application":[126],"method,":[131],"class":[134],"controllers":[138,161],"obtained":[140],"with":[144],"totally":[145],"unknown":[146],"dynamics.":[147],"learning-based":[149],"show":[155],"how":[156],"learn":[158],"robust":[159],"directly":[162],"from":[163],"real-time":[164],"data.":[165],"Finally,":[166],"two":[167],"examples":[168],"given":[170],"illustrate":[172],"efficacy":[174],"methodology.":[178]},"counts_by_year":[{"year":2026,"cited_by_count":13},{"year":2025,"cited_by_count":28},{"year":2024,"cited_by_count":36},{"year":2023,"cited_by_count":27},{"year":2022,"cited_by_count":14},{"year":2021,"cited_by_count":8}],"updated_date":"2026-07-14T08:27:34.040176","created_date":"2025-10-10T00:00:00"}
