{"id":"https://openalex.org/W2042882799","doi":"https://doi.org/10.1109/robot.2010.5509336","title":"Reinforcement learning of motor skills in high dimensions: A path integral approach","display_name":"Reinforcement learning of motor skills in high dimensions: A path integral approach","publication_year":2010,"publication_date":"2010-05-01","ids":{"openalex":"https://openalex.org/W2042882799","doi":"https://doi.org/10.1109/robot.2010.5509336","mag":"2042882799"},"language":"en","primary_location":{"id":"doi:10.1109/robot.2010.5509336","is_oa":false,"landing_page_url":"https://doi.org/10.1109/robot.2010.5509336","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2010 IEEE International Conference on Robotics and Automation","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5044505993","display_name":"Evangelos A. Theodorou","orcid":"https://orcid.org/0000-0002-0834-5738"},"institutions":[{"id":"https://openalex.org/I1174212","display_name":"University of Southern California","ror":"https://ror.org/03taz7m60","country_code":"US","type":"education","lineage":["https://openalex.org/I1174212"]},{"id":"https://openalex.org/I4210138427","display_name":"Computational Physics (United States)","ror":"https://ror.org/03grsxg62","country_code":"US","type":"company","lineage":["https://openalex.org/I4210138427"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Evangelos Theodorou","raw_affiliation_strings":["Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","institution_ids":["https://openalex.org/I4210138427"]},{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA","institution_ids":["https://openalex.org/I1174212"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5077279214","display_name":"Jonas Buchli","orcid":"https://orcid.org/0000-0001-7494-8492"},"institutions":[{"id":"https://openalex.org/I1174212","display_name":"University of Southern California","ror":"https://ror.org/03taz7m60","country_code":"US","type":"education","lineage":["https://openalex.org/I1174212"]},{"id":"https://openalex.org/I4210138427","display_name":"Computational Physics (United States)","ror":"https://ror.org/03grsxg62","country_code":"US","type":"company","lineage":["https://openalex.org/I4210138427"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Jonas Buchli","raw_affiliation_strings":["Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","institution_ids":["https://openalex.org/I4210138427"]},{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA","institution_ids":["https://openalex.org/I1174212"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5029642293","display_name":"Stefan Schaal","orcid":"https://orcid.org/0000-0001-5660-1874"},"institutions":[{"id":"https://openalex.org/I1174212","display_name":"University of Southern California","ror":"https://ror.org/03taz7m60","country_code":"US","type":"education","lineage":["https://openalex.org/I1174212"]},{"id":"https://openalex.org/I4210138427","display_name":"Computational Physics (United States)","ror":"https://ror.org/03grsxg62","country_code":"US","type":"company","lineage":["https://openalex.org/I4210138427"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Stefan Schaal","raw_affiliation_strings":["Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory in Computer Science, Biomedical Engineering, Neuroscience, University of Southern California, USA","institution_ids":["https://openalex.org/I4210138427"]},{"raw_affiliation_string":"Computational Learning and Motor Control Laboratory, in Computer Science, Neuroscience, and Biomedical Engineering, at the University of Southern California, USA","institution_ids":["https://openalex.org/I1174212"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":257,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":"2397","last_page":"2403"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12794","display_name":"Adaptive Dynamic Programming Control","score":0.9878000020980835,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12101","display_name":"Advanced Bandit Algorithms Research","score":0.9789000153541565,"subfield":{"id":"https://openalex.org/subfields/1803","display_name":"Management Science and Operations Research"},"field":{"id":"https://openalex.org/fields/18","display_name":"Decision Sciences"},"domain":{"id":"https://openalex.org/domains/2","display_name":"Social Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.9256943464279175},{"id":"https://openalex.org/keywords/parameterized-complexity","display_name":"Parameterized complexity","score":0.6833593845367432},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6388013958930969},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.5886709094047546},{"id":"https://openalex.org/keywords/scalability","display_name":"Scalability","score":0.5747875571250916},{"id":"https://openalex.org/keywords/path","display_name":"Path (computing)","score":0.5513395071029663},{"id":"https://openalex.org/keywords/robotics","display_name":"Robotics","score":0.4354536831378937},{"id":"https://openalex.org/keywords/temporal-difference-learning","display_name":"Temporal difference learning","score":0.4288158714771271},{"id":"https://openalex.org/keywords/path-integral-formulation","display_name":"Path integral formulation","score":0.42076390981674194},{"id":"https://openalex.org/keywords/optimal-control","display_name":"Optimal control","score":0.4185332953929901},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.39499834179878235},{"id":"https://openalex.org/keywords/mathematical-optimization","display_name":"Mathematical optimization","score":0.33391714096069336},{"id":"https://openalex.org/keywords/robot","display_name":"Robot","score":0.29024380445480347},{"id":"https://openalex.org/keywords/algorithm","display_name":"Algorithm","score":0.26067569851875305},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.24396491050720215}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.9256943464279175},{"id":"https://openalex.org/C165464430","wikidata":"https://www.wikidata.org/wiki/Q1570441","display_name":"Parameterized complexity","level":2,"score":0.6833593845367432},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6388013958930969},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.5886709094047546},{"id":"https://openalex.org/C48044578","wikidata":"https://www.wikidata.org/wiki/Q727490","display_name":"Scalability","level":2,"score":0.5747875571250916},{"id":"https://openalex.org/C2777735758","wikidata":"https://www.wikidata.org/wiki/Q817765","display_name":"Path (computing)","level":2,"score":0.5513395071029663},{"id":"https://openalex.org/C34413123","wikidata":"https://www.wikidata.org/wiki/Q170978","display_name":"Robotics","level":3,"score":0.4354536831378937},{"id":"https://openalex.org/C196340769","wikidata":"https://www.wikidata.org/wiki/Q7698910","display_name":"Temporal difference learning","level":3,"score":0.4288158714771271},{"id":"https://openalex.org/C154018700","wikidata":"https://www.wikidata.org/wiki/Q898323","display_name":"Path integral formulation","level":3,"score":0.42076390981674194},{"id":"https://openalex.org/C91575142","wikidata":"https://www.wikidata.org/wiki/Q1971426","display_name":"Optimal control","level":2,"score":0.4185332953929901},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.39499834179878235},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.33391714096069336},{"id":"https://openalex.org/C90509273","wikidata":"https://www.wikidata.org/wiki/Q11012","display_name":"Robot","level":2,"score":0.29024380445480347},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.26067569851875305},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.24396491050720215},{"id":"https://openalex.org/C77088390","wikidata":"https://www.wikidata.org/wiki/Q8513","display_name":"Database","level":1,"score":0.0},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0},{"id":"https://openalex.org/C199360897","wikidata":"https://www.wikidata.org/wiki/Q9143","display_name":"Programming language","level":1,"score":0.0},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0},{"id":"https://openalex.org/C84114770","wikidata":"https://www.wikidata.org/wiki/Q46344","display_name":"Quantum","level":2,"score":0.0}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1109/robot.2010.5509336","is_oa":false,"landing_page_url":"https://doi.org/10.1109/robot.2010.5509336","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2010 IEEE International Conference on Robotics and Automation","raw_type":"proceedings-article"},{"id":"pmh:oai:CiteSeerX.psu:10.1.1.159.2785","is_oa":false,"landing_page_url":"http://citeseerx.ist.psu.edu/viewdoc/summary?doi=10.1.1.159.2785","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"http://www-clmc.usc.edu/publications//E/PI2.pdf","raw_type":"text"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":46,"referenced_works":["https://openalex.org/W171226173","https://openalex.org/W1555368087","https://openalex.org/W1558902221","https://openalex.org/W1586251222","https://openalex.org/W1613865483","https://openalex.org/W1670925118","https://openalex.org/W2026659355","https://openalex.org/W2043962331","https://openalex.org/W2080039641","https://openalex.org/W2080759927","https://openalex.org/W2093524643","https://openalex.org/W2107662876","https://openalex.org/W2108936969","https://openalex.org/W2119717200","https://openalex.org/W2120070743","https://openalex.org/W2121412035","https://openalex.org/W2123157758","https://openalex.org/W2123967136","https://openalex.org/W2124758444","https://openalex.org/W2125612430","https://openalex.org/W2127107099","https://openalex.org/W2131490088","https://openalex.org/W2137808289","https://openalex.org/W2141436719","https://openalex.org/W2145060720","https://openalex.org/W2154032554","https://openalex.org/W2155027007","https://openalex.org/W2155772159","https://openalex.org/W2161872510","https://openalex.org/W2164569010","https://openalex.org/W2167804690","https://openalex.org/W2168921921","https://openalex.org/W2170316986","https://openalex.org/W2172968643","https://openalex.org/W2964109529","https://openalex.org/W3099235411","https://openalex.org/W3103182070","https://openalex.org/W4210791600","https://openalex.org/W4246729560","https://openalex.org/W6633459510","https://openalex.org/W6635158341","https://openalex.org/W6636453219","https://openalex.org/W6637173446","https://openalex.org/W6678111950","https://openalex.org/W6678157427","https://openalex.org/W6678376813"],"related_works":["https://openalex.org/W2051058708","https://openalex.org/W1494268238","https://openalex.org/W154868527","https://openalex.org/W1983207144","https://openalex.org/W2490706771","https://openalex.org/W2480116122","https://openalex.org/W4255576661","https://openalex.org/W1516574938","https://openalex.org/W2625725254","https://openalex.org/W2563912921"],"abstract_inverted_index":{"Reinforcement":[0],"learning":[1,11,33,80,97,110,119],"(RL)":[2],"is":[3],"one":[4,153],"of":[5,59,88,128,154],"the":[6,28,57,76,126,155],"most":[7,156],"general":[8],"approaches":[9],"to":[10,15,27,49,113,162],"control.":[12],"Its":[13],"applicability":[14],"complex":[16],"motor":[17],"systems,":[18],"however,":[19],"has":[20],"been":[21],"largely":[22],"impossible":[23],"so":[24],"far":[25],"due":[26],"computational":[29],"difficulties":[30],"that":[31,137],"reinforcement":[32],"encounters":[34],"in":[35,69,131,167],"high":[36],"dimensional":[37],"continuous":[38],"state-action":[39],"spaces.":[40],"In":[41],"this":[42],"paper,":[43],"we":[44],"derive":[45],"a":[46,118,122,132],"novel":[47],"approach":[48],"RL":[50,166],"for":[51,79,165],"parameterized":[52],"control":[53,62,71,115],"policies":[54],"based":[55],"on":[56,121],"framework":[58],"stochastic":[60],"optimal":[61,70],"with":[63,143],"path":[64],"integrals.":[65],"While":[66],"solidly":[67],"grounded":[68],"theory":[72],"and":[73,84,111,160],"estimation":[74],"theory,":[75],"update":[77],"equations":[78],"are":[81,99],"surprisingly":[82],"simple":[83],"have":[85],"no":[86],"danger":[87],"numerical":[89],"instabilities":[90],"as":[91],"neither":[92],"matrix":[93],"inversions":[94],"nor":[95],"gradient":[96],"rates":[98],"required.":[100],"Empirical":[101],"evaluations":[102],"demonstrate":[103],"significant":[104],"performance":[105],"improvements":[106],"over":[107],"gradient-based":[108],"policy":[109],"scalability":[112],"high-dimensional":[114],"problems.":[116],"Finally,":[117],"experiment":[120],"robot":[123],"dog":[124],"illustrates":[125],"functionality":[127],"our":[129,138],"algorithm":[130],"real-world":[133],"scenario.":[134],"We":[135],"believe":[136],"new":[139],"algorithm,":[140],"Policy":[141],"Improvement":[142],"Path":[144],"Integrals":[145],"(PI":[146],"<sup":[147],"xmlns:mml=\"http://www.w3.org/1998/Math/MathML\"":[148],"xmlns:xlink=\"http://www.w3.org/1999/xlink\">2</sup>":[149],"),":[150],"offers":[151],"currently":[152],"efficient,":[157],"numerically":[158],"robust,":[159],"easy":[161],"implement":[163],"algorithms":[164],"robotics.":[168]},"counts_by_year":[{"year":2026,"cited_by_count":2},{"year":2025,"cited_by_count":6},{"year":2024,"cited_by_count":11},{"year":2023,"cited_by_count":14},{"year":2022,"cited_by_count":11},{"year":2021,"cited_by_count":25},{"year":2020,"cited_by_count":18},{"year":2019,"cited_by_count":21},{"year":2018,"cited_by_count":17},{"year":2017,"cited_by_count":11},{"year":2016,"cited_by_count":13},{"year":2015,"cited_by_count":14},{"year":2014,"cited_by_count":31},{"year":2013,"cited_by_count":22},{"year":2012,"cited_by_count":20}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
