{"id":"https://openalex.org/W2909152095","doi":"https://doi.org/10.1109/iros.2018.8593728","title":"Accelerating Goal-Directed Reinforcement Learning by Model Characterization","display_name":"Accelerating Goal-Directed Reinforcement Learning by Model Characterization","publication_year":2018,"publication_date":"2018-10-01","ids":{"openalex":"https://openalex.org/W2909152095","doi":"https://doi.org/10.1109/iros.2018.8593728","mag":"2909152095"},"language":"en","primary_location":{"id":"doi:10.1109/iros.2018.8593728","is_oa":false,"landing_page_url":"https://doi.org/10.1109/iros.2018.8593728","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2018 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5006168104","display_name":"Shoubhik Debnath","orcid":null},"institutions":[{"id":"https://openalex.org/I4210127875","display_name":"Nvidia (United States)","ror":"https://ror.org/03jdj4y14","country_code":"US","type":"company","lineage":["https://openalex.org/I4210127875"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Shoubhik Debnath","raw_affiliation_strings":["NVIDIA Corporation, Santa Clara, CA, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"NVIDIA Corporation, Santa Clara, CA, USA","institution_ids":["https://openalex.org/I4210127875"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5077367921","display_name":"Gaurav S. Sukhatme","orcid":"https://orcid.org/0000-0003-2408-474X"},"institutions":[{"id":"https://openalex.org/I1174212","display_name":"University of Southern California","ror":"https://ror.org/03taz7m60","country_code":"US","type":"education","lineage":["https://openalex.org/I1174212"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Gaurav Sukhatme","raw_affiliation_strings":["Department of Computer Science at the University of Southern California, Los Angeles, CA, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science at the University of Southern California, Los Angeles, CA, USA","institution_ids":["https://openalex.org/I1174212"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5101917996","display_name":"Lantao Liu","orcid":"https://orcid.org/0000-0002-6796-6817"},"institutions":[{"id":"https://openalex.org/I4210119109","display_name":"Indiana University Bloomington","ror":"https://ror.org/02k40bc56","country_code":"US","type":"education","lineage":["https://openalex.org/I4210119109","https://openalex.org/I592451"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Lantao Liu","raw_affiliation_strings":["Intelligent Systems Engineering, Department at Indiana University - Bloomington, Bloomington, IN, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Intelligent Systems Engineering, Department at Indiana University - Bloomington, Bloomington, IN, USA","institution_ids":["https://openalex.org/I4210119109"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":3,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":1,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":"121","issue":null,"first_page":"1","last_page":"9"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9994999766349792,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9994999766349792,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11975","display_name":"Evolutionary Algorithms and Applications","score":0.9787999987602234,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11663","display_name":"Viral Infectious Diseases and Gene Expression in Insects","score":0.9779999852180481,"subfield":{"id":"https://openalex.org/subfields/1312","display_name":"Molecular Biology"},"field":{"id":"https://openalex.org/fields/13","display_name":"Biochemistry, Genetics and Molecular Biology"},"domain":{"id":"https://openalex.org/domains/1","display_name":"Life Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.9217270612716675},{"id":"https://openalex.org/keywords/leverage","display_name":"Leverage (statistics)","score":0.8001425266265869},{"id":"https://openalex.org/keywords/reachability","display_name":"Reachability","score":0.7499926090240479},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7240926623344421},{"id":"https://openalex.org/keywords/sample-complexity","display_name":"Sample complexity","score":0.5007617473602295},{"id":"https://openalex.org/keywords/temporal-difference-learning","display_name":"Temporal difference learning","score":0.49202391505241394},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.48407211899757385},{"id":"https://openalex.org/keywords/reinforcement","display_name":"Reinforcement","score":0.469959020614624},{"id":"https://openalex.org/keywords/sample","display_name":"Sample (material)","score":0.4346194863319397},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.39320844411849976},{"id":"https://openalex.org/keywords/theoretical-computer-science","display_name":"Theoretical computer science","score":0.18365582823753357},{"id":"https://openalex.org/keywords/engineering","display_name":"Engineering","score":0.08135655522346497}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.9217270612716675},{"id":"https://openalex.org/C153083717","wikidata":"https://www.wikidata.org/wiki/Q6535263","display_name":"Leverage (statistics)","level":2,"score":0.8001425266265869},{"id":"https://openalex.org/C136643341","wikidata":"https://www.wikidata.org/wiki/Q1361526","display_name":"Reachability","level":2,"score":0.7499926090240479},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7240926623344421},{"id":"https://openalex.org/C2778445095","wikidata":"https://www.wikidata.org/wiki/Q18354077","display_name":"Sample complexity","level":2,"score":0.5007617473602295},{"id":"https://openalex.org/C196340769","wikidata":"https://www.wikidata.org/wiki/Q7698910","display_name":"Temporal difference learning","level":3,"score":0.49202391505241394},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.48407211899757385},{"id":"https://openalex.org/C67203356","wikidata":"https://www.wikidata.org/wiki/Q1321905","display_name":"Reinforcement","level":2,"score":0.469959020614624},{"id":"https://openalex.org/C198531522","wikidata":"https://www.wikidata.org/wiki/Q485146","display_name":"Sample (material)","level":2,"score":0.4346194863319397},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.39320844411849976},{"id":"https://openalex.org/C80444323","wikidata":"https://www.wikidata.org/wiki/Q2878974","display_name":"Theoretical computer science","level":1,"score":0.18365582823753357},{"id":"https://openalex.org/C127413603","wikidata":"https://www.wikidata.org/wiki/Q11023","display_name":"Engineering","level":0,"score":0.08135655522346497},{"id":"https://openalex.org/C66938386","wikidata":"https://www.wikidata.org/wiki/Q633538","display_name":"Structural engineering","level":1,"score":0.0},{"id":"https://openalex.org/C185592680","wikidata":"https://www.wikidata.org/wiki/Q2329","display_name":"Chemistry","level":0,"score":0.0},{"id":"https://openalex.org/C43617362","wikidata":"https://www.wikidata.org/wiki/Q170050","display_name":"Chromatography","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/iros.2018.8593728","is_oa":false,"landing_page_url":"https://doi.org/10.1109/iros.2018.8593728","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2018 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":29,"referenced_works":["https://openalex.org/W186800770","https://openalex.org/W1491843047","https://openalex.org/W1545472701","https://openalex.org/W1595634327","https://openalex.org/W1858246706","https://openalex.org/W1977655452","https://openalex.org/W1981745143","https://openalex.org/W1982948368","https://openalex.org/W1997477668","https://openalex.org/W2012587148","https://openalex.org/W2048226872","https://openalex.org/W2052117683","https://openalex.org/W2072752020","https://openalex.org/W2100677568","https://openalex.org/W2107726111","https://openalex.org/W2121733891","https://openalex.org/W2122701159","https://openalex.org/W2159420891","https://openalex.org/W2313532661","https://openalex.org/W2743381431","https://openalex.org/W2752814746","https://openalex.org/W2756350131","https://openalex.org/W2949608212","https://openalex.org/W2951948137","https://openalex.org/W2953084784","https://openalex.org/W2962872206","https://openalex.org/W3022678080","https://openalex.org/W3128263354","https://openalex.org/W3139377883"],"related_works":["https://openalex.org/W4400868993","https://openalex.org/W2145363145","https://openalex.org/W2341346307","https://openalex.org/W2154399718","https://openalex.org/W4321463377","https://openalex.org/W4384574988","https://openalex.org/W2768629321","https://openalex.org/W2130711276","https://openalex.org/W2877093712","https://openalex.org/W2116157560"],"abstract_inverted_index":{"We":[0,15],"propose":[1],"a":[2,19,26,40,57],"hybrid":[3,97],"approach":[4],"aimed":[5],"at":[6],"improving":[7],"the":[8,86],"sample":[9],"efficiency":[10],"in":[11],"goal-directed":[12],"reinforcement":[13,30,52,88],"learning.":[14,31,53],"do":[16],"this":[17,35],"via":[18],"two-step":[20],"mechanism":[21],"where":[22],"firstly,":[23],"we":[24,33,60],"approximate":[25,36],"model":[27,37],"from":[28,85],"Model-Free":[29],"Then,":[32],"leverage":[34],"along":[38],"with":[39,100],"notion":[41],"of":[42],"reachability":[43],"using":[44],"Mean":[45,66,73],"First":[46,67,74],"Passage":[47,68,75],"Times":[48],"to":[49,120],"perform":[50],"Model-Based":[51],"Built":[54],"on":[55],"such":[56],"novel":[58],"observation,":[59],"design":[61],"two":[62],"new":[63],"algorithms":[64],"-":[65],"Time":[69,76],"based":[70,77],"Q-Learning":[71],"(MFPT-Q)and":[72],"DYNA":[78],"(MFPT-DYNA),":[79],"that":[80,95],"have":[81,93],"been":[82],"fundamentally":[83],"modified":[84],"state-of-the-art":[87,107],"learning":[89],"techniques.":[90],"Preliminary":[91],"results":[92],"shown":[94],"our":[96],"approaches":[98],"converge":[99],"much":[101,112,116],"fewer":[102,113,117],"iterations":[103],"than":[104],"their":[105],"corresponding":[106],"counterparts":[108],"and":[109,115],"therefore":[110],"requiring":[111],"samples":[114],"training":[118],"trials":[119],"converge.":[121]},"counts_by_year":[{"year":2020,"cited_by_count":1}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
