{"id":"https://openalex.org/W2904789544","doi":"https://doi.org/10.1609/aaai.v33i01.33013647","title":"Off-Policy Deep Reinforcement Learning by Bootstrapping the Covariate Shift","display_name":"Off-Policy Deep Reinforcement Learning by Bootstrapping the Covariate Shift","publication_year":2019,"publication_date":"2019-07-17","ids":{"openalex":"https://openalex.org/W2904789544","doi":"https://doi.org/10.1609/aaai.v33i01.33013647","mag":"2904789544"},"language":"en","primary_location":{"id":"doi:10.1609/aaai.v33i01.33013647","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v33i01.33013647","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"diamond","oa_url":"https://doi.org/10.1609/aaai.v33i01.33013647","any_repository_has_fulltext":null},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5022214962","display_name":"Carles Gelada","orcid":null},"institutions":[{"id":"https://openalex.org/I1291425158","display_name":"Google (United States)","ror":"https://ror.org/00njsd438","country_code":"US","type":"company","lineage":["https://openalex.org/I1291425158","https://openalex.org/I4210128969"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Carles Gelada","raw_affiliation_strings":["Google Brain"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Google Brain","institution_ids":["https://openalex.org/I1291425158"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5001087292","display_name":"Marc G. Bellemare","orcid":"https://orcid.org/0000-0002-6096-0105"},"institutions":[{"id":"https://openalex.org/I1291425158","display_name":"Google (United States)","ror":"https://ror.org/00njsd438","country_code":"US","type":"company","lineage":["https://openalex.org/I1291425158","https://openalex.org/I4210128969"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Marc G. Bellemare","raw_affiliation_strings":["Google Brain"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Google Brain","institution_ids":["https://openalex.org/I1291425158"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":[],"corresponding_institution_ids":["https://openalex.org/I1291425158"],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":true,"cited_by_count":61,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":"33","issue":"01","first_page":"3647","last_page":"3655"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9887999892234802,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9887999892234802,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.7393138408660889},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6562108993530273},{"id":"https://openalex.org/keywords/normalization","display_name":"Normalization (sociology)","score":0.5500966310501099},{"id":"https://openalex.org/keywords/markov-decision-process","display_name":"Markov decision process","score":0.5397655963897705},{"id":"https://openalex.org/keywords/bellman-equation","display_name":"Bellman equation","score":0.5186375975608826},{"id":"https://openalex.org/keywords/interpretability","display_name":"Interpretability","score":0.44169700145721436},{"id":"https://openalex.org/keywords/mathematical-optimization","display_name":"Mathematical optimization","score":0.4195806086063385},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.40580007433891296},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.2277756631374359},{"id":"https://openalex.org/keywords/markov-process","display_name":"Markov process","score":0.16307193040847778},{"id":"https://openalex.org/keywords/statistics","display_name":"Statistics","score":0.10411441326141357}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.7393138408660889},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6562108993530273},{"id":"https://openalex.org/C136886441","wikidata":"https://www.wikidata.org/wiki/Q926129","display_name":"Normalization (sociology)","level":2,"score":0.5500966310501099},{"id":"https://openalex.org/C106189395","wikidata":"https://www.wikidata.org/wiki/Q176789","display_name":"Markov decision process","level":3,"score":0.5397655963897705},{"id":"https://openalex.org/C14646407","wikidata":"https://www.wikidata.org/wiki/Q1430750","display_name":"Bellman equation","level":2,"score":0.5186375975608826},{"id":"https://openalex.org/C2781067378","wikidata":"https://www.wikidata.org/wiki/Q17027399","display_name":"Interpretability","level":2,"score":0.44169700145721436},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.4195806086063385},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.40580007433891296},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.2277756631374359},{"id":"https://openalex.org/C159886148","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov process","level":2,"score":0.16307193040847778},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.10411441326141357},{"id":"https://openalex.org/C19165224","wikidata":"https://www.wikidata.org/wiki/Q23404","display_name":"Anthropology","level":1,"score":0.0},{"id":"https://openalex.org/C144024400","wikidata":"https://www.wikidata.org/wiki/Q21201","display_name":"Sociology","level":0,"score":0.0}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1609/aaai.v33i01.33013647","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v33i01.33013647","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},{"id":"pmh:oai:ojs.aaai.org:article/4246","is_oa":true,"landing_page_url":"https://ojs.aaai.org/index.php/AAAI/article/view/4246","pdf_url":"https://ojs.aaai.org/index.php/AAAI/article/download/4246/4124","source":null,"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"2159-5399","raw_type":"info:eu-repo/semantics/article"}],"best_oa_location":{"id":"doi:10.1609/aaai.v33i01.33013647","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v33i01.33013647","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},"sustainable_development_goals":[{"id":"https://metadata.un.org/sdg/16","score":0.7900000214576721,"display_name":"Peace, Justice and strong institutions"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":46,"referenced_works":["https://openalex.org/W1514587017","https://openalex.org/W1547925194","https://openalex.org/W1646707810","https://openalex.org/W1730555343","https://openalex.org/W2068778050","https://openalex.org/W2075268401","https://openalex.org/W2109090232","https://openalex.org/W2119567691","https://openalex.org/W2121863487","https://openalex.org/W2134042548","https://openalex.org/W2139418546","https://openalex.org/W2145339207","https://openalex.org/W2151613683","https://openalex.org/W2155968351","https://openalex.org/W2155969318","https://openalex.org/W2201581102","https://openalex.org/W2411690432","https://openalex.org/W2514813345","https://openalex.org/W2551887912","https://openalex.org/W2739473244","https://openalex.org/W2746553466","https://openalex.org/W2787613197","https://openalex.org/W2950872548","https://openalex.org/W2952509347","https://openalex.org/W2963394426","https://openalex.org/W2963674921","https://openalex.org/W3041202696","https://openalex.org/W3103780890","https://openalex.org/W4293542549","https://openalex.org/W4297795563","https://openalex.org/W4298876402","https://openalex.org/W4300198501","https://openalex.org/W6637597983","https://openalex.org/W6649218630","https://openalex.org/W6667527893","https://openalex.org/W6676399722","https://openalex.org/W6677916085","https://openalex.org/W6682433095","https://openalex.org/W6683410978","https://openalex.org/W6687681856","https://openalex.org/W6704298589","https://openalex.org/W6734043786","https://openalex.org/W6751955673","https://openalex.org/W6780394890","https://openalex.org/W7001894244","https://openalex.org/W7065010408"],"related_works":["https://openalex.org/W2386410636","https://openalex.org/W4308702637","https://openalex.org/W3045510440","https://openalex.org/W2172425052","https://openalex.org/W2373808749","https://openalex.org/W4385342861","https://openalex.org/W2341346307","https://openalex.org/W3115089987","https://openalex.org/W4287549028","https://openalex.org/W1583080569"],"abstract_inverted_index":{"In":[0],"this":[1,21],"paper":[2],"we":[3,173,189,206],"revisit":[4],"the":[5,26,62,68,71,75,112,141,157,165,169,184,202],"method":[6],"of":[7,36,74,114,156,195,201],"off-policy":[8,37,76,162],"corrections":[9],"for":[10,143,210],"reinforcement":[11],"learning":[12,77],"(COP-TD)":[13],"pioneered":[14],"by":[15,103],"Hallak":[16,40],"et":[17,41],"al.":[18],"(2017).":[19],"Under":[20],"method,":[22],"online":[23,138],"updates":[24],"to":[25,31,51,85,177],"value":[27],"function":[28,53],"are":[29],"reweighted":[30],"avoid":[32],"divergence":[33],"issues":[34,102],"typical":[35],"learning.":[38],"While":[39],"al.\u2019s":[42],"solution":[43],"is":[44,79,82],"appealing,":[45],"it":[46,56,81,119],"cannot":[47],"easily":[48],"be":[49,86,93,136,178],"transferred":[50],"nonlinear":[52],"approximation.":[54],"First,":[55],"requires":[57],"a":[58,87,105,123,191],"projection":[59,146],"step":[60],"onto":[61],"probability":[63],"simplex;":[64],"second,":[65],"even":[66],"though":[67],"operator":[69],"describing":[70],"expected":[72],"behavior":[73,113],"algorithm":[78],"convergent,":[80],"not":[83],"known":[84],"contraction":[88],"mapping,":[89],"and":[90,117,139],"hence,":[91],"may":[92],"more":[94,192],"unstable":[95],"in":[96,160,181,198],"practice.":[97],"We":[98,110,126,148],"address":[99],"these":[100],"two":[101,158],"introducing":[104],"discount":[106],"factor":[107],"into":[108],"COP-TD.":[109],"analyze":[111],"discounted":[115,175,196],"COP-TD":[116,176,197],"find":[118,174,207],"better":[120,179],"behaved":[121,180],"from":[122,168],"theoretical":[124],"perspective.":[125],"also":[127],"propose":[128],"an":[129,144,153,161],"alternative":[130],"soft":[131,185],"normalization":[132,186],"penalty":[133],"that":[134],"can":[135],"minimized":[137],"obviates":[140],"need":[142],"explicit":[145],"step.":[147],"complement":[149],"our":[150,211],"analysis":[151],"with":[152],"empirical":[154],"evaluation":[155,194],"techniques":[159],"setting":[163],"on":[164],"game":[166],"Pong":[167],"Atari":[170,203],"domain":[171],"where":[172,205],"practice":[182],"than":[183],"penalty.":[187],"Finally,":[188],"perform":[190],"extensive":[193],"5":[199],"games":[200],"domain,":[204],"performance":[208],"gains":[209],"approach.":[212]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":3},{"year":2023,"cited_by_count":7},{"year":2022,"cited_by_count":6},{"year":2021,"cited_by_count":16},{"year":2020,"cited_by_count":18},{"year":2019,"cited_by_count":10}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
