{"id":"https://openalex.org/W4385325575","doi":"https://doi.org/10.1109/tai.2023.3297988","title":"Generalized Maximum Entropy Reinforcement Learning via Reward Shaping","display_name":"Generalized Maximum Entropy Reinforcement Learning via Reward Shaping","publication_year":2023,"publication_date":"2023-07-27","ids":{"openalex":"https://openalex.org/W4385325575","doi":"https://doi.org/10.1109/tai.2023.3297988"},"language":"en","primary_location":{"id":"doi:10.1109/tai.2023.3297988","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tai.2023.3297988","pdf_url":null,"source":{"id":"https://openalex.org/S4210169448","display_name":"IEEE Transactions on Artificial Intelligence","issn_l":"2691-4581","issn":["2691-4581"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Artificial Intelligence","raw_type":"journal-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5000442596","display_name":"Tao Feng","orcid":"https://orcid.org/0000-0002-2507-3180"},"institutions":[{"id":"https://openalex.org/I4210135257","display_name":"Volvo (United States)","ror":"https://ror.org/03e7zen21","country_code":"US","type":"company","lineage":["https://openalex.org/I1340210623","https://openalex.org/I4210135257"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Feng Tao","raw_affiliation_strings":["Volvo Car Technology USA LLC, Sunnyvale, CA, USA"],"raw_orcid":"https://orcid.org/0000-0002-2507-3180","affiliations":[{"raw_affiliation_string":"Volvo Car Technology USA LLC, Sunnyvale, CA, USA","institution_ids":["https://openalex.org/I4210135257"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5101438487","display_name":"MingKang Wu","orcid":"https://orcid.org/0009-0004-2960-0787"},"institutions":[{"id":"https://openalex.org/I45438204","display_name":"The University of Texas at San Antonio","ror":"https://ror.org/01kd65564","country_code":"US","type":"education","lineage":["https://openalex.org/I45438204"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Mingkang Wu","raw_affiliation_strings":["Department of Electrical Engineering, University of Texas, San Antonio, TX, USA"],"raw_orcid":"https://orcid.org/0009-0004-2960-0787","affiliations":[{"raw_affiliation_string":"Department of Electrical Engineering, University of Texas, San Antonio, TX, USA","institution_ids":["https://openalex.org/I45438204"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5014808989","display_name":"Yongcan Cao","orcid":"https://orcid.org/0000-0003-3383-0185"},"institutions":[{"id":"https://openalex.org/I45438204","display_name":"The University of Texas at San Antonio","ror":"https://ror.org/01kd65564","country_code":"US","type":"education","lineage":["https://openalex.org/I45438204"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Yongcan Cao","raw_affiliation_strings":["Department of Electrical Engineering, University of Texas, San Antonio, TX, USA"],"raw_orcid":"https://orcid.org/0000-0003-3383-0185","affiliations":[{"raw_affiliation_string":"Department of Electrical Engineering, University of Texas, San Antonio, TX, USA","institution_ids":["https://openalex.org/I45438204"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":0.7808,"has_fulltext":false,"cited_by_count":7,"citation_normalized_percentile":{"value":0.65665666,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":94,"max":98},"biblio":{"volume":"5","issue":"4","first_page":"1563","last_page":"1572"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T11601","display_name":"Neuroscience and Neural Engineering","score":0.9236000180244446,"subfield":{"id":"https://openalex.org/subfields/2804","display_name":"Cellular and Molecular Neuroscience"},"field":{"id":"https://openalex.org/fields/28","display_name":"Neuroscience"},"domain":{"id":"https://openalex.org/domains/1","display_name":"Life Sciences"}},"topics":[{"id":"https://openalex.org/T11601","display_name":"Neuroscience and Neural Engineering","score":0.9236000180244446,"subfield":{"id":"https://openalex.org/subfields/2804","display_name":"Cellular and Molecular Neuroscience"},"field":{"id":"https://openalex.org/fields/28","display_name":"Neuroscience"},"domain":{"id":"https://openalex.org/domains/1","display_name":"Life Sciences"}},{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9211999773979187,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.7738882303237915},{"id":"https://openalex.org/keywords/reinforcement","display_name":"Reinforcement","score":0.6477431654930115},{"id":"https://openalex.org/keywords/principle-of-maximum-entropy","display_name":"Principle of maximum entropy","score":0.476742148399353},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.3929729461669922},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.33710721135139465},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.3295390009880066},{"id":"https://openalex.org/keywords/psychology","display_name":"Psychology","score":0.28919148445129395},{"id":"https://openalex.org/keywords/social-psychology","display_name":"Social psychology","score":0.12597009539604187}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.7738882303237915},{"id":"https://openalex.org/C67203356","wikidata":"https://www.wikidata.org/wiki/Q1321905","display_name":"Reinforcement","level":2,"score":0.6477431654930115},{"id":"https://openalex.org/C9679016","wikidata":"https://www.wikidata.org/wiki/Q1417473","display_name":"Principle of maximum entropy","level":2,"score":0.476742148399353},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.3929729461669922},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.33710721135139465},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.3295390009880066},{"id":"https://openalex.org/C15744967","wikidata":"https://www.wikidata.org/wiki/Q9418","display_name":"Psychology","level":0,"score":0.28919148445129395},{"id":"https://openalex.org/C77805123","wikidata":"https://www.wikidata.org/wiki/Q161272","display_name":"Social psychology","level":1,"score":0.12597009539604187}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/tai.2023.3297988","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tai.2023.3297988","pdf_url":null,"source":{"id":"https://openalex.org/S4210169448","display_name":"IEEE Transactions on Artificial Intelligence","issn_l":"2691-4581","issn":["2691-4581"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Artificial Intelligence","raw_type":"journal-article"}],"best_oa_location":null,"sustainable_development_goals":[{"score":0.5099999904632568,"display_name":"Reduced inequalities","id":"https://metadata.un.org/sdg/10"}],"awards":[{"id":"https://openalex.org/G7298247483","display_name":null,"funder_award_id":"N000142212474","funder_id":"https://openalex.org/F4320337345","funder_display_name":"Office of Naval Research"},{"id":"https://openalex.org/G8052522079","display_name":null,"funder_award_id":"W911NF-21-1-0103","funder_id":"https://openalex.org/F4320338281","funder_display_name":"Army Research Office"}],"funders":[{"id":"https://openalex.org/F4320337345","display_name":"Office of Naval Research","ror":"https://ror.org/00rk2pe57"},{"id":"https://openalex.org/F4320338281","display_name":"Army Research Office","ror":"https://ror.org/05epdh915"}],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":46,"referenced_works":["https://openalex.org/W32403112","https://openalex.org/W64088143","https://openalex.org/W1191599655","https://openalex.org/W1977655452","https://openalex.org/W2044287460","https://openalex.org/W2119717200","https://openalex.org/W2123859855","https://openalex.org/W2145339207","https://openalex.org/W2157340968","https://openalex.org/W2158782408","https://openalex.org/W2736601468","https://openalex.org/W2914261249","https://openalex.org/W2950564923","https://openalex.org/W2965140792","https://openalex.org/W2967727187","https://openalex.org/W2973229164","https://openalex.org/W2996037775","https://openalex.org/W3109546547","https://openalex.org/W3127686539","https://openalex.org/W3133092968","https://openalex.org/W3157410348","https://openalex.org/W3160598148","https://openalex.org/W4214717370","https://openalex.org/W6627932998","https://openalex.org/W6628764021","https://openalex.org/W6631943919","https://openalex.org/W6632235069","https://openalex.org/W6638018090","https://openalex.org/W6638088447","https://openalex.org/W6677067356","https://openalex.org/W6683195989","https://openalex.org/W6683204974","https://openalex.org/W6692846177","https://openalex.org/W6734517396","https://openalex.org/W6741002519","https://openalex.org/W6747473740","https://openalex.org/W6749859622","https://openalex.org/W6751285671","https://openalex.org/W6756287877","https://openalex.org/W6757058172","https://openalex.org/W6758978475","https://openalex.org/W6763990646","https://openalex.org/W6767569084","https://openalex.org/W6772005887","https://openalex.org/W6780559895","https://openalex.org/W6922480057"],"related_works":["https://openalex.org/W2920061524","https://openalex.org/W1979597421","https://openalex.org/W4310083477","https://openalex.org/W2007980826","https://openalex.org/W4245490552","https://openalex.org/W4225152035","https://openalex.org/W1977959518","https://openalex.org/W2038908348","https://openalex.org/W2061531152","https://openalex.org/W4312713068"],"abstract_inverted_index":{"Entropy":[0],"regularization":[1,31,67],"is":[2,118,142],"a":[3,15,92,112,191,204,211],"commonly":[4],"used":[5],"technique":[6],"in":[7,69,74,83,217,240],"reinforcement":[8,50,161,172,207,233],"learning":[9,162,173,208,234],"to":[10,120,144,168,220],"improve":[11],"exploration":[12],"and":[13,37,150,156,178,202],"cultivate":[14],"better":[16],"pre-trained":[17],"policy":[18,40,71,99,135,194],"for":[19],"later":[20],"adaptation.":[21],"Recent":[22],"studies":[23,54,214],"further":[24,189],"show":[25],"that":[26,95,153,222],"the":[27,34,39,44,57,61,70,75,81,84,97,102,107,115,121,126,132,138,159,169,182,199],"use":[28],"of":[29,46,114,131,184,242],"entropy":[30,48,59,82,100,110,136,186,232],"can":[32,164,225],"smooth":[33],"optimization":[35,41],"landscape":[36],"simplify":[38],"process,":[42],"indicating":[43],"value":[45],"integrating":[47,80],"into":[49,101],"learning.":[51],"However,":[52],"existing":[53,170,228],"only":[55],"consider":[56],"policy's":[58],"at":[60,137],"current":[62,127],"state":[63,117,140,151],"as":[64,176],"an":[65,227],"extra":[66],"term":[68],"gradient":[72,195],"or":[73],"objective":[76],"function":[77,149,152],"without":[78],"formally":[79],"reward":[85,94,103,123,201],"function.":[86,104],"In":[87,105],"this":[88],"paper,":[89],"we":[90],"propose":[91,203],"shaped":[93,200],"includes":[96],"agent's":[98,108,133],"particular,":[106],"expected":[109,134],"over":[111],"distribution":[113,141],"next":[116,139],"added":[119],"immediate":[122],"associated":[124],"with":[125],"state.":[128],"The":[129],"addition":[130],"shown":[143],"yield":[145],"new":[146,160,205],"soft":[147,192],"Q":[148],"are":[154,215],"concise":[155],"modular.":[157],"Moreover,":[158],"framework":[163],"be":[165],"easily":[166],"applied":[167],"standard":[171],"algorithms,":[174],"such":[175],"DQN":[177],"PPO,":[179],"while":[180],"inheriting":[181],"benefits":[183],"employing":[185],"regularization.":[187],"We":[188],"present":[190],"stochastic":[193],"theorem":[196],"based":[197],"on":[198],"practical":[206],"algorithm.":[209],"Finally,":[210],"few":[212],"experimental":[213],"conducted":[216],"MuJoCo":[218],"environment":[219],"demonstrate":[221],"our":[223],"method":[224],"outperform":[226],"state-of-the-art":[229],"off-policy":[230],"maximum":[231],"approach":[235],"Soft":[236],"Actor-Critic":[237],"by":[238],"5-150%":[239],"terms":[241],"average":[243],"return.":[244]},"counts_by_year":[{"year":2026,"cited_by_count":2},{"year":2025,"cited_by_count":3},{"year":2024,"cited_by_count":2}],"updated_date":"2026-08-01T09:00:35.917206","created_date":"2025-10-10T00:00:00"}
