{"id":"https://openalex.org/W4386126338","doi":"https://doi.org/10.1007/s10994-023-06368-z","title":"Cautious policy programming: exploiting KL regularization for monotonic policy improvement in reinforcement learning","display_name":"Cautious policy programming: exploiting KL regularization for monotonic policy improvement in reinforcement learning","publication_year":2023,"publication_date":"2023-08-24","ids":{"openalex":"https://openalex.org/W4386126338","doi":"https://doi.org/10.1007/s10994-023-06368-z"},"language":"en","primary_location":{"id":"doi:10.1007/s10994-023-06368-z","is_oa":true,"landing_page_url":"https://doi.org/10.1007/s10994-023-06368-z","pdf_url":"https://link.springer.com/content/pdf/10.1007/s10994-023-06368-z.pdf","source":{"id":"https://openalex.org/S62148650","display_name":"Machine Learning","issn_l":"0885-6125","issn":["0885-6125","1573-0565"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319900","host_organization_name":"Springer Science+Business Media","host_organization_lineage":["https://openalex.org/P4310319900","https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Science+Business Media","Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Machine Learning","raw_type":"journal-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"hybrid","oa_url":"https://link.springer.com/content/pdf/10.1007/s10994-023-06368-z.pdf","any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5021499919","display_name":"Lingwei Zhu","orcid":"https://orcid.org/0000-0002-9514-6760"},"institutions":[{"id":"https://openalex.org/I154425047","display_name":"University of Alberta","ror":"https://ror.org/0160cpw27","country_code":"CA","type":"education","lineage":["https://openalex.org/I154425047"]}],"countries":["CA"],"is_corresponding":true,"raw_author_name":"Lingwei Zhu","raw_affiliation_strings":["Department of Computing Science, University of Alberta, Edmonton, Alberta, T6G R23, Canada","Department of Computing Science, University of Alberta, Edmonton, Canada"],"raw_orcid":"https://orcid.org/0000-0002-9514-6760","affiliations":[{"raw_affiliation_string":"Department of Computing Science, University of Alberta, Edmonton, Alberta, T6G R23, Canada","institution_ids":["https://openalex.org/I154425047"]},{"raw_affiliation_string":"Department of Computing Science, University of Alberta, Edmonton, Canada","institution_ids":["https://openalex.org/I154425047"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5042074952","display_name":"Takamitsu Matsubara","orcid":"https://orcid.org/0000-0003-3545-4814"},"institutions":[{"id":"https://openalex.org/I75917431","display_name":"Nara Institute of Science and Technology","ror":"https://ror.org/05bhada84","country_code":"JP","type":"education","lineage":["https://openalex.org/I75917431"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Takamitsu Matsubara","raw_affiliation_strings":["Division of Information Science, Nara Institute of Science and Technology, Ikoma City, Nara, 630-0192, Japan","Division of Information Science, Nara Institute of Science and Technology, Ikoma City, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Division of Information Science, Nara Institute of Science and Technology, Ikoma City, Nara, 630-0192, Japan","institution_ids":["https://openalex.org/I75917431"]},{"raw_affiliation_string":"Division of Information Science, Nara Institute of Science and Technology, Ikoma City, Japan","institution_ids":["https://openalex.org/I75917431"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":2,"corresponding_author_ids":["https://openalex.org/A5021499919"],"corresponding_institution_ids":["https://openalex.org/I154425047"],"apc_list":{"value":2390,"currency":"EUR","value_usd":2990},"apc_paid":{"value":2390,"currency":"EUR","value_usd":2990},"fwci":0.1586,"has_fulltext":true,"cited_by_count":1,"citation_normalized_percentile":{"value":0.52715813,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":90,"max":94},"biblio":{"volume":"112","issue":"11","first_page":"4527","last_page":"4562"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11975","display_name":"Evolutionary Algorithms and Applications","score":0.9851999878883362,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12794","display_name":"Adaptive Dynamic Programming Control","score":0.9847999811172485,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.8270271420478821},{"id":"https://openalex.org/keywords/monotonic-function","display_name":"Monotonic function","score":0.6907098293304443},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6458886861801147},{"id":"https://openalex.org/keywords/mathematical-optimization","display_name":"Mathematical optimization","score":0.5798251628875732},{"id":"https://openalex.org/keywords/regularization","display_name":"Regularization (linguistics)","score":0.5446088910102844},{"id":"https://openalex.org/keywords/bellman-equation","display_name":"Bellman equation","score":0.5292213559150696},{"id":"https://openalex.org/keywords/entropy","display_name":"Entropy (arrow of time)","score":0.5216996073722839},{"id":"https://openalex.org/keywords/maximization","display_name":"Maximization","score":0.4968555271625519},{"id":"https://openalex.org/keywords/upper-and-lower-bounds","display_name":"Upper and lower bounds","score":0.41074150800704956},{"id":"https://openalex.org/keywords/exploit","display_name":"Exploit","score":0.4104551076889038},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.3753851056098938},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.258246511220932}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.8270271420478821},{"id":"https://openalex.org/C72169020","wikidata":"https://www.wikidata.org/wiki/Q194404","display_name":"Monotonic function","level":2,"score":0.6907098293304443},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6458886861801147},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.5798251628875732},{"id":"https://openalex.org/C2776135515","wikidata":"https://www.wikidata.org/wiki/Q17143721","display_name":"Regularization (linguistics)","level":2,"score":0.5446088910102844},{"id":"https://openalex.org/C14646407","wikidata":"https://www.wikidata.org/wiki/Q1430750","display_name":"Bellman equation","level":2,"score":0.5292213559150696},{"id":"https://openalex.org/C106301342","wikidata":"https://www.wikidata.org/wiki/Q4117933","display_name":"Entropy (arrow of time)","level":2,"score":0.5216996073722839},{"id":"https://openalex.org/C2776330181","wikidata":"https://www.wikidata.org/wiki/Q18358244","display_name":"Maximization","level":2,"score":0.4968555271625519},{"id":"https://openalex.org/C77553402","wikidata":"https://www.wikidata.org/wiki/Q13222579","display_name":"Upper and lower bounds","level":2,"score":0.41074150800704956},{"id":"https://openalex.org/C165696696","wikidata":"https://www.wikidata.org/wiki/Q11287","display_name":"Exploit","level":2,"score":0.4104551076889038},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.3753851056098938},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.258246511220932},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0},{"id":"https://openalex.org/C134306372","wikidata":"https://www.wikidata.org/wiki/Q7754","display_name":"Mathematical analysis","level":1,"score":0.0},{"id":"https://openalex.org/C38652104","wikidata":"https://www.wikidata.org/wiki/Q3510521","display_name":"Computer security","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1007/s10994-023-06368-z","is_oa":true,"landing_page_url":"https://doi.org/10.1007/s10994-023-06368-z","pdf_url":"https://link.springer.com/content/pdf/10.1007/s10994-023-06368-z.pdf","source":{"id":"https://openalex.org/S62148650","display_name":"Machine Learning","issn_l":"0885-6125","issn":["0885-6125","1573-0565"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319900","host_organization_name":"Springer Science+Business Media","host_organization_lineage":["https://openalex.org/P4310319900","https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Science+Business Media","Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Machine Learning","raw_type":"journal-article"}],"best_oa_location":{"id":"doi:10.1007/s10994-023-06368-z","is_oa":true,"landing_page_url":"https://doi.org/10.1007/s10994-023-06368-z","pdf_url":"https://link.springer.com/content/pdf/10.1007/s10994-023-06368-z.pdf","source":{"id":"https://openalex.org/S62148650","display_name":"Machine Learning","issn_l":"0885-6125","issn":["0885-6125","1573-0565"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319900","host_organization_name":"Springer Science+Business Media","host_organization_lineage":["https://openalex.org/P4310319900","https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Science+Business Media","Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Machine Learning","raw_type":"journal-article"},"sustainable_development_goals":[{"score":0.5400000214576721,"display_name":"Peace, Justice and strong institutions","id":"https://metadata.un.org/sdg/16"}],"awards":[{"id":"https://openalex.org/G6558811686","display_name":"Creation of a technological platform for robot reinforcement learning with safety and reliability","funder_award_id":"21H03522","funder_id":"https://openalex.org/F4320334764","funder_display_name":"Japan Society for the Promotion of Science"},{"id":"https://openalex.org/G8355167600","display_name":"application of scalable safe reinforcement learning to high-risk robotics","funder_award_id":"21J15633","funder_id":"https://openalex.org/F4320334764","funder_display_name":"Japan Society for the Promotion of Science"}],"funders":[{"id":"https://openalex.org/F4320334764","display_name":"Japan Society for the Promotion of Science","ror":"https://ror.org/00hhkn466"}],"has_content":{"pdf":true,"grobid_xml":false},"content_urls":{"pdf":"https://content.openalex.org/works/W4386126338.pdf"},"referenced_works_count":71,"referenced_works":["https://openalex.org/W786265215","https://openalex.org/W1514587017","https://openalex.org/W1575592356","https://openalex.org/W1889629917","https://openalex.org/W1984901446","https://openalex.org/W2115738253","https://openalex.org/W2121863487","https://openalex.org/W2124477018","https://openalex.org/W2142899754","https://openalex.org/W2145060720","https://openalex.org/W2145339207","https://openalex.org/W2150468603","https://openalex.org/W2260756217","https://openalex.org/W2334782222","https://openalex.org/W2367669080","https://openalex.org/W2377531001","https://openalex.org/W2397607997","https://openalex.org/W2402108766","https://openalex.org/W2518143623","https://openalex.org/W2593044849","https://openalex.org/W2594103415","https://openalex.org/W2604272474","https://openalex.org/W2612690371","https://openalex.org/W2626384641","https://openalex.org/W2727576081","https://openalex.org/W2730929966","https://openalex.org/W2751530711","https://openalex.org/W2763081248","https://openalex.org/W2781726626","https://openalex.org/W2787938642","https://openalex.org/W2811292668","https://openalex.org/W2900582619","https://openalex.org/W2903383131","https://openalex.org/W2921114252","https://openalex.org/W2949676527","https://openalex.org/W2964986650","https://openalex.org/W2965050397","https://openalex.org/W2972166034","https://openalex.org/W2990138404","https://openalex.org/W2990747716","https://openalex.org/W2998538471","https://openalex.org/W3004881034","https://openalex.org/W3014137283","https://openalex.org/W3024652448","https://openalex.org/W3032398409","https://openalex.org/W3037603015","https://openalex.org/W3039845099","https://openalex.org/W3102159535","https://openalex.org/W3168986090","https://openalex.org/W3195641273","https://openalex.org/W3216772467","https://openalex.org/W3217314940","https://openalex.org/W4205182066","https://openalex.org/W4250589301","https://openalex.org/W4307347247","https://openalex.org/W6604666559","https://openalex.org/W6630551017","https://openalex.org/W6631190155","https://openalex.org/W6638018090","https://openalex.org/W6674995601","https://openalex.org/W6676516282","https://openalex.org/W6676898700","https://openalex.org/W6677916085","https://openalex.org/W6682330108","https://openalex.org/W6740092555","https://openalex.org/W6743613440","https://openalex.org/W6751495484","https://openalex.org/W6769081937","https://openalex.org/W6774591755","https://openalex.org/W6780995382","https://openalex.org/W6903351479"],"related_works":["https://openalex.org/W4380682190","https://openalex.org/W4315701745","https://openalex.org/W1990290471","https://openalex.org/W2945307361","https://openalex.org/W2102386043","https://openalex.org/W2005710836","https://openalex.org/W2386410636","https://openalex.org/W3038962357","https://openalex.org/W2025663273","https://openalex.org/W3099153698"],"abstract_inverted_index":{"Abstract":[0],"In":[1],"this":[2,63],"paper,":[3],"we":[4,34,89],"propose":[5,91],"cautious":[6],"policy":[7,23,42,49,75,79],"programming":[8],"(CPP),":[9],"a":[10,36,67,74,92],"novel":[11,93],"value-based":[12],"reinforcement":[13],"learning":[14],"(RL)":[15],"algorithm":[16,111],"that":[17,44,85,96,108],"exploits":[18],"the":[19,29,47,71,109],"idea":[20],"of":[21,31,41,73],"monotonic":[22],"improvement":[24,43],"during":[25],"learning.":[26],"Based":[27],"on":[28,46,54],"nature":[30],"entropy-regularized":[32],"RL,":[33],"derive":[35],"new":[37],"entropy-regularization-aware":[38],"lower":[39,64],"bound":[40,65],"depends":[45],"expected":[48],"advantage":[50],"function":[51],"but":[52],"not":[53],"state-action-space-wise":[55],"maximization":[56],"as":[57,66],"in":[58,101,118],"prior":[59],"work.":[60],"CPP":[61,98],"leverages":[62],"criterion":[68],"for":[69,77],"adjusting":[70],"degree":[72],"update":[76],"alleviating":[78],"oscillation.":[80],"Different":[81],"from":[82],"similar":[83],"algorithms":[84],"are":[86],"mostly":[87],"theory-oriented,":[88],"also":[90],"interpolation":[94],"scheme":[95],"makes":[97],"better":[99],"scale":[100],"high":[102],"dimensional":[103],"control":[104,122],"problems.":[105],"We":[106],"demonstrate":[107],"proposed":[110],"can":[112],"trade":[113],"off":[114],"performance":[115],"and":[116,124],"stability":[117],"both":[119],"didactic":[120],"classic":[121],"problems":[123],"challenging":[125],"high-dimensional":[126],"Atari":[127],"games.":[128]},"counts_by_year":[{"year":2024,"cited_by_count":1}],"updated_date":"2026-06-13T06:13:01.061226","created_date":"2025-10-10T00:00:00"}
