{"id":"https://openalex.org/W4409759757","doi":"https://doi.org/10.1109/aciiw63320.2024.00013","title":"Safe Reinforcement Learning for Collaborative Robots in Dynamic Human Environments","display_name":"Safe Reinforcement Learning for Collaborative Robots in Dynamic Human Environments","publication_year":2024,"publication_date":"2024-09-15","ids":{"openalex":"https://openalex.org/W4409759757","doi":"https://doi.org/10.1109/aciiw63320.2024.00013"},"language":"en","primary_location":{"id":"doi:10.1109/aciiw63320.2024.00013","is_oa":false,"landing_page_url":"https://doi.org/10.1109/aciiw63320.2024.00013","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2024 12th International Conference on Affective Computing and Intelligent Interaction Workshops and Demos (ACIIW)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5117291123","display_name":"Sundas Rafat Mulkana","orcid":null},"institutions":[{"id":"https://openalex.org/I7882870","display_name":"University of Glasgow","ror":"https://ror.org/00vtgdb53","country_code":"GB","type":"education","lineage":["https://openalex.org/I7882870"]}],"countries":["GB"],"is_corresponding":true,"raw_author_name":"Sundas Rafat Mulkana","raw_affiliation_strings":["School of Computing Science, University of Glasgow"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"School of Computing Science, University of Glasgow","institution_ids":["https://openalex.org/I7882870"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":["https://openalex.org/A5117291123"],"corresponding_institution_ids":["https://openalex.org/I7882870"],"apc_list":null,"apc_paid":null,"fwci":1.5289,"has_fulltext":false,"cited_by_count":2,"citation_normalized_percentile":{"value":0.86172457,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":91,"max":97},"biblio":{"volume":null,"issue":null,"first_page":"61","last_page":"65"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10709","display_name":"Social Robot Interaction and HRI","score":0.9994000196456909,"subfield":{"id":"https://openalex.org/subfields/3207","display_name":"Social Psychology"},"field":{"id":"https://openalex.org/fields/32","display_name":"Psychology"},"domain":{"id":"https://openalex.org/domains/2","display_name":"Social Sciences"}},"topics":[{"id":"https://openalex.org/T10709","display_name":"Social Robot Interaction and HRI","score":0.9994000196456909,"subfield":{"id":"https://openalex.org/subfields/3207","display_name":"Social Psychology"},"field":{"id":"https://openalex.org/fields/32","display_name":"Psychology"},"domain":{"id":"https://openalex.org/domains/2","display_name":"Social Sciences"}},{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9991999864578247,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10653","display_name":"Robot Manipulation and Learning","score":0.9983999729156494,"subfield":{"id":"https://openalex.org/subfields/2207","display_name":"Control and Systems Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.8062137365341187},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.680427074432373},{"id":"https://openalex.org/keywords/robot","display_name":"Robot","score":0.6720410585403442},{"id":"https://openalex.org/keywords/human\u2013computer-interaction","display_name":"Human\u2013computer interaction","score":0.5870077610015869},{"id":"https://openalex.org/keywords/human\u2013robot-interaction","display_name":"Human\u2013robot interaction","score":0.4425595700740814},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.3388732075691223}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.8062137365341187},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.680427074432373},{"id":"https://openalex.org/C90509273","wikidata":"https://www.wikidata.org/wiki/Q11012","display_name":"Robot","level":2,"score":0.6720410585403442},{"id":"https://openalex.org/C107457646","wikidata":"https://www.wikidata.org/wiki/Q207434","display_name":"Human\u2013computer interaction","level":1,"score":0.5870077610015869},{"id":"https://openalex.org/C145460709","wikidata":"https://www.wikidata.org/wiki/Q859951","display_name":"Human\u2013robot interaction","level":3,"score":0.4425595700740814},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.3388732075691223}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/aciiw63320.2024.00013","is_oa":false,"landing_page_url":"https://doi.org/10.1109/aciiw63320.2024.00013","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2024 12th International Conference on Affective Computing and Intelligent Interaction Workshops and Demos (ACIIW)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":37,"referenced_works":["https://openalex.org/W2032568497","https://openalex.org/W2095899461","https://openalex.org/W2724918885","https://openalex.org/W2792441573","https://openalex.org/W2963293747","https://openalex.org/W2982551402","https://openalex.org/W2998275384","https://openalex.org/W3034226296","https://openalex.org/W3095669803","https://openalex.org/W3105064013","https://openalex.org/W3174521520","https://openalex.org/W3184813433","https://openalex.org/W3187113103","https://openalex.org/W3195968524","https://openalex.org/W3206921000","https://openalex.org/W4206903421","https://openalex.org/W4214717370","https://openalex.org/W4280525627","https://openalex.org/W4285102237","https://openalex.org/W4285102519","https://openalex.org/W4293370597","https://openalex.org/W4302287685","https://openalex.org/W4312433264","https://openalex.org/W4315606134","https://openalex.org/W4321392130","https://openalex.org/W4386051596","https://openalex.org/W4392163586","https://openalex.org/W4394018687","https://openalex.org/W4401417352","https://openalex.org/W6650105374","https://openalex.org/W6739585900","https://openalex.org/W6785204124","https://openalex.org/W6801971982","https://openalex.org/W6804456054","https://openalex.org/W6838633502","https://openalex.org/W6846211847","https://openalex.org/W6857551013"],"related_works":["https://openalex.org/W2978665606","https://openalex.org/W4287179229","https://openalex.org/W3205513966","https://openalex.org/W3120459843","https://openalex.org/W4366547574","https://openalex.org/W3200191727","https://openalex.org/W189465620","https://openalex.org/W4366818884","https://openalex.org/W2248308732","https://openalex.org/W4392651289"],"abstract_inverted_index":{"Collaborative":[0],"robots":[1,27,125,179],"trained":[2],"using":[3],"Reinforcement":[4],"Learning":[5],"(RL)":[6],"techniques":[7],"to":[8,40,81,88,108,118,133,157,171],"perform":[9],"complex":[10],"tasks":[11,46,97,113],"have":[12],"shown":[13],"promising":[14],"results":[15],"in":[16,32,43,52,94,126,143,180],"simulations":[17],"and":[18,49,63,70,102,183],"controlled":[19],"environments.":[20,185],"However,":[21],"the":[22,120,127,174],"actions":[23,78],"of":[24,38,123,167,176],"such":[25,59,98],"autonomous":[26],"are":[28],"not":[29,74],"always":[30],"predictable":[31],"real-world":[33],"settings.":[34],"This":[35,130],"poses":[36],"risks":[37],"injury":[39],"humans,":[41],"particularly":[42],"joint":[44,95,145],"action":[45,96,146],"where":[47],"human":[48],"robot":[50,141,160],"come":[51],"close":[53],"contact.":[54],"Conventional":[55],"safe":[56,82,121,135,140,177],"RL":[57,107,136],"models,":[58],"as":[60,99],"reward":[61],"shaping":[62],"constraint":[64],"RL,":[65,85],"prioritize":[66],"safety":[67,90],"during":[68],"learning":[69],"deployment,":[71],"but":[72],"do":[73],"guarantee":[75],"that":[76,138],"all":[77],"will":[79],"lead":[80],"states.":[83],"Shielded":[84],"which":[86],"seeks":[87],"provide":[89],"guarantees,":[91],"also":[92],"struggles":[93],"collaborative":[100,163,178],"assembly":[101],"object":[103],"handover.":[104],"Improving":[105],"shielded":[106],"account":[109],"for":[110,162],"close-contact":[111],"human-robot":[112,144,155],"presents":[114],"a":[115],"potential":[116],"solution":[117],"facilitate":[119],"application":[122],"RL-based":[124],"real":[128],"world.":[129],"research":[131,169],"aims":[132],"develop":[134],"models":[137,153],"ensure":[139],"motion":[142],"tasks.":[147,164],"Additionally,":[148],"it":[149],"investigates":[150],"how":[151],"these":[152],"affect":[154],"interaction":[156],"identify":[158],"human-preferred":[159],"motions":[161],"The":[165],"outcomes":[166],"this":[168],"aim":[170],"significantly":[172],"advance":[173],"integration":[175],"social,":[181],"healthcare,":[182],"industrial":[184]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":1}],"updated_date":"2026-07-29T14:22:42.915294","created_date":"2025-10-10T00:00:00"}
