{"id":"https://openalex.org/W2909661184","doi":"https://doi.org/10.1109/iros.2018.8593881","title":"Reinforcement Learning with Symbolic Input-Output Models","display_name":"Reinforcement Learning with Symbolic Input-Output Models","publication_year":2018,"publication_date":"2018-10-01","ids":{"openalex":"https://openalex.org/W2909661184","doi":"https://doi.org/10.1109/iros.2018.8593881","mag":"2909661184"},"language":"en","primary_location":{"id":"doi:10.1109/iros.2018.8593881","is_oa":false,"landing_page_url":"https://doi.org/10.1109/iros.2018.8593881","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2018 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5033348430","display_name":"Erik Derner","orcid":"https://orcid.org/0000-0002-7588-7668"},"institutions":[{"id":"https://openalex.org/I44504214","display_name":"Czech Technical University in Prague","ror":"https://ror.org/03kqpb082","country_code":"CZ","type":"education","lineage":["https://openalex.org/I44504214"]}],"countries":["CZ"],"is_corresponding":false,"raw_author_name":"Erik Derner","raw_affiliation_strings":["Department of Control Engineering, Czech Technical University in Prague, Czech Republic"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Control Engineering, Czech Technical University in Prague, Czech Republic","institution_ids":["https://openalex.org/I44504214"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5004892921","display_name":"Ji\u0159\u0131\u0301 Kubal\u0131\u0301k","orcid":"https://orcid.org/0000-0002-6965-6142"},"institutions":[{"id":"https://openalex.org/I44504214","display_name":"Czech Technical University in Prague","ror":"https://ror.org/03kqpb082","country_code":"CZ","type":"education","lineage":["https://openalex.org/I44504214"]}],"countries":["CZ"],"is_corresponding":false,"raw_author_name":"Jiri Kubalik","raw_affiliation_strings":["Czech Institute of Informatics, Robotics and Cybernetics, Czech Technical University in Prague, Czech Republic"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Czech Institute of Informatics, Robotics and Cybernetics, Czech Technical University in Prague, Czech Republic","institution_ids":["https://openalex.org/I44504214"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5084264842","display_name":"Robert Babu\u0161ka","orcid":"https://orcid.org/0000-0001-9578-8598"},"institutions":[{"id":"https://openalex.org/I98358874","display_name":"Delft University of Technology","ror":"https://ror.org/02e2c7k09","country_code":"NL","type":"education","lineage":["https://openalex.org/I98358874"]}],"countries":["NL"],"is_corresponding":false,"raw_author_name":"Robert Babuska","raw_affiliation_strings":["Cognitive Robotics, Delft University of Technology, The Netherlands"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Cognitive Robotics, Delft University of Technology, The Netherlands","institution_ids":["https://openalex.org/I98358874"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":11,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9973999857902527,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9973999857902527,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10791","display_name":"Advanced Control Systems Optimization","score":0.9944999814033508,"subfield":{"id":"https://openalex.org/subfields/2207","display_name":"Control and Systems Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10876","display_name":"Fault Detection and Control Systems","score":0.9926999807357788,"subfield":{"id":"https://openalex.org/subfields/2207","display_name":"Control and Systems Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.7869130373001099},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6946707367897034},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.44297415018081665},{"id":"https://openalex.org/keywords/reinforcement","display_name":"Reinforcement","score":0.42639997601509094},{"id":"https://openalex.org/keywords/psychology","display_name":"Psychology","score":0.09952875971794128},{"id":"https://openalex.org/keywords/social-psychology","display_name":"Social psychology","score":0.06088072061538696}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.7869130373001099},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6946707367897034},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.44297415018081665},{"id":"https://openalex.org/C67203356","wikidata":"https://www.wikidata.org/wiki/Q1321905","display_name":"Reinforcement","level":2,"score":0.42639997601509094},{"id":"https://openalex.org/C15744967","wikidata":"https://www.wikidata.org/wiki/Q9418","display_name":"Psychology","level":0,"score":0.09952875971794128},{"id":"https://openalex.org/C77805123","wikidata":"https://www.wikidata.org/wiki/Q161272","display_name":"Social psychology","level":1,"score":0.06088072061538696}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/iros.2018.8593881","is_oa":false,"landing_page_url":"https://doi.org/10.1109/iros.2018.8593881","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2018 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[{"display_name":"Peace, Justice and strong institutions","score":0.5099999904632568,"id":"https://metadata.un.org/sdg/16"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":22,"referenced_works":["https://openalex.org/W197747737","https://openalex.org/W199153045","https://openalex.org/W1552830313","https://openalex.org/W1689445748","https://openalex.org/W1757796397","https://openalex.org/W1966086707","https://openalex.org/W1968962398","https://openalex.org/W2108051346","https://openalex.org/W2120346334","https://openalex.org/W2121863487","https://openalex.org/W2140135625","https://openalex.org/W2145339207","https://openalex.org/W2173248099","https://openalex.org/W2766046980","https://openalex.org/W2769736739","https://openalex.org/W2790924949","https://openalex.org/W2892053860","https://openalex.org/W2963864421","https://openalex.org/W4214717370","https://openalex.org/W4298857966","https://openalex.org/W6677737365","https://openalex.org/W6680657880"],"related_works":["https://openalex.org/W4210912933","https://openalex.org/W1562959674","https://openalex.org/W2923653485","https://openalex.org/W2952472710","https://openalex.org/W4255994452","https://openalex.org/W4206669594","https://openalex.org/W3037422413","https://openalex.org/W2959276766","https://openalex.org/W2794820824","https://openalex.org/W4226176818"],"abstract_inverted_index":{"It":[0],"is":[1,19,49,94],"well":[2],"known":[3],"that":[4,150],"reinforcement":[5],"learning":[6,46],"(RL)":[7],"can":[8,110,154],"benefit":[9],"from":[10,26,70,119],"the":[11,27,38,54,59,66,71,84,101,127],"use":[12,42,80],"of":[13,57,83,122],"a":[14,133,137,143,157,163],"dynamic":[15],"prediction":[16],"model":[17,167],"which":[18],"learned":[20],"on":[21,65,129,142,162],"data":[22],"samples":[23],"collected":[24],"online":[25],"process":[28,166],"to":[29,79,96,106],"be":[30,63],"controlled.":[31],"Most":[32],"RL":[33],"algorithms":[34],"are":[35],"formulated":[36],"in":[37,53],"state-space":[39,43,47],"domain":[40],"and":[41,100,114,136,141,168],"models.":[44],"However,":[45],"models":[48,82,99,113],"difficult,":[50],"mainly":[51],"because":[52],"vast":[55],"majority":[56],"problems":[58],"full":[60],"state":[61],"cannot":[62],"measured":[64],"system":[67],"or":[68],"reconstructed":[69],"measurements.":[72],"To":[73],"circumvent":[74],"this":[75,107],"limitation,":[76],"we":[77,109],"propose":[78],"input-output":[81,165],"NARX":[85],"(nonlinear":[86],"autoregressive":[87],"with":[88],"exogenous":[89],"input)":[90],"type.":[91],"Symbolic":[92],"regression":[93],"employed":[95],"construct":[97],"parsimonious":[98],"corresponding":[102],"value":[103,169],"functions.":[104],"Thanks":[105],"approach,":[108],"learn":[111],"accurate":[112],"compute":[115],"optimal":[116],"policies":[117],"even":[118],"small":[120],"amounts":[121],"training":[123],"data.":[124],"We":[125],"demonstrate":[126],"approach":[128],"two":[130],"simulated":[131],"examples,":[132],"hopping":[134],"robot":[135,139],"1-DOF":[138],"arm,":[140],"real":[144],"inverted":[145],"pendulum":[146],"system.":[147],"Results":[148],"show":[149],"our":[151],"proposed":[152],"method":[153],"reliably":[155],"determine":[156],"good":[158],"control":[159],"policy":[160],"based":[161],"symbolic":[164],"function.":[170]},"counts_by_year":[{"year":2022,"cited_by_count":1},{"year":2021,"cited_by_count":5},{"year":2020,"cited_by_count":4},{"year":2019,"cited_by_count":1}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
