{"id":"https://openalex.org/W4412459491","doi":"https://doi.org/10.1016/j.neunet.2025.107870","title":"Experimental data-efficient reinforcement learning with an ensemble of surrogate models","display_name":"Experimental data-efficient reinforcement learning with an ensemble of surrogate models","publication_year":2025,"publication_date":"2025-07-16","ids":{"openalex":"https://openalex.org/W4412459491","doi":"https://doi.org/10.1016/j.neunet.2025.107870","pmid":"https://pubmed.ncbi.nlm.nih.gov/40694893"},"language":"en","primary_location":{"id":"doi:10.1016/j.neunet.2025.107870","is_oa":true,"landing_page_url":"https://doi.org/10.1016/j.neunet.2025.107870","pdf_url":null,"source":{"id":"https://openalex.org/S123019304","display_name":"Neural Networks","issn_l":"0893-6080","issn":["0893-6080","1879-2782"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310320990","host_organization_name":"Elsevier BV","host_organization_lineage":["https://openalex.org/P4310320990"],"host_organization_lineage_names":["Elsevier BV"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Neural Networks","raw_type":"journal-article"},"type":"article","indexed_in":["crossref","pubmed"],"open_access":{"is_oa":true,"oa_status":"hybrid","oa_url":"https://doi.org/10.1016/j.neunet.2025.107870","any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5086467288","display_name":"Jiazhou Jiang","orcid":null},"institutions":[{"id":"https://openalex.org/I78757542","display_name":"University of Newcastle Australia","ror":"https://ror.org/00eae9z71","country_code":"AU","type":"education","lineage":["https://openalex.org/I78757542"]}],"countries":["AU"],"is_corresponding":false,"raw_author_name":"Jiazhou Jiang","raw_affiliation_strings":["School of Engineering, University of Newcastle, Callaghan, NSW, 2308, Australia. Electronic address: jiazhou.jiang@uon.edu.au"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"School of Engineering, University of Newcastle, Callaghan, NSW, 2308, Australia. Electronic address: jiazhou.jiang@uon.edu.au","institution_ids":["https://openalex.org/I78757542"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5100370053","display_name":"Zhiyong Chen","orcid":"https://orcid.org/0000-0002-2033-4249"},"institutions":[{"id":"https://openalex.org/I78757542","display_name":"University of Newcastle Australia","ror":"https://ror.org/00eae9z71","country_code":"AU","type":"education","lineage":["https://openalex.org/I78757542"]}],"countries":["AU"],"is_corresponding":true,"raw_author_name":"Zhiyong Chen","raw_affiliation_strings":["School of Engineering, University of Newcastle, Callaghan, NSW, 2308, Australia. Electronic address: zhiyong.chen@newcastle.edu.au"],"raw_orcid":"https://orcid.org/0000-0002-2033-4249","affiliations":[{"raw_affiliation_string":"School of Engineering, University of Newcastle, Callaghan, NSW, 2308, Australia. Electronic address: zhiyong.chen@newcastle.edu.au","institution_ids":["https://openalex.org/I78757542"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":["https://openalex.org/A5100370053"],"corresponding_institution_ids":["https://openalex.org/I78757542"],"apc_list":{"value":3350,"currency":"USD","value_usd":3350},"apc_paid":{"value":3350,"currency":"USD","value_usd":3350},"fwci":5.584,"has_fulltext":false,"cited_by_count":4,"citation_normalized_percentile":{"value":0.95619207,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":95,"max":98},"biblio":{"volume":"192","issue":null,"first_page":"107870","last_page":"107870"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9995999932289124,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9995999932289124,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11975","display_name":"Evolutionary Algorithms and Applications","score":0.9961000084877014,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10848","display_name":"Advanced Multi-Objective Optimization Algorithms","score":0.9954000115394592,"subfield":{"id":"https://openalex.org/subfields/1703","display_name":"Computational Theory and Mathematics"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.6894867420196533},{"id":"https://openalex.org/keywords/surrogate-model","display_name":"Surrogate model","score":0.6286120414733887},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.6258562207221985},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6078546047210693},{"id":"https://openalex.org/keywords/ensemble-learning","display_name":"Ensemble learning","score":0.5675658583641052},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.5468853712081909}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.6894867420196533},{"id":"https://openalex.org/C131675550","wikidata":"https://www.wikidata.org/wiki/Q7646884","display_name":"Surrogate model","level":2,"score":0.6286120414733887},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.6258562207221985},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6078546047210693},{"id":"https://openalex.org/C45942800","wikidata":"https://www.wikidata.org/wiki/Q245652","display_name":"Ensemble learning","level":2,"score":0.5675658583641052},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.5468853712081909}],"mesh":[{"descriptor_ui":"D000069550","descriptor_name":"Machine Learning","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true},{"descriptor_ui":"D000069550","descriptor_name":"Machine Learning","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true},{"descriptor_ui":"D000465","descriptor_name":"Algorithms","qualifier_ui":null,"qualifier_name":null,"is_major_topic":false},{"descriptor_ui":"D000465","descriptor_name":"Algorithms","qualifier_ui":null,"qualifier_name":null,"is_major_topic":false},{"descriptor_ui":"D006801","descriptor_name":"Humans","qualifier_ui":null,"qualifier_name":null,"is_major_topic":false},{"descriptor_ui":"D006801","descriptor_name":"Humans","qualifier_ui":null,"qualifier_name":null,"is_major_topic":false},{"descriptor_ui":"D012054","descriptor_name":"Reinforcement, Psychology","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true},{"descriptor_ui":"D012054","descriptor_name":"Reinforcement, Psychology","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true},{"descriptor_ui":"D016571","descriptor_name":"Neural Networks, Computer","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true},{"descriptor_ui":"D016571","descriptor_name":"Neural Networks, Computer","qualifier_ui":null,"qualifier_name":null,"is_major_topic":true}],"locations_count":2,"locations":[{"id":"doi:10.1016/j.neunet.2025.107870","is_oa":true,"landing_page_url":"https://doi.org/10.1016/j.neunet.2025.107870","pdf_url":null,"source":{"id":"https://openalex.org/S123019304","display_name":"Neural Networks","issn_l":"0893-6080","issn":["0893-6080","1879-2782"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310320990","host_organization_name":"Elsevier BV","host_organization_lineage":["https://openalex.org/P4310320990"],"host_organization_lineage_names":["Elsevier BV"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Neural Networks","raw_type":"journal-article"},{"id":"pmid:40694893","is_oa":false,"landing_page_url":"https://pubmed.ncbi.nlm.nih.gov/40694893","pdf_url":null,"source":{"id":"https://openalex.org/S4306525036","display_name":"PubMed","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I1299303238","host_organization_name":"National Institutes of Health","host_organization_lineage":["https://openalex.org/I1299303238"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Neural networks : the official journal of the International Neural Network Society","raw_type":null}],"best_oa_location":{"id":"doi:10.1016/j.neunet.2025.107870","is_oa":true,"landing_page_url":"https://doi.org/10.1016/j.neunet.2025.107870","pdf_url":null,"source":{"id":"https://openalex.org/S123019304","display_name":"Neural Networks","issn_l":"0893-6080","issn":["0893-6080","1879-2782"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310320990","host_organization_name":"Elsevier BV","host_organization_lineage":["https://openalex.org/P4310320990"],"host_organization_lineage_names":["Elsevier BV"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Neural Networks","raw_type":"journal-article"},"sustainable_development_goals":[{"score":0.46000000834465027,"display_name":"Reduced inequalities","id":"https://metadata.un.org/sdg/10"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":54,"referenced_works":["https://openalex.org/W834081922","https://openalex.org/W1979769287","https://openalex.org/W2068823081","https://openalex.org/W2155007355","https://openalex.org/W2257979135","https://openalex.org/W2766046980","https://openalex.org/W2890208753","https://openalex.org/W2911918032","https://openalex.org/W2963317745","https://openalex.org/W2988142477","https://openalex.org/W2989847975","https://openalex.org/W3025606523","https://openalex.org/W3028766998","https://openalex.org/W3033909669","https://openalex.org/W3092490845","https://openalex.org/W3118210634","https://openalex.org/W3149322120","https://openalex.org/W3155729845","https://openalex.org/W3198149942","https://openalex.org/W3201700917","https://openalex.org/W3206768895","https://openalex.org/W4283751222","https://openalex.org/W4286377487","https://openalex.org/W4288319859","https://openalex.org/W4298206671","https://openalex.org/W4299828299","https://openalex.org/W4300799055","https://openalex.org/W4365806381","https://openalex.org/W4385764347","https://openalex.org/W4401522318","https://openalex.org/W6623316541","https://openalex.org/W6680657880","https://openalex.org/W6682849425","https://openalex.org/W6683153233","https://openalex.org/W6696324988","https://openalex.org/W6740801417","https://openalex.org/W6742380005","https://openalex.org/W6747473740","https://openalex.org/W6748839928","https://openalex.org/W6751494529","https://openalex.org/W6754184789","https://openalex.org/W6754471908","https://openalex.org/W6756256016","https://openalex.org/W6764053384","https://openalex.org/W6777091672","https://openalex.org/W6777656069","https://openalex.org/W6783909840","https://openalex.org/W6791413555","https://openalex.org/W6797408030","https://openalex.org/W6797410853","https://openalex.org/W6810556818","https://openalex.org/W6811195552","https://openalex.org/W6851826667","https://openalex.org/W7046505727"],"related_works":["https://openalex.org/W2961085424","https://openalex.org/W4306674287","https://openalex.org/W4387369504","https://openalex.org/W4394896187","https://openalex.org/W3170094116","https://openalex.org/W4386462264","https://openalex.org/W3107602296","https://openalex.org/W4364306694","https://openalex.org/W4312192474","https://openalex.org/W4283697347"],"abstract_inverted_index":{"Model-based":[0],"reinforcement":[1,86,153],"learning":[2,87,154],"methods":[3],"enhance":[4],"sample":[5],"efficiency":[6],"by":[7,72],"generating":[8],"synthetic":[9,95],"data":[10,148],"during":[11],"training.":[12],"However,":[13],"modeling":[14],"errors":[15],"can":[16],"undermine":[17],"training,":[18],"leading":[19],"to":[20,24,30,52,126],"failures":[21],"when":[22,135],"applied":[23],"the":[25,33,54,94,99,108,121,146],"actual":[26],"environment,":[27],"especially":[28],"due":[29],"discrepancies":[31],"in":[32,120,137],"learned":[34],"dynamics.":[35],"In":[36],"this":[37],"paper,":[38],"we":[39],"propose":[40],"a":[41],"new":[42],"ensemble":[43],"of":[44,107,145],"double":[45,109],"surrogate":[46,110],"models":[47,75,83,111],"created":[48],"using":[49],"symbolic":[50],"regression":[51,68],"uncover":[53],"fundamental":[55],"physical":[56],"principles":[57],"governing":[58],"system":[59],"behavior,":[60],"thereby":[61],"enabling":[62],"more":[63],"data-efficient":[64],"real-world":[65],"applications.":[66],"Symbolic":[67],"enhances":[69],"model":[70,113],"interpretability":[71],"producing":[73],"straightforward":[74],"that":[76,123],"generalize":[77],"well":[78],"with":[79,85,89],"limited":[80],"data.":[81,104],"These":[82],"interact":[84],"algorithms,":[88],"interactions":[90],"occurring":[91],"solely":[92],"within":[93],"models,":[96],"significantly":[97],"reducing":[98],"need":[100],"for":[101,151],"real":[102,138],"experimental":[103,147],"The":[105],"structure":[106],"mitigates":[112],"bias,":[114],"preventing":[115],"agents":[116],"from":[117],"exploiting":[118],"inaccuracies":[119],"environment":[122],"could":[124],"lead":[125],"poor":[127],"performance.":[128],"Our":[129],"approach":[130],"demonstrates":[131],"comparable":[132],"training":[133],"performance":[134],"validated":[136],"environments,":[139],"requiring":[140],"less":[141],"than":[142],"1":[143],"%":[144],"typically":[149],"needed":[150],"conventional":[152],"algorithms.":[155]},"counts_by_year":[{"year":2026,"cited_by_count":2},{"year":2025,"cited_by_count":2}],"updated_date":"2026-01-20T17:24:06.736184","created_date":"2025-10-10T00:00:00"}
