{"id":"https://openalex.org/W1997350370","doi":"https://doi.org/10.1080/08839510802170538","title":"REINFORCEMENT LEARNING FOR POMDP USING STATE CLASSIFICATION","display_name":"REINFORCEMENT LEARNING FOR POMDP USING STATE CLASSIFICATION","publication_year":2008,"publication_date":"2008-08-19","ids":{"openalex":"https://openalex.org/W1997350370","doi":"https://doi.org/10.1080/08839510802170538","mag":"1997350370"},"language":"en","primary_location":{"id":"doi:10.1080/08839510802170538","is_oa":true,"landing_page_url":"https://doi.org/10.1080/08839510802170538","pdf_url":"https://www.tandfonline.com/doi/pdf/10.1080/08839510802170538?download=true","source":{"id":"https://openalex.org/S125501549","display_name":"Applied Artificial Intelligence","issn_l":"0883-9514","issn":["0883-9514","1087-6545"],"is_oa":false,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310320547","host_organization_name":"Taylor & Francis","host_organization_lineage":["https://openalex.org/P4310320547"],"host_organization_lineage_names":["Taylor & Francis"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Applied Artificial Intelligence","raw_type":"journal-article"},"type":"article","indexed_in":["crossref","doaj"],"open_access":{"is_oa":true,"oa_status":"bronze","oa_url":"https://www.tandfonline.com/doi/pdf/10.1080/08839510802170538?download=true","any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5113476798","display_name":"Le Tien Dung","orcid":null},"institutions":[{"id":"https://openalex.org/I171481255","display_name":"Shibaura Institute of Technology","ror":"https://ror.org/020wjcq07","country_code":"JP","type":"education","lineage":["https://openalex.org/I171481255"]}],"countries":["JP"],"is_corresponding":true,"raw_author_name":"Le Tien Dung","raw_affiliation_strings":["Graduate School of Engineering, Shibaura Institute of Technology, Saitama, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Graduate School of Engineering, Shibaura Institute of Technology, Saitama, Japan","institution_ids":["https://openalex.org/I171481255"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5026428065","display_name":"Takashi Komeda","orcid":null},"institutions":[{"id":"https://openalex.org/I171481255","display_name":"Shibaura Institute of Technology","ror":"https://ror.org/020wjcq07","country_code":"JP","type":"education","lineage":["https://openalex.org/I171481255"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Takashi Komeda","raw_affiliation_strings":["Faculty of System Engineering, Shibaura Institute of Technology, Saitama, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Faculty of System Engineering, Shibaura Institute of Technology, Saitama, Japan","institution_ids":["https://openalex.org/I171481255"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5105359615","display_name":"Motoki Takagi","orcid":"https://orcid.org/0000-0002-3534-2735"},"institutions":[{"id":"https://openalex.org/I171481255","display_name":"Shibaura Institute of Technology","ror":"https://ror.org/020wjcq07","country_code":"JP","type":"education","lineage":["https://openalex.org/I171481255"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Motoki Takagi","raw_affiliation_strings":["Faculty of System Engineering, Shibaura Institute of Technology, Saitama, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Faculty of System Engineering, Shibaura Institute of Technology, Saitama, Japan","institution_ids":["https://openalex.org/I171481255"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":["https://openalex.org/A5113476798"],"corresponding_institution_ids":["https://openalex.org/I171481255"],"apc_list":{"value":2195,"currency":"USD","value_usd":2195},"apc_paid":null,"fwci":3.3602,"has_fulltext":true,"cited_by_count":22,"citation_normalized_percentile":{"value":0.9249433,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":89,"max":98},"biblio":{"volume":"22","issue":"7-8","first_page":"761","last_page":"779"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10320","display_name":"Neural Networks and Applications","score":0.9976000189781189,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10320","display_name":"Neural Networks and Applications","score":0.9976000189781189,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9965000152587891,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12761","display_name":"Data Stream Mining Techniques","score":0.9955999851226807,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.8960533142089844},{"id":"https://openalex.org/keywords/partially-observable-markov-decision-process","display_name":"Partially observable Markov decision process","score":0.8858222961425781},{"id":"https://openalex.org/keywords/markov-decision-process","display_name":"Markov decision process","score":0.8565970659255981},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.8369530439376831},{"id":"https://openalex.org/keywords/observable","display_name":"Observable","score":0.6827753782272339},{"id":"https://openalex.org/keywords/recurrent-neural-network","display_name":"Recurrent neural network","score":0.6302703022956848},{"id":"https://openalex.org/keywords/state-space","display_name":"State space","score":0.5963284373283386},{"id":"https://openalex.org/keywords/q-learning","display_name":"Q-learning","score":0.5739333033561707},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.5718308091163635},{"id":"https://openalex.org/keywords/state","display_name":"State (computer science)","score":0.5266651511192322},{"id":"https://openalex.org/keywords/markov-process","display_name":"Markov process","score":0.4119079113006592},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.40788882970809937},{"id":"https://openalex.org/keywords/markov-chain","display_name":"Markov chain","score":0.3681352436542511},{"id":"https://openalex.org/keywords/mathematical-optimization","display_name":"Mathematical optimization","score":0.34453582763671875},{"id":"https://openalex.org/keywords/artificial-neural-network","display_name":"Artificial neural network","score":0.27695077657699585},{"id":"https://openalex.org/keywords/markov-model","display_name":"Markov model","score":0.2727193236351013},{"id":"https://openalex.org/keywords/algorithm","display_name":"Algorithm","score":0.16877275705337524},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.09786608815193176},{"id":"https://openalex.org/keywords/statistics","display_name":"Statistics","score":0.06958097219467163}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.8960533142089844},{"id":"https://openalex.org/C17098449","wikidata":"https://www.wikidata.org/wiki/Q176814","display_name":"Partially observable Markov decision process","level":4,"score":0.8858222961425781},{"id":"https://openalex.org/C106189395","wikidata":"https://www.wikidata.org/wiki/Q176789","display_name":"Markov decision process","level":3,"score":0.8565970659255981},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.8369530439376831},{"id":"https://openalex.org/C32848918","wikidata":"https://www.wikidata.org/wiki/Q845789","display_name":"Observable","level":2,"score":0.6827753782272339},{"id":"https://openalex.org/C147168706","wikidata":"https://www.wikidata.org/wiki/Q1457734","display_name":"Recurrent neural network","level":3,"score":0.6302703022956848},{"id":"https://openalex.org/C72434380","wikidata":"https://www.wikidata.org/wiki/Q230930","display_name":"State space","level":2,"score":0.5963284373283386},{"id":"https://openalex.org/C188116033","wikidata":"https://www.wikidata.org/wiki/Q2664563","display_name":"Q-learning","level":3,"score":0.5739333033561707},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.5718308091163635},{"id":"https://openalex.org/C48103436","wikidata":"https://www.wikidata.org/wiki/Q599031","display_name":"State (computer science)","level":2,"score":0.5266651511192322},{"id":"https://openalex.org/C159886148","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov process","level":2,"score":0.4119079113006592},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.40788882970809937},{"id":"https://openalex.org/C98763669","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov chain","level":2,"score":0.3681352436542511},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.34453582763671875},{"id":"https://openalex.org/C50644808","wikidata":"https://www.wikidata.org/wiki/Q192776","display_name":"Artificial neural network","level":2,"score":0.27695077657699585},{"id":"https://openalex.org/C163836022","wikidata":"https://www.wikidata.org/wiki/Q6771326","display_name":"Markov model","level":3,"score":0.2727193236351013},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.16877275705337524},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.09786608815193176},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.06958097219467163},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1080/08839510802170538","is_oa":true,"landing_page_url":"https://doi.org/10.1080/08839510802170538","pdf_url":"https://www.tandfonline.com/doi/pdf/10.1080/08839510802170538?download=true","source":{"id":"https://openalex.org/S125501549","display_name":"Applied Artificial Intelligence","issn_l":"0883-9514","issn":["0883-9514","1087-6545"],"is_oa":false,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310320547","host_organization_name":"Taylor & Francis","host_organization_lineage":["https://openalex.org/P4310320547"],"host_organization_lineage_names":["Taylor & Francis"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Applied Artificial Intelligence","raw_type":"journal-article"}],"best_oa_location":{"id":"doi:10.1080/08839510802170538","is_oa":true,"landing_page_url":"https://doi.org/10.1080/08839510802170538","pdf_url":"https://www.tandfonline.com/doi/pdf/10.1080/08839510802170538?download=true","source":{"id":"https://openalex.org/S125501549","display_name":"Applied Artificial Intelligence","issn_l":"0883-9514","issn":["0883-9514","1087-6545"],"is_oa":false,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310320547","host_organization_name":"Taylor & Francis","host_organization_lineage":["https://openalex.org/P4310320547"],"host_organization_lineage_names":["Taylor & Francis"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Applied Artificial Intelligence","raw_type":"journal-article"},"sustainable_development_goals":[{"score":0.800000011920929,"display_name":"Peace, Justice and strong institutions","id":"https://metadata.un.org/sdg/16"}],"awards":[],"funders":[],"has_content":{"pdf":true,"grobid_xml":true},"content_urls":{"pdf":"https://content.openalex.org/works/W1997350370.pdf","grobid_xml":"https://content.openalex.org/works/W1997350370.grobid-xml"},"referenced_works_count":33,"referenced_works":["https://openalex.org/W1492414751","https://openalex.org/W1499371387","https://openalex.org/W1508856789","https://openalex.org/W1515851193","https://openalex.org/W1570106308","https://openalex.org/W1570690983","https://openalex.org/W1587318060","https://openalex.org/W1595483645","https://openalex.org/W1687873425","https://openalex.org/W1981276685","https://openalex.org/W2048984163","https://openalex.org/W2064675550","https://openalex.org/W2096533821","https://openalex.org/W2107726111","https://openalex.org/W2115121720","https://openalex.org/W2116165835","https://openalex.org/W2121863487","https://openalex.org/W2123663688","https://openalex.org/W2136848157","https://openalex.org/W2144357723","https://openalex.org/W2147107577","https://openalex.org/W2147568880","https://openalex.org/W2159204968","https://openalex.org/W2171625236","https://openalex.org/W2172246523","https://openalex.org/W2913668833","https://openalex.org/W2914656440","https://openalex.org/W2999905431","https://openalex.org/W3011120880","https://openalex.org/W4212774754","https://openalex.org/W4214717370","https://openalex.org/W4234266784","https://openalex.org/W4285719527"],"related_works":["https://openalex.org/W2096013579","https://openalex.org/W1589140671","https://openalex.org/W1760611253","https://openalex.org/W52153049","https://openalex.org/W2951545791","https://openalex.org/W1515117609","https://openalex.org/W2294884454","https://openalex.org/W2364406457","https://openalex.org/W2173087131","https://openalex.org/W1997350370"],"abstract_inverted_index":{"Reinforcement":[0],"learning":[1,17,46,74,144],"(RL)":[2],"has":[3],"been":[4],"widely":[5],"used":[6,40,103,115],"to":[7,41,64,104,116,138,147],"solve":[8,19],"problems":[9,50,129],"with":[10,142],"a":[11,33,57,66,72,98,140,152],"little":[12],"feedback":[13],"from":[14],"environment.":[15],"Q":[16,43,99],"can":[18,38],"Markov":[20,29],"decision":[21,30],"processes":[22,31],"(MDPs)":[23],"quite":[24],"well.":[25],"For":[26],"partially":[27],"observable":[28,90,109],"(POMDPs),":[32],"recurrent":[34],"neural":[35],"network":[36],"(RNN)":[37],"be":[39],"approximate":[42,117],"values.":[44],"However,":[45],"time":[47],"for":[48,69,119],"these":[49],"is":[51,84,102,114],"typically":[52],"very":[53],"long.":[54],"We":[55],"present":[56],"new":[58],"combination":[59],"of":[60,107,123],"RL":[61],"and":[62,93,111],"RNN":[63,113],"find":[65],"good":[67],"policy":[68,141],"POMDPs":[70],"in":[71,125],"shorter":[73],"time.":[75],"This":[76],"method":[77,134,149],"contains":[78],"two":[79,87,126],"phases:":[80],"firstly,":[81],"state":[82,91,95],"space":[83],"divided":[85],"into":[86],"groups":[88],"(fully":[89],"group":[92],"hidden":[94,120],"group);":[96],"secondly,":[97],"value":[100],"table":[101],"store":[105],"values":[106,118],"fully":[108],"states":[110],"an":[112,136],"states.":[121],"Results":[122],"experiments":[124],"grid":[127],"world":[128],"show":[130],"that":[131],"the":[132,148],"proposed":[133],"enables":[135],"agent":[137],"acquire":[139],"better":[143],"performance":[145],"compared":[146],"using":[150],"only":[151],"RNN.":[153]},"counts_by_year":[{"year":2025,"cited_by_count":2},{"year":2024,"cited_by_count":2},{"year":2022,"cited_by_count":1},{"year":2021,"cited_by_count":1},{"year":2020,"cited_by_count":2},{"year":2018,"cited_by_count":1},{"year":2014,"cited_by_count":3},{"year":2013,"cited_by_count":4}],"updated_date":"2026-07-29T09:40:50.615796","created_date":"2025-10-10T00:00:00"}
