{"id":"https://openalex.org/W4401943714","doi":"https://doi.org/10.1109/cog60054.2024.10645662","title":"Multi-DMC: Deep Monte-Carlo with Multi-Stage Learning in the Card Game UNO","display_name":"Multi-DMC: Deep Monte-Carlo with Multi-Stage Learning in the Card Game UNO","publication_year":2024,"publication_date":"2024-08-05","ids":{"openalex":"https://openalex.org/W4401943714","doi":"https://doi.org/10.1109/cog60054.2024.10645662"},"language":"en","primary_location":{"id":"doi:10.1109/cog60054.2024.10645662","is_oa":false,"landing_page_url":"https://doi.org/10.1109/cog60054.2024.10645662","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2024 IEEE Conference on Games (CoG)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5086697078","display_name":"Xueqing Yang","orcid":"https://orcid.org/0000-0002-5816-8477"},"institutions":[{"id":"https://openalex.org/I91935597","display_name":"University of South China","ror":"https://ror.org/03mqfn238","country_code":"CN","type":"education","lineage":["https://openalex.org/I91935597"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Xueqing Yang","raw_affiliation_strings":["University of South China,School of Computer,Hengyang,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"University of South China,School of Computer,Hengyang,China","institution_ids":["https://openalex.org/I91935597"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5100373399","display_name":"Xiaofei Liu","orcid":"https://orcid.org/0000-0003-1733-4712"},"institutions":[{"id":"https://openalex.org/I91935597","display_name":"University of South China","ror":"https://ror.org/03mqfn238","country_code":"CN","type":"education","lineage":["https://openalex.org/I91935597"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Xiaofei Liu","raw_affiliation_strings":["University of South China,School of Computer,Hengyang,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"University of South China,School of Computer,Hengyang,China","institution_ids":["https://openalex.org/I91935597"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5087013512","display_name":"Wenbin Lin","orcid":"https://orcid.org/0000-0002-4282-066X"},"institutions":[{"id":"https://openalex.org/I149735164","display_name":"Hengyang Normal University","ror":"https://ror.org/006bvjm48","country_code":"CN","type":"education","lineage":["https://openalex.org/I149735164"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Wenbin Lin","raw_affiliation_strings":["School of Mathematics and Physics,School of Computer,Hengyang,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"School of Mathematics and Physics,School of Computer,Hengyang,China","institution_ids":["https://openalex.org/I149735164"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":0.7112,"has_fulltext":false,"cited_by_count":2,"citation_normalized_percentile":{"value":0.69602253,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":91,"max":97},"biblio":{"volume":null,"issue":null,"first_page":"1","last_page":"8"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T11574","display_name":"Artificial Intelligence in Games","score":0.9980999827384949,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T11574","display_name":"Artificial Intelligence in Games","score":0.9980999827384949,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9835000038146973,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T12101","display_name":"Advanced Bandit Algorithms Research","score":0.9767000079154968,"subfield":{"id":"https://openalex.org/subfields/1803","display_name":"Management Science and Operations Research"},"field":{"id":"https://openalex.org/fields/18","display_name":"Decision Sciences"},"domain":{"id":"https://openalex.org/domains/2","display_name":"Social Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/monte-carlo-method","display_name":"Monte Carlo method","score":0.7545807957649231},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7186322212219238},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.42532503604888916},{"id":"https://openalex.org/keywords/statistics","display_name":"Statistics","score":0.14251229166984558},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.1350698471069336}],"concepts":[{"id":"https://openalex.org/C19499675","wikidata":"https://www.wikidata.org/wiki/Q232207","display_name":"Monte Carlo method","level":2,"score":0.7545807957649231},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7186322212219238},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.42532503604888916},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.14251229166984558},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.1350698471069336}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/cog60054.2024.10645662","is_oa":false,"landing_page_url":"https://doi.org/10.1109/cog60054.2024.10645662","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2024 IEEE Conference on Games (CoG)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[{"score":0.5,"id":"https://metadata.un.org/sdg/17","display_name":"Partnerships for the goals"}],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":39,"referenced_works":["https://openalex.org/W2106261932","https://openalex.org/W2113228754","https://openalex.org/W2131600418","https://openalex.org/W2145339207","https://openalex.org/W2514260088","https://openalex.org/W2574978968","https://openalex.org/W2588283865","https://openalex.org/W2736601468","https://openalex.org/W2766447205","https://openalex.org/W2773381986","https://openalex.org/W2800288401","https://openalex.org/W2802963644","https://openalex.org/W2809129829","https://openalex.org/W2896791808","https://openalex.org/W2902907165","https://openalex.org/W2918733506","https://openalex.org/W2960876848","https://openalex.org/W2972894440","https://openalex.org/W2998299793","https://openalex.org/W3006861906","https://openalex.org/W3013828496","https://openalex.org/W3041349757","https://openalex.org/W3118881636","https://openalex.org/W3197103019","https://openalex.org/W3213749364","https://openalex.org/W4206506568","https://openalex.org/W4214717370","https://openalex.org/W4221158332","https://openalex.org/W4291414411","https://openalex.org/W4296513000","https://openalex.org/W4312709557","https://openalex.org/W6677916085","https://openalex.org/W6741002519","https://openalex.org/W6759618127","https://openalex.org/W6775289199","https://openalex.org/W6796814964","https://openalex.org/W6804066874","https://openalex.org/W6810521076","https://openalex.org/W6922480057"],"related_works":["https://openalex.org/W4391375266","https://openalex.org/W2748952813","https://openalex.org/W2390279801","https://openalex.org/W4391913857","https://openalex.org/W2358668433","https://openalex.org/W4396701345","https://openalex.org/W2376932109","https://openalex.org/W2001405890","https://openalex.org/W4396696052","https://openalex.org/W4402327032"],"abstract_inverted_index":{"Games":[0],"can":[1],"serve":[2],"as":[3,19],"important":[4],"benchmarks":[5],"for":[6],"evaluating":[7],"artificial":[8],"intelligence":[9],"(AI),":[10],"and":[11,25,42,78,124,137],"many":[12],"game":[13],"AIs":[14],"have":[15],"been":[16],"invented,":[17],"such":[18],"AlphaGo,":[20],"Libratus,":[21],"OpenAI":[22],"Five,":[23],"Suphx,":[24],"DouZero.":[26],"For":[27],"UNO,":[28],"a":[29,105,129],"popular":[30],"shedding-type":[31],"card":[32],"game,":[33,60],"the":[34,50,58,75,89,98,121],"AI":[35],"faces":[36],"imperfect-information":[37],"challenges,":[38],"large":[39],"state":[40],"space,":[41],"sparse":[43],"rewards.":[44],"The":[45,155],"existing":[46,122],"methods":[47],"only":[48],"demonstrate":[49],"promising":[51],"feasibility":[52],"of":[53],"classic":[54],"reinforcement":[55],"learning":[56],"in":[57,109,117],"UNO":[59,118],"which":[61,103],"is":[62,104,148,158],"still":[63],"far":[64],"from":[65],"being":[66],"able":[67],"to":[68],"compete":[69],"with":[70,85],"human":[71],"players.":[72],"Based":[73],"on":[74,97],"Monte-Carlo":[76],"method":[77],"deep":[79],"neural":[80],"networks,":[81],"we":[82,144],"combine":[83],"it":[84,125],"Multi-Stage":[86],"Learning":[87],"during":[88],"training":[90,114],"process,":[91],"called":[92],"Multi-DMC.":[93],"We":[94],"conduct":[95],"experiments":[96],"Two-vs-Two":[99],"cooperative":[100],"competition":[101],"version":[102,108],"commonly":[106],"used":[107],"tournaments.":[110],"With":[111],"4":[112],"billion":[113],"steps,":[115],"Multi-DMC":[116,147],"outperforms":[119],"all":[120],"baselines,":[123],"achieves":[126],"more":[127,149],"than":[128,151],"$99":[130],"\\%$":[131,139],"winning":[132],"rate":[133],"against":[134,140],"Random":[135],"agents":[136],"$85":[138],"Rule-Based":[141],"agents.":[142],"Meanwhile,":[143],"find":[145],"that":[146],"competitive":[150],"pure":[152],"self-game":[153],"DMC-Self.":[154],"project":[156],"page":[157],"at":[159],"https://github.com/LinsLab/UNO.":[160]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":1}],"updated_date":"2026-07-29T14:22:42.915294","created_date":"2025-10-10T00:00:00"}
