{"id":"https://openalex.org/W7139026625","doi":"https://doi.org/10.1609/aaai.v40i40.40666","title":"CP-Search: A Chain Progressive Search Training Framework Incentivizing the Cognitive Behaviors for Searching in LLMs","display_name":"CP-Search: A Chain Progressive Search Training Framework Incentivizing the Cognitive Behaviors for Searching in LLMs","publication_year":2026,"publication_date":"2026-03-14","ids":{"openalex":"https://openalex.org/W7139026625","doi":"https://doi.org/10.1609/aaai.v40i40.40666"},"language":"en","primary_location":{"id":"doi:10.1609/aaai.v40i40.40666","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v40i40.40666","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"diamond","oa_url":"https://doi.org/10.1609/aaai.v40i40.40666","any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5129886370","display_name":"Zehua Wang","orcid":null},"institutions":[{"id":"https://openalex.org/I158809036","display_name":"Shenzhen Institute of Information Technology","ror":"https://ror.org/03wrf9427","country_code":"CN","type":"education","lineage":["https://openalex.org/I158809036"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Zehua Wang","raw_affiliation_strings":["Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China\nGuangdong Provincial Key Laboratory of Intelligent Information Processing"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China\nGuangdong Provincial Key Laboratory of Intelligent Information Processing","institution_ids":["https://openalex.org/I158809036"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5003730507","display_name":"S. G. Li","orcid":null},"institutions":[{"id":"https://openalex.org/I158809036","display_name":"Shenzhen Institute of Information Technology","ror":"https://ror.org/03wrf9427","country_code":"CN","type":"education","lineage":["https://openalex.org/I158809036"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Shipeng Li","raw_affiliation_strings":["Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China","institution_ids":["https://openalex.org/I158809036"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5021991117","display_name":"Buzhou Tang","orcid":null},"institutions":[{"id":"https://openalex.org/I4210136793","display_name":"Peng Cheng Laboratory","ror":"https://ror.org/03qdqbt06","country_code":"CN","type":"facility","lineage":["https://openalex.org/I4210136793"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Buzhou Tang","raw_affiliation_strings":["Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China\nGuangdong Provincial Key Laboratory of Intelligent Information Processing\nPeng Cheng Laboratory, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science, Harbin Institue of Technology (Shenzhen), Shenzhen, China\nGuangdong Provincial Key Laboratory of Intelligent Information Processing\nPeng Cheng Laboratory, Shenzhen, China","institution_ids":["https://openalex.org/I4210136793"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":0,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":"40","issue":"40","first_page":"33755","last_page":"33763"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10028","display_name":"Topic Modeling","score":0.7124000191688538,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10028","display_name":"Topic Modeling","score":0.7124000191688538,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11714","display_name":"Multimodal Machine Learning Applications","score":0.0940999984741211,"subfield":{"id":"https://openalex.org/subfields/1707","display_name":"Computer Vision and Pattern Recognition"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10286","display_name":"Information Retrieval and Search Behavior","score":0.023800000548362732,"subfield":{"id":"https://openalex.org/subfields/1710","display_name":"Information Systems"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/consistency","display_name":"Consistency (knowledge bases)","score":0.6965000033378601},{"id":"https://openalex.org/keywords/reinforcement-learning","display_name":"Reinforcement learning","score":0.6154000163078308},{"id":"https://openalex.org/keywords/process","display_name":"Process (computing)","score":0.5651999711990356},{"id":"https://openalex.org/keywords/cognition","display_name":"Cognition","score":0.5432999730110168},{"id":"https://openalex.org/keywords/markov-chain","display_name":"Markov chain","score":0.5056999921798706},{"id":"https://openalex.org/keywords/markov-decision-process","display_name":"Markov decision process","score":0.4602999985218048},{"id":"https://openalex.org/keywords/cognitive-load","display_name":"Cognitive load","score":0.3617999851703644},{"id":"https://openalex.org/keywords/opportunistic-reasoning","display_name":"Opportunistic reasoning","score":0.3276999890804291}],"concepts":[{"id":"https://openalex.org/C2776436953","wikidata":"https://www.wikidata.org/wiki/Q5163215","display_name":"Consistency (knowledge bases)","level":2,"score":0.6965000033378601},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.638700008392334},{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.6154000163078308},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.5777000188827515},{"id":"https://openalex.org/C98045186","wikidata":"https://www.wikidata.org/wiki/Q205663","display_name":"Process (computing)","level":2,"score":0.5651999711990356},{"id":"https://openalex.org/C169900460","wikidata":"https://www.wikidata.org/wiki/Q2200417","display_name":"Cognition","level":2,"score":0.5432999730110168},{"id":"https://openalex.org/C98763669","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov chain","level":2,"score":0.5056999921798706},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.4941999912261963},{"id":"https://openalex.org/C106189395","wikidata":"https://www.wikidata.org/wiki/Q176789","display_name":"Markov decision process","level":3,"score":0.4602999985218048},{"id":"https://openalex.org/C61641136","wikidata":"https://www.wikidata.org/wiki/Q1107019","display_name":"Cognitive load","level":3,"score":0.3617999851703644},{"id":"https://openalex.org/C86827895","wikidata":"https://www.wikidata.org/wiki/Q7098582","display_name":"Opportunistic reasoning","level":4,"score":0.3276999890804291},{"id":"https://openalex.org/C159886148","wikidata":"https://www.wikidata.org/wiki/Q176645","display_name":"Markov process","level":2,"score":0.3125},{"id":"https://openalex.org/C161407221","wikidata":"https://www.wikidata.org/wiki/Q4382939","display_name":"Cognitive model","level":3,"score":0.304500013589859},{"id":"https://openalex.org/C67203356","wikidata":"https://www.wikidata.org/wiki/Q1321905","display_name":"Reinforcement","level":2,"score":0.2994999885559082},{"id":"https://openalex.org/C40506919","wikidata":"https://www.wikidata.org/wiki/Q7452469","display_name":"Sequence learning","level":2,"score":0.2921999990940094},{"id":"https://openalex.org/C20162079","wikidata":"https://www.wikidata.org/wiki/Q1151406","display_name":"Case-based reasoning","level":2,"score":0.2858999967575073},{"id":"https://openalex.org/C37335422","wikidata":"https://www.wikidata.org/wiki/Q6888134","display_name":"Model-based reasoning","level":3,"score":0.28360000252723694},{"id":"https://openalex.org/C115086926","wikidata":"https://www.wikidata.org/wiki/Q17004651","display_name":"Causal reasoning","level":3,"score":0.2809999883174896},{"id":"https://openalex.org/C2777211547","wikidata":"https://www.wikidata.org/wiki/Q17141490","display_name":"Training (meteorology)","level":2,"score":0.26460000872612},{"id":"https://openalex.org/C2776214188","wikidata":"https://www.wikidata.org/wiki/Q408386","display_name":"Inference","level":2,"score":0.2628999948501587}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1609/aaai.v40i40.40666","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v40i40.40666","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},{"id":"pmh:oai:ojs.aaai.org:article/40666","is_oa":false,"landing_page_url":"https://ojs.aaai.org/index.php/AAAI/article/view/40666","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"2159-5399","raw_type":"info:eu-repo/semantics/article"}],"best_oa_location":{"id":"doi:10.1609/aaai.v40i40.40666","is_oa":true,"landing_page_url":"https://doi.org/10.1609/aaai.v40i40.40666","pdf_url":null,"source":{"id":"https://openalex.org/S4210191458","display_name":"Proceedings of the AAAI Conference on Artificial Intelligence","issn_l":"2159-5399","issn":["2159-5399","2374-3468"],"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/P4310320058","host_organization_name":"Association for the Advancement of Artificial Intelligence","host_organization_lineage":["https://openalex.org/P4310320058"],"host_organization_lineage_names":["Association for the Advancement of Artificial Intelligence"],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Proceedings of the AAAI Conference on Artificial Intelligence","raw_type":"journal-article"},"sustainable_development_goals":[{"score":0.6329156756401062,"id":"https://metadata.un.org/sdg/16","display_name":"Peace, Justice and strong institutions"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":0,"referenced_works":[],"related_works":[],"abstract_inverted_index":{"Retrieval-Augmented":[0],"Generation":[1],"(RAG)":[2],"has":[3,42],"been":[4],"demonstrated":[5,43],"to":[6,65,98,146],"effectively":[7],"mitigate":[8],"the":[9,66,73,100,110,122,130,148,171],"knowledge":[10],"recency":[11],"issue":[12],"in":[13,29,45,56,78,104,164,179,196],"Large":[14],"Language":[15],"Models":[16],"(LLMs)":[17],"while":[18],"significantly":[19,169,191],"reducing":[20],"hallucinations.":[21],"However,":[22],"existing":[23,193],"RAG":[24,194],"methods":[25,195],"exhibit":[26,53],"insufficient":[27],"capability":[28,103],"modeling":[30],"reasoning":[31,36,48,59,173,199],"paths":[32],"for":[33,153],"complex":[34,105,197],"multi-hop":[35,186,198],"tasks.":[37,200],"While":[38],"Reinforcement":[39],"Learning":[40],"(RL)":[41],"success":[44],"enhancing":[46],"model":[47,149],"ability,":[49],"Token-level":[50],"RL":[51],"frameworks":[52],"inherent":[54],"limitations":[55],"maintaining":[57],"coherent":[58],"trajectories.":[60],"This":[61,107],"approach":[62],"remains":[63],"susceptible":[64],"compounding":[67],"accumulation":[68],"of":[69,129],"contextual":[70],"errors":[71],"during":[72],"retrieval":[74,102,112,124],"process,":[75],"ultimately":[76],"resulting":[77],"erroneous":[79],"output":[80],"generation.":[81],"To":[82],"address":[83],"this":[84],"challenge,":[85],"we":[86],"propose":[87],"Chain":[88],"Progressive":[89],"Search":[90],"(CP-Search),":[91],"a":[92,115,137,159],"novel":[93],"two-stage":[94],"training":[95],"framework":[96,108],"designed":[97],"enhance":[99],"model's":[101,123,172],"scenarios.":[106],"models":[109],"entire":[111],"process":[113],"as":[114],"Retrieval-level":[116],"Markov":[117],"Decision":[118],"Process,":[119],"systematically":[120],"optimizing":[121],"behavior":[125],"at":[126],"each":[127],"step":[128],"chained":[131,180],"retrieval.":[132,181],"Specifically,":[133],"CP-Search":[134,168,190],"first":[135],"constructs":[136],"retrieval-cognitive":[138],"behavioral":[139],"dataset":[140],"and":[141,175],"employs":[142],"Supervised":[143],"Fine-Tuning":[144],"(SFT)":[145],"endow":[147],"with":[150],"cognitive":[151],"behaviors":[152],"searching.":[154],"More":[155],"importantly,":[156],"by":[157],"introducing":[158],"dense":[160],"progressive":[161],"procedural":[162],"reward":[163],"reinforcement":[165],"learning":[166],"training,":[167],"improves":[170],"consistency":[174],"feedback":[176],"correction":[177],"ability":[178],"Experiments":[182],"conducted":[183],"on":[184],"multiple":[185],"datasets":[187],"demonstrate":[188],"that":[189],"outperforms":[192]},"counts_by_year":[],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2026-03-20T00:00:00"}
