{"id":"https://openalex.org/W7155058325","doi":"https://doi.org/10.48550/arxiv.2604.17415","title":"Reward Score Matching: Unifying Reward-based Fine-tuning for Flow and Diffusion Models","display_name":"Reward Score Matching: Unifying Reward-based Fine-tuning for Flow and Diffusion Models","publication_year":2026,"publication_date":"2026-04-19","ids":{"openalex":"https://openalex.org/W7155058325","doi":"https://doi.org/10.48550/arxiv.2604.17415"},"language":null,"primary_location":{"id":"doi:10.48550/arxiv.2604.17415","is_oa":true,"landing_page_url":"https://doi.org/10.48550/arxiv.2604.17415","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":null,"is_accepted":false,"is_published":false,"raw_source_name":null,"raw_type":"article"},"type":"preprint","indexed_in":["datacite"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://doi.org/10.48550/arxiv.2604.17415","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5134124293","display_name":"Jeongjae Lee","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Lee, Jeongjae","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5134154052","display_name":"Jinho Chang","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Chang, Jinho","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5134173854","display_name":"Jeongsol Kim","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Kim, Jeongsol","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"last","author":{"id":"https://openalex.org/A5134176686","display_name":"Jong Chul Ye","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Ye, Jong Chul","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]}],"institutions":[],"countries_distinct_count":0,"institutions_distinct_count":4,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":0,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10775","display_name":"Generative Adversarial Networks and Image Synthesis","score":0.27250000834465027,"subfield":{"id":"https://openalex.org/subfields/1707","display_name":"Computer Vision and Pattern Recognition"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10775","display_name":"Generative Adversarial Networks and Image Synthesis","score":0.27250000834465027,"subfield":{"id":"https://openalex.org/subfields/1707","display_name":"Computer Vision and Pattern Recognition"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11612","display_name":"Stochastic Gradient Optimization Techniques","score":0.10320000350475311,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10203","display_name":"Recommender Systems and Techniques","score":0.07329999655485153,"subfield":{"id":"https://openalex.org/subfields/1710","display_name":"Information Systems"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/matching","display_name":"Matching (statistics)","score":0.704800009727478},{"id":"https://openalex.org/keywords/unification","display_name":"Unification","score":0.641700029373169},{"id":"https://openalex.org/keywords/estimator","display_name":"Estimator","score":0.6380000114440918},{"id":"https://openalex.org/keywords/flow","display_name":"Flow (mathematics)","score":0.45489999651908875},{"id":"https://openalex.org/keywords/generative-grammar","display_name":"Generative grammar","score":0.436599999666214},{"id":"https://openalex.org/keywords/core","display_name":"Core (optical fiber)","score":0.4244999885559082},{"id":"https://openalex.org/keywords/generative-model","display_name":"Generative model","score":0.3862000107765198},{"id":"https://openalex.org/keywords/differentiable-function","display_name":"Differentiable function","score":0.37619999051094055}],"concepts":[{"id":"https://openalex.org/C165064840","wikidata":"https://www.wikidata.org/wiki/Q1321061","display_name":"Matching (statistics)","level":2,"score":0.704800009727478},{"id":"https://openalex.org/C96146094","wikidata":"https://www.wikidata.org/wiki/Q609057","display_name":"Unification","level":2,"score":0.641700029373169},{"id":"https://openalex.org/C185429906","wikidata":"https://www.wikidata.org/wiki/Q1130160","display_name":"Estimator","level":2,"score":0.6380000114440918},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.5467000007629395},{"id":"https://openalex.org/C38349280","wikidata":"https://www.wikidata.org/wiki/Q1434290","display_name":"Flow (mathematics)","level":2,"score":0.45489999651908875},{"id":"https://openalex.org/C39890363","wikidata":"https://www.wikidata.org/wiki/Q36108","display_name":"Generative grammar","level":2,"score":0.436599999666214},{"id":"https://openalex.org/C2164484","wikidata":"https://www.wikidata.org/wiki/Q5170150","display_name":"Core (optical fiber)","level":2,"score":0.4244999885559082},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.39480000734329224},{"id":"https://openalex.org/C167966045","wikidata":"https://www.wikidata.org/wiki/Q5532625","display_name":"Generative model","level":3,"score":0.3862000107765198},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.38260000944137573},{"id":"https://openalex.org/C202615002","wikidata":"https://www.wikidata.org/wiki/Q783507","display_name":"Differentiable function","level":2,"score":0.37619999051094055},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.3580000102519989},{"id":"https://openalex.org/C2776459999","wikidata":"https://www.wikidata.org/wiki/Q2119376","display_name":"Fidelity","level":2,"score":0.35519999265670776},{"id":"https://openalex.org/C2780009758","wikidata":"https://www.wikidata.org/wiki/Q6804172","display_name":"Measure (data warehouse)","level":2,"score":0.34529998898506165},{"id":"https://openalex.org/C176217482","wikidata":"https://www.wikidata.org/wiki/Q860554","display_name":"Metric (unit)","level":2,"score":0.3343000113964081},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.29750001430511475},{"id":"https://openalex.org/C69357855","wikidata":"https://www.wikidata.org/wiki/Q163214","display_name":"Diffusion","level":2,"score":0.29269999265670776},{"id":"https://openalex.org/C22367795","wikidata":"https://www.wikidata.org/wiki/Q7625208","display_name":"Structured prediction","level":2,"score":0.28119999170303345},{"id":"https://openalex.org/C34559072","wikidata":"https://www.wikidata.org/wiki/Q2334061","display_name":"Design of experiments","level":2,"score":0.2782999873161316},{"id":"https://openalex.org/C189950617","wikidata":"https://www.wikidata.org/wiki/Q937228","display_name":"Property (philosophy)","level":2,"score":0.2644999921321869},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.26089999079704285},{"id":"https://openalex.org/C2780801425","wikidata":"https://www.wikidata.org/wiki/Q5164392","display_name":"Construct (python library)","level":2,"score":0.2529999911785126}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.48550/arxiv.2604.17415","is_oa":true,"landing_page_url":"https://doi.org/10.48550/arxiv.2604.17415","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":null,"is_accepted":false,"is_published":null,"raw_source_name":null,"raw_type":"article"}],"best_oa_location":{"id":"doi:10.48550/arxiv.2604.17415","is_oa":true,"landing_page_url":"https://doi.org/10.48550/arxiv.2604.17415","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":null,"is_accepted":false,"is_published":false,"raw_source_name":null,"raw_type":"article"},"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":0,"referenced_works":[],"related_works":[],"abstract_inverted_index":{"Reward-based":[0],"fine-tuning":[1,128],"steers":[2],"a":[3,36,54,122,131],"pretrained":[4,18],"diffusion":[5],"or":[6],"flow-based":[7],"generative":[8],"model":[9],"toward":[10],"higher-reward":[11],"samples":[12],"while":[13],"remaining":[14],"close":[15],"to":[16,64],"the":[17,58,65,68,72,81],"model.":[19],"Although":[20],"existing":[21,85],"methods":[22,62,129],"are":[23],"derived":[24],"from":[25,92],"different":[26],"perspectives,":[27],"we":[28,40,105],"show":[29],"that":[30,95],"many":[31],"can":[32],"be":[33],"written":[34],"under":[35],"common":[37],"framework,":[38],"which":[39],"call":[41],"reward":[42,116],"score":[43,51],"matching":[44,52],"(RSM).":[45],"Under":[46],"this":[47,103],"view,":[48],"alignment":[49,117],"becomes":[50],"against":[53],"value-guided":[55],"target,":[56],"and":[57,71,87,114,135],"main":[59],"differences":[60],"across":[61,76,111],"reduce":[63],"construction":[66],"of":[67,84,126],"value-guidance":[69],"estimator":[70],"effective":[73],"optimization":[74,90],"strength":[75],"timesteps.":[77],"This":[78],"unification":[79],"clarifies":[80],"bias-variance-compute":[82],"tradeoffs":[83],"designs,":[86],"distinguishes":[88],"core":[89],"components":[91],"auxiliary":[93],"mechanisms":[94],"add":[96],"complexity":[97],"without":[98],"clear":[99],"benefit.":[100],"Guided":[101],"by":[102],"perspective,":[104],"develop":[106],"simpler,":[107],"more":[108,133,136],"efficient":[109],"redesigns":[110],"representative":[112],"differentiable":[113],"black-box":[115],"tasks.":[118],"Overall,":[119],"RSM":[120],"turns":[121],"seemingly":[123],"fragmented":[124],"collection":[125],"reward-based":[127],"into":[130],"smaller,":[132],"interpretable,":[134],"actionable":[137],"design":[138],"space.":[139],"Code":[140],"is":[141],"available":[142],"at":[143],"https://github.com/jaylee2000/rsm":[144]},"counts_by_year":[],"updated_date":"2026-06-11T09:08:48.828518","created_date":"2026-04-22T00:00:00"}
