{"id":"https://openalex.org/W4225326921","doi":"https://doi.org/10.1109/icassp43922.2022.9747909","title":"Source Separation By Steering Pretrained Music Models","display_name":"Source Separation By Steering Pretrained Music Models","publication_year":2022,"publication_date":"2022-04-27","ids":{"openalex":"https://openalex.org/W4225326921","doi":"https://doi.org/10.1109/icassp43922.2022.9747909"},"language":"en","primary_location":{"id":"doi:10.1109/icassp43922.2022.9747909","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp43922.2022.9747909","pdf_url":null,"source":{"id":"https://openalex.org/S4363607702","display_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5023355857","display_name":"Ethan Manilow","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Ethan Manilow","raw_affiliation_strings":["Northwestern University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern University","institution_ids":[]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5075935325","display_name":"Patrick O\u2019Reilly","orcid":"https://orcid.org/0000-0002-6205-2354"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Patrick O'Reilly","raw_affiliation_strings":["Northwestern University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern University","institution_ids":[]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5023673004","display_name":"Prem Seetharaman","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Prem Seetharaman","raw_affiliation_strings":["Descript, Inc"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Descript, Inc","institution_ids":[]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5102878078","display_name":"Bryan Pardo","orcid":"https://orcid.org/0000-0002-1427-6492"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Bryan Pardo","raw_affiliation_strings":["Northwestern University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern University","institution_ids":[]}]}],"institutions":[],"countries_distinct_count":0,"institutions_distinct_count":0,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":0.7383,"has_fulltext":false,"cited_by_count":4,"citation_normalized_percentile":{"value":0.67349746,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":89,"max":97},"biblio":{"volume":null,"issue":null,"first_page":"126","last_page":"130"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11349","display_name":"Music Technology and Sound Studies","score":0.9987000226974487,"subfield":{"id":"https://openalex.org/subfields/1707","display_name":"Computer Vision and Pattern Recognition"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/separation","display_name":"Separation (statistics)","score":0.621497631072998},{"id":"https://openalex.org/keywords/source-separation","display_name":"Source separation","score":0.6161060929298401},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.5887942314147949},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.3976135551929474},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.12824735045433044}],"concepts":[{"id":"https://openalex.org/C2776061190","wikidata":"https://www.wikidata.org/wiki/Q7451805","display_name":"Separation (statistics)","level":2,"score":0.621497631072998},{"id":"https://openalex.org/C2776864781","wikidata":"https://www.wikidata.org/wiki/Q52617913","display_name":"Source separation","level":2,"score":0.6161060929298401},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.5887942314147949},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.3976135551929474},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.12824735045433044}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp43922.2022.9747909","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp43922.2022.9747909","pdf_url":null,"source":{"id":"https://openalex.org/S4363607702","display_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":45,"referenced_works":["https://openalex.org/W1561135842","https://openalex.org/W2027884847","https://openalex.org/W2127851351","https://openalex.org/W2127870748","https://openalex.org/W2296573765","https://openalex.org/W2414894569","https://openalex.org/W2950060770","https://openalex.org/W2963992487","https://openalex.org/W2964237233","https://openalex.org/W2971074500","https://openalex.org/W2972411915","https://openalex.org/W2981892560","https://openalex.org/W2984935418","https://openalex.org/W2990594533","https://openalex.org/W2998490864","https://openalex.org/W3015213533","https://openalex.org/W3015289235","https://openalex.org/W3029858316","https://openalex.org/W3104704316","https://openalex.org/W3160751854","https://openalex.org/W3166396011","https://openalex.org/W3180355996","https://openalex.org/W3180663620","https://openalex.org/W3197781391","https://openalex.org/W4287180975","https://openalex.org/W4287802874","https://openalex.org/W4288089799","https://openalex.org/W4292779060","https://openalex.org/W4298310324","https://openalex.org/W6633727632","https://openalex.org/W6678969435","https://openalex.org/W6697668227","https://openalex.org/W6715395060","https://openalex.org/W6762931180","https://openalex.org/W6763945542","https://openalex.org/W6766320909","https://openalex.org/W6769478989","https://openalex.org/W6769627184","https://openalex.org/W6776218486","https://openalex.org/W6778572914","https://openalex.org/W6778883912","https://openalex.org/W6791353385","https://openalex.org/W6795742094","https://openalex.org/W6798597329","https://openalex.org/W6801319400"],"related_works":["https://openalex.org/W4391375266","https://openalex.org/W2748952813","https://openalex.org/W2390279801","https://openalex.org/W2358668433","https://openalex.org/W4396701345","https://openalex.org/W2071676784","https://openalex.org/W2376932109","https://openalex.org/W2001405890","https://openalex.org/W4396696052","https://openalex.org/W2077498359"],"abstract_inverted_index":{"We":[0,120],"showcase":[1],"a":[2,33,49,69,161],"method":[3],"that":[4,53],"repurposes":[5],"deep":[6],"models":[7,184],"trained":[8],"for":[9,15,64,72,160,185],"music":[10,13,51,138,183],"generation":[11,24],"and":[12,68,105,129,142,176],"tagging":[14,144],"audio":[16,23,38,45,67],"source":[17,55,75,150,189],"separation,":[18],"without":[19],"any":[20,167],"retraining.":[21],"An":[22],"model":[25,101],"is":[26,46,76],"conditioned":[27],"on":[28,108,148],"an":[29,73],"input":[30],"mixture,":[31],"producing":[32],"latent":[34,85,114],"encoding":[35],"of":[36,87,98,136,164,180],"the":[37,61,65,83,88,96,99,103,111,125,174],"used":[39,77],"to":[40,48,78,116,173],"generate":[41],"audio.":[42],"This":[43,91,170],"generated":[44,66],"fed":[47],"pretrained":[50,126,137,182],"tagger":[52],"creates":[54],"labels.":[56],"The":[57],"cross-entropy":[58],"loss":[59],"between":[60],"tag":[62],"distribution":[63,71],"predefined":[70],"isolated":[74],"guide":[79],"gradient":[80],"ascent":[81],"in":[82],"(unchanging)":[84],"space":[86,115],"generative":[89,100,112,127],"model.":[90],"system":[92],"does":[93],"not":[94],"update":[95],"weights":[97],"or":[102],"tagger,":[104],"only":[106],"relies":[107],"moving":[109],"through":[110],"model\u2019s":[113],"produce":[117,157],"separated":[118],"sources.":[119],"use":[121],"OpenAI\u2019s":[122],"Jukebox":[123],"as":[124],"model,":[128],"we":[130],"couple":[131],"it":[132],"with":[133],"four":[134],"kinds":[135],"taggers":[139],"(two":[140],"architectures":[141],"two":[143,149],"datasets).":[145],"Experimental":[146],"results":[147],"separation":[151,158],"datasets,":[152],"show":[153],"this":[154],"approach":[155],"can":[156],"estimates":[159],"wider":[162],"variety":[163],"sources":[165],"than":[166],"tested":[168],"system.":[169],"work":[171],"points":[172],"vast":[175],"heretofore":[177],"untapped":[178],"potential":[179],"large":[181],"audio-to-audio":[186],"tasks":[187],"like":[188],"separation.":[190]},"counts_by_year":[{"year":2025,"cited_by_count":3},{"year":2022,"cited_by_count":1}],"updated_date":"2026-07-29T14:22:42.915294","created_date":"2025-10-10T00:00:00"}
