{"id":"https://openalex.org/W3094893851","doi":"https://doi.org/10.21437/interspeech.2020-1089","title":"Neural Speech Separation Using Spatially Distributed Microphones","display_name":"Neural Speech Separation Using Spatially Distributed Microphones","publication_year":2020,"publication_date":"2020-10-25","ids":{"openalex":"https://openalex.org/W3094893851","doi":"https://doi.org/10.21437/interspeech.2020-1089","mag":"3094893851"},"language":"en","primary_location":{"id":"doi:10.21437/interspeech.2020-1089","is_oa":false,"landing_page_url":"https://doi.org/10.21437/interspeech.2020-1089","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Interspeech 2020","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5100408281","display_name":"Dongmei Wang","orcid":"https://orcid.org/0000-0002-6930-0066"},"institutions":[{"id":"https://openalex.org/I1290206253","display_name":"Microsoft (United States)","ror":"https://ror.org/00d0nc645","country_code":"US","type":"company","lineage":["https://openalex.org/I1290206253"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Dongmei Wang","raw_affiliation_strings":["Microsoft, One Microsoft Way, Redmond, WA, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Microsoft, One Microsoft Way, Redmond, WA, USA","institution_ids":["https://openalex.org/I1290206253"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5100345056","display_name":"Zhuo Chen","orcid":"https://orcid.org/0000-0001-7799-5163"},"institutions":[{"id":"https://openalex.org/I1290206253","display_name":"Microsoft (United States)","ror":"https://ror.org/00d0nc645","country_code":"US","type":"company","lineage":["https://openalex.org/I1290206253"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Zhuo Chen","raw_affiliation_strings":["Microsoft, One Microsoft Way, Redmond, WA, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Microsoft, One Microsoft Way, Redmond, WA, USA","institution_ids":["https://openalex.org/I1290206253"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5101618071","display_name":"Takuya Yoshioka","orcid":"https://orcid.org/0009-0003-7791-3545"},"institutions":[{"id":"https://openalex.org/I1290206253","display_name":"Microsoft (United States)","ror":"https://ror.org/00d0nc645","country_code":"US","type":"company","lineage":["https://openalex.org/I1290206253"]}],"countries":["US"],"is_corresponding":true,"raw_author_name":"Takuya Yoshioka","raw_affiliation_strings":["Microsoft, One Microsoft Way, Redmond, WA, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Microsoft, One Microsoft Way, Redmond, WA, USA","institution_ids":["https://openalex.org/I1290206253"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":["https://openalex.org/A5101618071"],"corresponding_institution_ids":["https://openalex.org/I1290206253"],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":35,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":"339","last_page":"343"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11233","display_name":"Advanced Adaptive Filtering Techniques","score":0.9926999807357788,"subfield":{"id":"https://openalex.org/subfields/2206","display_name":"Computational Mechanics"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9836000204086304,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7235375046730042},{"id":"https://openalex.org/keywords/separation","display_name":"Separation (statistics)","score":0.6787939071655273},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.5226616263389587},{"id":"https://openalex.org/keywords/artificial-neural-network","display_name":"Artificial neural network","score":0.44166430830955505},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.28971266746520996},{"id":"https://openalex.org/keywords/machine-learning","display_name":"Machine learning","score":0.09855416417121887}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7235375046730042},{"id":"https://openalex.org/C2776061190","wikidata":"https://www.wikidata.org/wiki/Q7451805","display_name":"Separation (statistics)","level":2,"score":0.6787939071655273},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.5226616263389587},{"id":"https://openalex.org/C50644808","wikidata":"https://www.wikidata.org/wiki/Q192776","display_name":"Artificial neural network","level":2,"score":0.44166430830955505},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.28971266746520996},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.09855416417121887}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.21437/interspeech.2020-1089","is_oa":false,"landing_page_url":"https://doi.org/10.21437/interspeech.2020-1089","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Interspeech 2020","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":23,"referenced_works":["https://openalex.org/W1494198834","https://openalex.org/W2042860487","https://openalex.org/W2117678320","https://openalex.org/W2141411743","https://openalex.org/W2288645994","https://openalex.org/W2293634267","https://openalex.org/W2517616541","https://openalex.org/W2591810467","https://openalex.org/W2594789261","https://openalex.org/W2631415506","https://openalex.org/W2734774145","https://openalex.org/W2803322398","https://openalex.org/W2889503488","https://openalex.org/W2889664202","https://openalex.org/W2943162839","https://openalex.org/W2964058413","https://openalex.org/W2972693890","https://openalex.org/W3004309045","https://openalex.org/W3008283340","https://openalex.org/W3015372568","https://openalex.org/W3016232124","https://openalex.org/W3104196160","https://openalex.org/W4385245566"],"related_works":["https://openalex.org/W2899084033","https://openalex.org/W2748952813","https://openalex.org/W2390279801","https://openalex.org/W2358668433","https://openalex.org/W2071676784","https://openalex.org/W2376932109","https://openalex.org/W2382290278","https://openalex.org/W2350741829","https://openalex.org/W2130043461","https://openalex.org/W4292513318"],"abstract_inverted_index":{"This":[0],"paper":[1],"proposes":[2],"a":[3,50,69,82,93],"neural":[4,41],"network":[5,52,108,123],"based":[6,43,91],"speech":[7,39,138,158],"separation":[8,40,159],"method":[9,153],"using":[10],"spatially":[11],"distributed":[12],"microphones.Unlike":[13],"with":[14,81,141],"traditional":[15],"microphone":[16],"array":[17],"settings,":[18],"neither":[19],"the":[20,34,73,78,151],"number":[21,84],"of":[22,36,85,120],"microphones":[23],"nor":[24],"their":[25],"spatial":[26],"arrangement":[27],"is":[28,54],"known":[29],"in":[30],"advance,":[31],"which":[32,131],"hinders":[33],"use":[35],"conventional":[37],"multi-channel":[38,157],"networks":[42],"on":[44,92],"fixed":[45],"size":[46],"input.To":[47],"overcome":[48],"this,":[49],"novel":[51],"architecture":[53],"proposed":[55,107,152],"that":[56,150],"interleaves":[57],"inter-channel":[58,65],"processing":[59,63,66,88],"layers":[60,67,89,121],"and":[61,101,113],"temporal":[62,87],"layers.The":[64],"apply":[68],"selfattention":[70],"mechanism":[71],"along":[72],"channel":[74,105],"dimension":[75],"to":[76,103,135],"exploit":[77],"information":[79,110],"obtained":[80],"varying":[83],"microphones.The":[86],"are":[90,132],"bidirectional":[94],"long":[95],"short":[96],"term":[97],"memory":[98],"(BLSTM)":[99],"model":[100],"applied":[102],"each":[104,129],"independently.The":[106],"leverages":[109],"across":[111],"time":[112],"space":[114],"by":[115],"stacking":[116],"these":[117],"two":[118],"kinds":[119],"alternately.Our":[122],"estimates":[124],"time-frequency":[125],"(TF)":[126],"masks":[127],"for":[128],"speaker,":[130],"then":[133],"used":[134],"generate":[136],"enhanced":[137],"signals":[139],"either":[140],"TF":[142],"masking":[143],"or":[144],"beamforming.Speech":[145],"recognition":[146],"experimental":[147],"results":[148],"show":[149],"significantly":[154],"outperforms":[155],"baseline":[156],"systems.":[160]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":2},{"year":2024,"cited_by_count":3},{"year":2023,"cited_by_count":6},{"year":2022,"cited_by_count":9},{"year":2021,"cited_by_count":9},{"year":2020,"cited_by_count":5}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
