{"id":"https://openalex.org/W4224919629","doi":"https://doi.org/10.1109/icassp43922.2022.9747765","title":"TEA-PSE: Tencent-Ethereal-Audio-Lab Personalized Speech Enhancement System for ICASSP 2022 DNS Challenge","display_name":"TEA-PSE: Tencent-Ethereal-Audio-Lab Personalized Speech Enhancement System for ICASSP 2022 DNS Challenge","publication_year":2022,"publication_date":"2022-04-27","ids":{"openalex":"https://openalex.org/W4224919629","doi":"https://doi.org/10.1109/icassp43922.2022.9747765"},"language":"en","primary_location":{"id":"doi:10.1109/icassp43922.2022.9747765","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp43922.2022.9747765","pdf_url":null,"source":{"id":"https://openalex.org/S4363607702","display_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5015448985","display_name":"Yukai Ju","orcid":null},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]},{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Yukai Ju","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5035226758","display_name":"Wei Rao","orcid":"https://orcid.org/0000-0002-7237-0874"},"institutions":[{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Wei Rao","raw_affiliation_strings":["Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","institution_ids":["https://openalex.org/I2250653659"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5001664777","display_name":"Xiaopeng Yan","orcid":null},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]},{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Xiaopeng Yan","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5010250251","display_name":"Yihui Fu","orcid":null},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Yihui Fu","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5017208328","display_name":"Shubo Lv","orcid":null},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]},{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Shubo Lv","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5102891716","display_name":"Luyao Cheng","orcid":"https://orcid.org/0009-0006-1311-8448"},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Luyao Cheng","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5084128157","display_name":"Yannan Wang","orcid":"https://orcid.org/0000-0001-7248-4954"},"institutions":[{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Yannan Wang","raw_affiliation_strings":["Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","institution_ids":["https://openalex.org/I2250653659"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5100668966","display_name":"Lei Xie","orcid":"https://orcid.org/0000-0001-8234-0823"},"institutions":[{"id":"https://openalex.org/I17145004","display_name":"Northwestern Polytechnical University","ror":"https://ror.org/01y0j0j86","country_code":"CN","type":"education","lineage":["https://openalex.org/I17145004"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Lei Xie","raw_affiliation_strings":["Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xi&#x2019;an,China","institution_ids":["https://openalex.org/I17145004"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5078353046","display_name":"Shidong Shang","orcid":null},"institutions":[{"id":"https://openalex.org/I2250653659","display_name":"Tencent (China)","ror":"https://ror.org/00hhjss72","country_code":"CN","type":"company","lineage":["https://openalex.org/I2250653659"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Shidong Shang","raw_affiliation_strings":["Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Tencent Corporation,Tencent Ethereal Audio Lab,Shenzhen,China","institution_ids":["https://openalex.org/I2250653659"]},{"raw_affiliation_string":"Tencent Ethereal Audio Lab, Tencent Corporation, Shenzhen, China","institution_ids":["https://openalex.org/I2250653659"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":9,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":4.2364,"has_fulltext":false,"cited_by_count":36,"citation_normalized_percentile":{"value":0.95845137,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":94,"max":100},"biblio":{"volume":null,"issue":null,"first_page":"9291","last_page":"9295"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9995999932289124,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9965000152587891,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.767082929611206},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.7291445732116699},{"id":"https://openalex.org/keywords/speech-enhancement","display_name":"Speech enhancement","score":0.6609783172607422},{"id":"https://openalex.org/keywords/noise","display_name":"Noise (video)","score":0.5086994767189026},{"id":"https://openalex.org/keywords/set","display_name":"Set (abstract data type)","score":0.4315020740032196},{"id":"https://openalex.org/keywords/test-set","display_name":"Test set","score":0.41984423995018005},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.323464035987854},{"id":"https://openalex.org/keywords/noise-reduction","display_name":"Noise reduction","score":0.2958058714866638}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.767082929611206},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.7291445732116699},{"id":"https://openalex.org/C2776182073","wikidata":"https://www.wikidata.org/wiki/Q7575395","display_name":"Speech enhancement","level":3,"score":0.6609783172607422},{"id":"https://openalex.org/C99498987","wikidata":"https://www.wikidata.org/wiki/Q2210247","display_name":"Noise (video)","level":3,"score":0.5086994767189026},{"id":"https://openalex.org/C177264268","wikidata":"https://www.wikidata.org/wiki/Q1514741","display_name":"Set (abstract data type)","level":2,"score":0.4315020740032196},{"id":"https://openalex.org/C169903167","wikidata":"https://www.wikidata.org/wiki/Q3985153","display_name":"Test set","level":2,"score":0.41984423995018005},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.323464035987854},{"id":"https://openalex.org/C163294075","wikidata":"https://www.wikidata.org/wiki/Q581861","display_name":"Noise reduction","level":2,"score":0.2958058714866638},{"id":"https://openalex.org/C115961682","wikidata":"https://www.wikidata.org/wiki/Q860623","display_name":"Image (mathematics)","level":2,"score":0.0},{"id":"https://openalex.org/C199360897","wikidata":"https://www.wikidata.org/wiki/Q9143","display_name":"Programming language","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp43922.2022.9747765","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp43922.2022.9747765","pdf_url":null,"source":{"id":"https://openalex.org/S4363607702","display_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"conference"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[{"display_name":"Peace, Justice and strong institutions","score":0.7300000190734863,"id":"https://metadata.un.org/sdg/16"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":40,"referenced_works":["https://openalex.org/W2219249508","https://openalex.org/W2593116425","https://openalex.org/W2752782242","https://openalex.org/W2792764867","https://openalex.org/W2808631503","https://openalex.org/W2928165649","https://openalex.org/W2936774411","https://openalex.org/W2951130829","https://openalex.org/W2952218014","https://openalex.org/W2964054038","https://openalex.org/W2973062255","https://openalex.org/W2991361823","https://openalex.org/W3016361963","https://openalex.org/W3024869864","https://openalex.org/W3094806148","https://openalex.org/W3096090308","https://openalex.org/W3096408984","https://openalex.org/W3097653961","https://openalex.org/W3099330747","https://openalex.org/W3103434036","https://openalex.org/W3120336970","https://openalex.org/W3161140524","https://openalex.org/W3162534564","https://openalex.org/W3194338569","https://openalex.org/W3195288392","https://openalex.org/W3196570692","https://openalex.org/W3197260772","https://openalex.org/W3198543387","https://openalex.org/W3198680319","https://openalex.org/W3206706278","https://openalex.org/W4225302959","https://openalex.org/W4225905067","https://openalex.org/W4232282348","https://openalex.org/W6688816777","https://openalex.org/W6734260513","https://openalex.org/W6749825310","https://openalex.org/W6761176859","https://openalex.org/W6794354017","https://openalex.org/W6802436868","https://openalex.org/W6810523761"],"related_works":["https://openalex.org/W1600259599","https://openalex.org/W1630865680","https://openalex.org/W3096184950","https://openalex.org/W2770665941","https://openalex.org/W2782782444","https://openalex.org/W2342291550","https://openalex.org/W2252747487","https://openalex.org/W4307413935","https://openalex.org/W3158697290","https://openalex.org/W4318718858"],"abstract_inverted_index":{"This":[0],"paper":[1],"describes":[2],"Tencent":[3],"Ethereal":[4],"Audio":[5],"Lab":[6],"\u2013":[7],"Northwestern":[8],"Polytechnical":[9],"University":[10],"personalized":[11],"speech":[12,41,65,82,120,141,149],"enhancement":[13,42,66],"(TEA-PSE)":[14],"system":[15,30,158],"submitted":[16],"to":[17,61,92,136],"track":[18,193],"2":[19],"of":[20,79,155,178],"the":[21,33,45,63,77,80,89,101,125,132,174,179,183],"ICASSP":[22],"2022":[23],"Deep":[24],"Noise":[25],"Suppression":[26],"(DNS)":[27],"challenge.":[28],"Our":[29,157],"specifically":[31],"combines":[32],"dual-stage":[34,58],"network":[35,49,59,109],"which":[36,50,85,144,181],"is":[37,83,86,128,142,145],"a":[38,94,112],"superior":[39],"real-time":[40],"framework":[43],"with":[44,88],"ECAPA-TDNN":[46],"speaker":[47,55],"embedding":[48],"achieves":[51],"state-of-the-art":[52],"performance":[53,151],"in":[54,73,104,161,169,192],"verification.":[56],"The":[57],"aims":[60],"decouple":[62],"primal":[64],"problem":[67],"into":[68],"multiple":[69],"easier":[70],"sub-problems.":[71],"Specifically,":[72],"stage":[74,105],"1,":[75],"only":[76],"magnitude":[78],"target":[81,140],"estimated,":[84],"incorporated":[87],"noisy":[90],"phase":[91,126],"obtain":[93],"coarse":[95],"complex":[96],"spectrum":[97],"estimation.":[98],"To":[99],"facilitate":[100],"formal":[102],"estimation,":[103],"2,":[106],"an":[107],"auxiliary":[108],"serves":[110],"as":[111],"post-processing":[113],"module,":[114],"where":[115],"residual":[116],"noise":[117],"and":[118,124,152,167,189],"interfering":[119],"are":[121],"further":[122],"suppressed":[123],"information":[127],"effectively":[129],"modified.":[130],"With":[131],"asymmetric":[133],"loss":[134],"function":[135],"penalize":[137],"over-suppression,":[138],"more":[139],"preserved,":[143],"helpful":[146],"for":[147],"both":[148],"recognition":[150],"subjective":[153],"sense":[154],"hearing.":[156],"reaches":[159],"3.97":[160],"overall":[162],"audio":[163],"quality":[164],"(OVRL)":[165],"MOS":[166],"0.69":[168],"word":[170],"accuracy":[171],"(WAcc)":[172],"on":[173],"blind":[175],"test":[176],"set":[177],"challenge,":[180],"outperforms":[182],"DNS":[184],"baseline":[185],"by":[186],"0.57":[187],"OVRL":[188],"ranks":[190],"1st":[191],"2.":[194]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":6},{"year":2024,"cited_by_count":11},{"year":2023,"cited_by_count":16},{"year":2022,"cited_by_count":2}],"updated_date":"2026-06-11T09:08:48.828518","created_date":"2025-10-10T00:00:00"}
