{"id":"https://openalex.org/W3163019736","doi":"https://doi.org/10.1109/icassp39728.2021.9413436","title":"End-To-End Speaker Diarization as Post-Processing","display_name":"End-To-End Speaker Diarization as Post-Processing","publication_year":2021,"publication_date":"2021-05-13","ids":{"openalex":"https://openalex.org/W3163019736","doi":"https://doi.org/10.1109/icassp39728.2021.9413436","mag":"3163019736"},"language":"en","primary_location":{"id":"doi:10.1109/icassp39728.2021.9413436","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp39728.2021.9413436","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5026324656","display_name":"Shota Horiguchi","orcid":"https://orcid.org/0000-0002-3166-4956"},"institutions":[{"id":"https://openalex.org/I65143321","display_name":"Hitachi (Japan)","ror":"https://ror.org/02exqgm79","country_code":"JP","type":"company","lineage":["https://openalex.org/I65143321"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Shota Horiguchi","raw_affiliation_strings":["Research &amp; Development Group,Hitachi, Ltd,Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Research &amp; Development Group,Hitachi, Ltd,Japan","institution_ids":["https://openalex.org/I65143321"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5059858850","display_name":"Leibny Paola Garcia","orcid":"https://orcid.org/0000-0002-7449-5726"},"institutions":[{"id":"https://openalex.org/I145311948","display_name":"Johns Hopkins University","ror":"https://ror.org/00za53h95","country_code":"US","type":"education","lineage":["https://openalex.org/I145311948"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Paola Garcia","raw_affiliation_strings":["Johns Hopkins University,Center for Language and Speech Processing,USA","Center for Language and Speech Processing, Johns Hopkins University, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Johns Hopkins University,Center for Language and Speech Processing,USA","institution_ids":["https://openalex.org/I145311948"]},{"raw_affiliation_string":"Center for Language and Speech Processing, Johns Hopkins University, USA","institution_ids":["https://openalex.org/I145311948"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5044818016","display_name":"Yusuke Fujita","orcid":"https://orcid.org/0000-0002-6523-8146"},"institutions":[{"id":"https://openalex.org/I65143321","display_name":"Hitachi (Japan)","ror":"https://ror.org/02exqgm79","country_code":"JP","type":"company","lineage":["https://openalex.org/I65143321"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Yusuke Fujita","raw_affiliation_strings":["Research &amp; Development Group,Hitachi, Ltd,Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Research &amp; Development Group,Hitachi, Ltd,Japan","institution_ids":["https://openalex.org/I65143321"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5001291873","display_name":"Shinji Watanabe","orcid":"https://orcid.org/0000-0002-5970-8631"},"institutions":[{"id":"https://openalex.org/I145311948","display_name":"Johns Hopkins University","ror":"https://ror.org/00za53h95","country_code":"US","type":"education","lineage":["https://openalex.org/I145311948"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Shinji Watanabe","raw_affiliation_strings":["Johns Hopkins University,Center for Language and Speech Processing,USA","Center for Language and Speech Processing, Johns Hopkins University, USA"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Johns Hopkins University,Center for Language and Speech Processing,USA","institution_ids":["https://openalex.org/I145311948"]},{"raw_affiliation_string":"Center for Language and Speech Processing, Johns Hopkins University, USA","institution_ids":["https://openalex.org/I145311948"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5076987349","display_name":"Kenji Nagamatsu","orcid":null},"institutions":[{"id":"https://openalex.org/I65143321","display_name":"Hitachi (Japan)","ror":"https://ror.org/02exqgm79","country_code":"JP","type":"company","lineage":["https://openalex.org/I65143321"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Kenji Nagamatsu","raw_affiliation_strings":["Research &amp; Development Group,Hitachi, Ltd,Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Research &amp; Development Group,Hitachi, Ltd,Japan","institution_ids":["https://openalex.org/I65143321"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":4.982,"has_fulltext":false,"cited_by_count":40,"citation_normalized_percentile":{"value":0.96316052,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":94,"max":99},"biblio":{"volume":null,"issue":null,"first_page":"7188","last_page":"7192"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9973000288009644,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9973000288009644,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10403","display_name":"Phonetics and Phonology Research","score":0.9901000261306763,"subfield":{"id":"https://openalex.org/subfields/3205","display_name":"Experimental and Cognitive Psychology"},"field":{"id":"https://openalex.org/fields/32","display_name":"Psychology"},"domain":{"id":"https://openalex.org/domains/2","display_name":"Social Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9776999950408936,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/end-to-end-principle","display_name":"End-to-end principle","score":0.6646676063537598},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6383712887763977},{"id":"https://openalex.org/keywords/speaker-diarisation","display_name":"Speaker diarisation","score":0.5970564484596252},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.5469737648963928},{"id":"https://openalex.org/keywords/speaker-recognition","display_name":"Speaker recognition","score":0.39708763360977173},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.24137148261070251}],"concepts":[{"id":"https://openalex.org/C74296488","wikidata":"https://www.wikidata.org/wiki/Q2527392","display_name":"End-to-end principle","level":2,"score":0.6646676063537598},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6383712887763977},{"id":"https://openalex.org/C149838564","wikidata":"https://www.wikidata.org/wiki/Q7574248","display_name":"Speaker diarisation","level":3,"score":0.5970564484596252},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.5469737648963928},{"id":"https://openalex.org/C133892786","wikidata":"https://www.wikidata.org/wiki/Q1145189","display_name":"Speaker recognition","level":2,"score":0.39708763360977173},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.24137148261070251}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp39728.2021.9413436","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp39728.2021.9413436","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":39,"referenced_works":["https://openalex.org/W1965819578","https://openalex.org/W2081074144","https://openalex.org/W2083751884","https://openalex.org/W2219249508","https://openalex.org/W2460742184","https://openalex.org/W2696967604","https://openalex.org/W2805869973","https://openalex.org/W2889418727","https://openalex.org/W2891054259","https://openalex.org/W2896538040","https://openalex.org/W2900212944","https://openalex.org/W2963403868","https://openalex.org/W2972449503","https://openalex.org/W2972680151","https://openalex.org/W2972949456","https://openalex.org/W2981103786","https://openalex.org/W2989863749","https://openalex.org/W3007483572","https://openalex.org/W3008357631","https://openalex.org/W3010196324","https://openalex.org/W3015544392","https://openalex.org/W3015780472","https://openalex.org/W3016031604","https://openalex.org/W3016244460","https://openalex.org/W3024085360","https://openalex.org/W3025260599","https://openalex.org/W3025757635","https://openalex.org/W3033627755","https://openalex.org/W3095212884","https://openalex.org/W3096090308","https://openalex.org/W3131736947","https://openalex.org/W4237168004","https://openalex.org/W4385245566","https://openalex.org/W6688816777","https://openalex.org/W6739901393","https://openalex.org/W6768584815","https://openalex.org/W6774249064","https://openalex.org/W6774558098","https://openalex.org/W6779069803"],"related_works":["https://openalex.org/W2206035908","https://openalex.org/W1491159402","https://openalex.org/W4297807400","https://openalex.org/W2249138175","https://openalex.org/W4389984014","https://openalex.org/W2144208207","https://openalex.org/W1509309911","https://openalex.org/W1599425004","https://openalex.org/W2118860825","https://openalex.org/W2096510939"],"abstract_inverted_index":{"This":[0],"paper":[1],"investigates":[2],"the":[3,24,44,57,78,102,115,119,122,127,134,139,142],"utilization":[4],"of":[5,12,23,26,70,80,101,121,141],"an":[6],"end-to-end":[7,48,96],"diarization":[8,17,49,97],"model":[9],"as":[10,59,99],"post-processing":[11,100],"conventional":[13],"clustering-based":[14,107],"diarization.":[15],"Clustering-based":[16],"methods":[18,50,64,144],"partition":[19],"frames":[20],"into":[21],"clusters":[22],"number":[25,69,79],"speakers;":[27],"thus,":[28],"they":[29,72],"typically":[30],"cannot":[31],"handle":[32,52],"overlapping":[33,53],"speech":[34,54],"because":[35],"each":[36,87],"frame":[37],"is":[38,82],"assigned":[39],"to":[40,92,125],"one":[41],"speaker.":[42],"On":[43],"other":[45],"hand,":[46],"some":[47,63],"can":[51,65],"by":[55,105],"treating":[56],"problem":[58],"multi-label":[60],"classification.":[61],"Although":[62],"treat":[66],"a":[67,94,106],"flexible":[68],"speakers,":[71],"do":[73],"not":[74],"perform":[75],"well":[76],"when":[77],"speakers":[81,113,124],"large.":[83],"To":[84],"compensate":[85],"for":[86],"other\u2019s":[88],"weakness,":[89],"we":[90],"propose":[91],"use":[93],"two-speaker":[95],"method":[98],"results":[103,116,120,131],"obtained":[104],"method.":[108],"We":[109],"iteratively":[110],"select":[111],"two":[112,123],"from":[114],"and":[117,148],"update":[118],"improve":[126],"overlapped":[128],"region.":[129],"Experimental":[130],"show":[132],"that":[133],"proposed":[135],"algorithm":[136],"consistently":[137],"improved":[138],"performance":[140],"state-of-the-art":[143],"across":[145],"CALLHOME,":[146],"AMI,":[147],"DIHARD":[149],"II":[150],"datasets.":[151]},"counts_by_year":[{"year":2026,"cited_by_count":1},{"year":2025,"cited_by_count":5},{"year":2024,"cited_by_count":9},{"year":2023,"cited_by_count":6},{"year":2022,"cited_by_count":9},{"year":2021,"cited_by_count":10}],"updated_date":"2026-07-29T14:22:42.915294","created_date":"2025-10-10T00:00:00"}
