{"id":"https://openalex.org/W4372348072","doi":"https://doi.org/10.1109/icassp49357.2023.10096464","title":"Zero-Shot Personalized Lip-To-Speech Synthesis with Face Image Based Voice Control","display_name":"Zero-Shot Personalized Lip-To-Speech Synthesis with Face Image Based Voice Control","publication_year":2023,"publication_date":"2023-05-05","ids":{"openalex":"https://openalex.org/W4372348072","doi":"https://doi.org/10.1109/icassp49357.2023.10096464"},"language":"en","primary_location":{"id":"doi:10.1109/icassp49357.2023.10096464","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp49357.2023.10096464","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5004434747","display_name":"Zheng-Yan Sheng","orcid":"https://orcid.org/0009-0002-0638-5530"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Zheng-Yan Sheng","raw_affiliation_strings":["University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","institution_ids":["https://openalex.org/I126520041"]},{"raw_affiliation_string":"National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China","institution_ids":["https://openalex.org/I126520041"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5045907056","display_name":"Yang Ai","orcid":"https://orcid.org/0000-0001-6668-022X"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Yang Ai","raw_affiliation_strings":["University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","institution_ids":["https://openalex.org/I126520041"]},{"raw_affiliation_string":"National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China","institution_ids":["https://openalex.org/I126520041"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5059767940","display_name":"Zhen-Hua Ling","orcid":"https://orcid.org/0000-0001-7853-5273"},"institutions":[{"id":"https://openalex.org/I126520041","display_name":"University of Science and Technology of China","ror":"https://ror.org/04c4dkn09","country_code":"CN","type":"education","lineage":["https://openalex.org/I126520041","https://openalex.org/I19820366"]}],"countries":["CN"],"is_corresponding":false,"raw_author_name":"Zhen-Hua Ling","raw_affiliation_strings":["University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"University of Science and Technology of China,National Engineering Research Center of Speech and Language Information Processing,Hefei,P.R. China","institution_ids":["https://openalex.org/I126520041"]},{"raw_affiliation_string":"National Engineering Research Center of Speech and Language Information Processing, University of Science and Technology of China, Hefei, P.R. China","institution_ids":["https://openalex.org/I126520041"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":[],"corresponding_institution_ids":["https://openalex.org/I126520041"],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":6,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":"1","last_page":"5"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11448","display_name":"Face recognition and analysis","score":0.9993000030517578,"subfield":{"id":"https://openalex.org/subfields/1707","display_name":"Computer Vision and Pattern Recognition"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9959999918937683,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7797533273696899},{"id":"https://openalex.org/keywords/speaker-recognition","display_name":"Speaker recognition","score":0.7048425078392029},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.6865763664245605},{"id":"https://openalex.org/keywords/speech-synthesis","display_name":"Speech synthesis","score":0.6317296028137207},{"id":"https://openalex.org/keywords/autoencoder","display_name":"Autoencoder","score":0.6010087132453918},{"id":"https://openalex.org/keywords/face","display_name":"Face (sociological concept)","score":0.5381762385368347},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.4851025342941284},{"id":"https://openalex.org/keywords/identity","display_name":"Identity (music)","score":0.45879289507865906},{"id":"https://openalex.org/keywords/speaker-diarisation","display_name":"Speaker diarisation","score":0.426777184009552},{"id":"https://openalex.org/keywords/control","display_name":"Control (management)","score":0.41296151280403137},{"id":"https://openalex.org/keywords/deep-learning","display_name":"Deep learning","score":0.2519495487213135},{"id":"https://openalex.org/keywords/linguistics","display_name":"Linguistics","score":0.11954078078269958}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7797533273696899},{"id":"https://openalex.org/C133892786","wikidata":"https://www.wikidata.org/wiki/Q1145189","display_name":"Speaker recognition","level":2,"score":0.7048425078392029},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.6865763664245605},{"id":"https://openalex.org/C14999030","wikidata":"https://www.wikidata.org/wiki/Q16346","display_name":"Speech synthesis","level":2,"score":0.6317296028137207},{"id":"https://openalex.org/C101738243","wikidata":"https://www.wikidata.org/wiki/Q786435","display_name":"Autoencoder","level":3,"score":0.6010087132453918},{"id":"https://openalex.org/C2779304628","wikidata":"https://www.wikidata.org/wiki/Q3503480","display_name":"Face (sociological concept)","level":2,"score":0.5381762385368347},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.4851025342941284},{"id":"https://openalex.org/C2778355321","wikidata":"https://www.wikidata.org/wiki/Q17079427","display_name":"Identity (music)","level":2,"score":0.45879289507865906},{"id":"https://openalex.org/C149838564","wikidata":"https://www.wikidata.org/wiki/Q7574248","display_name":"Speaker diarisation","level":3,"score":0.426777184009552},{"id":"https://openalex.org/C2775924081","wikidata":"https://www.wikidata.org/wiki/Q55608371","display_name":"Control (management)","level":2,"score":0.41296151280403137},{"id":"https://openalex.org/C108583219","wikidata":"https://www.wikidata.org/wiki/Q197536","display_name":"Deep learning","level":2,"score":0.2519495487213135},{"id":"https://openalex.org/C41895202","wikidata":"https://www.wikidata.org/wiki/Q8162","display_name":"Linguistics","level":1,"score":0.11954078078269958},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0},{"id":"https://openalex.org/C138885662","wikidata":"https://www.wikidata.org/wiki/Q5891","display_name":"Philosophy","level":0,"score":0.0},{"id":"https://openalex.org/C24890656","wikidata":"https://www.wikidata.org/wiki/Q82811","display_name":"Acoustics","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp49357.2023.10096464","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp49357.2023.10096464","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[{"id":"https://metadata.un.org/sdg/4","score":0.4699999988079071,"display_name":"Quality Education"}],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":26,"referenced_works":["https://openalex.org/W1522301498","https://openalex.org/W2015143272","https://openalex.org/W2115252128","https://openalex.org/W2138621090","https://openalex.org/W2187089797","https://openalex.org/W2532494225","https://openalex.org/W2808631503","https://openalex.org/W2916104401","https://openalex.org/W2963839617","https://openalex.org/W2964350391","https://openalex.org/W2972563022","https://openalex.org/W3016011581","https://openalex.org/W3035626590","https://openalex.org/W3096650361","https://openalex.org/W3097777922","https://openalex.org/W3157840621","https://openalex.org/W3162707322","https://openalex.org/W3163271138","https://openalex.org/W3211862173","https://openalex.org/W4224926225","https://openalex.org/W4225299282","https://openalex.org/W4283809657","https://openalex.org/W4296069328","https://openalex.org/W6631190155","https://openalex.org/W6677618333","https://openalex.org/W6803359641"],"related_works":["https://openalex.org/W2206035908","https://openalex.org/W2149220986","https://openalex.org/W1493012537","https://openalex.org/W4247736853","https://openalex.org/W2162158162","https://openalex.org/W1999004162","https://openalex.org/W2125642021","https://openalex.org/W1521049138","https://openalex.org/W2023466863","https://openalex.org/W2696990509"],"abstract_inverted_index":{"Lip-to-Speech":[0],"(Lip2Speech)":[1],"synthesis,":[2],"which":[3,76,96],"predicts":[4],"corresponding":[5],"speech":[6,49,107],"from":[7,46],"talking":[8],"face":[9,77,174],"images,":[10],"has":[11],"witnessed":[12],"significant":[13],"progress":[14],"with":[15,147,172],"various":[16],"models":[17],"and":[18,50,92,145],"training":[19],"strategies":[20],"in":[21,75],"a":[22,69,173],"series":[23],"of":[24,58,105,122,135,150],"independent":[25],"studies.":[26],"However,":[27],"existing":[28],"studies":[29],"can":[30],"not":[31],"achieve":[32],"voice":[33,103,128,182],"control":[34,79,101,181],"under":[35],"zero-shot":[36,70,168],"condition,":[37],"because":[38],"extra":[39],"speaker":[40,61,80,90,98,124],"embeddings":[41,99,125],"need":[42],"to":[43,87,100,118,180],"be":[44],"extracted":[45],"natural":[47,144],"reference":[48,178],"are":[51,142],"unavailable":[52],"when":[53],"only":[54],"the":[55,89,102,120,133,136,148,154,164],"silent":[56],"video":[57,152],"an":[59],"unseen":[60,109],"is":[62,85],"given.":[63],"In":[64],"this":[65,161],"paper,":[66],"we":[67,112],"propose":[68,113],"personalized":[71,169],"Lip2Speech":[72,170],"synthesis":[73,171],"method,":[74],"images":[78],"identities.":[81],"A":[82],"variational":[83],"autoencoder":[84],"adopted":[86],"disentangle":[88],"identity":[91],"linguistic":[93],"content":[94],"representations,":[95],"enables":[97],"characteristics":[104],"synthetic":[106,140],"for":[108],"speakers.":[110],"Furthermore,":[111],"associated":[114],"cross-modal":[115],"representation":[116],"learning":[117],"promote":[119],"ability":[121],"face-based":[123],"(FSE)":[126],"on":[127,167],"control.":[129],"Extensive":[130],"experiments":[131],"verify":[132],"effectiveness":[134],"proposed":[137],"method":[138],"whose":[139],"utterances":[141],"more":[143],"matching":[146],"personality":[149],"input":[151],"than":[153,177],"compared":[155],"methods.":[156],"To":[157],"our":[158],"best":[159],"knowledge,":[160],"paper":[162],"makes":[163],"first":[165],"attempt":[166],"image":[175],"rather":[176],"audio":[179],"characteristics.":[183]},"counts_by_year":[{"year":2026,"cited_by_count":2},{"year":2025,"cited_by_count":1},{"year":2024,"cited_by_count":1},{"year":2023,"cited_by_count":2}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
