{"id":"https://openalex.org/W2576152124","doi":"https://doi.org/10.1109/taslp.2017.2651398","title":"Extending the Cascaded Gaussian Mixture Regression Framework for Cross-Speaker Acoustic-Articulatory Mapping","display_name":"Extending the Cascaded Gaussian Mixture Regression Framework for Cross-Speaker Acoustic-Articulatory Mapping","publication_year":2017,"publication_date":"2017-01-11","ids":{"openalex":"https://openalex.org/W2576152124","doi":"https://doi.org/10.1109/taslp.2017.2651398","mag":"2576152124"},"language":"en","primary_location":{"id":"doi:10.1109/taslp.2017.2651398","is_oa":false,"landing_page_url":"https://doi.org/10.1109/taslp.2017.2651398","pdf_url":null,"source":{"id":"https://openalex.org/S4210169297","display_name":"IEEE/ACM Transactions on Audio Speech and Language Processing","issn_l":"2329-9290","issn":["2329-9290","2329-9304"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE/ACM Transactions on Audio, Speech, and Language Processing","raw_type":"journal-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5020392160","display_name":"Laurent Girin","orcid":"https://orcid.org/0000-0002-9214-8760"},"institutions":[{"id":"https://openalex.org/I1326498283","display_name":"Institut national de recherche en sciences et technologies du num\u00e9rique","ror":"https://ror.org/02kvxyf05","country_code":"FR","type":"government","lineage":["https://openalex.org/I1326498283"]},{"id":"https://openalex.org/I4210101348","display_name":"Centre Inria de l'Universit\u00e9 Grenoble Alpes","ror":"https://ror.org/00n8d6z93","country_code":"FR","type":"facility","lineage":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]},{"id":"https://openalex.org/I4210124956","display_name":"GIPSA-Lab","ror":"https://ror.org/02wrme198","country_code":"FR","type":"facility","lineage":["https://openalex.org/I106785703","https://openalex.org/I1294671590","https://openalex.org/I4210124956","https://openalex.org/I899635006","https://openalex.org/I899635006"]},{"id":"https://openalex.org/I899635006","display_name":"Universit\u00e9 Grenoble Alpes","ror":"https://ror.org/02rx3b187","country_code":"FR","type":"education","lineage":["https://openalex.org/I899635006"]}],"countries":["FR"],"is_corresponding":false,"raw_author_name":"Laurent Girin","raw_affiliation_strings":["INRIA Grenoble Rh\u00f4ne-Alpes, Montbonnot, France","University of Grenoble Alpes, GIPSA-lab, Grenoble, France","PERCEPTION  - Interpretation and Modelling of Images and Videos (INRIA Rh\u00f4ne-Alpes 655 avenue de l'Europe 38330 Montbonnot, France - France)","GIPSA-CRISSP - GIPSA - Cognitive Robotics, Interactive Systems, & Speech Processing (GIPSA-lab, 11 rue des Math\u00e9matiques, Grenoble Campus BP46, F-38402 SAINT MARTIN D'HERES CEDEX - France)"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"INRIA Grenoble Rh\u00f4ne-Alpes, Montbonnot, France","institution_ids":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]},{"raw_affiliation_string":"University of Grenoble Alpes, GIPSA-lab, Grenoble, France","institution_ids":["https://openalex.org/I4210124956","https://openalex.org/I899635006"]},{"raw_affiliation_string":"PERCEPTION  - Interpretation and Modelling of Images and Videos (INRIA Rh\u00f4ne-Alpes 655 avenue de l'Europe 38330 Montbonnot, France - France)","institution_ids":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]},{"raw_affiliation_string":"GIPSA-CRISSP - GIPSA - Cognitive Robotics, Interactive Systems, & Speech Processing (GIPSA-lab, 11 rue des Math\u00e9matiques, Grenoble Campus BP46, F-38402 SAINT MARTIN D'HERES CEDEX - France)","institution_ids":["https://openalex.org/I4210124956"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5060372031","display_name":"Thomas Hueber","orcid":"https://orcid.org/0000-0002-8296-5177"},"institutions":[{"id":"https://openalex.org/I1294671590","display_name":"Centre National de la Recherche Scientifique","ror":"https://ror.org/02feahw73","country_code":"FR","type":"government","lineage":["https://openalex.org/I1294671590"]},{"id":"https://openalex.org/I4210124956","display_name":"GIPSA-Lab","ror":"https://ror.org/02wrme198","country_code":"FR","type":"facility","lineage":["https://openalex.org/I106785703","https://openalex.org/I1294671590","https://openalex.org/I4210124956","https://openalex.org/I899635006","https://openalex.org/I899635006"]},{"id":"https://openalex.org/I899635006","display_name":"Universit\u00e9 Grenoble Alpes","ror":"https://ror.org/02rx3b187","country_code":"FR","type":"education","lineage":["https://openalex.org/I899635006"]}],"countries":["FR"],"is_corresponding":false,"raw_author_name":"Thomas Hueber","raw_affiliation_strings":["CNRS, University of Grenoble Alpes, GIPSA-lab, Grenoble, France","GIPSA-CRISSP - GIPSA - Cognitive Robotics, Interactive Systems, & Speech Processing (GIPSA-lab, 11 rue des Math\u00e9matiques, Grenoble Campus BP46, F-38402 SAINT MARTIN D'HERES CEDEX - France)"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"CNRS, University of Grenoble Alpes, GIPSA-lab, Grenoble, France","institution_ids":["https://openalex.org/I1294671590","https://openalex.org/I4210124956","https://openalex.org/I899635006"]},{"raw_affiliation_string":"GIPSA-CRISSP - GIPSA - Cognitive Robotics, Interactive Systems, & Speech Processing (GIPSA-lab, 11 rue des Math\u00e9matiques, Grenoble Campus BP46, F-38402 SAINT MARTIN D'HERES CEDEX - France)","institution_ids":["https://openalex.org/I4210124956"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5066621495","display_name":"Xavier Alameda-Pineda","orcid":"https://orcid.org/0000-0002-5354-1084"},"institutions":[{"id":"https://openalex.org/I1326498283","display_name":"Institut national de recherche en sciences et technologies du num\u00e9rique","ror":"https://ror.org/02kvxyf05","country_code":"FR","type":"government","lineage":["https://openalex.org/I1326498283"]},{"id":"https://openalex.org/I193223587","display_name":"University of Trento","ror":"https://ror.org/05trd4x28","country_code":"IT","type":"education","lineage":["https://openalex.org/I193223587"]},{"id":"https://openalex.org/I4210101348","display_name":"Centre Inria de l'Universit\u00e9 Grenoble Alpes","ror":"https://ror.org/00n8d6z93","country_code":"FR","type":"facility","lineage":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]}],"countries":["FR","IT"],"is_corresponding":false,"raw_author_name":"Xavier Alameda-Pineda","raw_affiliation_strings":["INRIA Grenoble Rh\u00f4ne-Alpes, Montbonnot, France","PERCEPTION  - Interpretation and Modelling of Images and Videos (INRIA Rh\u00f4ne-Alpes 655 avenue de l'Europe 38330 Montbonnot, France - France)","UNITN - Universit\u00e0 degli Studi di Trento =  University of Trento (via Calepina, 14 - I-38122 Trento - Italy)"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"INRIA Grenoble Rh\u00f4ne-Alpes, Montbonnot, France","institution_ids":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]},{"raw_affiliation_string":"PERCEPTION  - Interpretation and Modelling of Images and Videos (INRIA Rh\u00f4ne-Alpes 655 avenue de l'Europe 38330 Montbonnot, France - France)","institution_ids":["https://openalex.org/I1326498283","https://openalex.org/I4210101348"]},{"raw_affiliation_string":"UNITN - Universit\u00e0 degli Studi di Trento =  University of Trento (via Calepina, 14 - I-38122 Trento - Italy)","institution_ids":["https://openalex.org/I193223587"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":6,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":0.6909,"has_fulltext":false,"cited_by_count":6,"citation_normalized_percentile":{"value":0.78335907,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":89,"max":96},"biblio":{"volume":"25","issue":"3","first_page":"662","last_page":"673"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9997000098228455,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9993000030517578,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.6229928135871887},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.5520492196083069},{"id":"https://openalex.org/keywords/inversion","display_name":"Inversion (geology)","score":0.5092965364456177},{"id":"https://openalex.org/keywords/gaussian-process","display_name":"Gaussian process","score":0.45447617769241333},{"id":"https://openalex.org/keywords/mixture-model","display_name":"Mixture model","score":0.43046835064888},{"id":"https://openalex.org/keywords/gaussian","display_name":"Gaussian","score":0.396784245967865},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.2631242871284485},{"id":"https://openalex.org/keywords/physics","display_name":"Physics","score":0.11782407760620117}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.6229928135871887},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.5520492196083069},{"id":"https://openalex.org/C1893757","wikidata":"https://www.wikidata.org/wiki/Q3653001","display_name":"Inversion (geology)","level":3,"score":0.5092965364456177},{"id":"https://openalex.org/C61326573","wikidata":"https://www.wikidata.org/wiki/Q1496376","display_name":"Gaussian process","level":3,"score":0.45447617769241333},{"id":"https://openalex.org/C61224824","wikidata":"https://www.wikidata.org/wiki/Q2260434","display_name":"Mixture model","level":2,"score":0.43046835064888},{"id":"https://openalex.org/C163716315","wikidata":"https://www.wikidata.org/wiki/Q901177","display_name":"Gaussian","level":2,"score":0.396784245967865},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.2631242871284485},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.11782407760620117},{"id":"https://openalex.org/C151730666","wikidata":"https://www.wikidata.org/wiki/Q7205","display_name":"Paleontology","level":1,"score":0.0},{"id":"https://openalex.org/C86803240","wikidata":"https://www.wikidata.org/wiki/Q420","display_name":"Biology","level":0,"score":0.0},{"id":"https://openalex.org/C109007969","wikidata":"https://www.wikidata.org/wiki/Q749565","display_name":"Structural basin","level":2,"score":0.0},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1109/taslp.2017.2651398","is_oa":false,"landing_page_url":"https://doi.org/10.1109/taslp.2017.2651398","pdf_url":null,"source":{"id":"https://openalex.org/S4210169297","display_name":"IEEE/ACM Transactions on Audio Speech and Language Processing","issn_l":"2329-9290","issn":["2329-9290","2329-9304"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE/ACM Transactions on Audio, Speech, and Language Processing","raw_type":"journal-article"},{"id":"pmh:oai:HAL:hal-01485540v1","is_oa":false,"landing_page_url":"https://hal.science/hal-01485540","pdf_url":null,"source":{"id":"https://openalex.org/S4306402512","display_name":"HAL (Le Centre pour la Communication Scientifique Directe)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I1294671590","host_organization_name":"Centre National de la Recherche Scientifique","host_organization_lineage":["https://openalex.org/I1294671590"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"IEEE/ACM Transactions on Audio, Speech and Language Processing, 2017, 25 (3), pp.662-673. &#x27E8;10.1109/TASLP.2017.2651398&#x27E9;","raw_type":"info:eu-repo/semantics/article"}],"best_oa_location":null,"sustainable_development_goals":[{"display_name":"Quality Education","id":"https://metadata.un.org/sdg/4","score":0.5400000214576721}],"awards":[],"funders":[],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":36,"referenced_works":["https://openalex.org/W40881502","https://openalex.org/W60202269","https://openalex.org/W1506806321","https://openalex.org/W1520325657","https://openalex.org/W1562842113","https://openalex.org/W1579271636","https://openalex.org/W1866403196","https://openalex.org/W1945090009","https://openalex.org/W1973878427","https://openalex.org/W1982854652","https://openalex.org/W2037740282","https://openalex.org/W2064234620","https://openalex.org/W2073739879","https://openalex.org/W2084414314","https://openalex.org/W2100969003","https://openalex.org/W2116257577","https://openalex.org/W2117853077","https://openalex.org/W2120605154","https://openalex.org/W2153991100","https://openalex.org/W2154543878","https://openalex.org/W2156142001","https://openalex.org/W2159801939","https://openalex.org/W2165108269","https://openalex.org/W2293512268","https://openalex.org/W2430424407","https://openalex.org/W2460967393","https://openalex.org/W2488678869","https://openalex.org/W2509129594","https://openalex.org/W2595142274","https://openalex.org/W3100003654","https://openalex.org/W6601659613","https://openalex.org/W6602447114","https://openalex.org/W6631002676","https://openalex.org/W6640608080","https://openalex.org/W6697107752","https://openalex.org/W6734797129"],"related_works":["https://openalex.org/W1952261593","https://openalex.org/W1975321310","https://openalex.org/W2990323019","https://openalex.org/W2014494654","https://openalex.org/W1516392727","https://openalex.org/W1578916557","https://openalex.org/W1992295166","https://openalex.org/W2143508933","https://openalex.org/W2169866437","https://openalex.org/W1964286703"],"abstract_inverted_index":{"This":[0],"paper":[1],"addresses":[2],"the":[3,15,31,35,56,77,102,115,124,135,149,154,160,164,171,175,194,203,210,217,221],"adaptation":[4,78,186],"of":[5,10,17,25,34,47,94,163,179,193,216],"an":[6,190],"acoustic-articulatory":[7,92,144,199],"inversion":[8,93],"model":[9,44,129,138],"a":[11,22,41,69],"reference":[12,36,89,155],"speaker":[13,37,64,90],"to":[14,134,182,213],"voice":[16],"another":[18,128],"source":[19,87,150],"speaker,":[20],"using":[21],"limited":[23,185],"amount":[24],"audio-only":[26],"data.":[27,187],"In":[28,98,119],"this":[29,120,137],"study,":[30],"articulatory-acoustic":[32],"relationship":[33],"is":[38,53],"modeled":[39],"by":[40,55],"Gaussian":[42,58],"mixture":[43,59],"and":[45,88,91,153,201,223],"inference":[46],"articulatory":[48],"data":[49,52,177,200],"from":[50],"acoustic":[51],"made":[54],"associated":[57],"regression":[60],"(GMR).":[61],"To":[62],"address":[63],"adaptation,":[65],"we":[66,100,122],"previously":[67],"proposed":[68,101],"general":[70],"framework":[71,126],"called":[72,130],"Cascaded-GMR":[73],"(C-GMR)":[74],"which":[75,108],"decomposes":[76],"process":[79],"into":[80],"two":[81],"consecutive":[82],"steps:":[83],"spectral":[84,96],"conversion":[85],"between":[86,148],"converted":[95],"trajectories.":[97],"particular,":[99],"integrated":[103],"C-GMR":[104,125,218],"technique":[105],"(IC-GMR)":[106],"in":[107,114],"both":[109,197],"steps":[110],"are":[111],"tied":[112],"together":[113],"same":[116],"probabilistic":[117],"model.":[118],"paper,":[121],"extend":[123],"with":[127,184],"Joint-GMR":[131],"(J-GMR).":[132],"Contrary":[133],"IC-GMR,":[136,222],"aims":[139],"at":[140],"exploiting":[141],"all":[142],"potential":[143],"relationships,":[145],"including":[146],"those":[147],"speaker's":[151,156],"acoustics":[152],"articulation.":[157],"We":[158,188,208],"present":[159],"full":[161],"derivation":[162],"exact":[165],"expectation-maximization":[166],"(EM)":[167],"training":[168],"algorithm":[169],"for":[170],"J-GMR.":[172],"It":[173],"exploits":[174],"missing":[176],"methodology":[178],"machine":[180],"learning":[181],"deal":[183],"provide":[189],"extensive":[191],"evaluation":[192],"J-GMR":[195,211],"on":[196,202],"synthetic":[198],"multispeaker":[204],"MOCHA":[205],"EMA":[206],"database.":[207],"compare":[209],"performance":[212],"other":[214],"models":[215],"framework,":[219],"notably":[220],"discuss":[224],"their":[225],"respective":[226],"merits.":[227]},"counts_by_year":[{"year":2022,"cited_by_count":1},{"year":2021,"cited_by_count":2},{"year":2019,"cited_by_count":2},{"year":2017,"cited_by_count":1}],"updated_date":"2026-07-25T09:21:30.201066","created_date":"2025-10-10T00:00:00"}
