{"id":"https://openalex.org/W2395849284","doi":"https://doi.org/10.1109/icassp.2016.7472736","title":"A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis","display_name":"A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis","publication_year":2016,"publication_date":"2016-03-01","ids":{"openalex":"https://openalex.org/W2395849284","doi":"https://doi.org/10.1109/icassp.2016.7472736","mag":"2395849284"},"language":"en","primary_location":{"id":"doi:10.1109/icassp.2016.7472736","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2016.7472736","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://www.research.ed.ac.uk/en/publications/eabdc1dc-b33c-4460-a915-f27a153dcde9","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5062895056","display_name":"Shinji Takaki","orcid":"https://orcid.org/0000-0001-7294-7699"},"institutions":[{"id":"https://openalex.org/I184597095","display_name":"National Institute of Informatics","ror":"https://ror.org/04ksd4g47","country_code":"JP","type":"facility","lineage":["https://openalex.org/I1319490839","https://openalex.org/I184597095","https://openalex.org/I4210158934"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Shinji Takaki","raw_affiliation_strings":["National Institute of Informatics, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"National Institute of Informatics, Japan","institution_ids":["https://openalex.org/I184597095"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5007639385","display_name":"Junichi Yamagishi","orcid":"https://orcid.org/0000-0003-2752-3955"},"institutions":[{"id":"https://openalex.org/I184597095","display_name":"National Institute of Informatics","ror":"https://ror.org/04ksd4g47","country_code":"JP","type":"facility","lineage":["https://openalex.org/I1319490839","https://openalex.org/I184597095","https://openalex.org/I4210158934"]},{"id":"https://openalex.org/I98677209","display_name":"University of Edinburgh","ror":"https://ror.org/01nrxwf90","country_code":"GB","type":"education","lineage":["https://openalex.org/I98677209"]}],"countries":["GB","JP"],"is_corresponding":false,"raw_author_name":"Junichi Yamagishi","raw_affiliation_strings":["National Institute of Informatics, Japan","The Centre for Speech Technology Research (CSTR), University of Edinburgh, United Kingdom"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"National Institute of Informatics, Japan","institution_ids":["https://openalex.org/I184597095"]},{"raw_affiliation_string":"The Centre for Speech Technology Research (CSTR), University of Edinburgh, United Kingdom","institution_ids":["https://openalex.org/I98677209"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":true,"cited_by_count":34,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":"5535","last_page":"5539"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9968000054359436,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.996399998664856,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/autoencoder","display_name":"Autoencoder","score":0.8323705792427063},{"id":"https://openalex.org/keywords/spectral-envelope","display_name":"Spectral envelope","score":0.7973483800888062},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.727030873298645},{"id":"https://openalex.org/keywords/feature-extraction","display_name":"Feature extraction","score":0.6442972421646118},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.6133919358253479},{"id":"https://openalex.org/keywords/parametric-statistics","display_name":"Parametric statistics","score":0.6085713505744934},{"id":"https://openalex.org/keywords/fast-fourier-transform","display_name":"Fast Fourier transform","score":0.605568528175354},{"id":"https://openalex.org/keywords/encoder","display_name":"Encoder","score":0.5491037368774414},{"id":"https://openalex.org/keywords/feature","display_name":"Feature (linguistics)","score":0.5383648872375488},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.5293567776679993},{"id":"https://openalex.org/keywords/pattern-recognition","display_name":"Pattern recognition (psychology)","score":0.4770866334438324},{"id":"https://openalex.org/keywords/speech-processing","display_name":"Speech processing","score":0.45963698625564575},{"id":"https://openalex.org/keywords/linear-predictive-coding","display_name":"Linear predictive coding","score":0.45832881331443787},{"id":"https://openalex.org/keywords/speech-synthesis","display_name":"Speech synthesis","score":0.43168970942497253},{"id":"https://openalex.org/keywords/deep-learning","display_name":"Deep learning","score":0.2967529892921448},{"id":"https://openalex.org/keywords/algorithm","display_name":"Algorithm","score":0.23292335867881775},{"id":"https://openalex.org/keywords/mathematics","display_name":"Mathematics","score":0.17040792107582092}],"concepts":[{"id":"https://openalex.org/C101738243","wikidata":"https://www.wikidata.org/wiki/Q786435","display_name":"Autoencoder","level":3,"score":0.8323705792427063},{"id":"https://openalex.org/C54926389","wikidata":"https://www.wikidata.org/wiki/Q7575188","display_name":"Spectral envelope","level":2,"score":0.7973483800888062},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.727030873298645},{"id":"https://openalex.org/C52622490","wikidata":"https://www.wikidata.org/wiki/Q1026626","display_name":"Feature extraction","level":2,"score":0.6442972421646118},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.6133919358253479},{"id":"https://openalex.org/C117251300","wikidata":"https://www.wikidata.org/wiki/Q1849855","display_name":"Parametric statistics","level":2,"score":0.6085713505744934},{"id":"https://openalex.org/C75172450","wikidata":"https://www.wikidata.org/wiki/Q623950","display_name":"Fast Fourier transform","level":2,"score":0.605568528175354},{"id":"https://openalex.org/C118505674","wikidata":"https://www.wikidata.org/wiki/Q42586063","display_name":"Encoder","level":2,"score":0.5491037368774414},{"id":"https://openalex.org/C2776401178","wikidata":"https://www.wikidata.org/wiki/Q12050496","display_name":"Feature (linguistics)","level":2,"score":0.5383648872375488},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.5293567776679993},{"id":"https://openalex.org/C153180895","wikidata":"https://www.wikidata.org/wiki/Q7148389","display_name":"Pattern recognition (psychology)","level":2,"score":0.4770866334438324},{"id":"https://openalex.org/C61328038","wikidata":"https://www.wikidata.org/wiki/Q3358061","display_name":"Speech processing","level":2,"score":0.45963698625564575},{"id":"https://openalex.org/C59883199","wikidata":"https://www.wikidata.org/wiki/Q1826438","display_name":"Linear predictive coding","level":3,"score":0.45832881331443787},{"id":"https://openalex.org/C14999030","wikidata":"https://www.wikidata.org/wiki/Q16346","display_name":"Speech synthesis","level":2,"score":0.43168970942497253},{"id":"https://openalex.org/C108583219","wikidata":"https://www.wikidata.org/wiki/Q197536","display_name":"Deep learning","level":2,"score":0.2967529892921448},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.23292335867881775},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.17040792107582092},{"id":"https://openalex.org/C111919701","wikidata":"https://www.wikidata.org/wiki/Q9135","display_name":"Operating system","level":1,"score":0.0},{"id":"https://openalex.org/C41895202","wikidata":"https://www.wikidata.org/wiki/Q8162","display_name":"Linguistics","level":1,"score":0.0},{"id":"https://openalex.org/C138885662","wikidata":"https://www.wikidata.org/wiki/Q5891","display_name":"Philosophy","level":0,"score":0.0},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.0}],"mesh":[],"locations_count":4,"locations":[{"id":"doi:10.1109/icassp.2016.7472736","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2016.7472736","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},{"id":"pmh:oai:pure.ed.ac.uk:openaire/eabdc1dc-b33c-4460-a915-f27a153dcde9","is_oa":true,"landing_page_url":"https://www.research.ed.ac.uk/en/publications/eabdc1dc-b33c-4460-a915-f27a153dcde9","pdf_url":null,"source":{"id":"https://openalex.org/S4306400321","display_name":"Edinburgh Research Explorer (University of Edinburgh)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I98677209","host_organization_name":"University of Edinburgh","host_organization_lineage":["https://openalex.org/I98677209"],"host_organization_lineage_names":[],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Takaki, S & Yamagishi, J 2016, A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis. in 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). Institute of Electrical and Electronics Engineers, pp. 5535-5539, 41st IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016, Shanghai, China, 20/03/16. https://doi.org/10.1109/ICASSP.2016.7472736","raw_type":"contributionToPeriodical"},{"id":"pmh:oai:pure.ed.ac.uk:publications/eabdc1dc-b33c-4460-a915-f27a153dcde9","is_oa":false,"landing_page_url":"http://ieeexplore.ieee.org/xpl/articleDetails.jsp?arnumber=7472736","pdf_url":null,"source":{"id":"https://openalex.org/S4406922455","display_name":"Edinburgh Research Explorer","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"","raw_type":""},{"id":"pmh:oai:pure.ed.ac.uk:publications/eabdc1dc-b33c-4460-a915-f27a153dcde9","is_oa":true,"landing_page_url":"https://hdl.handle.net/20.500.11820/eabdc1dc-b33c-4460-a915-f27a153dcde9","pdf_url":null,"source":{"id":"https://openalex.org/S4306400321","display_name":"Edinburgh Research Explorer (University of Edinburgh)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I98677209","host_organization_name":"University of Edinburgh","host_organization_lineage":["https://openalex.org/I98677209"],"host_organization_lineage_names":[],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Takaki, S & Yamagishi, J 2016, A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis. in 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). Institute of Electrical and Electronics Engineers, pp. 5535-5539, 41st IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016, Shanghai, China, 20/03/16. https://doi.org/10.1109/ICASSP.2016.7472736","raw_type":"contributionToPeriodical"}],"best_oa_location":{"id":"pmh:oai:pure.ed.ac.uk:openaire/eabdc1dc-b33c-4460-a915-f27a153dcde9","is_oa":true,"landing_page_url":"https://www.research.ed.ac.uk/en/publications/eabdc1dc-b33c-4460-a915-f27a153dcde9","pdf_url":null,"source":{"id":"https://openalex.org/S4306400321","display_name":"Edinburgh Research Explorer (University of Edinburgh)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I98677209","host_organization_name":"University of Edinburgh","host_organization_lineage":["https://openalex.org/I98677209"],"host_organization_lineage_names":[],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Takaki, S & Yamagishi, J 2016, A deep auto-encoder based low-dimensional feature extraction from FFT spectral envelopes for statistical parametric speech synthesis. in 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). Institute of Electrical and Electronics Engineers, pp. 5535-5539, 41st IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016, Shanghai, China, 20/03/16. https://doi.org/10.1109/ICASSP.2016.7472736","raw_type":"contributionToPeriodical"},"sustainable_development_goals":[{"display_name":"Quality Education","id":"https://metadata.un.org/sdg/4","score":0.5099999904632568}],"awards":[{"id":"https://openalex.org/G3007105200","display_name":"Deep architectures for statistical speech synthesis","funder_award_id":"EP/J002526/1","funder_id":"https://openalex.org/F4320334627","funder_display_name":"Engineering and Physical Sciences Research Council"},{"id":"https://openalex.org/G3790594077","display_name":"Natural Speech Technology","funder_award_id":"EP/I031022/1","funder_id":"https://openalex.org/F4320334627","funder_display_name":"Engineering and Physical Sciences Research Council"},{"id":"https://openalex.org/G4055593462","display_name":null,"funder_award_id":"EP/I031022/1","funder_id":"https://openalex.org/F4320334627","funder_display_name":"Engineering and Physical Sciences Research Council"},{"id":"https://openalex.org/G6082203376","display_name":null,"funder_award_id":"EP/J002526/1","funder_id":"https://openalex.org/F4320334627","funder_display_name":"Engineering and Physical Sciences Research Council"}],"funders":[{"id":"https://openalex.org/F4320334627","display_name":"Engineering and Physical Sciences Research Council","ror":"https://ror.org/0439y7842"}],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":40,"referenced_works":["https://openalex.org/W773905565","https://openalex.org/W1514737389","https://openalex.org/W1778816975","https://openalex.org/W1877553482","https://openalex.org/W1970088388","https://openalex.org/W1973681148","https://openalex.org/W2012086895","https://openalex.org/W2020024436","https://openalex.org/W2049686551","https://openalex.org/W2100495367","https://openalex.org/W2102003408","https://openalex.org/W2105099419","https://openalex.org/W2129142580","https://openalex.org/W2145203520","https://openalex.org/W2145247325","https://openalex.org/W2153750031","https://openalex.org/W2168013545","https://openalex.org/W2290318471","https://openalex.org/W2294797155","https://openalex.org/W2296581541","https://openalex.org/W2394662942","https://openalex.org/W2398742733","https://openalex.org/W2398826216","https://openalex.org/W2400654494","https://openalex.org/W2405774341","https://openalex.org/W2408093180","https://openalex.org/W2608305796","https://openalex.org/W2917245127","https://openalex.org/W4395699642","https://openalex.org/W6638023308","https://openalex.org/W6639363725","https://openalex.org/W6675380101","https://openalex.org/W6682857957","https://openalex.org/W6696843773","https://openalex.org/W6712239235","https://openalex.org/W6712460684","https://openalex.org/W6712560600","https://openalex.org/W6712631415","https://openalex.org/W6713548365","https://openalex.org/W6736620556"],"related_works":["https://openalex.org/W2383072803","https://openalex.org/W2069501481","https://openalex.org/W2536737918","https://openalex.org/W1842536210","https://openalex.org/W2068677590","https://openalex.org/W2619026611","https://openalex.org/W2096476291","https://openalex.org/W2223500991","https://openalex.org/W1796893744","https://openalex.org/W2155010696"],"abstract_inverted_index":{"In":[0,62],"the":[1],"state-of-the-art":[2],"statistical":[3,85],"parametric":[4,86],"speech":[5,9,52,87],"synthesis":[6,95],"system,":[7],"a":[8,43,51,93,110],"analysis":[10,53],"module,":[11],"e.g.":[12],"STRAIGHT":[13],"spectral":[14,24,34,44,82,106],"analysis,":[15],"is":[16,108],"generally":[17],"used":[18,37,48],"for":[19,38,84],"obtaining":[20],"accurate":[21],"and":[22,26,75],"stable":[23],"envelopes,":[25],"then":[27],"low-dimensional":[28,77,101],"acoustic":[29,40],"features":[30],"extracted":[31],"from":[32,59,104],"obtained":[33],"envelopes":[35,83,107],"are":[36],"training":[39],"models.":[41],"However,":[42],"envelope":[45],"estimation":[46],"algorithm":[47],"in":[49],"such":[50],"module":[54],"includes":[55],"various":[56],"processing":[57],"derived":[58],"human":[60],"knowledge.":[61],"this":[63],"paper,":[64],"we":[65],"present":[66],"our":[67],"investigation":[68],"of":[69],"deep":[70,98],"autoencoder":[71],"based,":[72],"non-linear,":[73],"data-driven":[74],"unsupervised":[76],"feature":[78,102],"extraction":[79,103],"using":[80,97],"FFT":[81,105],"synthesis.":[88],"Experimental":[89],"results":[90],"showed":[91],"that":[92],"text-to-speech":[94],"system":[96],"auto-encoder":[99],"based":[100],"indeed":[109],"promising":[111],"approach.":[112]},"counts_by_year":[{"year":2025,"cited_by_count":1},{"year":2023,"cited_by_count":1},{"year":2022,"cited_by_count":2},{"year":2021,"cited_by_count":3},{"year":2020,"cited_by_count":3},{"year":2019,"cited_by_count":2},{"year":2018,"cited_by_count":8},{"year":2017,"cited_by_count":11},{"year":2016,"cited_by_count":3}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
