{"id":"https://openalex.org/W2642153443","doi":"https://doi.org/10.1109/icassp.2017.7953156","title":"Feedback connection for deep neural network-based acoustic modeling","display_name":"Feedback connection for deep neural network-based acoustic modeling","publication_year":2017,"publication_date":"2017-03-01","ids":{"openalex":"https://openalex.org/W2642153443","doi":"https://doi.org/10.1109/icassp.2017.7953156","mag":"2642153443"},"language":"en","primary_location":{"id":"doi:10.1109/icassp.2017.7953156","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2017.7953156","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5021137751","display_name":"D\u0169ng Tr\u1ea7n Trung","orcid":"https://orcid.org/0000-0003-2015-3963"},"institutions":[{"id":"https://openalex.org/I2251713219","display_name":"NTT (Japan)","ror":"https://ror.org/00berct97","country_code":"JP","type":"company","lineage":["https://openalex.org/I2251713219"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Dung T. Tran","raw_affiliation_strings":["NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan","institution_ids":["https://openalex.org/I2251713219"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5023868166","display_name":"Marc Delcroix","orcid":"https://orcid.org/0000-0002-5175-7834"},"institutions":[{"id":"https://openalex.org/I2251713219","display_name":"NTT (Japan)","ror":"https://ror.org/00berct97","country_code":"JP","type":"company","lineage":["https://openalex.org/I2251713219"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Marc Delcroix","raw_affiliation_strings":["NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan","institution_ids":["https://openalex.org/I2251713219"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5101949937","display_name":"Atsunori Ogawa","orcid":"https://orcid.org/0000-0002-2888-101X"},"institutions":[{"id":"https://openalex.org/I2251713219","display_name":"NTT (Japan)","ror":"https://ror.org/00berct97","country_code":"JP","type":"company","lineage":["https://openalex.org/I2251713219"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Atsunori Ogawa","raw_affiliation_strings":["NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan","institution_ids":["https://openalex.org/I2251713219"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5039041764","display_name":"Christian Huemmer","orcid":"https://orcid.org/0000-0002-4341-4217"},"institutions":[{"id":"https://openalex.org/I181369854","display_name":"Friedrich-Alexander-Universit\u00e4t Erlangen-N\u00fcrnberg","ror":"https://ror.org/00f7hpc57","country_code":"DE","type":"education","lineage":["https://openalex.org/I181369854"]}],"countries":["DE"],"is_corresponding":false,"raw_author_name":"Christian Huemmer","raw_affiliation_strings":["Multimedia Communications and Signal Processing, University of Erlangen-Nuremberg, Erlangen, Germany"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Multimedia Communications and Signal Processing, University of Erlangen-Nuremberg, Erlangen, Germany","institution_ids":["https://openalex.org/I181369854"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5021240106","display_name":"Tomohiro Nakatani","orcid":"https://orcid.org/0000-0002-7487-7150"},"institutions":[{"id":"https://openalex.org/I2251713219","display_name":"NTT (Japan)","ror":"https://ror.org/00berct97","country_code":"JP","type":"company","lineage":["https://openalex.org/I2251713219"]}],"countries":["JP"],"is_corresponding":false,"raw_author_name":"Tomohiro Nakatani","raw_affiliation_strings":["NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"NTT Communication Science Laboratories, NTT corporation, Kyoto, Japan","institution_ids":["https://openalex.org/I2251713219"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":2,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":3,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":"4","issue":null,"first_page":"5240","last_page":"5244"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9998999834060669,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9995999932289124,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9993000030517578,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7766846418380737},{"id":"https://openalex.org/keywords/artificial-neural-network","display_name":"Artificial neural network","score":0.7073993682861328},{"id":"https://openalex.org/keywords/time-delay-neural-network","display_name":"Time delay neural network","score":0.6387763023376465},{"id":"https://openalex.org/keywords/convolutional-neural-network","display_name":"Convolutional neural network","score":0.6344816088676453},{"id":"https://openalex.org/keywords/deep-learning","display_name":"Deep learning","score":0.6217555403709412},{"id":"https://openalex.org/keywords/layer","display_name":"Layer (electronics)","score":0.5931161642074585},{"id":"https://openalex.org/keywords/echo-state-network","display_name":"Echo state network","score":0.5630532503128052},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.5540772080421448},{"id":"https://openalex.org/keywords/recurrent-neural-network","display_name":"Recurrent neural network","score":0.5090303421020508},{"id":"https://openalex.org/keywords/connection","display_name":"Connection (principal bundle)","score":0.4500404894351959},{"id":"https://openalex.org/keywords/pattern-recognition","display_name":"Pattern recognition (psychology)","score":0.43410587310791016},{"id":"https://openalex.org/keywords/network-architecture","display_name":"Network architecture","score":0.42565396428108215},{"id":"https://openalex.org/keywords/feature-extraction","display_name":"Feature extraction","score":0.4156845808029175},{"id":"https://openalex.org/keywords/deep-neural-networks","display_name":"Deep neural networks","score":0.41234180331230164},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.3549853563308716},{"id":"https://openalex.org/keywords/engineering","display_name":"Engineering","score":0.08058345317840576},{"id":"https://openalex.org/keywords/computer-network","display_name":"Computer network","score":0.05973786115646362}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7766846418380737},{"id":"https://openalex.org/C50644808","wikidata":"https://www.wikidata.org/wiki/Q192776","display_name":"Artificial neural network","level":2,"score":0.7073993682861328},{"id":"https://openalex.org/C175202392","wikidata":"https://www.wikidata.org/wiki/Q2434543","display_name":"Time delay neural network","level":3,"score":0.6387763023376465},{"id":"https://openalex.org/C81363708","wikidata":"https://www.wikidata.org/wiki/Q17084460","display_name":"Convolutional neural network","level":2,"score":0.6344816088676453},{"id":"https://openalex.org/C108583219","wikidata":"https://www.wikidata.org/wiki/Q197536","display_name":"Deep learning","level":2,"score":0.6217555403709412},{"id":"https://openalex.org/C2779227376","wikidata":"https://www.wikidata.org/wiki/Q6505497","display_name":"Layer (electronics)","level":2,"score":0.5931161642074585},{"id":"https://openalex.org/C172025690","wikidata":"https://www.wikidata.org/wiki/Q5332763","display_name":"Echo state network","level":4,"score":0.5630532503128052},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.5540772080421448},{"id":"https://openalex.org/C147168706","wikidata":"https://www.wikidata.org/wiki/Q1457734","display_name":"Recurrent neural network","level":3,"score":0.5090303421020508},{"id":"https://openalex.org/C13355873","wikidata":"https://www.wikidata.org/wiki/Q2920850","display_name":"Connection (principal bundle)","level":2,"score":0.4500404894351959},{"id":"https://openalex.org/C153180895","wikidata":"https://www.wikidata.org/wiki/Q7148389","display_name":"Pattern recognition (psychology)","level":2,"score":0.43410587310791016},{"id":"https://openalex.org/C193415008","wikidata":"https://www.wikidata.org/wiki/Q639681","display_name":"Network architecture","level":2,"score":0.42565396428108215},{"id":"https://openalex.org/C52622490","wikidata":"https://www.wikidata.org/wiki/Q1026626","display_name":"Feature extraction","level":2,"score":0.4156845808029175},{"id":"https://openalex.org/C2984842247","wikidata":"https://www.wikidata.org/wiki/Q197536","display_name":"Deep neural networks","level":3,"score":0.41234180331230164},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.3549853563308716},{"id":"https://openalex.org/C127413603","wikidata":"https://www.wikidata.org/wiki/Q11023","display_name":"Engineering","level":0,"score":0.08058345317840576},{"id":"https://openalex.org/C31258907","wikidata":"https://www.wikidata.org/wiki/Q1301371","display_name":"Computer network","level":1,"score":0.05973786115646362},{"id":"https://openalex.org/C185592680","wikidata":"https://www.wikidata.org/wiki/Q2329","display_name":"Chemistry","level":0,"score":0.0},{"id":"https://openalex.org/C66938386","wikidata":"https://www.wikidata.org/wiki/Q633538","display_name":"Structural engineering","level":1,"score":0.0},{"id":"https://openalex.org/C178790620","wikidata":"https://www.wikidata.org/wiki/Q11351","display_name":"Organic chemistry","level":1,"score":0.0}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp.2017.7953156","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2017.7953156","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[],"awards":[],"funders":[],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":25,"referenced_works":["https://openalex.org/W1571447047","https://openalex.org/W1603614877","https://openalex.org/W1970088388","https://openalex.org/W1985371235","https://openalex.org/W2012897754","https://openalex.org/W2033310064","https://openalex.org/W2062164080","https://openalex.org/W2075925017","https://openalex.org/W2079623482","https://openalex.org/W2105099419","https://openalex.org/W2123237149","https://openalex.org/W2160815625","https://openalex.org/W2165712214","https://openalex.org/W2237224976","https://openalex.org/W2288645994","https://openalex.org/W2289394825","https://openalex.org/W2397725648","https://openalex.org/W2398776621","https://openalex.org/W2400505028","https://openalex.org/W2506203739","https://openalex.org/W2683320792","https://openalex.org/W2964139918","https://openalex.org/W6653316155","https://openalex.org/W6690441418","https://openalex.org/W6712772800"],"related_works":["https://openalex.org/W1514460679","https://openalex.org/W2903992663","https://openalex.org/W4237814686","https://openalex.org/W2109916967","https://openalex.org/W3001462218","https://openalex.org/W2997427060","https://openalex.org/W99221663","https://openalex.org/W2890297197","https://openalex.org/W2101697354","https://openalex.org/W2953065922"],"abstract_inverted_index":{"The":[0],"use":[1,22],"of":[2,13,39,48,55,66,79,99,113,122,151],"auxiliary":[3,23,33,49,71,101],"features":[4,24,34,50,72,102],"is":[5,85],"an":[6],"effective":[7],"way":[8],"to":[9,62],"improve":[10,110],"the":[11,27,30,40,53,63,67,70,76,80,89,92,97,100,104,111,114,119,149,155],"performance":[12,112],"deep":[14,131,136],"neural":[15,132,137,140],"network":[16,81,116,133],"(DNN)-based":[17],"acoustic":[18,41],"models.":[19],"Most":[20],"approaches":[21],"that":[25,59],"represent":[26],"speaker":[28,90],"or":[29,91],"environment.":[31,93],"These":[32],"are":[35,73],"usually":[36],"computed":[37],"independently":[38],"model.":[42],"This":[43],"paper":[44],"investigates":[45],"a":[46,56],"types":[47],"obtained":[51],"from":[52,75,103],"output":[54],"hidden":[57,77],"layer":[58,65,78],"feeds":[60],"back":[61],"input":[64],"network.":[68],"Since":[69],"extracted":[74],"no":[82],"external":[83],"information":[84],"required":[86],"such":[87],"as":[88],"Experimentally,":[94],"by":[95],"forcing":[96],"extraction":[98],"same":[105],"networks,":[106,138],"we":[107],"can":[108],"further":[109],"overall":[115],"and":[117,142],"reduce":[118],"total":[120],"number":[121],"parameters":[123],"used.":[124],"We":[125,147],"tested":[126],"this":[127,152],"approach":[128,153],"with":[129],"different":[130],"architectures":[134],"including:":[135],"convolutional":[139,145],"networks":[141],"unfolded":[143],"recurrent":[144],"networks.":[146],"confirmed":[148],"effectiveness":[150],"on":[154],"CHiME3":[156],"dataset.":[157]},"counts_by_year":[{"year":2018,"cited_by_count":2},{"year":2017,"cited_by_count":1}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
