{"id":"https://openalex.org/W2972514683","doi":"https://doi.org/10.21437/interspeech.2019-2778","title":"Trainable Dynamic Subsampling for End-to-End Speech Recognition","display_name":"Trainable Dynamic Subsampling for End-to-End Speech Recognition","publication_year":2019,"publication_date":"2019-09-13","ids":{"openalex":"https://openalex.org/W2972514683","doi":"https://doi.org/10.21437/interspeech.2019-2778","mag":"2972514683"},"language":"en","primary_location":{"id":"doi:10.21437/interspeech.2019-2778","is_oa":false,"landing_page_url":"https://doi.org/10.21437/interspeech.2019-2778","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Interspeech 2019","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://www.research.ed.ac.uk/en/publications/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5003030422","display_name":"Shucong Zhang","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Shucong Zhang","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5042184515","display_name":"Erfan Loweimi","orcid":"https://orcid.org/0000-0002-8761-021X"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Erfan Loweimi","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5023350949","display_name":"Yumo Xu","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Yumo Xu","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5102911387","display_name":"Peter Bell","orcid":"https://orcid.org/0000-0002-9597-9615"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Peter Bell","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]},{"author_position":"last","author":{"id":"https://openalex.org/A5027442277","display_name":"Steve Renals","orcid":"https://orcid.org/0000-0002-8790-3389"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Steve Renals","raw_affiliation_strings":[],"raw_orcid":null,"affiliations":[]}],"institutions":[],"countries_distinct_count":0,"institutions_distinct_count":0,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":8,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":"1413","last_page":"1417"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9994999766349792,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9991000294685364,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/end-to-end-principle","display_name":"End-to-end principle","score":0.8644719123840332},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7244035005569458},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.6282849311828613},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.2767961025238037}],"concepts":[{"id":"https://openalex.org/C74296488","wikidata":"https://www.wikidata.org/wiki/Q2527392","display_name":"End-to-end principle","level":2,"score":0.8644719123840332},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7244035005569458},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.6282849311828613},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.2767961025238037}],"mesh":[],"locations_count":4,"locations":[{"id":"doi:10.21437/interspeech.2019-2778","is_oa":false,"landing_page_url":"https://doi.org/10.21437/interspeech.2019-2778","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"Interspeech 2019","raw_type":"proceedings-article"},{"id":"pmh:oai:pure.ed.ac.uk:openaire/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","is_oa":true,"landing_page_url":"https://www.research.ed.ac.uk/en/publications/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","pdf_url":null,"source":{"id":"https://openalex.org/S4406922455","display_name":"Edinburgh Research Explorer","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Zhang, S, Loweimi, E, Xu, Y, Bell, P & Renals, S 2019, Trainable Dynamic Subsampling for End-to-End Speech Recognition. in Proceedings Interspeech 2019. International Speech Communication Association, pp. 1413-1417, Interspeech 2019, Graz, Austria, 15/09/19. https://doi.org/10.21437/Interspeech.2019-2778","raw_type":"contributionToPeriodical"},{"id":"pmh:oai:pure.ed.ac.uk:publications/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","is_oa":false,"landing_page_url":"https://www.research.ed.ac.uk/portal/en/publications/trainable-dynamic-subsampling-for-endtoend-speech-recognition(c9f00e8a-71a7-4661-a351-de2fa3ecb33a).html","pdf_url":null,"source":{"id":"https://openalex.org/S4406922455","display_name":"Edinburgh Research Explorer","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"","raw_type":""},{"id":"pmh:oai:repository@napier.ac.uk:3585891","is_oa":false,"landing_page_url":"http://researchrepository.napier.ac.uk/Output/3585891","pdf_url":null,"source":{"id":"https://openalex.org/S4306402591","display_name":"Edinburgh Napier Research Repository (Edinburgh Napier University)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I251738","host_organization_name":"Edinburgh Napier University","host_organization_lineage":["https://openalex.org/I251738"],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":null,"raw_type":"Presentation / Conference Contribution"}],"best_oa_location":{"id":"pmh:oai:pure.ed.ac.uk:openaire/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","is_oa":true,"landing_page_url":"https://www.research.ed.ac.uk/en/publications/c9f00e8a-71a7-4661-a351-de2fa3ecb33a","pdf_url":null,"source":{"id":"https://openalex.org/S4406922455","display_name":"Edinburgh Research Explorer","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Zhang, S, Loweimi, E, Xu, Y, Bell, P & Renals, S 2019, Trainable Dynamic Subsampling for End-to-End Speech Recognition. in Proceedings Interspeech 2019. International Speech Communication Association, pp. 1413-1417, Interspeech 2019, Graz, Austria, 15/09/19. https://doi.org/10.21437/Interspeech.2019-2778","raw_type":"contributionToPeriodical"},"sustainable_development_goals":[],"awards":[{"id":"https://openalex.org/G3297227137","display_name":null,"funder_award_id":"EP/R012180/1","funder_id":"https://openalex.org/F4320334627","funder_display_name":"Engineering and Physical Sciences Research Council"}],"funders":[{"id":"https://openalex.org/F4320334627","display_name":"Engineering and Physical Sciences Research Council","ror":"https://ror.org/0439y7842"}],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":15,"referenced_works":["https://openalex.org/W1524333225","https://openalex.org/W1902237438","https://openalex.org/W1924770834","https://openalex.org/W1942035323","https://openalex.org/W2095705004","https://openalex.org/W2291022022","https://openalex.org/W2327501763","https://openalex.org/W2526425061","https://openalex.org/W2600628583","https://openalex.org/W2627092829","https://openalex.org/W2889163603","https://openalex.org/W2890297197","https://openalex.org/W2936706905","https://openalex.org/W2962826786","https://openalex.org/W2964308564"],"related_works":["https://openalex.org/W2899084033","https://openalex.org/W2748952813","https://openalex.org/W3179968364","https://openalex.org/W2390279801","https://openalex.org/W2358668433","https://openalex.org/W2376932109","https://openalex.org/W2151749779","https://openalex.org/W2382290278","https://openalex.org/W2938107654","https://openalex.org/W2350741829"],"abstract_inverted_index":{"Jointly":[0],"optimised":[1],"attention-based":[2],"encoder-decoder":[3],"models":[4,23],"have":[5],"yielded":[6],"impressive":[7],"speech":[8,45],"recognition":[9],"results.":[10],"The":[11,153],"recurrent":[12,53],"neural":[13],"network":[14],"(RNN)":[15],"encoder":[16,57,104,125,155],"is":[17,34,58,138],"a":[18,90,97,148,157,165],"key":[19],"component":[20],"in":[21,74],"such":[22],"\u2013":[24],"it":[25,33,140],"learns":[26],"the":[27,31,40,56,64,67,102,112,124,128,136,161],"hidden":[28],"representations":[29],"of":[30,44,55,66,120],"inputs.However,":[32],"difficult":[35],"for":[36,132],"RNNs":[37],"to":[38,72,107,126],"model":[39],"long":[41],"sequences":[42],"characteristic":[43],"recognition.":[46],"To":[47],"address":[48],"this,":[49],"subsampling":[50,78,92,170],"between":[51],"stacked":[52],"layers":[54],"commonly":[59],"employed.":[60],"This":[61],"method":[62],"reduces":[63],"length":[65],"input":[68],"sequence":[69],"and":[70,84,175,180],"leads":[71],"gains":[73],"accuracy.":[75],"However,":[76],"static":[77,169],"may":[79,115],"both":[80],"include":[81],"redundant":[82,109],"information":[83,131],"miss":[85],"relevant":[86,130],"information.<br/><br/>We":[87],"propose":[88],"using":[89],"dynamic":[91],"RNN":[93,100,150,177],"(dsRNN)":[94],"encoder.":[95],"Unlike":[96],"statically":[98],"subsampled":[99],"encoder,":[101],"dsRNN":[103,137,154],"can":[105],"learn":[106,127],"skip":[108,113],"frames.":[110],"Furthermore,":[111],"ratio":[114],"vary":[116],"at":[117],"different":[118],"stages":[119],"training,":[121],"thus":[122],"allowing":[123],"most":[129],"each":[133],"epoch.":[134],"Although":[135],"unidirectional,":[139],"yields":[141],"lower":[142],"phone":[143],"error":[144],"rates":[145],"(PERs)":[146],"than":[147],"bidirectional":[149,176],"on":[151,160],"TIMIT.":[152],"has":[156],"16.8%":[158],"PER":[159,182],"TIMIT":[162],"test":[163],"set,":[164],"considerable":[166],"improvement":[167],"over":[168],"methods":[171],"used":[172],"with":[173],"unidirectional":[174],"encoders":[178],"(23.5%":[179],"20.4%":[181],"respectively).<br/>":[183]},"counts_by_year":[{"year":2025,"cited_by_count":2},{"year":2023,"cited_by_count":1},{"year":2021,"cited_by_count":3},{"year":2020,"cited_by_count":2}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
