{"id":"https://openalex.org/W2936399153","doi":"https://doi.org/10.1109/icassp.2019.8682319","title":"Improving Speech Recognition Error Prediction for Modern and Off-the-shelf Speech Recognizers","display_name":"Improving Speech Recognition Error Prediction for Modern and Off-the-shelf Speech Recognizers","publication_year":2019,"publication_date":"2019-04-17","ids":{"openalex":"https://openalex.org/W2936399153","doi":"https://doi.org/10.1109/icassp.2019.8682319","mag":"2936399153"},"language":"en","primary_location":{"id":"doi:10.1109/icassp.2019.8682319","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2019.8682319","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5070013411","display_name":"Prashant Serai","orcid":"https://orcid.org/0000-0002-4672-4413"},"institutions":[{"id":"https://openalex.org/I52357470","display_name":"The Ohio State University","ror":"https://ror.org/00rs6vg23","country_code":"US","type":"education","lineage":["https://openalex.org/I52357470"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Prashant Serai","raw_affiliation_strings":["Department of Computer Science & Engineering, The Ohio State University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science & Engineering, The Ohio State University","institution_ids":["https://openalex.org/I52357470"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5062250845","display_name":"Peidong Wang","orcid":"https://orcid.org/0000-0002-7042-0209"},"institutions":[{"id":"https://openalex.org/I52357470","display_name":"The Ohio State University","ror":"https://ror.org/00rs6vg23","country_code":"US","type":"education","lineage":["https://openalex.org/I52357470"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Peidong Wang","raw_affiliation_strings":["Department of Computer Science & Engineering, The Ohio State University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science & Engineering, The Ohio State University","institution_ids":["https://openalex.org/I52357470"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5056667180","display_name":"Eric Fosler\u2010Lussier","orcid":"https://orcid.org/0000-0001-8004-5169"},"institutions":[{"id":"https://openalex.org/I52357470","display_name":"The Ohio State University","ror":"https://ror.org/00rs6vg23","country_code":"US","type":"education","lineage":["https://openalex.org/I52357470"]}],"countries":["US"],"is_corresponding":false,"raw_author_name":"Eric Fosler-Lussier","raw_affiliation_strings":["Department of Computer Science & Engineering, The Ohio State University"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Department of Computer Science & Engineering, The Ohio State University","institution_ids":["https://openalex.org/I52357470"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":[],"corresponding_institution_ids":["https://openalex.org/I52357470"],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":5,"citation_normalized_percentile":null,"cited_by_percentile_year":null,"biblio":{"volume":null,"issue":null,"first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9997000098228455,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9997000098228455,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":0.9939000010490417,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10181","display_name":"Natural Language Processing Techniques","score":0.9927999973297119,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.7829911112785339},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7786883115768433},{"id":"https://openalex.org/keywords/hidden-markov-model","display_name":"Hidden Markov model","score":0.4647260308265686},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.42626631259918213},{"id":"https://openalex.org/keywords/natural-language-processing","display_name":"Natural language processing","score":0.3868715763092041}],"concepts":[{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.7829911112785339},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7786883115768433},{"id":"https://openalex.org/C23224414","wikidata":"https://www.wikidata.org/wiki/Q176769","display_name":"Hidden Markov model","level":2,"score":0.4647260308265686},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.42626631259918213},{"id":"https://openalex.org/C204321447","wikidata":"https://www.wikidata.org/wiki/Q30642","display_name":"Natural language processing","level":1,"score":0.3868715763092041}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/icassp.2019.8682319","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2019.8682319","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"}],"best_oa_location":null,"sustainable_development_goals":[{"display_name":"Reduced inequalities","id":"https://metadata.un.org/sdg/10","score":0.7200000286102295}],"awards":[],"funders":[{"id":"https://openalex.org/F4320306076","display_name":"National Science Foundation","ror":"https://ror.org/021nxhr62"},{"id":"https://openalex.org/F4320309480","display_name":"Nvidia","ror":"https://ror.org/03jdj4y14"}],"has_content":{"grobid_xml":false,"pdf":false},"content_urls":null,"referenced_works_count":30,"referenced_works":["https://openalex.org/W21963070","https://openalex.org/W97072897","https://openalex.org/W99561634","https://openalex.org/W168280322","https://openalex.org/W200927961","https://openalex.org/W399167303","https://openalex.org/W1980280616","https://openalex.org/W1993607814","https://openalex.org/W2121642367","https://openalex.org/W2130942839","https://openalex.org/W2131342762","https://openalex.org/W2133564696","https://openalex.org/W2148182949","https://openalex.org/W2251321385","https://openalex.org/W2402827806","https://openalex.org/W2756554273","https://openalex.org/W2898989428","https://openalex.org/W2938630833","https://openalex.org/W2963211739","https://openalex.org/W2964308564","https://openalex.org/W6600897908","https://openalex.org/W6603931906","https://openalex.org/W6604094504","https://openalex.org/W6606774702","https://openalex.org/W6608167223","https://openalex.org/W6679434410","https://openalex.org/W6679436768","https://openalex.org/W6691770337","https://openalex.org/W6712871323","https://openalex.org/W6755848947"],"related_works":["https://openalex.org/W2748952813","https://openalex.org/W2136763963","https://openalex.org/W2109705048","https://openalex.org/W2940588515","https://openalex.org/W1909151225","https://openalex.org/W2160030256","https://openalex.org/W1521297879","https://openalex.org/W4253235840","https://openalex.org/W3151937861","https://openalex.org/W3204019825"],"abstract_inverted_index":{"Modeling":[0],"the":[1,55,67,73,105,116,130,134,143,163,185,191],"errors":[2,92,144],"of":[3,29,50,57,107,165],"a":[4,82,99,108,120,147,172,181],"speech":[5,12,90],"recognizer":[6],"can":[7],"help":[8],"simulate":[9],"errorful":[10],"recognized":[11],"data":[13,38,153],"from":[14],"plain":[15],"text,":[16],"which":[17],"has":[18],"proven":[19],"useful":[20],"for":[21,88],"tasks":[22],"like":[23],"discriminative":[24],"language":[25],"modeling,":[26],"improving":[27],"robustness":[28],"NLP":[30],"systems,":[31,53],"where":[32],"limited":[33],"or":[34],"even":[35],"no":[36],"audio":[37],"is":[39,65],"available":[40],"at":[41],"train":[42],"time.":[43],"Previous":[44],"work":[45],"typically":[46],"considered":[47],"replicating":[48],"behavior":[49,56,106,164],"GMM-HMM":[51],"based":[52,86],"but":[54],"more":[58],"modern":[59],"posterior-based":[60,109],"neural":[61],"network":[62],"acoustic":[63,110],"models":[64],"not":[66],"same":[68,159],"and":[69,155],"requires":[70],"adjustments":[71],"to":[72,125,161,190],"error":[74,135],"prediction":[75],"model.":[76,111],"In":[77],"this":[78],"work,":[79],"we":[80,97,113],"extend":[81],"prior":[83],"phonetic":[84],"confusion":[85,117,192],"model":[87,122,187],"predicting":[89,142],"recognition":[91],"in":[93,123,137],"two":[94,138],"ways:":[95,139],"first,":[96],"introduce":[98,126],"sampling-based":[100],"paradigm":[101],"that":[102,158],"better":[103],"simulates":[104],"Second,":[112],"investigate":[114],"replacing":[115],"matrix":[118],"with":[119],"sequence-to-sequence":[121],"order":[124],"context":[127],"dependency":[128],"into":[129],"prediction.":[131],"We":[132],"evaluate":[133],"predictors":[136],"first":[140],"by":[141,146],"made":[145],"Switchboard":[148],"ASR":[149,169],"system":[150,170],"on":[151,171],"unseen":[152],"(Fisher),":[154],"then":[156],"using":[157],"predictor":[160],"estimate":[162],"an":[166],"unrelated":[167],"cloud-based":[168],"novel":[173],"task.":[174],"Sampling":[175],"greatly":[176],"improves":[177],"predictive":[178],"accuracy":[179],"within":[180],"100-guess":[182],"paradigm,":[183],"while":[184],"sequence":[186],"performs":[188],"similarly":[189],"matrix.":[193]},"counts_by_year":[{"year":2024,"cited_by_count":1},{"year":2022,"cited_by_count":1},{"year":2021,"cited_by_count":1},{"year":2020,"cited_by_count":1},{"year":2019,"cited_by_count":1}],"updated_date":"2026-07-14T23:27:15.235271","created_date":"2025-10-10T00:00:00"}
