{"id":"https://openalex.org/W4413097434","doi":"https://doi.org/10.1186/s13636-025-00420-7","title":"Coded speech enhancement using auxiliary utterance-level information","display_name":"Coded speech enhancement using auxiliary utterance-level information","publication_year":2025,"publication_date":"2025-07-29","ids":{"openalex":"https://openalex.org/W4413097434","doi":"https://doi.org/10.1186/s13636-025-00420-7"},"language":"en","primary_location":{"id":"doi:10.1186/s13636-025-00420-7","is_oa":true,"landing_page_url":"https://doi.org/10.1186/s13636-025-00420-7","pdf_url":"https://link.springer.com/content/pdf/10.1186/s13636-025-00420-7.pdf","source":{"id":"https://openalex.org/S19605986","display_name":"EURASIP Journal on Audio Speech and Music Processing","issn_l":"1687-4714","issn":["1687-4714","1687-4722","3091-4523"],"is_oa":true,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310319965","host_organization_name":"Springer Nature","host_organization_lineage":["https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"EURASIP Journal on Audio, Speech, and Music Processing","raw_type":"journal-article"},"type":"article","indexed_in":["crossref","doaj"],"open_access":{"is_oa":true,"oa_status":"gold","oa_url":"https://link.springer.com/content/pdf/10.1186/s13636-025-00420-7.pdf","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5101890082","display_name":"Haixin Zhao","orcid":"https://orcid.org/0009-0005-3957-8859"},"institutions":[{"id":"https://openalex.org/I32597200","display_name":"Ghent University","ror":"https://ror.org/00cv9y106","country_code":"BE","type":"education","lineage":["https://openalex.org/I32597200"]}],"countries":["BE"],"is_corresponding":true,"raw_author_name":"Haixin Zhao","raw_affiliation_strings":["IDLab, Ghent University - Imec, Technologiepark-Zwijnaarde 122, 9052, Ghent, Belgium"],"raw_orcid":"https://orcid.org/0009-0005-3957-8859","affiliations":[{"raw_affiliation_string":"IDLab, Ghent University - Imec, Technologiepark-Zwijnaarde 122, 9052, Ghent, Belgium","institution_ids":["https://openalex.org/I32597200"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5081844255","display_name":"Nilesh Madhu","orcid":"https://orcid.org/0000-0001-9131-3309"},"institutions":[{"id":"https://openalex.org/I32597200","display_name":"Ghent University","ror":"https://ror.org/00cv9y106","country_code":"BE","type":"education","lineage":["https://openalex.org/I32597200"]}],"countries":["BE"],"is_corresponding":false,"raw_author_name":"Nilesh Madhu","raw_affiliation_strings":["IDLab, Ghent University - Imec, Technologiepark-Zwijnaarde 122, 9052, Ghent, Belgium"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"IDLab, Ghent University - Imec, Technologiepark-Zwijnaarde 122, 9052, Ghent, Belgium","institution_ids":["https://openalex.org/I32597200"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":["https://openalex.org/A5101890082"],"corresponding_institution_ids":["https://openalex.org/I32597200"],"apc_list":{"value":1635,"currency":"USD","value_usd":1635},"apc_paid":{"value":1635,"currency":"USD","value_usd":1635},"fwci":0.0,"has_fulltext":true,"cited_by_count":0,"citation_normalized_percentile":{"value":0.20923887,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":null,"biblio":{"volume":"2025","issue":"1","first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10283","display_name":"Hearing Loss and Rehabilitation","score":0.9980999827384949,"subfield":{"id":"https://openalex.org/subfields/2805","display_name":"Cognitive Neuroscience"},"field":{"id":"https://openalex.org/fields/28","display_name":"Neuroscience"},"domain":{"id":"https://openalex.org/domains/1","display_name":"Life Sciences"}},{"id":"https://openalex.org/T10822","display_name":"Acoustic Wave Phenomena Research","score":0.9973999857902527,"subfield":{"id":"https://openalex.org/subfields/2204","display_name":"Biomedical Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.8827784061431885},{"id":"https://openalex.org/keywords/pesq","display_name":"PESQ","score":0.5405169725418091},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.49835968017578125},{"id":"https://openalex.org/keywords/speech-coding","display_name":"Speech coding","score":0.43357881903648376},{"id":"https://openalex.org/keywords/speech-enhancement","display_name":"Speech enhancement","score":0.3783314824104309},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.27112656831741333},{"id":"https://openalex.org/keywords/noise-reduction","display_name":"Noise reduction","score":0.11117351055145264}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.8827784061431885},{"id":"https://openalex.org/C103734657","wikidata":"https://www.wikidata.org/wiki/Q2739975","display_name":"PESQ","level":4,"score":0.5405169725418091},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.49835968017578125},{"id":"https://openalex.org/C13895895","wikidata":"https://www.wikidata.org/wiki/Q3270773","display_name":"Speech coding","level":2,"score":0.43357881903648376},{"id":"https://openalex.org/C2776182073","wikidata":"https://www.wikidata.org/wiki/Q7575395","display_name":"Speech enhancement","level":3,"score":0.3783314824104309},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.27112656831741333},{"id":"https://openalex.org/C163294075","wikidata":"https://www.wikidata.org/wiki/Q581861","display_name":"Noise reduction","level":2,"score":0.11117351055145264}],"mesh":[],"locations_count":3,"locations":[{"id":"doi:10.1186/s13636-025-00420-7","is_oa":true,"landing_page_url":"https://doi.org/10.1186/s13636-025-00420-7","pdf_url":"https://link.springer.com/content/pdf/10.1186/s13636-025-00420-7.pdf","source":{"id":"https://openalex.org/S19605986","display_name":"EURASIP Journal on Audio Speech and Music Processing","issn_l":"1687-4714","issn":["1687-4714","1687-4722","3091-4523"],"is_oa":true,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310319965","host_organization_name":"Springer Nature","host_organization_lineage":["https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"EURASIP Journal on Audio, Speech, and Music Processing","raw_type":"journal-article"},{"id":"pmh:oai:archive.ugent.be:01K1Z1D634WNADYG3Q0HR2AQ1A","is_oa":true,"landing_page_url":"https://biblio.ugent.be/publication/01K1Z1D634WNADYG3Q0HR2AQ1A","pdf_url":"https://biblio.ugent.be/publication/01K1Z1D634WNADYG3Q0HR2AQ1A/file/01K1Z1DWB08Q4NC71J8C1E06E4.pdf","source":{"id":"https://openalex.org/S4306400478","display_name":"Ghent University Academic Bibliography (Ghent University)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I32597200","host_organization_name":"Ghent University","host_organization_lineage":["https://openalex.org/I32597200"],"host_organization_lineage_names":[],"type":"repository"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"ISSN: 1687-4722","raw_type":"info:eu-repo/semantics/article"},{"id":"pmh:oai:doaj.org/article:055b3eff572b4090b5c3c3bbe6eb8b5b","is_oa":true,"landing_page_url":"https://doaj.org/article/055b3eff572b4090b5c3c3bbe6eb8b5b","pdf_url":null,"source":{"id":"https://openalex.org/S4306401280","display_name":"DOAJ (DOAJ: Directory of Open Access Journals)","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":"cc-by-sa","license_id":"https://openalex.org/licenses/cc-by-sa","version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"EURASIP Journal on Audio, Speech, and Music Processing, Vol 2025, Iss 1, Pp 1-17 (2025)","raw_type":"article"}],"best_oa_location":{"id":"doi:10.1186/s13636-025-00420-7","is_oa":true,"landing_page_url":"https://doi.org/10.1186/s13636-025-00420-7","pdf_url":"https://link.springer.com/content/pdf/10.1186/s13636-025-00420-7.pdf","source":{"id":"https://openalex.org/S19605986","display_name":"EURASIP Journal on Audio Speech and Music Processing","issn_l":"1687-4714","issn":["1687-4714","1687-4722","3091-4523"],"is_oa":true,"is_in_doaj":true,"is_core":true,"host_organization":"https://openalex.org/P4310319965","host_organization_name":"Springer Nature","host_organization_lineage":["https://openalex.org/P4310319965"],"host_organization_lineage_names":["Springer Nature"],"type":"journal"},"license":"cc-by","license_id":"https://openalex.org/licenses/cc-by","version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"EURASIP Journal on Audio, Speech, and Music Processing","raw_type":"journal-article"},"sustainable_development_goals":[{"display_name":"Quality Education","score":0.5099999904632568,"id":"https://metadata.un.org/sdg/4"}],"awards":[],"funders":[],"has_content":{"pdf":true,"grobid_xml":false},"content_urls":{"pdf":"https://content.openalex.org/works/W4413097434.pdf"},"referenced_works_count":34,"referenced_works":["https://openalex.org/W1491878173","https://openalex.org/W1527477959","https://openalex.org/W1577541836","https://openalex.org/W2046820868","https://openalex.org/W2067295501","https://openalex.org/W2115781824","https://openalex.org/W2131094339","https://openalex.org/W2165291881","https://openalex.org/W2808920027","https://openalex.org/W2809824582","https://openalex.org/W2900004752","https://openalex.org/W2963189033","https://openalex.org/W2963902628","https://openalex.org/W3015780049","https://openalex.org/W3016057201","https://openalex.org/W3096468295","https://openalex.org/W3096893582","https://openalex.org/W3160129476","https://openalex.org/W3161950572","https://openalex.org/W3188073270","https://openalex.org/W3197042120","https://openalex.org/W3197312245","https://openalex.org/W3201698955","https://openalex.org/W4205788663","https://openalex.org/W4225302959","https://openalex.org/W4225860133","https://openalex.org/W4226360164","https://openalex.org/W4239527008","https://openalex.org/W4296068418","https://openalex.org/W4385807442","https://openalex.org/W4385822821","https://openalex.org/W4386764386","https://openalex.org/W4392903975","https://openalex.org/W4392943956"],"related_works":["https://openalex.org/W2058482658","https://openalex.org/W3016109656","https://openalex.org/W1973895194","https://openalex.org/W2546593254","https://openalex.org/W2166831097","https://openalex.org/W4386746628","https://openalex.org/W4388016426","https://openalex.org/W1980687383","https://openalex.org/W3209446892","https://openalex.org/W2564850605"],"abstract_inverted_index":{"Abstract":[0],"Numerous":[1],"post-processing":[2],"methods":[3,151],"have":[4],"been":[5],"proposed":[6,75,182],"to":[7,132,155,173,187,194,223],"improve":[8,156],"coded":[9],"speech":[10],"quality":[11],"and":[12,18,52,91,96,107,128,230],"intelligibility.":[13],"However,":[14],"achieving":[15],"state-of-the-art":[16],"enhancement":[17],"generalisation":[19],"across":[20,62,82,208],"varying":[21,209],"distortion":[22,110,216],"levels":[23],"remains":[24],"a":[25,33,42,188],"challenge.":[26],"To":[27,98],"bridge":[28],"this":[29],"gap,":[30],"we":[31,117],"propose":[32,118],"Lightweight":[34],"Causal-Transformer-based":[35],"Coded":[36],"Speech":[37],"Enhancement":[38],"(LCT-CSE)":[39],"model":[40,77,168,203],"employing":[41],"causal":[43],"frequency-time-frequency":[44],"(FTF)":[45],"transformer":[46],"block.":[47],"This":[48,238],"block":[49],"facilitates":[50],"temporal":[51],"spectral":[53],"sequential":[54],"modelling":[55],"using":[56],"transformers,":[57],"efficiently":[58],"exploiting":[59],"global":[60],"dependency":[61],"causal-context":[63],"TF":[64],"bins":[65],"while":[66,137],"minimising":[67],"computational":[68,164],"overhead.":[69,165],"Experimental":[70],"results":[71],"indicate":[72],"that":[73],"the":[74,79,114,138,181,195,200],"LCT-CSE":[76,115,202],"outperforms":[78],"considered":[80],"baselines":[81],"mainstream":[83],"lossy":[84],"audio":[85],"codecs,":[86],"including":[87],"Opus,":[88],"AMR-WB,":[89],"EVS":[90],"LC3+,":[92],"with":[93,161],"less":[94],"footprint":[95],"complexity.":[97],"further":[99,179,243],"utilise":[100],"auxiliary,":[101],"utterance-level":[102],"information":[103,120,183,247],"such":[104],"as":[105,133],"bitrate":[106,236],"other":[108,139],"general":[109],"characteristics,":[111],"building":[112],"upon":[113],"model,":[116],"two":[119,196],"incorporation":[121,184],"methods.":[122],"One":[123],"employs":[124],"one-hot":[125],"vector":[126],"representations":[127],"feature":[129],"fusions,":[130],"referred":[131],"1-hot":[134],"vector-based":[135],"modulation,":[136],"dynamically":[140],"switches":[141],"information-dependent":[142],"network":[143],"paths,":[144],"termed":[145],"dynamic":[146],"linear":[147],"modulation":[148],"(DLM).":[149],"These":[150],"can":[152,248],"be":[153,249],"used":[154,198],"performance":[157,172],"in":[158,225,228,232],"bitrate-information":[159],"utilisation,":[160],"negligible":[162],"additional":[163],"The":[166],"DLM":[167],"even":[169],"achieves":[170,220],"comparable":[171],"bitrate-specific":[174],"trained":[175],"(BST)":[176],"models.":[177],"We":[178],"extend":[180],"method,":[185],"DLM,":[186],"generalised":[189],"scenario,":[190],"tandem":[191,210],"coding.":[192],"Compared":[193],"practically":[197],"approaches,":[199],"DLM-based":[201],"consistently":[204],"exhibits":[205],"improved":[206],"generalisability":[207],"encoding":[211],"conditions,":[212],"based":[213],"on":[214],"derivative":[215],"information.":[217],"Specifically,":[218],"it":[219],"gains":[221],"up":[222],"0.74":[224],"PESQ,":[226],"7%":[227],"STOI,":[229],"0.18":[231],"MOS-SIG":[233],"under":[234],"various":[235],"conditions.":[237],"indicates":[239],"significant":[240],"potential":[241],"for":[242],"applications":[244],"where":[245],"auxiliary":[246],"utilised.":[250]},"counts_by_year":[],"updated_date":"2026-08-01T09:00:35.917206","created_date":"2025-10-10T00:00:00"}
