{"id":"https://openalex.org/W2404704342","doi":"https://doi.org/10.1109/icassp.2016.7472088","title":"Deep complementary bottleneck features for visual speech recognition","display_name":"Deep complementary bottleneck features for visual speech recognition","publication_year":2016,"publication_date":"2016-03-01","ids":{"openalex":"https://openalex.org/W2404704342","doi":"https://doi.org/10.1109/icassp.2016.7472088","mag":"2404704342"},"language":"en","primary_location":{"id":"doi:10.1109/icassp.2016.7472088","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2016.7472088","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},"type":"conference-paper","indexed_in":["crossref"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://ris.utwente.nl/ws/files/5317759/petridispantic_icassp2016.pdf","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5009475700","display_name":"Stavros Petridis","orcid":"https://orcid.org/0000-0001-7478-9479"},"institutions":[{"id":"https://openalex.org/I47508984","display_name":"Imperial College London","ror":"https://ror.org/041kmwe10","country_code":"GB","type":"education","lineage":["https://openalex.org/I47508984"]}],"countries":["GB"],"is_corresponding":false,"raw_author_name":"Stavros Petridis","raw_affiliation_strings":["Dept. of Computing, Imperial College London, London, UK"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Dept. of Computing, Imperial College London, London, UK","institution_ids":["https://openalex.org/I47508984"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5016033078","display_name":"Maja Panti\u0107","orcid":"https://orcid.org/0000-0002-3620-5986"},"institutions":[{"id":"https://openalex.org/I124357947","display_name":"University of London","ror":"https://ror.org/04cw6st05","country_code":"GB","type":"education","lineage":["https://openalex.org/I124357947"]},{"id":"https://openalex.org/I47508984","display_name":"Imperial College London","ror":"https://ror.org/041kmwe10","country_code":"GB","type":"education","lineage":["https://openalex.org/I47508984"]},{"id":"https://openalex.org/I94624287","display_name":"University of Twente","ror":"https://ror.org/006hf6230","country_code":"NL","type":"education","lineage":["https://openalex.org/I94624287"]}],"countries":["GB","NL"],"is_corresponding":false,"raw_author_name":"Maja Pantic","raw_affiliation_strings":["Dept. of Computing, UK / EEMCS, Imperial College London / Univ. of Twente, Netherlands"],"raw_orcid":null,"affiliations":[{"raw_affiliation_string":"Dept. of Computing, UK / EEMCS, Imperial College London / Univ. of Twente, Netherlands","institution_ids":["https://openalex.org/I124357947","https://openalex.org/I47508984","https://openalex.org/I94624287"]}]}],"institutions":[],"countries_distinct_count":2,"institutions_distinct_count":3,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":10.7741,"has_fulltext":true,"cited_by_count":116,"citation_normalized_percentile":{"value":0.98799896,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":90,"max":100},"biblio":{"volume":null,"issue":null,"first_page":"2304","last_page":"2308"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10860","display_name":"Speech and Audio Processing","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11309","display_name":"Music and Audio Processing","score":0.9998000264167786,"subfield":{"id":"https://openalex.org/subfields/1711","display_name":"Signal Processing"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10201","display_name":"Speech Recognition and Synthesis","score":0.9986000061035156,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/discrete-cosine-transform","display_name":"Discrete cosine transform","score":0.8917148113250732},{"id":"https://openalex.org/keywords/bottleneck","display_name":"Bottleneck","score":0.8812521696090698},{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.7769582271575928},{"id":"https://openalex.org/keywords/autoencoder","display_name":"Autoencoder","score":0.7325543165206909},{"id":"https://openalex.org/keywords/discriminative-model","display_name":"Discriminative model","score":0.701329231262207},{"id":"https://openalex.org/keywords/artificial-intelligence","display_name":"Artificial intelligence","score":0.6995335221290588},{"id":"https://openalex.org/keywords/speech-recognition","display_name":"Speech recognition","score":0.623317301273346},{"id":"https://openalex.org/keywords/deep-learning","display_name":"Deep learning","score":0.6125104427337646},{"id":"https://openalex.org/keywords/pattern-recognition","display_name":"Pattern recognition (psychology)","score":0.5300217270851135},{"id":"https://openalex.org/keywords/decoding-methods","display_name":"Decoding methods","score":0.45799949765205383},{"id":"https://openalex.org/keywords/encoding","display_name":"Encoding (memory)","score":0.41054394841194153},{"id":"https://openalex.org/keywords/image","display_name":"Image (mathematics)","score":0.23544952273368835},{"id":"https://openalex.org/keywords/algorithm","display_name":"Algorithm","score":0.09612324833869934}],"concepts":[{"id":"https://openalex.org/C2221639","wikidata":"https://www.wikidata.org/wiki/Q2877","display_name":"Discrete cosine transform","level":3,"score":0.8917148113250732},{"id":"https://openalex.org/C2780513914","wikidata":"https://www.wikidata.org/wiki/Q18210350","display_name":"Bottleneck","level":2,"score":0.8812521696090698},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.7769582271575928},{"id":"https://openalex.org/C101738243","wikidata":"https://www.wikidata.org/wiki/Q786435","display_name":"Autoencoder","level":3,"score":0.7325543165206909},{"id":"https://openalex.org/C97931131","wikidata":"https://www.wikidata.org/wiki/Q5282087","display_name":"Discriminative model","level":2,"score":0.701329231262207},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.6995335221290588},{"id":"https://openalex.org/C28490314","wikidata":"https://www.wikidata.org/wiki/Q189436","display_name":"Speech recognition","level":1,"score":0.623317301273346},{"id":"https://openalex.org/C108583219","wikidata":"https://www.wikidata.org/wiki/Q197536","display_name":"Deep learning","level":2,"score":0.6125104427337646},{"id":"https://openalex.org/C153180895","wikidata":"https://www.wikidata.org/wiki/Q7148389","display_name":"Pattern recognition (psychology)","level":2,"score":0.5300217270851135},{"id":"https://openalex.org/C57273362","wikidata":"https://www.wikidata.org/wiki/Q576722","display_name":"Decoding methods","level":2,"score":0.45799949765205383},{"id":"https://openalex.org/C125411270","wikidata":"https://www.wikidata.org/wiki/Q18653","display_name":"Encoding (memory)","level":2,"score":0.41054394841194153},{"id":"https://openalex.org/C115961682","wikidata":"https://www.wikidata.org/wiki/Q860623","display_name":"Image (mathematics)","level":2,"score":0.23544952273368835},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.09612324833869934},{"id":"https://openalex.org/C149635348","wikidata":"https://www.wikidata.org/wiki/Q193040","display_name":"Embedded system","level":1,"score":0.0}],"mesh":[],"locations_count":2,"locations":[{"id":"doi:10.1109/icassp.2016.7472088","is_oa":false,"landing_page_url":"https://doi.org/10.1109/icassp.2016.7472088","pdf_url":null,"source":null,"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","raw_type":"proceedings-article"},{"id":"pmh:oai:ris.utwente.nl:openaire_cris_publications/8c6caabd-366b-42fa-a763-227a0f94ed07","is_oa":true,"landing_page_url":"https://research.utwente.nl/en/publications/8c6caabd-366b-42fa-a763-227a0f94ed07","pdf_url":"https://ris.utwente.nl/ws/files/5317759/petridispantic_icassp2016.pdf","source":{"id":"https://openalex.org/S4406922991","display_name":"University of Twente Research Information","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Petridis, S & Pantic, M 2016, Deep Complementary Bottleneck Features for Visual Speech Recognition. in Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016. IEEE International Conference on Acoustics, Speech and Signal Processing, IEEE, Danvers, MA, USA, pp. 2304-2308, IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016, Shanghai, China, 20/03/16. https://doi.org/10.1109/ICASSP.2016.7472088","raw_type":"info:eu-repo/semantics/conferenceObject"}],"best_oa_location":{"id":"pmh:oai:ris.utwente.nl:openaire_cris_publications/8c6caabd-366b-42fa-a763-227a0f94ed07","is_oa":true,"landing_page_url":"https://research.utwente.nl/en/publications/8c6caabd-366b-42fa-a763-227a0f94ed07","pdf_url":"https://ris.utwente.nl/ws/files/5317759/petridispantic_icassp2016.pdf","source":{"id":"https://openalex.org/S4406922991","display_name":"University of Twente Research Information","issn_l":null,"issn":null,"is_oa":false,"is_in_doaj":false,"is_core":false,"host_organization":null,"host_organization_name":null,"host_organization_lineage":[],"host_organization_lineage_names":[],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false,"raw_source_name":"Petridis, S & Pantic, M 2016, Deep Complementary Bottleneck Features for Visual Speech Recognition. in Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016. IEEE International Conference on Acoustics, Speech and Signal Processing, IEEE, Danvers, MA, USA, pp. 2304-2308, IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2016, Shanghai, China, 20/03/16. https://doi.org/10.1109/ICASSP.2016.7472088","raw_type":"info:eu-repo/semantics/conferenceObject"},"sustainable_development_goals":[{"display_name":"Reduced inequalities","score":0.75,"id":"https://metadata.un.org/sdg/10"}],"awards":[],"funders":[],"has_content":{"pdf":true,"grobid_xml":true},"content_urls":{"pdf":"https://content.openalex.org/works/W2404704342.pdf","grobid_xml":"https://content.openalex.org/works/W2404704342.grobid-xml"},"referenced_works_count":23,"referenced_works":["https://openalex.org/W44815768","https://openalex.org/W304834817","https://openalex.org/W1588214556","https://openalex.org/W1970088388","https://openalex.org/W2022799064","https://openalex.org/W2076462394","https://openalex.org/W2095705004","https://openalex.org/W2100495367","https://openalex.org/W2105099419","https://openalex.org/W2113814270","https://openalex.org/W2121684305","https://openalex.org/W2123237149","https://openalex.org/W2136155248","https://openalex.org/W2138527036","https://openalex.org/W2184188583","https://openalex.org/W2397233148","https://openalex.org/W2406846463","https://openalex.org/W6610843619","https://openalex.org/W6674330103","https://openalex.org/W6678292227","https://openalex.org/W6686207219","https://openalex.org/W6712469390","https://openalex.org/W6713551954"],"related_works":["https://openalex.org/W1657880117","https://openalex.org/W2595172197","https://openalex.org/W2159052453","https://openalex.org/W3013693939","https://openalex.org/W2566616303","https://openalex.org/W2127970246","https://openalex.org/W3131327266","https://openalex.org/W2084856301","https://openalex.org/W1001352512","https://openalex.org/W4382618745"],"abstract_inverted_index":{"Deep":[0],"bottleneck":[1,39,75,99,113,122],"features":[2,41,100,107,123,159],"(DBNFs)":[3],"have":[4],"been":[5],"used":[6,134],"successfully":[7],"in":[8,77,111,117,155,165],"the":[9,47,54,81,84,87,98,112,121,137,141,146,161,174],"past":[10],"for":[11,22,60],"acoustic":[12],"speech":[13,24,62],"recognition":[14,25,63],"from":[15,65],"audio.":[16],"However,":[17],"research":[18],"on":[19,43,145],"extracting":[20],"DBNFs":[21,59],"visual":[23,40,61],"is":[26,53,143],"very":[27],"limited.":[28],"In":[29],"this":[30,52],"work,":[31],"we":[32],"present":[33],"an":[34,166],"approach":[35],"to":[36,79,119,125,135,171],"extract":[37],"deep":[38,44,71],"based":[42],"autoencoders.":[45],"To":[46],"best":[48,162],"of":[49,83,169],"our":[50],"knowledge,":[51],"first":[55,68],"work":[56],"that":[57],"extracts":[58],"directly":[64],"pixels.":[66],"We":[67],"train":[69],"a":[70,74],"autoencoder":[72],"with":[73,157],"layer":[76,114],"order":[78,118],"reduce":[80],"dimensionality":[82],"image.":[85],"Then":[86],"autoencoder's":[88],"decoding":[89],"layers":[90,95],"are":[91,108,133],"replaced":[92],"by":[93],"classification":[94],"which":[96],"make":[97,120],"more":[101],"discriminative.":[102],"Discrete":[103],"Cosine":[104],"Transform":[105],"(DCT)":[106],"also":[109],"appended":[110],"during":[115],"training":[116],"complementary":[124,153],"DCT":[126,158,175],"features.":[127],"Long-Short":[128],"Term":[129],"Memory":[130],"(LSTM)":[131],"networks":[132],"model":[136],"temporal":[138],"dynamics":[139],"and":[140,148],"performance":[142,163],"evaluated":[144],"OuluVS":[147],"AVLetters":[149],"databases.":[150],"The":[151],"extracted":[152],"DBNF":[154],"combination":[156],"achieve":[160],"resulting":[164],"absolute":[167],"improvement":[168],"up":[170],"5%":[172],"over":[173],"baseline.":[176]},"counts_by_year":[{"year":2025,"cited_by_count":6},{"year":2024,"cited_by_count":12},{"year":2023,"cited_by_count":8},{"year":2022,"cited_by_count":10},{"year":2021,"cited_by_count":16},{"year":2020,"cited_by_count":15},{"year":2019,"cited_by_count":17},{"year":2018,"cited_by_count":17},{"year":2017,"cited_by_count":14},{"year":2016,"cited_by_count":1}],"updated_date":"2026-07-29T14:22:42.915294","created_date":"2025-10-10T00:00:00"}
