{"doi":"10.1109/taffc.2021.3114365","title":"Survey of Deep Representation Learning for Speech Emotion Recognition","abstract":null,"journal":"IEEE Transactions on Affective Computing","year":2023,"id":684408,"datarank":0.7433740586401892,"base_score":4.955827057601261,"endowment":4.955827057601261,"self_citation_contribution":0.7433740586401892,"citation_network_contribution":0.0,"self_endowment_contribution":0.7433740586401892,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":141,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":null,"is_data_producer":false,"deposit_databanks":null,"is_oa":false,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":null,"fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":1014285,"name":"Rajib Rana","orcid":"0000-0002-0506-2409","position":1,"is_corresponding":false},{"id":1786247,"name":"Sara Khalifa","orcid":"0000-0002-3417-2834","position":2,"is_corresponding":false},{"id":1786248,"name":"Raja Jurdak","orcid":"0000-0001-7517-0782","position":3,"is_corresponding":false},{"id":1510447,"name":"Junaid Qadir","orcid":"0000-0001-9466-2475","position":4,"is_corresponding":false},{"id":97861,"name":"Björn W. Schuller","orcid":"0000-0002-6478-8699","position":5,"is_corresponding":false},{"id":1510445,"name":"Siddique Latif","orcid":"0000-0001-5662-4777","position":0,"is_corresponding":false}],"reference_count":0,"raw_metadata":{"has_enrichment":true,"resolved":true,"title":"Survey of Deep Representation Learning for Speech Emotion Recognition","abstract":"Traditionally, speech emotion recognition (SER) research has relied on manually handcrafted acoustic features using feature engineering. However, the design of handcrafted features for complex SER tasks requires significant manual effort, which impedes generalisability and slows the pace of innovation. This has motivated the adoption of representation learning techniques that can automatically learn an intermediate representation of the input signal without any manual feature engineering. Representation learning has led to improved SER performance and enabled rapid innovation. Its effectiveness has further increased with advances in deep learning (DL), which has facilitateddeep representation learningwhere hierarchical representations are automatically learned in a data-driven manner. This article presents the first comprehensive survey on the important topic of deep representation learning for SER. We highlight various techniques, related challenges and identify important future areas of research. Our survey bridges the gap in the literature since existing surveys either focus on SER with hand-engineered features or representation learning in the general setting without focusing on SER.","is_dataset_classified":null,"base_score":4.955827057601261,"endowment":4.955827057601261,"datacite_reuse_total":0,"file_count":0,"downloads":0,"views":0,"has_version_chain":false,"is_dataset":false,"is_oa":false,"pmid":"26207759","pmcid":null,"openalex_id":"https://openalex.org/W3199964822","authors":[],"funders":[],"total_grants":0,"fwci":10.1749,"citation_percentile":0.98915166,"influential_citations":0,"citation_trend":[{"year":2022,"count":11},{"year":2023,"count":23},{"year":2024,"count":39},{"year":2025,"count":43},{"year":2026,"count":25}],"oa_status":"green","license":"https://doi.org/10.15223/policy-029","oa_locations":[{"url":"https://opus.bibliothek.uni-augsburg.de/opus4/files/91554/91554.pdf","host_type":"repository"},{"url":"https://opus.bibliothek.uni-augsburg.de/opus4/files/91554/91554.pdf","host_type":"repository"},{"url":"http://xplorestaging.ieee.org/ielx7/5165369/10138707/09543566.pdf?arnumber=9543566","host_type":"publisher"},{"url":"https://opus.bibliothek.uni-augsburg.de/opus4/frontdoor/index/index/docId/91554","host_type":"repository"},{"url":"https://doi.org/10.1109/taffc.2021.3114365","host_type":"journal"},{"url":"https://eprints.qut.edu.au/213410/1/Emotional_Representations_Review_minor_revision_2_3.pdf","host_type":"repository"},{"url":"https://ieeexplore.ieee.org/document/9543566","host_type":"repository"},{"url":"https://mediatum.ub.tum.de/1773113","host_type":"repository"},{"url":"http://hdl.handle.net/10576/66063","host_type":"repository"}],"fields_of_study":["Speech and Audio Processing","Music and Audio Processing","Speech Recognition and Synthesis"],"mesh_terms":[],"keywords":["Feature engineering","Representation (politics)","Feature learning","Deep learning","Computer science","Artificial intelligence","Pace","Feature (linguistics)","Focus (optics)","Natural language processing","Machine learning","Linguistics"],"sdg_mappings":[{"sdg_number":0,"sdg_label":"Industry, innovation and infrastructure"}],"linked_datasets":[],"clinical_trials":[],"software_tools":[],"database_accessions":[],"source":"live","citation_network_status":"fetched"},"created_at":"2026-08-18T13:59:49.425861Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}