@techreport{99f8092e7f6040b09afe400894a302a8,
title = "Learning spectro-temporal features with 3D CNNs for speech emotion recognition",
abstract = "In this paper, we propose to use deep 3-dimensional convolutional networks (3D CNNs) in order to address the challenge of modelling spectro-temporal dynamics for speech emotion recognition (SER). Compared to a hybrid of Convolutional Neural Network and Long-Short-Term-Memory (CNN-LSTM), our proposed 3D CNNs simultaneously extract short-term and long-term spectral features with a moderate number of parameters. We evaluated our proposed and other state-of-the-art methods in a speaker-independent manner using aggregated corpora that give a large and diverse set of speakers. We found that 1) shallow temporal and moderately deep spectral kernels of a homogeneous architecture are optimal for the task; and 2) our 3D CNNs are more effective for spectro-temporal feature learning compared to other methods. Finally, we visualised the feature space obtained with our proposed method using t-distributed stochastic neighbour embedding (T-SNE) and could observe distinct clusters of emotions.",
keywords = "cs.CL, cs.CV",
author = "Jaebok Kim and Truong, \{Khiet P.\} and Gwenn Englebienne and Vanessa Evers",
note = "ACII, 2017, San Antonio.",
year = "2017",
month = aug,
day = "14",
doi = "10.48550/arXiv.1708.05071",
language = "English",
publisher = "ArXiv.org",
type = "WorkingPaper",
institution = "ArXiv.org",
}