{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T15:18:43Z","timestamp":1777130323046,"version":"3.51.4"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,11,3]],"date-time":"2023-11-03T00:00:00Z","timestamp":1698969600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,3]],"date-time":"2023-11-03T00:00:00Z","timestamp":1698969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Evolving Systems"],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1007\/s12530-023-09550-9","type":"journal-article","created":{"date-parts":[[2023,11,3]],"date-time":"2023-11-03T11:01:38Z","timestamp":1699009298000},"page":"541-554","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Speech emotion classification using feature-level and classifier-level fusion"],"prefix":"10.1007","volume":"15","author":[{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0001-8076-8295","authenticated-orcid":false,"given":"Siba Prasad","family":"Mishra","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pankaj","family":"Warule","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suman","family":"Deb","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,3]]},"reference":[{"issue":"10","key":"9550_CR1","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid O, Mohamed A-R, Jiang H, Deng L, Penn G, Yu D (2014) Convolutional neural networks for speech recognition. IEEE\/ACM Trans Audio Speech Language Process 22(10):1533\u20131545","journal-title":"IEEE\/ACM Trans Audio Speech Language Process"},{"key":"9550_CR2","doi-asserted-by":"publisher","first-page":"49265","DOI":"10.1109\/ACCESS.2022.3172954","volume":"10","author":"AA Abdelhamid","year":"2022","unstructured":"Abdelhamid AA, El-Kenawy E-SM, Alotaibi B, Amer GM, Abdelkader MY, Ibrahim A, Eid MM (2022) Robust speech emotion recognition using CNN+ lSTM based on stochastic fractal search optimization algorithm. IEEE Access 10:49265\u201349284","journal-title":"IEEE Access"},{"key":"9550_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2021.108046","volume":"179","author":"J Ancilin","year":"2021","unstructured":"Ancilin J, Milton A (2021) Improved speech emotion recognition with Mel frequency magnitude coefficient. Appl Acoust 179:108046","journal-title":"Appl Acoust"},{"key":"9550_CR4","doi-asserted-by":"publisher","first-page":"36018","DOI":"10.1109\/ACCESS.2022.3163856","volume":"10","author":"F Andayani","year":"2022","unstructured":"Andayani F, Theng LB, Tsun MT, Chua C (2022) Hybrid lSTM-transformer model for emotion recognition from speech audio files. IEEE Access 10:36018\u201336027","journal-title":"IEEE Access"},{"key":"9550_CR5","doi-asserted-by":"crossref","unstructured":"Badshah A\u00a0M, Ahmad J, Rahim v, Baik S\u00a0W (2017) Speech emotion recognition from spectrograms with deep convolutional neural network. In: 2017 international conference on platform technology and service (PlatCon), IEEE, pp 1\u20135","DOI":"10.1109\/PlatCon.2017.7883728"},{"key":"9550_CR6","doi-asserted-by":"crossref","unstructured":"Bansal M, Yadav S, Vishwakarma D\u00a0K (2021) A language-independent speech sentiment analysis using prosodic features. In: 2021 5th International Conference on Computing Methodologies and Communication (ICCMC), IEEE, pp 1210\u20131216","DOI":"10.1109\/ICCMC51019.2021.9418357"},{"issue":"10","key":"9550_CR7","doi-asserted-by":"publisher","first-page":"1440","DOI":"10.1109\/LSP.2018.2860246","volume":"25","author":"M Chen","year":"2018","unstructured":"Chen M, He X, Yang J, Zhang H (2018) 3-d convolutional recurrent neural networks with attention model for speech emotion recognition. IEEE Signal Process Lett 25(10):1440\u20131444","journal-title":"IEEE Signal Process Lett"},{"key":"9550_CR8","doi-asserted-by":"publisher","first-page":"34862","DOI":"10.1109\/ACCESS.2019.2902870","volume":"7","author":"G-H Choi","year":"2019","unstructured":"Choi G-H, Bak E-S, Pan S-B (2019) User identification system using 2d resized spectrogram features of ECG. IEEE Access 7:34862\u201334873","journal-title":"IEEE Access"},{"key":"9550_CR9","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1016\/j.compeleceng.2016.09.027","volume":"55","author":"S Deb","year":"2016","unstructured":"Deb S, Dandapat S (2016) Classification of speech under stress using harmonic peak to energy ratio. Comput Electric Eng 55:12\u201323","journal-title":"Comput Electric Eng"},{"key":"9550_CR10","doi-asserted-by":"crossref","unstructured":"Deb S, Dandapat S (2016) Emotion classification using residual sinusoidal peak amplitude. In: 2016 International conference on signal processing and communications (SPCOM), IEEE, pp 1\u20135","DOI":"10.1109\/SPCOM.2016.7746697"},{"key":"9550_CR11","doi-asserted-by":"crossref","unstructured":"Deb S, Dandapat S (2017) Exploration of phase information for speech emotion classification. In: 2017 Twenty-third National Conference on Communications (NCC), IEEE, pp 1\u20135","DOI":"10.1109\/NCC.2017.8077114"},{"key":"9550_CR12","doi-asserted-by":"crossref","unstructured":"Dolka H, VM AX, Juliet S (2021) Speech emotion recognition using ann on mfcc features. In: 2021 3rd International Conference on Signal Processing and Communication (ICPSC), IEEE, pp 431\u2013435","DOI":"10.1109\/ICSPC51351.2021.9451810"},{"key":"9550_CR13","doi-asserted-by":"crossref","unstructured":"Ezzameli K, Mahersia H (2023) Emotion recognition from unimodal to multimodal analysis: a review. Inf Fusion 101847","DOI":"10.1016\/j.inffus.2023.101847"},{"key":"9550_CR14","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2020.102951","volume":"110","author":"MS Fahad","year":"2021","unstructured":"Fahad MS, Ranjan A, Yadav J, Deepak A (2021) A survey of speech emotion recognition in natural environment. Digital Signal Process 110:102951","journal-title":"Digital Signal Process"},{"key":"9550_CR15","doi-asserted-by":"crossref","unstructured":"Fu W, Yang X, Wang Y (2010) Heart sound diagnosis based on DTW and MFCC. In: 2010 3rd International Congress on Image and Signal Processing, Vol.\u00a06, IEEE, pp 2920\u20132923","DOI":"10.1109\/CISP.2010.5646678"},{"key":"9550_CR16","doi-asserted-by":"crossref","unstructured":"Huang Z, Dong M, Mao Q, Zhan Y (2014) Speech emotion recognition using CNN. In: Proceedings of the 22nd ACM international conference on Multimedia, pp 801\u2013804","DOI":"10.1145\/2647868.2654984"},{"key":"9550_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2020.101894","volume":"59","author":"D Issa","year":"2020","unstructured":"Issa D, Demirci MF, Yazici A (2020) Speech emotion recognition with deep convolutional neural networks. Biomed Signal Process Control 59:101894","journal-title":"Biomed Signal Process Control"},{"key":"9550_CR18","unstructured":"Ittichaichareon C, Suksri S, Yingthawornsuk T (2012) Speech recognition using mfcc. In: International conference on computer graphics, simulation and modeling, Vol.\u00a09"},{"issue":"1","key":"9550_CR19","doi-asserted-by":"publisher","first-page":"183","DOI":"10.3390\/s20010183","volume":"20","author":"S Kwon","year":"2019","unstructured":"Kwon S (2019) A CNN-assisted enhanced audio signal processing for speech emotion recognition. Sensors 20(1):183","journal-title":"Sensors"},{"issue":"2","key":"9550_CR20","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1109\/TSA.2004.838534","volume":"13","author":"CM Lee","year":"2005","unstructured":"Lee CM, Narayanan SS (2005) Toward detecting emotions in spoken dialogs. IEEE Trans Speech Audio Process 13(2):293\u2013303","journal-title":"IEEE Trans Speech Audio Process"},{"key":"9550_CR21","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1016\/j.ins.2021.02.016","volume":"563","author":"Z-T Liu","year":"2021","unstructured":"Liu Z-T, Rehman A, Wu M, Cao W-H, Hao M (2021) Speech emotion recognition based on formant characteristics feature extraction and phoneme type convergence. Inf Sci 563:309\u2013325","journal-title":"Inf Sci"},{"key":"9550_CR22","doi-asserted-by":"crossref","unstructured":"Lukose S, Upadhya SS (2017) Music player based on emotion recognition of voice signals. 2017 International Conference on Intelligent Computing. Instrumentation and Control Technologies (ICICICT), IEEE, pp 1751\u20131754","DOI":"10.1109\/ICICICT1.2017.8342835"},{"key":"9550_CR23","doi-asserted-by":"crossref","unstructured":"Mekruksavanich S, Jitpattanakul A, Hnoohom N (2020) Negative emotion recognition using deep learning for Thai language. In: 2020 joint international conference on digital arts, media and technology with ECTI northern section conference on electrical, electronics, computer and telecommunications engineering (ECTI DAMT & NCON), IEEE, pp 71\u201374","DOI":"10.1109\/ECTIDAMTNCON48261.2020.9090768"},{"key":"9550_CR24","doi-asserted-by":"crossref","unstructured":"Milton A, Roy SS, Selvi ST (2013) Svm scheme for speech emotion recognition using MFCC feature. Int J Comput Appl 69(9)","DOI":"10.5120\/11872-7667"},{"key":"9550_CR25","doi-asserted-by":"crossref","unstructured":"Mishra S\u00a0P, Warule P, Deb S (2023) Deep learning based emotion classification using Mel frequency magnitude coefficient. In: 2023 1st International Conference on Innovations in High Speed Communication and Signal Processing (IHCSP), IEEE, pp 93\u201398","DOI":"10.1109\/IHCSP56702.2023.10127148"},{"key":"9550_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107141","volume":"103","author":"AB Nassif","year":"2021","unstructured":"Nassif AB, Shahin I, Hamsa S, Nemmour N, Hirose K (2021) Casa-based speaker identification using cascaded GMM-CNN classifier in noisy and emotional talking conditions. Appl Soft Comput 103:107141","journal-title":"Appl Soft Comput"},{"key":"9550_CR27","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1016\/j.apacoust.2018.08.003","volume":"142","author":"T \u00d6zseven","year":"2018","unstructured":"\u00d6zseven T (2018) Investigation of the effect of spectrogram images and different texture analysis methods on speech emotion recognition. Appl Acoust 142:70\u201377","journal-title":"Appl Acoust"},{"key":"9550_CR28","doi-asserted-by":"crossref","unstructured":"Pandey SK, Shekhawat HS, Prasanna SM (2019) Deep learning techniques for speech emotion recognition: a review. In: 2019 29th International Conference Radioelektronika (RADIOELEKTRONIKA), IEEE, pp 1\u20136","DOI":"10.1109\/RADIOELEK.2019.8733432"},{"key":"9550_CR29","doi-asserted-by":"publisher","first-page":"79861","DOI":"10.1109\/ACCESS.2020.2990405","volume":"8","author":"M Sajjad","year":"2020","unstructured":"Sajjad M, Kwon S et al (2020) Clustering-based speech emotion recognition by incorporating learned features and deep Bilstm. IEEE Access 8:79861\u201379875","journal-title":"IEEE Access"},{"key":"9550_CR30","doi-asserted-by":"crossref","unstructured":"Satt A, Rozenberg S, Hoory R (2017) Efficient emotion recognition from speech using deep learning on spectrograms. In: Interspeech, pp 1089\u20131093","DOI":"10.21437\/Interspeech.2017-200"},{"issue":"9\u201310","key":"9550_CR31","doi-asserted-by":"publisher","first-page":"1062","DOI":"10.1016\/j.specom.2011.01.011","volume":"53","author":"B Schuller","year":"2011","unstructured":"Schuller B, Batliner A, Steidl S, Seppi D (2011) Recognising realistic emotions and affect in speech: state of the art and lessons learnt from the first challenge. Speech Commun 53(9\u201310):1062\u20131087","journal-title":"Speech Commun"},{"key":"9550_CR32","doi-asserted-by":"publisher","first-page":"190784","DOI":"10.1109\/ACCESS.2020.3031763","volume":"8","author":"Y\u00dc S\u00f6nmez","year":"2020","unstructured":"S\u00f6nmez Y\u00dc, Varol A (2020) A speech emotion recognition model based on multi-level local binary and local ternary patterns. IEEE Access 8:190784\u2013190796","journal-title":"IEEE Access"},{"issue":"4","key":"9550_CR33","doi-asserted-by":"publisher","first-page":"931","DOI":"10.1007\/s10772-018-9551-4","volume":"21","author":"L Sun","year":"2018","unstructured":"Sun L, Chen J, Xie K, Gu T (2018) Deep and shallow features fusion based on deep convolutional neural network for speech emotion recognition. Int J Speech Technol 21(4):931\u2013940","journal-title":"Int J Speech Technol"},{"key":"9550_CR34","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1016\/j.specom.2019.10.004","volume":"115","author":"L Sun","year":"2019","unstructured":"Sun L, Zou B, Fu S, Chen J, Wang F (2019) Speech emotion recognition based on DNN-decision tree SVM model. Speech Commun 115:29\u201337","journal-title":"Speech Commun"},{"issue":"1","key":"9550_CR35","first-page":"19","volume":"1","author":"V Tiwari","year":"2010","unstructured":"Tiwari V (2010) Mfcc and its applications in speaker recognition. Int J Emerg Technol 1(1):19\u201322","journal-title":"Int J Emerg Technol"},{"key":"9550_CR36","doi-asserted-by":"crossref","unstructured":"Valles D, Matin R (2021) An audio processing approach using ensemble learning for speech-emotion recognition for children with ASD. In: 2021 IEEE World AI IoT Congress (AIIoT), IEEE, pp 0055\u20130061","DOI":"10.1109\/AIIoT52608.2021.9454174"},{"key":"9550_CR37","unstructured":"Ververidis D, Kotropoulos C (2003) A state of the art review on emotional speech databases. In: Proceedings of 1st Richmedia Conference, Citeseer, pp 109\u2013119"},{"key":"9550_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.104653","volume":"83","author":"P Warule","year":"2023","unstructured":"Warule P, Mishra SP, Deb S, Krajewski J (2023) Sinusoidal model-based diagnosis of the common cold from the speech signal. Biomed Signal Process Control 83:104653","journal-title":"Biomed Signal Process Control"},{"key":"9550_CR39","doi-asserted-by":"crossref","unstructured":"Warule P, Mishra S\u00a0P, Deb S (2022) Classification of cold and non-cold speech using vowel-like region segments. In: 2022 IEEE International Conference on Signal Processing and Communications (SPCOM), IEEE, pp 1\u20135","DOI":"10.1109\/SPCOM55316.2022.9840775"},{"key":"9550_CR40","doi-asserted-by":"crossref","unstructured":"Warule P, Mishra S\u00a0P, Deb S (2023) Time-frequency analysis of speech signal using chirplet transform for automatic diagnosis of Parkinson\u2019s disease. Biomed Eng Lett 1\u201311","DOI":"10.1109\/LSENS.2023.3311670"},{"key":"9550_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2020.107721","volume":"173","author":"S Yildirim","year":"2021","unstructured":"Yildirim S, Kaya Y, K\u0131l\u0131\u00e7 F (2021) A modified feature selection method based on metaheuristic algorithms for speech emotion recognition. Appl Acoust 173:107721","journal-title":"Appl Acoust"},{"issue":"5","key":"9550_CR42","doi-asserted-by":"publisher","first-page":"620","DOI":"10.1109\/LSP.2014.2311435","volume":"21","author":"L Z\u00e3o","year":"2014","unstructured":"Z\u00e3o L, Cavalcante D, Coelho R (2014) Time-frequency feature and AMS-GMM mask for acoustic emotion classification. IEEE Signal Process Lett 21(5):620\u2013624","journal-title":"IEEE Signal Process Lett"},{"issue":"3","key":"9550_CR43","doi-asserted-by":"publisher","first-page":"3705","DOI":"10.1007\/s11042-017-5539-3","volume":"78","author":"Y Zeng","year":"2019","unstructured":"Zeng Y, Mao H, Peng D, Yi Z (2019) Spectrogram based multi-task audio classification. Multimed Tools Appl 78(3):3705\u20133722","journal-title":"Multimed Tools Appl"},{"key":"9550_CR44","doi-asserted-by":"publisher","first-page":"312","DOI":"10.1016\/j.bspc.2018.08.035","volume":"47","author":"J Zhao","year":"2019","unstructured":"Zhao J, Mao X, Chen L (2019) Speech emotion recognition using deep 1d & 2d CNN lSTM networks. Biomed Signal Process Control 47:312\u2013323","journal-title":"Biomed Signal Process Control"}],"container-title":["Evolving Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s12530-023-09550-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/article\/10.1007\/s12530-023-09550-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s12530-023-09550-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,28]],"date-time":"2024-03-28T12:24:15Z","timestamp":1711628655000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/10.1007\/s12530-023-09550-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,3]]},"references-count":44,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,4]]}},"alternative-id":["9550"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/s12530-023-09550-9","relation":{},"ISSN":["1868-6478","1868-6486"],"issn-type":[{"value":"1868-6478","type":"print"},{"value":"1868-6486","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,3]]},"assertion":[{"value":"27 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 November 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}