{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T20:37:25Z","timestamp":1782506245123,"version":"3.54.5"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,11,11]],"date-time":"2023-11-11T00:00:00Z","timestamp":1699660800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,11]],"date-time":"2023-11-11T00:00:00Z","timestamp":1699660800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s11760-023-02845-z","type":"journal-article","created":{"date-parts":[[2023,11,11]],"date-time":"2023-11-11T19:02:13Z","timestamp":1699729333000},"page":"1365-1373","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Improving speech command recognition through decision-level fusion of deep filtered speech cues"],"prefix":"10.1007","volume":"18","author":[{"given":"Sunakshi","family":"Mehra","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Virender","family":"Ranga","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ritu","family":"Agarwal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,11,11]]},"reference":[{"issue":"1","key":"2845_CR1","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.: Dropout: a simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(1), 1929\u20131958 (2014)","journal-title":"J. Mach. Learn. Res."},{"issue":"1","key":"2845_CR2","doi-asserted-by":"publisher","first-page":"18","DOI":"10.5815\/ijmecs.2020.01.03","volume":"12","author":"A Iqbal","year":"2020","unstructured":"Iqbal, A., Aftab, S.: A classification framework for software defect prediction using multi-filter feature selection technique and MLP. Int. J. Mod. Educ. Comput. Sci. 12(1), 18 (2020)","journal-title":"Int. J. Mod. Educ. Comput. Sci."},{"key":"2845_CR3","doi-asserted-by":"crossref","unstructured":"Hermansky, H.: The modulation spectrum in the automatic recognition of speech. In: 1997 IEEE Workshop on Automatic Speech Recognition and Understanding Proceedings, pp. 140\u2013147. IEEE (1997)","DOI":"10.1109\/ASRU.1997.658998"},{"key":"2845_CR4","doi-asserted-by":"crossref","unstructured":"Sadhu, S., Hermansky, H.: Importance of different temporal modulations of speech: a tale of two perspectives. In: ICASSP 2023\u20132023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10095972"},{"key":"2845_CR5","doi-asserted-by":"crossref","unstructured":"Narayanan, A., Wang, D.: Ideal ratio mask estimation using deep neural networks for robust speech recognition. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 7092\u20137096. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639038"},{"key":"2845_CR6","doi-asserted-by":"publisher","first-page":"1778","DOI":"10.1109\/TASLP.2020.2998279","volume":"28","author":"Z-Q Wang","year":"2020","unstructured":"Wang, Z.-Q., Wang, P., Wang, DeLiang: Complex spectral mapping for single-and multi-channel speech enhancement and robust ASR. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 28, 1778\u20131787 (2020)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"2845_CR7","doi-asserted-by":"crossref","unstructured":"Luo, Y., Mesgarani, N.: Tasnet: time-domain audio separation network for real-time, single-channel speech separation. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 696\u2013700. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8462116"},{"key":"2845_CR8","doi-asserted-by":"crossref","unstructured":"Liu, Ze, Yutong Lin, Yue Cao, Han Hu, Yixuan Wei, Zheng Zhang, Stephen Lin, and Baining Guo. \"Swin transformer: Hierarchical vision transformer using shifted windows.\" In\u00a0Proceedings of the IEEE\/CVF international conference on computer vision, pp. 10012\u201310022. 2021.","DOI":"10.1109\/ICCV48922.2021.00986"},{"issue":"1\u20132","key":"2845_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1561\/2000000001","volume":"1","author":"LR Rabiner","year":"2007","unstructured":"Rabiner, L.R., Schafer, R.W.: Introduction to digital speech processing. Found. Trendsin\u00ae Signal Process. 1(1\u20132), 1\u2013194 (2007)","journal-title":"Found. Trendsin\u00ae Signal Process."},{"key":"2845_CR10","volume-title":"Digital Image Processing","author":"RC Gonzalez","year":"2008","unstructured":"Gonzalez, R.C., Woods, R.E.: Digital Image Processing. Prentice hall, Upper Saddle River (2008)"},{"key":"2845_CR11","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2022.108761","volume":"193","author":"Y Korkmaz","year":"2022","unstructured":"Korkmaz, Y., Boyac\u0131, A.: A comprehensive Turkish accent\/dialect recognition system using acoustic perceptual formants. Appl. Acoust. 193, 108761 (2022)","journal-title":"Appl. Acoust."},{"issue":"1","key":"2845_CR12","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2014","unstructured":"Xu, Y., Jun, Du., Dai, L.-R., Lee, C.-H.: A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 23(1), 7\u201319 (2014)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"2845_CR13","doi-asserted-by":"crossref","unstructured":"Chen, Z.: Noise reduction of bird calls based on a combination of spectral subtraction, Wiener filtering, and Kalman filtering. In: 2022 IEEE 5th International Conference on Automation, Electronics and Electrical Engineering (AUTEEE), pp. 512\u2013517. IEEE (2022)","DOI":"10.1109\/AUTEEE56487.2022.9994282"},{"key":"2845_CR14","doi-asserted-by":"publisher","first-page":"2267","DOI":"10.1109\/TASLP.2021.3091805","volume":"29","author":"S Liu","year":"2021","unstructured":"Liu, S., Geng, M., Shoukang, Hu., Xie, X., Cui, M., Jianwei, Yu., Liu, X., Meng, H.: Recent progress in the CUHK dysarthric speech recognition system. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 29, 2267\u20132281 (2021)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"2845_CR15","doi-asserted-by":"crossref","unstructured":"Xiong, F., Barker, J., Christensen, H.: Phonetic analysis of dysarthric speech tempo and applications to robust personalised dysarthric speech recognition. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5836\u20135840. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8683091"},{"key":"2845_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2022.104408","volume":"80","author":"Y Korkmaz","year":"2023","unstructured":"Korkmaz, Y., Boyac\u0131, A.: Hybrid voice activity detection system based on LSTM and auditory speech features. Biomed. Signal Process. Control 80, 104408 (2023)","journal-title":"Biomed. Signal Process. Control"},{"key":"2845_CR17","unstructured":"Aks\u00ebnova, A., Chen, Z., Chiu, C.C., van Esch, D., Golik, G., Han, W., King , L. et al.: Accented speech recognition: benchmarking, pre-training, and diverse data. arXiv preprint arXiv:2205.08014\u00a0(2022)"},{"key":"2845_CR18","doi-asserted-by":"crossref","unstructured":"Korkmaz, Y., Boyac\u0131, A.: Analysis of speaker's gender effects in voice onset time of Turkish stop consonants. In: 2018 6th International Symposium on Digital Forensic and Security (ISDFS), pp. 1\u20135. IEEE (2018)","DOI":"10.1109\/ISDFS.2018.8355341"},{"key":"2845_CR19","doi-asserted-by":"crossref","unstructured":"Sadhu, S., Li, R., Hermansky, H.: M-vectors: sub-band based energy modulation features for multi-stream automatic speech recognition. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6545\u20136549. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682710"},{"key":"2845_CR20","doi-asserted-by":"crossref","unstructured":"Das, P., Bhattacharjee, U.: Robust speaker verification using GFCC and joint factor analysis. In: Fifth International Conference on Computing, Communications and Networking Technologies (ICCCNT), pp. 1\u20134. IEEE (2014)","DOI":"10.1109\/ICCCNT.2014.6963092"},{"key":"2845_CR21","doi-asserted-by":"crossref","unstructured":"Shi, X., Yang, H., Zhou, P.: Robust speaker recognition based on improved GFCC. In: 2016 2nd IEEE international conference on computer and communications (ICCC), pp. 1927\u20131931. IEEE (2016)","DOI":"10.1109\/CompComm.2016.7925037"},{"issue":"7","key":"2845_CR22","doi-asserted-by":"publisher","first-page":"1315","DOI":"10.1109\/TASLP.2016.2545928","volume":"24","author":"C Kim","year":"2016","unstructured":"Kim, C., Stern, R.M.: Power-normalized cepstral coefficients (PNCC) for robust speech recognition. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 24(7), 1315\u20131329 (2016)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"issue":"1","key":"2845_CR23","first-page":"39","volume":"38","author":"A Badi","year":"2019","unstructured":"Badi, A., Ko, K., Ko, H.: Bird sounds classification by combining PNCC and robust Mel-log filter bank features. J. Acoust. Soc. Korea 38(1), 39\u201346 (2019)","journal-title":"J. Acoust. Soc. Korea"},{"issue":"1","key":"2845_CR24","first-page":"1261","volume":"29","author":"V Passricha","year":"2019","unstructured":"Passricha, V., Aggarwal, R.K.: A hybrid of deep CNN and bidirectional LSTM for automatic speech recognition. J. Intell. Syst. 29(1), 1261\u20131274 (2019)","journal-title":"J. Intell. Syst."},{"issue":"10","key":"2845_CR25","doi-asserted-by":"publisher","first-page":"3683","DOI":"10.3390\/s22103683","volume":"22","author":"A Mukhamadiyev","year":"2022","unstructured":"Mukhamadiyev, A., Khujayarov, I., Djuraev, O., Cho, J.: Automatic speech recognition method based on deep learning approaches for Uzbek language. Sensors 22(10), 3683 (2022)","journal-title":"Sensors"},{"key":"2845_CR26","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: 2011 IEEE Workshop on Automatic Speech Recognition & Understanding, pp. 24\u201329. IEEE (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"2845_CR27","doi-asserted-by":"crossref","unstructured":"Uebel, L.F., Woodland, P.C.: An investigation into vocal tract length normalisation. In: Sixth European Conference on Speech Communication and Technology (1999)","DOI":"10.21437\/Eurospeech.1999-553"},{"issue":"2","key":"2845_CR28","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"MJF Gales","year":"1998","unstructured":"Gales, M.J.F.: Maximum likelihood linear transformations for HMM-based speech recognition. Comput. Speech Lang. 12(2), 75\u201398 (1998)","journal-title":"Comput. Speech Lang."},{"key":"2845_CR29","doi-asserted-by":"crossref","unstructured":"Mehra, S., Susan, S.: Improving word recognition in speech transcriptions by decision-level fusion of stemming and two-way phoneme pruning. In: Advanced Computing: 10th International Conference, IACC 2020, Panaji, Goa, India, December 5\u20136, 2020, Revised Selected Papers, Part I 10, pp. 256\u2013266. Springer, Singapore (2021)","DOI":"10.1007\/978-981-16-0401-0_19"},{"issue":"4","key":"2845_CR30","doi-asserted-by":"publisher","first-page":"673","DOI":"10.26599\/TST.2022.9010038","volume":"28","author":"Q Zhang","year":"2023","unstructured":"Zhang, Q., Zhang, H., Zhou, K., Zhang, Le.: Developing a physiological signal-based, mean threshold and decision-level fusion algorithm (PMD) for emotion recognition. Tsinghua Sci. Technol. 28(4), 673\u2013685 (2023)","journal-title":"Tsinghua Sci. Technol."},{"key":"2845_CR31","doi-asserted-by":"crossref","unstructured":"Mehra, S., Susan, S.: Deep fusion framework for speech command recognition using acoustic and linguistic features. Multimed. Tools Appl., 1\u201325 (2023)","DOI":"10.1007\/s11042-023-15118-1"},{"key":"2845_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3584861","volume":"22","author":"R Das","year":"2023","unstructured":"Das, R., Singh, T.D.: Image-text multimodal sentiment analysis framework of assamese news articles using late fusion. ACM Trans. Asian Low-Resour. Lang. Inf. Process. 22, 1\u201330 (2023)","journal-title":"ACM Trans. Asian Low-Resour. Lang. Inf. Process."},{"key":"2845_CR33","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1016\/j.inffus.2022.09.012","volume":"90","author":"J Zhu","year":"2023","unstructured":"Zhu, J., Huang, C., De Meo, P.: DFMKE: a dual fusion multi-modal knowledge graph embedding framework for entity alignment. Inf. Fusion 90, 111\u2013119 (2023)","journal-title":"Inf. Fusion"},{"key":"2845_CR34","doi-asserted-by":"crossref","unstructured":"Mehra, S., Susan, S.: Early fusion of phone embeddings for recognition of low-resourced accented speech. In: 2022 4th International Conference on Artificial Intelligence and Speech Technology (AIST), pp. 1\u20135. IEEE (2022)","DOI":"10.1109\/AIST55798.2022.10064735"},{"key":"2845_CR35","unstructured":"Warden, P.: Speech commands: a dataset for limited-vocabulary speech recognition. arXiv preprint arXiv:1804.03209\u00a0(2018)"},{"key":"2845_CR36","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M. et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929\u00a0(2020)"},{"issue":"7","key":"2845_CR37","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. J. Mach. Learn. Res. 12(7), 2121\u20132159 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"2845_CR38","doi-asserted-by":"crossref","unstructured":"Haque, M.A., Verma, A., Alex, J.S.R., Venkatesan, N.: Experimental evaluation of CNN architecture for speech recognition. In: First International Conference on Sustainable Technologies for Computational Intelligence: Proceedings of ICTSCI 2019, pp. 507\u2013514. Springer, Singapore (2020)","DOI":"10.1007\/978-981-15-0029-9_40"},{"issue":"1","key":"2845_CR39","doi-asserted-by":"publisher","first-page":"27","DOI":"10.21608\/ejle.2020.47685.1015","volume":"8","author":"ER Abdelmaksoud","year":"2021","unstructured":"Abdelmaksoud, E.R., Hassen, A., Hassan, N., Hesham, M.: Convolutional neural network for Arabic speech recognition. Egypt. J. Lang. Eng. 8(1), 27\u201338 (2021)","journal-title":"Egypt. J. Lang. Eng."},{"key":"2845_CR40","doi-asserted-by":"crossref","unstructured":"McDermott, E., Sak, H., Variani, E.: A density ratio approach to language model fusion in end-to-end automatic speech recognition. In: 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 434\u2013441. IEEE (2019)","DOI":"10.1109\/ASRU46091.2019.9003790"},{"key":"2845_CR41","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/s10772-018-09573-7","volume":"22","author":"T Zia","year":"2019","unstructured":"Zia, T., Zahid, U.: Long short-term memory recurrent neural network architectures for Urdu acoustic modeling. Int. J. Speech Technol. 22, 21\u201330 (2019)","journal-title":"Int. J. Speech Technol."},{"key":"2845_CR42","doi-asserted-by":"crossref","unstructured":"Lezhenin, I., Bogach, N., Pyshkin, E.: Urban sound classification using long short-term memory neural network. In: 2019 federated conference on computer science and information systems (FedCSIS), pp. 57\u201360 (2019)","DOI":"10.15439\/2019F185"},{"key":"2845_CR43","doi-asserted-by":"publisher","first-page":"10767","DOI":"10.1109\/ACCESS.2019.2891838","volume":"7","author":"M Zeng","year":"2019","unstructured":"Zeng, M., Xiao, N.: Effective combination of DenseNet and BiLSTM for keyword spotting. IEEE Access 7, 10767\u201310775 (2019)","journal-title":"IEEE Access"},{"key":"2845_CR44","unstructured":"De Andrade, D.C., Leo, S., Viana, M.L.D.S., Bernkopf, C.: A neural attention model for speech command recognition.\u00a0arXiv preprint arXiv:1808.08929\u00a0(2018)"},{"key":"2845_CR45","unstructured":"Wei, Y., Gong, Z., Yang, S., Ye, K., Wen, Y.: EdgeCRNN: an edge-computing oriented model of acoustic feature enhancement for keyword spotting. J. Ambient Intell. Humaniz. Comput., 1\u201311 (2022)"},{"key":"2845_CR46","doi-asserted-by":"crossref","unstructured":"Cances, L., Pellegrini, T.: Comparison of deep co-training and mean-teacher approaches for semi-supervised audio tagging. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 361\u2013365. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9415116"},{"key":"2845_CR47","unstructured":"Higy, B., Bell, P.: Few-shot learning with attention-based sequence-to-sequence models. arXiv preprint arXiv:1811.03519\u00a0(2018)."}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02845-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/article\/10.1007\/s11760-023-02845-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02845-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T21:32:37Z","timestamp":1730496757000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/10.1007\/s11760-023-02845-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,11]]},"references-count":47,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["2845"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/s11760-023-02845-z","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,11]]},"assertion":[{"value":"10 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 September 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors affirm that they do not possess any identifiable financial or personal affiliations that might have influenced the conclusions presented in this study.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The article does not include any studies or investigations that involve human or animal participants.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}