{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T16:18:12Z","timestamp":1781108292626,"version":"3.54.1"},"reference-count":102,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2021,11,5]],"date-time":"2021-11-05T00:00:00Z","timestamp":1636070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,11,5]],"date-time":"2021-11-05T00:00:00Z","timestamp":1636070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2022,4]]},"DOI":"10.1007\/s10462-021-10083-3","type":"journal-article","created":{"date-parts":[[2021,11,5]],"date-time":"2021-11-05T03:02:31Z","timestamp":1636081351000},"page":"3431-3455","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["A study on the challenges and opportunities of speech recognition for Bengali language"],"prefix":"10.1007","volume":"55","author":[{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0001-5738-1631","authenticated-orcid":false,"given":"M. F.","family":"Mridha","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abu Quwsar","family":"Ohi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Md Abdul","family":"Hamid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muhammad Mostafa","family":"Monowar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,11,5]]},"reference":[{"key":"10083_CR1","doi-asserted-by":"crossref","unstructured":"Ahmed M, Shill PC, Islam K, Mollah MAS, Akhand M (2015) Acoustic modeling using deep belief network for bangla speech recognition. In: 2015 18th international conference on computer and information technology (ICCIT), pp 306\u2013311. IEEE","DOI":"10.1109\/ICCITechn.2015.7488087"},{"key":"10083_CR2","unstructured":"Ahmed S, Sadeq N, Shubha SS, Islam MN, Adnan MA, Islam MZ (2020) Preparation of bangla speech corpus from publicly available audio & text. In: Proceedings of The 12th language resources and evaluation conference, pp 6586\u20136592"},{"key":"10083_CR3","doi-asserted-by":"crossref","unstructured":"Al Amin MA, Islam MT, Kibria S, Rahman MS (2019) Continuous bengali speech recognition based on deep neural network. In: 2019 international conference on electrical, computer and communication engineering (ECCE), pp 1\u20136. IEEE","DOI":"10.1109\/ECACE.2019.8679341"},{"key":"10083_CR4","unstructured":"Alam F (2018) Development of annotated bangla speech corpora. https:\/\/2.zoppoz.workers.dev:443\/https\/data.mendeley.com\/datasets\/c79z6gz9rm\/1"},{"key":"10083_CR5","unstructured":"Alam F, Habib S, Sultana DA, Khan M (2010) Development of annotated bangla speech corpora"},{"key":"10083_CR6","first-page":"51","volume":"50","author":"MA Ali","year":"2013","unstructured":"Ali MA, Hossain M, Bhuiyan MN (2013) Automatic speech recognition technique for bangla words. Int J Adv Sci Technol 50:51\u201360","journal-title":"Int J Adv Sci Technol"},{"key":"10083_CR7","doi-asserted-by":"crossref","unstructured":"Arslan RS, Bari\u015e\u00c7I N (2020) A detailed survey of turkish automatic speech recognition. Turk J Electr Eng Comput Sci 28(6):3253\u20133269","DOI":"10.3906\/elk-2001-38"},{"key":"10083_CR8","doi-asserted-by":"crossref","unstructured":"Audhkhasi K, Kingsbury B, Ramabhadran B, Saon G, Picheny M (2018) Building competitive direct acoustics-to-word models for english conversational speech recognition. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4759\u20134763. IEEE","DOI":"10.1109\/ICASSP.2018.8461935"},{"issue":"1","key":"10083_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.5815\/ijigsp.2020.01.01","volume":"12","author":"SR Aura","year":"2020","unstructured":"Aura SR, Rahimi MJ, Baroi OL (2020) Analysis of the error pattern of hmm based bangla asr. Int J Image Graph Signal Process 12(1):1","journal-title":"Int J Image Graph Signal Process"},{"key":"10083_CR10","unstructured":"Baevski A, Zhou H, Mohamed A, Auli M (2020) wav2vec 2.0: a framework for self-supervised learning of speech representations. In Larochelle H, Ranzato M, Hadsell R, Balcan MF, Lin H (eds) Advances in neural information processing systems. Curran Associates, Inc., vol 134, pp 12449\u201312460"},{"key":"10083_CR11","first-page":"1","volume":"18","author":"S Balakrishnama","year":"1998","unstructured":"Balakrishnama S, Ganapathiraju A (1998) Linear discriminant analysis-a brief tutorial. Inst Signal Inf Process 18:1\u20138","journal-title":"Inst Signal Inf Process"},{"key":"10083_CR12","unstructured":"Bangladesh BE (1995) Bangla Academy Journal. Number v. 21, no. 2 - v. 22, no. 2. Bangla Academy. https:\/\/2.zoppoz.workers.dev:443\/https\/books.google.com.bd\/books?id=33xjAAAAMAAJ"},{"issue":"10\u201311","key":"10083_CR13","doi-asserted-by":"publisher","first-page":"763","DOI":"10.1016\/j.specom.2007.02.006","volume":"49","author":"M Benzeghiba","year":"2007","unstructured":"Benzeghiba M, De Mori R, Deroo O, Dupont S, Erbes T, Jouvet D, Fissore L, Laface P, Mertins A, Ris C et al (2007) Automatic speech recognition and speech variability: a review. Speech Commun 49(10\u201311):763\u2013786","journal-title":"Speech Commun"},{"key":"10083_CR14","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/j.specom.2013.07.008","volume":"56","author":"L Besacier","year":"2014","unstructured":"Besacier L, Barnard E, Karpov A, Schultz T (2014) Automatic speech recognition for under-resourced languages: a survey. Speech Commun 56:85\u2013100","journal-title":"Speech Commun"},{"key":"10083_CR15","doi-asserted-by":"crossref","unstructured":"Bhowmik T, Mandal SKD (2019) Prosodic word boundary detection from bengali continuous speech. Lang Resour Eval pp 1\u201319","DOI":"10.1007\/s10579-019-09478-0"},{"key":"10083_CR16","doi-asserted-by":"crossref","unstructured":"Bhowmik T, Choudhury A, Mandal SKD (2017) Deep neural network based recognition and classification of bengali phonemes: a case study of bengali unconstrained speech. In: International conference on next generation computing technologies, pp 750\u2013760. Springer","DOI":"10.1007\/978-981-10-8657-1_58"},{"key":"10083_CR17","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1016\/j.procs.2017.12.114","volume":"125","author":"T Bhowmik","year":"2018","unstructured":"Bhowmik T, Chowdhury A, Mandal SKD (2018) Deep neural network based place and manner of articulation detection and classification for bengali continuous speech. Procedia Comput Sci 125:895\u2013901","journal-title":"Procedia Comput Sci"},{"key":"10083_CR18","doi-asserted-by":"crossref","unstructured":"Bird JJ, Wanner E, Ek\u00e1rt A, Faria DR (2019) Phoneme aware speech recognition through evolutionary optimisation. In: Proceedings of the genetic and evolutionary computation conference companion, pp 362\u2013363","DOI":"10.1145\/3319619.3321951"},{"key":"10083_CR19","unstructured":"Bourlard HA, Morgan N (2012) Connectionist speech recognition: a hybrid approach, volume 247. Springer Science & Business Media"},{"key":"10083_CR20","unstructured":"Cakrabart\u012b U (1992) B\u0101ml\u0101 b\u0101kyera padagucchera samgathana. Pram\u0101 Prak\u0101\u015ban\u012b. https:\/\/2.zoppoz.workers.dev:443\/https\/books.google.com.bd\/books?id=kukbAAAAIAAJ"},{"key":"10083_CR21","unstructured":"Chatterji S (1988) https:\/\/2.zoppoz.workers.dev:443\/https\/books.google.com.bd\/books?id=NJgyAAAAIAAJ"},{"key":"10083_CR22","unstructured":"Chorowski JK, Bahdanau D, Serdyuk D, Cho K, Bengio Y (2015) Attention-based models for speech recognition. In: Advances in neural information processing systems, pp 577\u2013585"},{"key":"10083_CR23","doi-asserted-by":"crossref","unstructured":"Chowdhury MSA, Khan MF (2019) Linear predictor coefficient, power spectral analysis and two-layer feed forward network for bangla speech recognition. In: 2019 IEEE international conference on system, computation, automation and networking (ICSCAN), pp 1\u20136. IEEE","DOI":"10.1109\/ICSCAN.2019.8878709"},{"key":"10083_CR24","unstructured":"Chung J, Gulcehre C, Cho K, Bengio Y (2014) Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv:1412.3555"},{"key":"10083_CR25","doi-asserted-by":"crossref","unstructured":"Das B, Mandal S, Mitra P (2011) Bengali speech corpus for continuous auutomatic speech recognition system. In: 2011 international conference on speech database and assessments (Oriental COCOSDA), pp 51\u201355. IEEE","DOI":"10.1109\/ICSDA.2011.6085979"},{"key":"10083_CR26","unstructured":"Das B, Mandal S, Mitra P (2021) SHRUTI bengali continuous ASR speech corpus. https:\/\/2.zoppoz.workers.dev:443\/https\/cse.iitkgp.ac.in\/~pabitra\/shruti_corpus.html"},{"issue":"6","key":"10083_CR27","first-page":"1","volume":"1","author":"N Dave","year":"2013","unstructured":"Dave N (2013) Feature extraction methods lpc, plp and mfcc in speech recognition. Int J Adv Res Eng Technol 1(6):1\u20134","journal-title":"Int J Adv Res Eng Technol"},{"issue":"4","key":"10083_CR28","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N Dehak","year":"2010","unstructured":"Dehak N, Kenny PJ, Dehak R, Dumouchel P, Ouellet P (2010) Front-end factor analysis for speaker verification. IEEE Trans Audio Speech Lang Process 19(4):788\u2013798","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"10083_CR29","doi-asserted-by":"crossref","unstructured":"Dong L, Xu S, Xu B (2018) Speech-transformer: a no-recurrence sequence-to-sequence model for speech recognition. In: 2018 IEEE international conference on acoustics, speech and signal Processing (ICASSP), pp 5884\u20135888. IEEE","DOI":"10.1109\/ICASSP.2018.8462506"},{"issue":"3","key":"10083_CR30","first-page":"16","volume":"10","author":"SK Gaikwad","year":"2010","unstructured":"Gaikwad SK, Gawali BW, Yannawar P (2010) A review on speech recognition technique. Int J Comput Appl 10(3):16\u201324","journal-title":"Int J Comput Appl"},{"key":"10083_CR31","doi-asserted-by":"crossref","unstructured":"Gales M, Young S, et\u00a0al. (2008) The application of hidden markov models in speech recognition. Found Trends\u00ae Signal Process 1(3):195\u2013304","DOI":"10.1561\/2000000004"},{"issue":"2","key":"10083_CR32","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"MJ Gales","year":"1998","unstructured":"Gales MJ (1998) Maximum likelihood linear transformations for hmm-based speech recognition. Comput Speech Llang 12(2):75\u201398","journal-title":"Comput Speech Llang"},{"key":"10083_CR33","unstructured":"Gales MJ, Knill KM, Ragni A, Rath SP (2014) Speech recognition and keyword spotting for low-resource languages: Babel project research at cued. In: Fourth international workshop on spoken language technologies for under-resourced languages (SLTU-2014), pp 16\u201323. International Speech Communication Association (ISCA)"},{"key":"10083_CR34","unstructured":"Gales MJ, Knill KM, Ragni A, Rath SP (2021) IARPA babel bengali language pack. https:\/\/2.zoppoz.workers.dev:443\/https\/catalog.ldc.upenn.edu\/LDC2016S08"},{"key":"10083_CR35","unstructured":"Google. Large Bengali ASR training data set. https:\/\/2.zoppoz.workers.dev:443\/http\/www.openslr.org\/53\/"},{"key":"10083_CR36","unstructured":"Graves A, Jaitly N (2014) Towards end-to-end speech recognition with recurrent neural networks. In: International conference on machine learning, pp 1764\u20131772"},{"key":"10083_CR37","doi-asserted-by":"crossref","unstructured":"Graves A, Jaitly N, Mohamed A-R (2013) Hybrid speech recognition with deep bidirectional lstm. In: 2013 IEEE workshop on automatic speech recognition and understanding, pp 273\u2013278. IEEE","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"10083_CR38","doi-asserted-by":"crossref","unstructured":"Haeb-Umbach R, Ney H (1992) Linear discriminant analysis for improved large vocabulary continuous speech recognition. In: Proceedings of ICASSP, volume\u00a01, pp 13\u201316. USA: ICASSP","DOI":"10.1109\/ICASSP.1992.225984"},{"key":"10083_CR39","unstructured":"Hannun A, Case C, Casper J, Catanzaro B, Diamos G, Elsen E, Prenger R, Satheesh S, Sengupta S, Coates A, et\u00a0al. (2014) Deep speech: scaling up end-to-end speech recognition. arXiv:1412.5567"},{"key":"10083_CR40","doi-asserted-by":"crossref","unstructured":"Haque MA, Verma A, Alex JSR, Venkatesan N (2020) Experimental evaluation of cnn architecture for speech recognition. In: First international conference on sustainable technologies for computational intelligence, pp 507\u2013514. Springer","DOI":"10.1007\/978-981-15-0029-9_40"},{"key":"10083_CR41","doi-asserted-by":"crossref","unstructured":"Hassan F, Khan MSA, Kotwal MRA, Huda MN (2012) Gender independent bangla automatic speech recognition. In: 2012 international conference on informatics, electronics & vision (ICIEV), pp 144\u2013148. IEEE","DOI":"10.1109\/ICIEV.2012.6317500"},{"key":"10083_CR42","unstructured":"Hassan MR, Nath B, Bhuiyan MA (2003) Bengali phoneme recognition: a new approach. In: Proceedings of 6th international conference on computer and information technology (ICCIT03)"},{"key":"10083_CR43","doi-asserted-by":"crossref","unstructured":"Hermansky H, Fousek P (2005) Multi-resolution rasta filtering for tandem-based asr. Technical report, IDIAP","DOI":"10.21437\/Interspeech.2005-184"},{"issue":"4","key":"10083_CR44","doi-asserted-by":"publisher","first-page":"578","DOI":"10.1109\/89.326616","volume":"2","author":"H Hermansky","year":"1994","unstructured":"Hermansky H, Morgan N (1994) Rasta processing of speech. IEEE Trans Speech Audio Process 2(4):578\u2013589","journal-title":"IEEE Trans Speech Audio Process"},{"issue":"5","key":"10083_CR45","doi-asserted-by":"publisher","first-page":"5947","DOI":"10.4249\/scholarpedia.5947","volume":"4","author":"GE Hinton","year":"2009","unstructured":"Hinton GE (2009) Deep belief networks. Scholarpedia 4(5):5947","journal-title":"Scholarpedia"},{"issue":"8","key":"10083_CR46","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"10083_CR47","unstructured":"Hossain M, Rahman M, Prodhan UK, Khan M, et\u00a0al. (2013) Implementation of back-propagation neural network for isolated bangla speech recognition. arXiv:1308.3785"},{"key":"10083_CR48","unstructured":"Houque A (2006) Bengali segmented speech recognition system. Undergraduate thesis, BRAC University, Bangladesh"},{"key":"10083_CR49","doi-asserted-by":"crossref","unstructured":"Hsu J-Y, Chen Y-J, Lee H-Y (2020) Meta learning for end-to-end low-resource speech recognition. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7844\u20137848. IEEE","DOI":"10.1109\/ICASSP40776.2020.9053112"},{"key":"10083_CR50","doi-asserted-by":"crossref","unstructured":"Irie K, T\u00fcske Z, Alkhouli T, Schl\u00fcter R, Ney H (2016) Lstm, gru, highway and a bit of attention: an empirical overview for language modeling in speech recognition. In: Interspeech, pp 3519\u20133523","DOI":"10.21437\/Interspeech.2016-491"},{"key":"10083_CR51","unstructured":"Islam MR, Sohail ASM, Sadid MWH, Mottalib M (2005) Bangla speech recognition using three layer back-propagation neural network. In: Proceedings of the national conference on computer processing of Bangla (NCCPB), Dhaka"},{"key":"10083_CR52","unstructured":"Ittichaichareon C, Suksri S, Yingthawornsuk T (2012) Speech recognition using mfcc. In: International conference on computer graphics, simulation and modeling (ICGSM\u20192012), pp 28\u201329"},{"key":"10083_CR53","unstructured":"Karim R, Rahman MS, Iqbal MZ (2002) Recognition of spoken letters in bangla. In: Proceedings of 5th international conference on computer and information technology (ICCIT02)"},{"key":"10083_CR54","unstructured":"Khan MF, Debnath DRC (2002) Comparative study of feature extraction methods for bangla phoneme recognition. In: 5th ICCIT, pp 27\u201328"},{"key":"10083_CR55","doi-asserted-by":"crossref","unstructured":"Khan S, Pal M, Basu J, Bepari MS, Roy R (2018) Assessing performance of bengali speech recognizers under real world conditions using gmm-hmm and dnn based methods. In: SLTU, pp 192\u2013196","DOI":"10.21437\/SLTU.2018-40"},{"key":"10083_CR56","doi-asserted-by":"crossref","unstructured":"Kotwal MRA, Banik M, Eity QN, Huda MN, Muhammad G, Alotaibi YA (2010) Bangla phoneme recognition for asr using multilayer neural network. In: 2010 13th international conference on computer and information technology (ICCIT), pp 103\u2013107. IEEE","DOI":"10.1109\/ICCITECHN.2010.5723837"},{"issue":"6","key":"10083_CR57","doi-asserted-by":"publisher","first-page":"1005","DOI":"10.1016\/j.sigpro.2004.03.004","volume":"84","author":"O-W Kwon","year":"2004","unstructured":"Kwon O-W, Lee T-W (2004) Phoneme recognition using ica-based feature extraction and transformation. Signal Process 84(6):1005\u20131019","journal-title":"Signal Process"},{"issue":"1","key":"10083_CR58","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1109\/29.45616","volume":"38","author":"K-F Lee","year":"1990","unstructured":"Lee K-F, Hon H-W, Reddy R (1990) An overview of the sphinx speech recognition system. IEEE Trans Acoust Speech Signal Process 38(1):35\u201345","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"key":"10083_CR59","doi-asserted-by":"crossref","unstructured":"Mandal S, Das B, Mitra P (2010) Shruti-ii: a vernacular speech recognition system in bengali and an application for visually impaired community. In: 2010 IEEE students technology symposium (TechSym), pp 229\u2013233. IEEE","DOI":"10.1109\/TECHSYM.2010.5469156"},{"key":"10083_CR60","doi-asserted-by":"crossref","unstructured":"Mandal S, Das B, Mitra P, Basu A (2011) Developing bengali speech corpus for phone recognizer using optimum text selection technique. In: 2011 international conference on asian language processing, pp 268\u2013271. IEEE","DOI":"10.1109\/IALP.2011.16"},{"issue":"7\u20138","key":"10083_CR61","doi-asserted-by":"publisher","first-page":"953","DOI":"10.1080\/01690965.2012.705006","volume":"27","author":"SL Mattys","year":"2012","unstructured":"Mattys SL, Davis MH, Bradlow AR, Scott SK (2012) Speech recognition in adverse conditions: a review. Lang Cognit Process 27(7\u20138):953\u2013978","journal-title":"Lang Cognit Process"},{"key":"10083_CR62","doi-asserted-by":"crossref","unstructured":"Molla K, Hirose K (2004) On the effectiveness of mfccs and their statistical distribution properties in speaker identification. In: 2004 IEEE symposium on virtual environments, human-computer interfaces and measurement systems, 2004.(VCIMS)., pp 136\u2013141. IEEE","DOI":"10.1109\/VECIMS.2004.1397204"},{"key":"10083_CR66","unstructured":"Nahid MMH (2018) Bengali speech recognition\u2014bangla real number audio dataset. https:\/\/2.zoppoz.workers.dev:443\/https\/data.mendeley.com\/datasets\/t33byr6cpt\/6"},{"key":"10083_CR63","doi-asserted-by":"crossref","unstructured":"Nahid MMH, Islam MA, Islam MS (2016) A noble approach for recognizing bangla real number automatically using cmu sphinx4. In: 2016 5th international conference on informatics, electronics and vision (ICIEV), pp 844\u2013849. IEEE","DOI":"10.1109\/ICIEV.2016.7760121"},{"key":"10083_CR64","doi-asserted-by":"crossref","unstructured":"Nahid MMH, Purkaystha B, Islam MS (2017) Bengali speech recognition: a double layered lstm-rnn approach. In: 2017 20th international conference of computer and information technology (ICCIT), pp 1\u20136. IEEE","DOI":"10.1109\/ICCITECHN.2017.8281848"},{"key":"10083_CR65","unstructured":"Nahid MMH, Islam MA, Purkaystha B, Islam MS (2018) Comprehending real numbers: development of bengali real number speech corpus"},{"key":"10083_CR67","doi-asserted-by":"crossref","unstructured":"Nivetha S (2020) A survey on speech feature extraction and classification techniques. In: 2020 international conference on inventive computation technologies (ICICT), pp 48\u201353. IEEE","DOI":"10.1109\/ICICT48043.2020.9112582"},{"key":"10083_CR68","doi-asserted-by":"publisher","first-page":"89619","DOI":"10.1109\/ACCESS.2021.3090109","volume":"9","author":"AQ Ohi","year":"2021","unstructured":"Ohi AQ, Mridha M, Hamid MA, Monowar MM (2021) Deep speaker recognition: process, progress, and challenges. IEEE Access 9:89619\u201389643","journal-title":"IEEE Access"},{"key":"10083_CR69","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: an asr corpus based on public domain audio books. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5206\u20135210. IEEE","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"10083_CR70","unstructured":"Parishad BV (2001) Indian Journal of Linguistics. Number v. 20. Bhasa Vidya Parishad. https:\/\/2.zoppoz.workers.dev:443\/https\/books.google.co.uk\/books?id=0yxhAAAAMAAJ"},{"key":"10083_CR71","doi-asserted-by":"crossref","unstructured":"Paul AK, Das D, Kamal MM (2009) Bangla speech recognition system using lpc and ann. In: 2009 seventh international conference on advances in pattern recognition, pp 171\u2013174. IEEE","DOI":"10.1109\/ICAPR.2009.80"},{"key":"10083_CR72","unstructured":"Placeway P, Chen S, Eskenazi M, Jain U, Parikh V, Raj B, Ravishankar M, Rosenfeld R, Seymore K, Siegler M, et\u00a0al. (1997) The 1996 hub-4 sphinx-3 system. In: Proceedings of DARPA speech recognition workshop, volume\u00a097. Citeseer"},{"key":"10083_CR73","unstructured":"Povey D, Ghoshal A, Boulianne G, Burget L, Glembek O, Goel N, Hannemann M, Motlicek P, Qian Y, Schwarz P, et\u00a0al. (2011) The kaldi speech recognition toolkit. In: IEEE 2011 workshop on automatic speech recognition and understanding, Number CONF. IEEE Signal Processing Society"},{"key":"10083_CR74","doi-asserted-by":"crossref","unstructured":"Rabiner LR (1990) Selected applications in speech recognition. Read Speech Recognit p 267","DOI":"10.1016\/B978-0-08-051584-7.50027-9"},{"key":"10083_CR75","unstructured":"Rahman K, Hossain M, Das D, Islam T, Ali M (2003) Continuous bangla speech recognition system. In: Proceedings of 6th international conference on computer and information technology (ICCIT03), pp 303\u2013307"},{"issue":"1","key":"10083_CR76","doi-asserted-by":"publisher","first-page":"67","DOI":"10.3329\/diujst.v5i1.4384","volume":"5","author":"MM Rahman","year":"2010","unstructured":"Rahman MM, Khan MF, Moni MA (2010) Speech recognition front-end for segmenting and clustering continuous bangla speech. Daffodil Int Univ J Sci Technol 5(1):67\u201372","journal-title":"Daffodil Int Univ J Sci Technol"},{"issue":"4","key":"10083_CR77","first-page":"5258","volume":"5","author":"C Rashmi","year":"2014","unstructured":"Rashmi C (2014) Review of algorithms and applications in speech recognition system. Int J Comput Sci Inf Technol 5(4):5258\u20135262","journal-title":"Int J Comput Sci Inf Technol"},{"issue":"2","key":"10083_CR78","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1109\/TETCI.2017.2762739","volume":"2","author":"M Ravanelli","year":"2018","unstructured":"Ravanelli M, Brakel P, Omologo M, Bengio Y (2018) Light gated recurrent units for speech recognition. IEEE Trans Emerg Topics Comput Intell 2(2):92\u2013102","journal-title":"IEEE Trans Emerg Topics Comput Intell"},{"issue":"4","key":"10083_CR79","doi-asserted-by":"publisher","first-page":"501","DOI":"10.1109\/PROC.1976.10158","volume":"64","author":"DR Reddy","year":"1976","unstructured":"Reddy DR (1976) Speech recognition by machine: a review. Proc IEEE 64(4):501\u2013531","journal-title":"Proc IEEE"},{"key":"10083_CR80","doi-asserted-by":"crossref","unstructured":"Reza M, Rashid W, Mostakim M (2017) Prodorshok i: a bengali isolated speech dataset for voice-based assistive technologies: a comparative analysis of the effects of data augmentation on hmm-gmm and dnn classifiers. In: 2017 IEEE region 10 humanitarian technology conference (R10-HTC), pp 396\u2013399. IEEE","DOI":"10.1109\/R10-HTC.2017.8288983"},{"key":"10083_CR81","doi-asserted-by":"crossref","unstructured":"Sahu P, Dua M, Kumar A (2018) Challenges and issues in adopting speech recognition. In: Speech and language processing for human-machine communications, pp 209\u2013215. Springer","DOI":"10.1007\/978-981-10-6626-9_23"},{"key":"10083_CR82","doi-asserted-by":"crossref","unstructured":"Saurav JR, Amin S, Kibria S, Rahman MS (2018) Bangla speech recognition for voice search. In: 2018 international conference on bangla speech and language processing (ICBSLP), pp 1\u20134. IEEE","DOI":"10.1109\/ICBSLP.2018.8554574"},{"key":"10083_CR83","doi-asserted-by":"publisher","first-page":"1381","DOI":"10.1016\/j.procs.2020.04.148","volume":"171","author":"R Sharmin","year":"2020","unstructured":"Sharmin R, Rahut SK, Huq MR (2020) Bengali spoken digit classification: a deep learning approach using convolutional neural network. Procedia Comput Sci 171:1381\u20131388","journal-title":"Procedia Comput Sci"},{"key":"10083_CR84","doi-asserted-by":"crossref","unstructured":"Singh A, Kadyan V, Kumar M, Bassan N (2019) Asroil: a comprehensive survey for automatic speech recognition of indian languages. Artif Intell Rev, pp 1\u201332","DOI":"10.1007\/s10462-019-09775-8"},{"key":"10083_CR85","unstructured":"Srivastava N, Mukhopadhyay R, Prajwal K, Jawahar C (2020) Indicspeech: text-to-speech corpus for indian languages. In: Proceedings of the 12th language resources and evaluation conference, pp 6417\u20136422"},{"key":"10083_CR86","unstructured":"Srivastava N, Mukhopadhyay R, Prajwal K, Jawahar C (2021) IndicSpeech: text-to-Speech Corpus for Indian Languages. https:\/\/2.zoppoz.workers.dev:443\/http\/cvit.iiit.ac.in\/research\/projects\/cvit-projects\/text-to-speech-dataset-for-indian-languages"},{"key":"10083_CR87","doi-asserted-by":"crossref","unstructured":"Sultana S, Akhand M, Das PK, Rahman MH (2012) Bangla speech-to-text conversion using sapi. In: 2012 international conference on computer and communication engineering (ICCCE), pp 385\u2013390. IEEE","DOI":"10.1109\/ICCCE.2012.6271216"},{"key":"10083_CR88","doi-asserted-by":"crossref","unstructured":"Sumit SH, Al\u00a0Muntasir T, Zaman MA, Nandi RN, Sourov T (2018) Noise robust end-to-end speech recognition for bangla language. In: 2018 international conference on bangla speech and language processing (ICBSLP), pp 1\u20135. IEEE","DOI":"10.1109\/ICBSLP.2018.8554871"},{"key":"10083_CR89","doi-asserted-by":"crossref","unstructured":"Sumon SA, Chowdhury J, Debnath S, Mohammed N, Momen S (2018) Bangla short speech commands recognition using convolutional neural networks. In: 2018 international conference on bangla speech and language processing (ICBSLP), pp 1\u20136. IEEE","DOI":"10.1109\/ICBSLP.2018.8554395"},{"key":"10083_CR90","unstructured":"SUST SUoST (2020) Pipilika: (Bengali Search Engine). Accessed April 1, 2020. https:\/\/2.zoppoz.workers.dev:443\/https\/www.pipilika.com\/"},{"key":"10083_CR91","doi-asserted-by":"crossref","unstructured":"Takiguchi T, Ariki Y (2007) PCA-based speech enhancement for distorted speech recognition. J Multimed 2(5)","DOI":"10.4304\/jmm.2.5.13-18"},{"key":"10083_CR92","unstructured":"Tebelskis J (1995) Speech recognition using neural networks. PhD thesis, Carnegie Mellon University"},{"issue":"1\u20134","key":"10083_CR93","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1016\/S0925-2312(00)00308-8","volume":"37","author":"E Trentin","year":"2001","unstructured":"Trentin E, Gori M (2001) A survey of hybrid ann\/hmm models for automatic speech recognition. Neurocomputing 37(1\u20134):91\u2013126","journal-title":"Neurocomputing"},{"key":"10083_CR94","unstructured":"Tunga, \u015aekhara S (1995) Bengali and other related dialects of south Assam. Mittal Publications, 1 edition"},{"issue":"1","key":"10083_CR95","first-page":"31","volume":"175","author":"AY Vadwala","year":"2017","unstructured":"Vadwala AY, Suthar KA, Karmakar YA, Pandya N, Patel B (2017) Survey paper on different speech recognition algorithm: challenges and techniques. Int J Comput Appl 175(1):31\u201336","journal-title":"Int J Comput Appl"},{"key":"10083_CR96","doi-asserted-by":"crossref","unstructured":"Variani E, Lei X, McDermott E, Moreno IL, Gonzalez-Dominguez J (2014) Deep neural networks for small footprint text-dependent speaker verification. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4052\u20134056. IEEE","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"10083_CR97","unstructured":"Walker W, Lamere P, Kwok P, Raj B, Singh R, Gouvea E, Wolf P, Woelfel J (2004) Sphinx-4: a flexible open source framework for speech recognition"},{"key":"10083_CR98","doi-asserted-by":"crossref","unstructured":"Westphal M (1997) The use of cepstral means in conversational speech recognition. In: Fifth European conference on speech communication and technology","DOI":"10.21437\/Eurospeech.1997-120"},{"issue":"6","key":"10083_CR99","doi-asserted-by":"publisher","first-page":"582","DOI":"10.1007\/BF02943243","volume":"16","author":"F Zheng","year":"2001","unstructured":"Zheng F, Zhang G, Song Z (2001) Comparison of different implementations of mfcc. J Comput Sci Technol 16(6):582\u2013589","journal-title":"J Comput Sci Technol"},{"key":"10083_CR100","doi-asserted-by":"crossref","unstructured":"Zinnat SB, Siddique RMA, Hossain MI, Abdullah DM, Huda MN (2014) Automatic word recognition for bangla spoken language. In: 2014 international conference on signal propagation and computer technology (ICSPCT 2014), pp 470\u2013475. IEEE","DOI":"10.1109\/ICSPCT.2014.6884886"},{"key":"10083_CR101","unstructured":"Zi\u00f3\u0142ko M, Samborski R, Ga\u0142ka J, Zi\u00f3\u0142ko B (2011) Wavelet-Fourier analysis for speaker recognition. In: 17th national conference on applications of mathematics in biology and medicine, vol 134, p 129"},{"key":"10083_CR102","doi-asserted-by":"publisher","first-page":"112840","DOI":"10.1016\/j.eswa.2019.112840","volume":"139","author":"T Zoughi","year":"2020","unstructured":"Zoughi T, Homayounpour MM, Deypir M (2020) Adaptive windows multiple deep residual networks for speech recognition. Expert Syst Appl 139:112840","journal-title":"Expert Syst Appl"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s10462-021-10083-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/article\/10.1007\/s10462-021-10083-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s10462-021-10083-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T14:50:24Z","timestamp":1726066224000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/10.1007\/s10462-021-10083-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,5]]},"references-count":102,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,4]]}},"alternative-id":["10083"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/s10462-021-10083-3","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,11,5]]},"assertion":[{"value":"5 November 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}