{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T16:41:05Z","timestamp":1774456865069,"version":"3.50.1"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1016\/j.jvcir.2024.104309","type":"journal-article","created":{"date-parts":[[2024,10,9]],"date-time":"2024-10-09T17:23:28Z","timestamp":1728494608000},"page":"104309","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":3,"special_numbering":"C","title":["Optimized deep learning enabled lecture audio video summarization"],"prefix":"10.1016","volume":"104","author":[{"given":"Preet","family":"Chandan Kaur","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dr. Leena","family":"Ragha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2024.104309_b0005","doi-asserted-by":"crossref","first-page":"104469","DOI":"10.1109\/ACCESS.2021.3099427","article-title":"FCN-LectureNet: Extractive summarization of whiteboard and chalkboard lecture videos","volume":"9","author":"Davila","year":"2021","journal-title":"IEEE Access"},{"issue":"3","key":"10.1016\/j.jvcir.2024.104309_b0010","doi-asserted-by":"crossref","first-page":"221","DOI":"10.1007\/s10032-019-00327-y","article-title":"Generalized framework for summarization of fixed-camera lecture videos by detecting and binarizing handwritten content","volume":"22","author":"Urala Kota","year":"2019","journal-title":"Int. J. Document Anal. Recognition (IJDAR)"},{"key":"10.1016\/j.jvcir.2024.104309_b0015","doi-asserted-by":"crossref","first-page":"64676","DOI":"10.1109\/ACCESS.2019.2916989","article-title":"Spatiotemporal modelling for video summarization using convolutional recurrent neural network","volume":"7","author":"Yuan","year":"2019","journal-title":"IEEE Access"},{"issue":"1","key":"10.1016\/j.jvcir.2024.104309_b0020","doi-asserted-by":"crossref","first-page":"857","DOI":"10.1007\/s11042-016-4300-7","article-title":"VISCOM: A robust video summarization approach using colour co-occurrence matrices","volume":"77","author":"Mussel Cirne","year":"2018","journal-title":"Multimed. Tools Appl."},{"issue":"3","key":"10.1016\/j.jvcir.2024.104309_b0025","doi-asserted-by":"crossref","first-page":"507","DOI":"10.1007\/s11760-018-1376-8","article-title":"Key frame extraction for video summarization using local description and repeatability graph clustering","volume":"13","author":"Gharbi","year":"2019","journal-title":"SIViP"},{"issue":"9","key":"10.1016\/j.jvcir.2024.104309_b0030","doi-asserted-by":"crossref","first-page":"14459","DOI":"10.1007\/s11042-020-10460-0","article-title":"GVSUM: Generic video summarization using deep visual features","volume":"80","author":"Basavarajaiah","year":"2021","journal-title":"Multimed. Tools Appl."},{"issue":"6","key":"10.1016\/j.jvcir.2024.104309_b0035","doi-asserted-by":"crossref","first-page":"3940","DOI":"10.1002\/ett.3940","article-title":"Network video summarization based on key frame extraction via superpixel segmentation","volume":"33","author":"Jin","year":"2022","journal-title":"Trans. Emerg. Telecommun. Technol."},{"issue":"6","key":"10.1016\/j.jvcir.2024.104309_b0040","doi-asserted-by":"crossref","first-page":"1702","DOI":"10.3390\/s20061702","article-title":"Scene classification for sports video summarization using transfer learning","volume":"20","author":"Rafiq","year":"2020","journal-title":"Sensors"},{"issue":"11","key":"10.1016\/j.jvcir.2024.104309_b0045","doi-asserted-by":"crossref","first-page":"8","DOI":"10.9734\/jerr\/2021\/v20i1117399","article-title":"Hybrid method of video shot segmentation based on YCbCr space color model","volume":"20","author":"Nayak","year":"2021","journal-title":"J. Eng. Res. Rep."},{"issue":"3","key":"10.1016\/j.jvcir.2024.104309_b0050","doi-asserted-by":"crossref","first-page":"2237","DOI":"10.1007\/s10462-019-09732-5","article-title":"Novel meta-heuristic bald eagle search optimisation algorithm","volume":"53","author":"Alsattar","year":"2020","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.jvcir.2024.104309_b0055","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1016\/j.matcom.2021.08.013","article-title":"Honey Badger Algorithm: New metaheuristic algorithm for solving optimization problems","volume":"192","author":"Hashim","year":"2022","journal-title":"Math. Comput. Simul"},{"key":"10.1016\/j.jvcir.2024.104309_b0060","doi-asserted-by":"crossref","unstructured":"Kumar, C., Rehman, F., Kumar, S. and Mehmood, A. and Shabir, G.,\u201cAnalysis of MFCC and BFCC in a Speaker Identification System\u201d, In proceedings of International Conference on Computing, Mathematics and Engineering Technologies, 2018.","DOI":"10.1109\/ICOMET.2018.8346330"},{"key":"10.1016\/j.jvcir.2024.104309_b0065","doi-asserted-by":"crossref","unstructured":"Hassan, A.R. and Haque, M.A., \u201cComputer-aided sleep apnea diagnosis from single-lead electrocardiogram using dual-tree complex wavelet transform and spectral features,\u201d In Proceedings of International Conference on Electrical & Electronic Engineering (ICEEE), pp. 49-52, 2015.","DOI":"10.1109\/CEEE.2015.7428289"},{"key":"10.1016\/j.jvcir.2024.104309_b0070","doi-asserted-by":"crossref","DOI":"10.1016\/j.enconman.2019.111793","article-title":"Deep residual network based fault detection and diagnosis of photovoltaic arrays using current-voltage curves and ambient conditions","volume":"198","author":"Chen","year":"2019","journal-title":"Energ. Conver. Manage."},{"key":"10.1016\/j.jvcir.2024.104309_b0075","unstructured":"Liu, T. and Kender, J.R, \u201cLecture videos for e-learning: Current research and challenges,\u201d In IEEE Sixth International Symposium on Multimedia Software Engineering, pp. 574-578, 2004."},{"issue":"3","key":"10.1016\/j.jvcir.2024.104309_b0080","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1109\/TLT.2008.22","article-title":"Browsing within lecture videos based on the chain index of speech transcription","volume":"1","author":"Repp","year":"2008","journal-title":"IEEE Trans. Learn. Technol."},{"key":"10.1016\/j.jvcir.2024.104309_b0085","doi-asserted-by":"crossref","unstructured":"Ngo, C.W., Wang, F. and Pong, T.C, \u201cStructuring lecture videos for distance learning applications,\u201d In Fifth International Symposium on Multimedia Software Engineering, pp. 215-222, 2003.","DOI":"10.1109\/MMSE.2003.1254444"},{"key":"10.1016\/j.jvcir.2024.104309_b0090","doi-asserted-by":"crossref","unstructured":"Davila, K. and Zanibbi, R., \u201cWhiteboard video summarization via spatio-temporal conflict minimization,\u201d In 14th IAPR International conference on document analysis and recognition (ICDAR), vol.1, pp. 355-362, 2017.","DOI":"10.1109\/ICDAR.2017.66"},{"issue":"5","key":"10.1016\/j.jvcir.2024.104309_b0095","doi-asserted-by":"crossref","first-page":"7067","DOI":"10.1007\/s11042-016-3353-y","article-title":"Robust handwriting extraction and lecture video summarization","volume":"76","author":"Lee","year":"2017","journal-title":"Multimed. Tools Appl."},{"key":"10.1016\/j.jvcir.2024.104309_b0100","doi-asserted-by":"crossref","DOI":"10.1155\/2022\/7453744","article-title":"An effective video summarization framework based on the object of interest using deep learning","author":"UlHaq","year":"2022","journal-title":"Math. Probl. Eng."},{"issue":"13","key":"10.1016\/j.jvcir.2024.104309_b0105","first-page":"30","article-title":"A survey on video summarization techniques","volume":"132","author":"Sebastian","year":"2015","journal-title":"Int. J. Comput. Appl"},{"key":"10.1016\/j.jvcir.2024.104309_b0110","doi-asserted-by":"crossref","unstructured":"Otani, M., Nakashima, Y., Rahtu, E., Heikkil\u00e4, J. and Yokoya, N., \u201cVideo summarization using deep semantic features,\u201d In Asian conference on computer vision, pp. 361-377, 2016.","DOI":"10.1007\/978-3-319-54193-8_23"},{"issue":"5","key":"10.1016\/j.jvcir.2024.104309_b0115","doi-asserted-by":"crossref","DOI":"10.1109\/LSP.2018.2817176","article-title":"LOOP descriptor: Local optimal-oriented pattern","volume":"25","author":"Chakraborti","year":"2018","journal-title":"IEEE Signal Process Lett."},{"key":"10.1016\/j.jvcir.2024.104309_b0120","unstructured":"Li, Su, and Yang, Y., \u201cPower-Scaled Spectral Flux and Peak-Valley Group-Delay Methods for RobustMusical Onset Detection\u201d, In ICMC,2014."},{"key":"10.1016\/j.jvcir.2024.104309_b0125","doi-asserted-by":"crossref","unstructured":"Lakshmi Prabha, N. S. and Majumder, S. \u201cFace Recognition System Invariant to Plastic Surgery\u201d, In Proceedings of 12th International Conference on Intelligent Systems Design and Applications (ISDA), IEEE, pp. 258-263, November 2012.","DOI":"10.1109\/ISDA.2012.6416547"},{"issue":"6","key":"10.1016\/j.jvcir.2024.104309_b0130","doi-asserted-by":"crossref","first-page":"1635","DOI":"10.1109\/TIP.2010.2042645","article-title":"Enhanced local texture feature sets for FaceRecognition under difficult lighting conditions","volume":"19","author":"Tan","year":"2010","journal-title":"IEEE Trans. Image Process."},{"issue":"4","key":"10.1016\/j.jvcir.2024.104309_b0135","doi-asserted-by":"crossref","first-page":"1051","DOI":"10.1109\/TBME.2014.2360154","article-title":"Small blob identification in medical images using regional features from optimum scale","volume":"62","author":"Zhang","year":"2014","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"10.1016\/j.jvcir.2024.104309_b0140","doi-asserted-by":"crossref","DOI":"10.1016\/j.apacoust.2019.107020","article-title":"Trends in audio signal feature extraction methods","volume":"158","author":"Sharma","year":"2020","journal-title":"Appl. Acoust."},{"issue":"7","key":"10.1016\/j.jvcir.2024.104309_b0145","doi-asserted-by":"crossref","first-page":"2877","DOI":"10.1109\/TIP.2014.2321495","article-title":"A novel local pattern descriptor\u2014local vector pattern in high-order derivative space for face recognition","volume":"23","author":"Fan","year":"2014","journal-title":"IEEE Trans. Image Process."},{"issue":"8","key":"10.1016\/j.jvcir.2024.104309_b0150","doi-asserted-by":"crossref","first-page":"5181","DOI":"10.1109\/TNNLS.2021.3119969","article-title":"AudioVisual video summarization","volume":"34","author":"Zhao","year":"2021","journal-title":"IEEE Trans. Neural Networks Learn. Syst."},{"issue":"5","key":"10.1016\/j.jvcir.2024.104309_b0155","doi-asserted-by":"crossref","first-page":"996","DOI":"10.1109\/TKDE.2018.2848260","article-title":"Read, watch, listen, and summarize: Multi-modal summarization for asynchronous text, image, audio and video","volume":"31","author":"Li","year":"2019","journal-title":"IEEE Trans. Knowl. Data Eng."},{"issue":"6","key":"10.1016\/j.jvcir.2024.104309_b0160","doi-asserted-by":"crossref","first-page":"7239","DOI":"10.1109\/TPAMI.2022.3223688","article-title":"Contrastive positive sample propagation along the audio-visual event line","volume":"45","author":"Zhou","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2024.104309_b0165","unstructured":"Zhou, J. et al., \u201cAudio\u2013Visual Segmentation,\u201d In Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds) Computer Vision \u2013 ECCV 2022. ECCV 2022. Lecture Notes in Computer Science, vol 13697, 2022."},{"key":"10.1016\/j.jvcir.2024.104309_b0170","unstructured":"Zhou, J., Guo, D., Zhong, Y., and Wang, M., Improving Audio-Visual Video Parsing with Pseudo Visual Labels, ArXiv, 2023."},{"key":"10.1016\/j.jvcir.2024.104309_b0175","doi-asserted-by":"crossref","unstructured":"Shen, Xuyang, Li, D., Zhou, J., Qin, Z., He, B., Han, X., Li, A., et al. \u201cFine-grained audible video description.\u201d In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10585-10596, 2023.","DOI":"10.1109\/CVPR52729.2023.01020"},{"key":"10.1016\/j.jvcir.2024.104309_b0180","unstructured":"Tian, Yapeng, Guan, C., Goodman, J., Moore, M. and Xu, C., \u201cAudio-visual interpretable and controllable video captioning,\u201d In IEEE Computer Society Conference on Computer Vision and Pattern Recognition workshops, 2019."},{"issue":"4","key":"10.1016\/j.jvcir.2024.104309_b0185","first-page":"3306","article-title":"Object-aware adaptive-positivity learning for audio-visual question answering","volume":"38","author":"Li","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.jvcir.2024.104309_b0190","doi-asserted-by":"crossref","unstructured":"Hershey, S., Chaudhuri, S., Ellis, D.P.W., Gemmeke, J.F., Jansen, A., Moore, R.C., Plakal, M., Platt, D., Saurous, R.A., Seybold, B., Slaney, M., Weiss, R.J. and Wilson, K. \u201cCNN architectures for large-scale audio classification,\u201d In2017 IEEE international conference on acoustics, speech and signal processing (icassp), pp. 131-135, 2017.","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"10.1016\/j.jvcir.2024.104309_b0195","doi-asserted-by":"crossref","unstructured":"Carreira, J. and Zisserman, A. \u201cQuo Vadis, action recognition? A new model and the kinetics dataset,\u201d InProceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299-6308, 2017.","DOI":"10.1109\/CVPR.2017.502"},{"key":"10.1016\/j.jvcir.2024.104309_b0200","article-title":"Attention-guided multi-granularity fusion model for video summarization","volume":"249","author":"Yunzuo","year":"2024","journal-title":"Expert Syst. Appl."},{"issue":"4406410","key":"10.1016\/j.jvcir.2024.104309_b0205","first-page":"1","article-title":"SFSANet: Multiscale object detection in remote sensing image based on semantic fusion and scale adaptability","volume":"62","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.jvcir.2024.104309_b0210","first-page":"1","article-title":"CFANet: Efficient detection of UAV image based on cross-layer feature aggregation","volume":"61","author":"Yunzuo","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.jvcir.2024.104309_b0215","doi-asserted-by":"crossref","first-page":"4183","DOI":"10.1109\/TMM.2023.3321394","article-title":"Multi-scale spatiotemporal feature fusion network for video saliency prediction","volume":"26","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Multimedia"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S1047320324002657?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S1047320324002657?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T19:39:27Z","timestamp":1730662767000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320324002657"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10]]},"references-count":43,"alternative-id":["S1047320324002657"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.jvcir.2024.104309","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2024,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Optimized deep learning enabled lecture audio video summarization","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.jvcir.2024.104309","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2024 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104309"}}