{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:40:54Z","timestamp":1784302854290,"version":"3.55.0"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1007\/s10489-024-05433-x","type":"journal-article","created":{"date-parts":[[2024,5,6]],"date-time":"2024-05-06T07:02:49Z","timestamp":1714978969000},"page":"6108-6124","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["DePAint: a decentralized safe multi-agent reinforcement learning algorithm considering peak and average constraints"],"prefix":"10.1007","volume":"54","author":[{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0009-0003-2074-1113","authenticated-orcid":false,"given":"Raheeb","family":"Hassan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0009-0009-2762-9734","authenticated-orcid":false,"given":"K.M. Shadman","family":"Wadith","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Md. Mamun or","family":"Rashid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Md. Mosaddek","family":"Khan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,5,6]]},"reference":[{"key":"5433_CR1","unstructured":"Amodei D, Olah C, Steinhardt J, Christiano P, Schulman J, Man\u00e9 D (2016) Concrete problems in ai safety. arXiv:1606.06565"},{"key":"5433_CR2","unstructured":"Shalev-Shwartz S, Shammah S, Shashua A (2016) Safe, multi-agent, reinforcement learning for autonomous driving. arXiv:1610.03295"},{"key":"5433_CR3","doi-asserted-by":"publisher","first-page":"108180","DOI":"10.1016\/j.cie.2022.108180","volume":"169","author":"M Alqahtani","year":"2022","unstructured":"Alqahtani M, Scott MJ, Hu M (2022) Dynamic energy scheduling and routing of a large fleet of electric vehicles using multi-agent reinforcement learning. Comput Ind Eng 169:108180","journal-title":"Comput Ind Eng"},{"key":"5433_CR4","unstructured":"Altman E (1995) Constrained markov decision processes. PhD thesis, INRIA"},{"key":"5433_CR5","unstructured":"Achiam J, Held D, Tamar A, Abbeel P (2017) Constrained policy optimization. In: International conference on machine learning, pp 22\u201331. PMLR"},{"key":"5433_CR6","unstructured":"Gu S, Kuba JG, Wen M, Chen R, Wang Z, Tian Z, Wang J, Knoll A, Yang Y (2021) Multi-agent constrained policy optimisation. arXiv:2110.02793"},{"key":"5433_CR7","doi-asserted-by":"crossref","unstructured":"Gronauer S, Diepold K (2021) Multi-agent deep reinforcement learning: a survey. Artificial Intelligence Review, pp\u00a01\u201349","DOI":"10.1007\/s10462-021-09996-w"},{"key":"5433_CR8","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Pieter Abbeel O, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. Advances in Neural Information Processing Systems 30"},{"key":"5433_CR9","unstructured":"Parnika P, Diddigi RB, Danda SKR, Bhatnagar S (2021) Attention Actor-Critic algorithm for Multi-Agent Constrained Co-operative Reinforcement Learning"},{"key":"5433_CR10","doi-asserted-by":"crossref","unstructured":"Lu S, Zhang K, Chen T, Basar T, Horesh L (2021) Decentralized policy gradient descent ascent for safe multi-agent reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence vol\u00a035, pp\u00a08767\u20138775","DOI":"10.1609\/aaai.v35i10.17062"},{"key":"5433_CR11","unstructured":"Bai Q, Aggarwal V, Gattami A (2020) Provably efficient model-free algorithm for mdps with peak constraints. arXiv:2003.05555"},{"key":"5433_CR12","unstructured":"Gattami A (2019) Reinforcement learning of markov decision processes with peak constraints. arXiv:1901.07839"},{"key":"5433_CR13","doi-asserted-by":"crossref","unstructured":"Geibel P (2006) Reinforcement learning for mdps with constraints. In: European conference on machine learning, pp 646\u2013653. Springer","DOI":"10.1007\/11871842_63"},{"key":"5433_CR14","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1613\/jair.1666","volume":"24","author":"P Geibel","year":"2005","unstructured":"Geibel P, Wysotzki F (2005) Risk-sensitive reinforcement learning applied to control under constraints. J Artif Intell Res 24:81\u2013108","journal-title":"J Artif Intell Res"},{"key":"5433_CR15","unstructured":"Chow Y, Nachum O, Duenez-Guzman E, Ghavamzadeh M (2018) A lyapunov-based approach to safe reinforcement learning. Advances in Neural Information Processing Systems 31"},{"key":"5433_CR16","unstructured":"Ding D, Wei X, Yang Z, Wang Z, Jovanovic M (2021) Provably efficient safe exploration via primal-dual policy optimization. In: International conference on artificial intelligence and statistics, pp 3304\u20133312. PMLR"},{"issue":"3","key":"5433_CR17","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1007\/s10994-016-5569-5","volume":"105","author":"L Prashanth","year":"2016","unstructured":"Prashanth L, Ghavamzadeh M (2016) Variance-constrained actor-critic algorithms for discounted and average reward mdps. Machine Learning 105(3):367\u2013417","journal-title":"Machine Learning"},{"key":"5433_CR18","doi-asserted-by":"crossref","unstructured":"Liu C, Geng N, Aggarwal V, Lan T, Yang Y, Xu M (2021) Cmix: Deep multi-agent reinforcement learning with peak and average constraints. In: Joint european conference on machine learning and knowledge discovery in databases, pp 157\u2013173. Springer","DOI":"10.1007\/978-3-030-86486-6_10"},{"key":"5433_CR19","unstructured":"Rashid T, Samvelyan M, Schroeder C, Farquhar G, Foerster J, Whiteson S (2018) Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning. In: International conference on machine learning, pp 4295\u20134304. PMLR"},{"key":"5433_CR20","doi-asserted-by":"crossref","unstructured":"Geng N, Bai Q, Liu C, Lan T, Aggarwal V, Yang Y, Xu M (2023) A reinforcement learning framework for vehicular network routing under peak and average constraints. IEEE Transactions on Vehicular Technology","DOI":"10.1109\/TVT.2023.3235946"},{"issue":"3","key":"5433_CR21","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/BF00992698","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins CJ, Dayan P (1992) Q-learning. Mach Learn 8(3):279\u2013292","journal-title":"Mach Learn"},{"key":"5433_CR22","unstructured":"Rummery GA, Niranjan M (1994) On-line Q-learning Using Connectionist Systems, vol 37. University of Cambridge, Department of Engineering Cambridge"},{"key":"5433_CR23","unstructured":"Bertsekas DP (2014) Constrained Optimization and Lagrange Multiplier Methods. Academic press (Massachusetts Institute of Technology)"},{"key":"5433_CR24","unstructured":"Beznosikov A, Gorbunov E, Berard H, Loizou N (2023) Stochastic gradient descent-ascent: Unified theory and new efficient methods. In: International conference on artificial intelligence and statistics, pp 172\u2013235. PMLR"},{"key":"5433_CR25","first-page":"25865","volume":"34","author":"W Xian","year":"2021","unstructured":"Xian W, Huang F, Zhang Y, Huang H (2021) A faster decentralized algorithm for nonconvex minimax problems. Adv Neural Inf Process Syst 34:25865\u201325877","journal-title":"Adv Neural Inf Process Syst"},{"key":"5433_CR26","unstructured":"Li B, Cen S, Chen Y, Chi Y (2020) Communication-efficient distributed optimization in networks with gradient tracking and variance reduction. In: International conference on artificial intelligence and statistics, pp 1662\u20131672. PMLR"},{"issue":"1","key":"5433_CR27","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/s10107-020-01487-0","volume":"187","author":"S Pu","year":"2021","unstructured":"Pu S, Nedi\u0107 A (2021) Distributed stochastic gradient tracking methods. Math Program 187(1):409\u2013457","journal-title":"Math Program"},{"key":"5433_CR28","unstructured":"Cutkosky A, Orabona F (2019) Momentum-based variance reduction in non-convex sgd. Advances in Neural Information Processing Systems 32"},{"key":"5433_CR29","unstructured":"Tran-Dinh Q, Pham NH, Phan DT, Nguyen LM (2019) Hybrid stochastic gradient descent algorithms for stochastic nonconvex optimization. arXiv:1905.05920"},{"key":"5433_CR30","first-page":"9377","volume":"36","author":"Z Jiang","year":"2022","unstructured":"Jiang Z, Lee XY, Tan SY, Tan KL, Balu A, Lee YM, Hegde C, Sarkar S (2022) Mdpgt: momentum-based decentralized policy gradient tracking. In: Proceedings of the AAAI conference on artificial intelligence 36:9377\u20139385","journal-title":"In: Proceedings of the AAAI conference on artificial intelligence"},{"key":"5433_CR31","doi-asserted-by":"crossref","unstructured":"Zhang K, Yang Z, Liu H, Zhang T, Basar T (2018) Fully decentralized multi-agent reinforcement learning with networked agents. In: International conference on machine learning, pp 5872\u20135881. PMLR","DOI":"10.1109\/CDC.2018.8619581"},{"key":"5433_CR32","unstructured":"Zhang G, Martens J, Grosse, RB (2019) Fast convergence of natural gradient descent for over-parameterized neural networks. Advances in Neural Information Processing Systems 32"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05433-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/article\/10.1007\/s10489-024-05433-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05433-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,15]],"date-time":"2024-06-15T12:10:24Z","timestamp":1718453424000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/link.springer.com\/10.1007\/s10489-024-05433-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":32,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2024,4]]}},"alternative-id":["5433"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1007\/s10489-024-05433-x","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4]]},"assertion":[{"value":"29 March 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 May 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing of interest"}},{"value":"In the research presented within this journal, we emphasize our commitment to ethical data practices. In the context of our research, it is essential to address the unique data dynamics inherent to reinforcement learning and simulations. Unlike traditional datasets, our study revolves around agent interactions within simulated environments, and as such, ethical considerations differ from those concerning personal or sensitive data.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and Informed Consent for Data Used"}}]}}