{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T14:04:44Z","timestamp":1779977084773,"version":"3.53.1"},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T00:00:00Z","timestamp":1779926400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T00:00:00Z","timestamp":1779926400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Data Sci Anal"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1007\/s41060-026-01148-z","type":"journal-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T13:28:32Z","timestamp":1779974912000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Spark-based DBSCAN++ for efficient density-based clustering"],"prefix":"10.1007","volume":"22","author":[{"given":"Abdalrahman","family":"Eltahir","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mahmoud","family":"Madi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Humaid","family":"Alhadidi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zaher","family":"AL Aghbari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,28]]},"reference":[{"issue":"20","key":"1148_CR1","doi-asserted-by":"publisher","first-page":"11122","DOI":"10.3390\/app132011122","volume":"13","author":"X An","year":"2023","unstructured":"An, X., Wang, Z., Wang, D., et al.: STRP-DBSCAN: a parallel DBSCAN algorithm based on spatial-temporal random partitioning for clustering trajectory data. Appl. Sci. 13(20), 11122 (2023). https:\/\/doi.org\/10.3390\/app132011122","journal-title":"Appl. Sci."},{"key":"1148_CR2","doi-asserted-by":"publisher","unstructured":"Arthur D, Vassilvitskii, S.: k-means++: The advantages of careful seeding. In: Proceedings of the Eighteenth Annual ACM-SIAM Symposium on Discrete Algorithms (SODA\u201907), pp 1027\u20131035 (2007). https:\/\/doi.org\/10.5555\/1283383.1283494","DOI":"10.5555\/1283383.1283494"},{"key":"1148_CR3","doi-asserted-by":"publisher","unstructured":"Aryal, A.M., Wang, S.: SparkSNN: A density-based clustering algorithm on Spark. In: 2018 IEEE 3rd International Conference on Big Data Analysis (ICBDA), pp 433\u2013437, (2018). https:\/\/doi.org\/10.1109\/ICBDA.2018.8367722","DOI":"10.1109\/ICBDA.2018.8367722"},{"issue":"4","key":"1148_CR4","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1016\/j.aej.2015.08.009","volume":"54","author":"AM Bakr","year":"2015","unstructured":"Bakr, A.M., Ghanem, N.M., Ismail, M.A.: Efficient incremental density-based algorithm for clustering large datasets. Alex. Eng. J. 54(4), 1147\u20131154 (2015). https:\/\/doi.org\/10.1016\/j.aej.2015.08.009","journal-title":"Alex. Eng. J."},{"key":"1148_CR5","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2025.114019","volume":"186","author":"NF Batista","year":"2025","unstructured":"Batista, N.F., Nunes, B.L., Naldi, M.C.: Efficient multiple density-based models over large datasets with data stream applications. Appl. Soft Comput. 186, 114019 (2025). https:\/\/doi.org\/10.1016\/j.asoc.2025.114019","journal-title":"Appl. Soft Comput."},{"issue":"9","key":"1148_CR6","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1145\/361002.361007","volume":"18","author":"JL Bentley","year":"1975","unstructured":"Bentley, J.L.: Multidimensional binary search trees used for associative searching. Commun. ACM 18(9), 509\u2013517 (1975). https:\/\/doi.org\/10.1145\/361002.361007","journal-title":"Commun. ACM"},{"key":"1148_CR7","doi-asserted-by":"publisher","unstructured":"Breunig, M.M., Kriegel, H.P., Kr\u00f6ger, P., et\u00a0al.: Data bubbles: quality preserving performance boosting for hierarchical clustering. In: Proceedings of the 2001 International Conference on Management of Data (SIGMOD\u201901), pp 79\u201390, (2001). https:\/\/doi.org\/10.1145\/376284.375672","DOI":"10.1145\/376284.375672"},{"key":"1148_CR8","doi-asserted-by":"publisher","unstructured":"Campello, RJGB., Moulavi, D., Sander, J.: Density-based clustering based on hierarchical density estimates. In: Advances in Knowledge Discovery and Data Mining, pp 160\u2013172 (2013). https:\/\/doi.org\/10.1007\/978-3-642-37456-2_14","DOI":"10.1007\/978-3-642-37456-2_14"},{"key":"1148_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107624","volume":"109","author":"Y Chen","year":"2021","unstructured":"Chen, Y., Zhou, L., Bouguila, N., et al.: BLOCK-DBSCAN: fast clustering for large scale data. Pattern Recogn. 109, 107624 (2021). https:\/\/doi.org\/10.1016\/j.patcog.2020.107624","journal-title":"Pattern Recogn."},{"issue":"3","key":"1148_CR10","doi-asserted-by":"publisher","first-page":"1249","DOI":"10.1007\/s11227-014-1225-7","volume":"70","author":"X Cui","year":"2014","unstructured":"Cui, X., Zhu, P., Yang, X., et al.: Optimized big data K-means clustering using MapReduce. J. Supercomput. 70(3), 1249\u20131259 (2014). https:\/\/doi.org\/10.1007\/s11227-014-1225-7","journal-title":"J. Supercomput."},{"key":"1148_CR11","doi-asserted-by":"publisher","unstructured":"Ester, M., Kriegel, H., Sander, J., et\u00a0al.: A density-based algorithm for discovering clusters in large spatial databases with noise. In: Proceedings of the Second International Conference on Knowledge Discovery and Data Mining (KDD\u201996), pp 226\u2013231 (1996). https:\/\/doi.org\/10.5555\/3001460.3001507","DOI":"10.5555\/3001460.3001507"},{"key":"1148_CR12","doi-asserted-by":"publisher","unstructured":"Gan, J., Tao, Y.: DBSCAN revisited: Mis-claim, un-fixability, and approximation. In: Proceedings of the 2015 ACM SIGMOD International Conference on Management of Data (SIGMOD\u201915), pp 519\u2013530 (2015). https:\/\/doi.org\/10.1145\/2723372.2737792","DOI":"10.1145\/2723372.2737792"},{"issue":"6","key":"1148_CR13","doi-asserted-by":"publisher","first-page":"6214","DOI":"10.1007\/s11227-020-03524-3","volume":"77","author":"N Gholizadeh","year":"2021","unstructured":"Gholizadeh, N., Saadatfar, H., Hanafi, N.: K-DBSCAN: an improved DBSCAN algorithm for big data. J. Supercomput. 77(6), 6214\u20136235 (2021). https:\/\/doi.org\/10.1007\/s11227-020-03524-3","journal-title":"J. Supercomput."},{"key":"1148_CR14","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/0304-3975(85)90224-5","volume":"38","author":"TF Gonzalez","year":"1985","unstructured":"Gonzalez, T.F.: Clustering to minimize the maximum intercluster distance. Theoret. Comput. Sci. 38, 293\u2013306 (1985). https:\/\/doi.org\/10.1016\/0304-3975(85)90224-5","journal-title":"Theoret. Comput. Sci."},{"key":"1148_CR15","doi-asserted-by":"publisher","unstructured":"Han, D., Agrawal, A., Liao, W., et\u00a0al.: A Fast DBSCAN Algorithm with Spark Implementation, Springer Singapore, Singapore, pp 173\u2013192 (2018a). https:\/\/doi.org\/10.1007\/978-981-10-8476-8_9","DOI":"10.1007\/978-981-10-8476-8_9"},{"key":"1148_CR16","doi-asserted-by":"publisher","unstructured":"Han, D., Agrawal, A., Liao, W., et\u00a0al.: Parallel DBSCAN algorithm using a data partitioning strategy with Spark implementation. In: 2018 IEEE International Conference on Big Data, pp 305\u2013312 (2018b). https:\/\/doi.org\/10.1109\/BigData.2018.8622258","DOI":"10.1109\/BigData.2018.8622258"},{"key":"1148_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.117501","volume":"203","author":"N Hanafi","year":"2022","unstructured":"Hanafi, N., Saadatfar, H.: A fast DBSCAN algorithm for big data based on efficient density calculation. Expert Syst. Appl. 203, 117501 (2022). https:\/\/doi.org\/10.1016\/j.eswa.2022.117501","journal-title":"Expert Syst. Appl."},{"issue":"1","key":"1148_CR18","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/s11704-013-3158-3","volume":"8","author":"Y He","year":"2014","unstructured":"He, Y., Tan, H., Luo, W., et al.: MR-DBSCAN: a scalable MapReduce-based DBSCAN algorithm for heavily skewed data. Front. Comp. Sci. 8(1), 83\u201399 (2014). https:\/\/doi.org\/10.1007\/s11704-013-3158-3","journal-title":"Front. Comp. Sci."},{"issue":"12","key":"1148_CR19","doi-asserted-by":"publisher","first-page":"1301","DOI":"10.3390\/rs9121301","volume":"9","author":"F Huang","year":"2017","unstructured":"Huang, F., Zhu, Q., Zhou, J., et al.: Research on the parallelization of the DBSCAN clustering algorithm for spatial data mining based on the Spark platform. Remote Sens. 9(12), 1301 (2017). https:\/\/doi.org\/10.3390\/rs9121301","journal-title":"Remote Sens."},{"key":"1148_CR20","unstructured":"Jang, J., Jiang, H.: DBSCAN++: Towards fast and scalable density clustering. In: Proceedings of the 36th International Conference on Machine Learning (ICML 2019), pp 3019\u20133029 (2019). https:\/\/proceedings.mlr.press\/v97\/jang19a.html"},{"key":"1148_CR21","doi-asserted-by":"publisher","unstructured":"Mk, K.: A scalable density based clustering method for large datasets with noise. (2022). https:\/\/doi.org\/10.2139\/ssrn.4111798","DOI":"10.2139\/ssrn.4111798"},{"key":"1148_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.cose.2025.104325","volume":"151","author":"VN K\u0131l\u0131\u00e7","year":"2025","unstructured":"K\u0131l\u0131\u00e7, V.N., E\u015fsiz, E.S.: K-salp swarm anomaly detection (K-sad): a novel clustering and threshold-based approach for cybersecurity applications. Computers & Security 151, 104325 (2025). https:\/\/doi.org\/10.1016\/j.cose.2025.104325","journal-title":"Computers & Security"},{"issue":"6","key":"1148_CR23","first-page":"100","volume":"2","author":"S Li","year":"2025","unstructured":"Li, S., Liu, K., Chen, X.: A context-aware personalized recommendation framework integrating user clustering and BERT-based sentiment analysis. J. Comput. Signal Sys. Res. 2(6), 100\u2013108 (2025)","journal-title":"J. Comput. Signal Sys. Res."},{"key":"1148_CR24","unstructured":"Litouka, A.: Spark DBSCAN: DBSCAN clustering algorithm on top of Apache Spark. (2014). https:\/\/github.com\/alitouka\/spark_dbscan, gitHub repository"},{"key":"1148_CR25","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1016\/j.patrec.2018.12.010","volume":"117","author":"D Luchi","year":"2019","unstructured":"Luchi, D., Loureiros Rodrigues, A., Miguel Varej\u00e3o, F.: Sampling approaches for applying DBSCAN to large datasets. Pattern Recogn. Lett. 117, 90\u201396 (2019). https:\/\/doi.org\/10.1016\/j.patrec.2018.12.010","journal-title":"Pattern Recogn. Lett."},{"key":"1148_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.125431","volume":"260","author":"C Luo","year":"2025","unstructured":"Luo, C., Zhang, J., Zhang, X.: Tensor multi-view clustering method for natural image segmentation. Expert Syst. Appl. 260, 125431 (2025). https:\/\/doi.org\/10.1016\/j.eswa.2024.125431","journal-title":"Expert Syst. Appl."},{"key":"1148_CR27","doi-asserted-by":"publisher","unstructured":"Luo, G., Luo, X., Gooch, T.F., et\u00a0al.: A parallel DBSCAN algorithm based on Spark. In: 2016 IEEE International Conferences on Big Data and Cloud Computing (BDCloud), Social Computing and Networking (SocialCom), and Sustainable Computing and Communications (SustainCom), pp 548\u2013553 (2016). https:\/\/doi.org\/10.1109\/BDCloud-SocialCom-SustainCom.2016.85","DOI":"10.1109\/BDCloud-SocialCom-SustainCom.2016.85"},{"key":"1148_CR28","doi-asserted-by":"publisher","unstructured":"Mathur, V., Mehta, J., Singh, S.: HCA-DBSCAN: Hypercube accelerated density based spatial clustering for applications with noise. (2019). https:\/\/doi.org\/10.48550\/arXiv.1912.00323","DOI":"10.48550\/arXiv.1912.00323"},{"key":"1148_CR29","doi-asserted-by":"publisher","unstructured":"Matloob, I., Khan, S., Rukaiya, R., et al.: Healthcare fraud detection using adaptive learning and deep learning techniques. Evol. Syst. 16(2), 72 (2025). https:\/\/doi.org\/10.1007\/s12530-025-09698-6","DOI":"10.1007\/s12530-025-09698-6"},{"issue":"85","key":"1148_CR30","first-page":"2825","volume":"12","author":"F Pedregosa","year":"2011","unstructured":"Pedregosa, F., Varoquaux, G., Gramfort, A., et al.: Scikit-learn: Machine learning in Python. J. Mach. Learn. Res. 12(85), 2825\u20132830 (2011). (http:\/\/jmlr.org\/papers\/v12\/pedregosa11a.html)","journal-title":"J. Mach. Learn. Res."},{"issue":"3","key":"1148_CR31","doi-asserted-by":"publisher","first-page":"2364","DOI":"10.59934\/jaiea.v4i3.1170","volume":"4","author":"LA Putri","year":"2025","unstructured":"Putri, L.A., Tsaqofah, M., Hasibuan, D.S., et al.: Application of K-means clustering algorithm for E-commerce data analysis. J. Artific. Intelli. Eng. Appl. 4(3), 2364\u20132367 (2025). https:\/\/doi.org\/10.59934\/jaiea.v4i3.1170","journal-title":"J. Artific. Intelli. Eng. Appl."},{"key":"1148_CR32","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1016\/0377-0427(87)90125-7","volume":"20","author":"PJ Rousseeuw","year":"1987","unstructured":"Rousseeuw, P.J.: Silhouettes: A graphical aid to the interpretation and validation of cluster analysis. J. Comput. Appl. Math. 20, 53\u201365 (1987). https:\/\/doi.org\/10.1016\/0377-0427(87)90125-7","journal-title":"J. Comput. Appl. Math."},{"key":"1148_CR33","doi-asserted-by":"publisher","unstructured":"Santos, J.A., Syed, T.I., Naldi, M.C., et al.: Hierarchical density-based clustering using mapreduce. IEEE Trans. Big Data 7(1), 102\u2013114 (2019). https:\/\/doi.org\/10.1109\/TBDATA.2019.2907624","DOI":"10.1109\/TBDATA.2019.2907624"},{"key":"1148_CR34","doi-asserted-by":"publisher","unstructured":"Song, H., Lee, J.: RP-DBSCAN: a superfast parallel DBSCAN algorithm based on random partitioning. In: Proceedings of the 2018 International Conference on Management of Data (SIGMOD\u201918), pp 1173\u20131187 (2018). https:\/\/doi.org\/10.1145\/3183713.3196887","DOI":"10.1145\/3183713.3196887"},{"key":"1148_CR35","doi-asserted-by":"publisher","unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., et\u00a0al.: Spark: Cluster computing with working sets. In: Proceedings of the 2nd USENIX conference on Hot topics in cloud computing (HotCloud\u201910), p\u00a010, (2010). https:\/\/doi.org\/10.5555\/1863103.1863113","DOI":"10.5555\/1863103.1863113"},{"issue":"11","key":"1148_CR36","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1145\/2934664","volume":"59","author":"M Zaharia","year":"2016","unstructured":"Zaharia, M., Xin, R.S., Wendell, P., et al.: Apache Spark: a unified engine for big data processing. Commun. ACM 59(11), 56\u201365 (2016). https:\/\/doi.org\/10.1145\/2934664","journal-title":"Commun. ACM"}],"container-title":["International Journal of Data Science and Analytics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41060-026-01148-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41060-026-01148-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41060-026-01148-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T13:28:35Z","timestamp":1779974915000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s41060-026-01148-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,28]]},"references-count":36,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,12]]}},"alternative-id":["1148"],"URL":"https:\/\/doi.org\/10.1007\/s41060-026-01148-z","relation":{},"ISSN":["2364-415X","2364-4168"],"issn-type":[{"value":"2364-415X","type":"print"},{"value":"2364-4168","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,28]]},"assertion":[{"value":"24 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"180"}}