{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T02:14:17Z","timestamp":1771467257225,"version":"3.50.1"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2018,8,23]],"date-time":"2018-08-23T00:00:00Z","timestamp":1534982400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"We Acknowledge the Data to Decisions CRC (D2D CRC) and the Cooperative Research Centres Program for funding this research."}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Distrib Parallel Databases"],"published-print":{"date-parts":[[2019,9]]},"DOI":"10.1007\/s10619-018-7245-1","type":"journal-article","created":{"date-parts":[[2018,8,23]],"date-time":"2018-08-23T11:53:38Z","timestamp":1535025218000},"page":"351-384","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":38,"title":["DataSynapse: A Social Data Curation Foundry"],"prefix":"10.1007","volume":"37","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5988-5494","authenticated-orcid":false,"given":"Amin","family":"Beheshti","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Boualem","family":"Benatallah","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alireza","family":"Tabebordbar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hamid Reza","family":"Motahari-Nezhad","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Moshe Chai","family":"Barukh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Reza","family":"Nouri","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,8,23]]},"reference":[{"key":"7245_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-1-4419-8462-3","volume-title":"An Introduction to Social Network Data Analytics","author":"CC Aggarwal","year":"2011","unstructured":"Aggarwal, C.C.: An Introduction to Social Network Data Analytics, pp. 1\u201315. Springer, Berlin (2011)"},{"key":"7245_CR2","unstructured":"Anderson, M.R., Antenucci, D., Bittorf, V., Burgess, M., Cafarella, M.J., Kumar, A., Niu, F. et al.: Brainwash: a data system for feature engineering. In: CIDR (2013)"},{"key":"7245_CR3","unstructured":"Beheshti, S.-M.-R., Nezhad, H.R.M., Benatallah, B.: Temporal provenance model (TPM): model and query language. CoRR, abs\/1211.5009 (2012)"},{"key":"7245_CR4","unstructured":"Beheshti, S.-M.-R. et al.: Galaxy: a platform for explorative analysis of open data sources. In: Proceedings of the 19th International Conference on Extending Database Technology, (EDBT), pp. 640\u2013643 (2016). \n                    https:\/\/dblp.org\/rec\/bibtex\/conf\/edbt\/BeheshtiBM16"},{"issue":"3","key":"7245_CR5","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1007\/s10619-014-7171-9","volume":"34","author":"S-M-R Beheshti","year":"2016","unstructured":"Beheshti, S.-M.-R., Benatallah, B., Motahari-Nezhad, H.R.: Scalable graph-based OLAP analytics over process execution data. Distrib. Parallel Databases 34(3), 379\u2013423 (2016)","journal-title":"Distrib. Parallel Databases"},{"key":"7245_CR6","doi-asserted-by":"crossref","unstructured":"Beheshti, S.-M.-R., Benatallah, B., Sakr, S., Grigori, D., Motahari-Nezhad, H.R., Barukh, M.C., Gater, A., Ryu, S.H.: Process Analytics\u2014Concepts and Techniques for Querying and Analyzing Process Data. Springer, Berlin (2016)","DOI":"10.1007\/978-3-319-25037-3"},{"issue":"2","key":"7245_CR7","first-page":"47","volume":"39","author":"PC Arocena","year":"2016","unstructured":"Arocena, P.C., Glavic, B., Mecca, G., Miller, R.J., Papotti, P., Santoro, D.: Benchmarking data curation systems. IEEE Data Eng. Bull. 39(2), 47\u201362 (2016)","journal-title":"IEEE Data Eng. Bull."},{"key":"7245_CR8","unstructured":"Beheshti, S.-M.-R., Tabebordbar, A., Benatallah, B., Nouri, R.: On automating basic data curation tasks. In: Proceedings of the 26th International Conference on World Wide Web Companion, Perth, Australia, April 3\u20137, 2017, pp. 165\u2013169 (2017)"},{"issue":"4","key":"7245_CR9","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1007\/s00607-016-0490-0","volume":"99","author":"S-M-R Beheshti","year":"2017","unstructured":"Beheshti, S.-M.-R., Benatallah, B., Venugopal, S., Ryu, S.H., Motahari-Nezhad, H.R., Wang, Wei: A systematic review and comparative analysis of cross-document coreference resolution methods and tools. Computing 99(4), 313\u2013349 (2017)","journal-title":"Computing"},{"key":"7245_CR10","unstructured":"Beheshti, A., Benatallah, B., Nouri, R., Chhieng, Van M., Xiong, H., Zhao, X.: Coredb: a data lake service. In: Proceedings of the 2017 ACM on Conference on Information and Knowledge Management, CIKM 2017, Singapore, November 06\u201310, 2017, pp. 2451\u20132454 (2017)"},{"key":"7245_CR11","unstructured":"Beheshti, A., Benatallah, B., Nouri, R., Tabebordbar, A.: Corekg: a knowledge lake service. PVLDB 11(12), 1942\u20131945 (2018). \n                    https:\/\/dblp.org\/rec\/bibtex\/journals\/pvldb\/BeheshtiBNT18"},{"key":"7245_CR12","doi-asserted-by":"crossref","unstructured":"Beheshti, A., Schiliro, F., Ghodratnama, S., Amouzgar, F., Benatallah, B., Yang, J., Sheng, Q.Z., Casati, F., Motahari-Nezhad, H.R.: iprocess: Enabling iot platforms in data-driven knowledge-intensive processes. In: Business Process Management Forum - BPM Forum 2018 (2018)","DOI":"10.1007\/978-3-319-98651-7_7"},{"key":"7245_CR13","unstructured":"Beheshti, A., Vaghani, K., Benatallah, B., Tabebordbar, A.: Crowdcorrect: A curation pipeline for social data cleansing and curation. In: Information Systems in the Big Data Era\u2014CAiSE Forum 2018, Tallinn, Estonia, June 11\u201315, 2018, Proceedings, pp. 24\u201338 (2018)"},{"issue":"3","key":"7245_CR14","first-page":"4","volume":"36","author":"X Chai","year":"2013","unstructured":"Chai, X., Deshpande, O., Garera, N., Gattani, A., Lam, W., Lamba, D.S., Liu, L., Tiwari, M., Tourn, M., Vacheri, Z., Prasad, S.T.S., Subramaniam, S., Harinarayan, V., Rajaraman, A., Ardalan, A., Das, S., Suganthan, G.C.P., Doan, A.: Social media analytics: the kosmix story. IEEE Data Eng. Bull. 36(3), 4\u201312 (2013)","journal-title":"IEEE Data Eng. Bull."},{"issue":"4","key":"7245_CR15","doi-asserted-by":"publisher","first-page":"1165","DOI":"10.2307\/41703503","volume":"36","author":"H Chen","year":"2012","unstructured":"Chen, H., Chiang, R.H.L., Storey, V.C.: Business intelligence and analytics: from big data to big impact. MIS Q. 36(4), 1165\u20131188 (2012)","journal-title":"MIS Q."},{"key":"7245_CR16","unstructured":"Chiticariu, L., Krishnamurthy, R., Li, Y., Raghavan, S., Reiss, F., Vaithyanathan, S.: Systemt: an algebraic approach to declarative information extraction. In: ACL 2010, Proceedings of the 48th Annual Meeting of the Association for Computational Linguistics, July 11\u201316, 2010, Uppsala, pp. 128\u2013137 (2010)"},{"key":"7245_CR17","doi-asserted-by":"crossref","unstructured":"Dean, J., Ghemawat, S.: Mapreduce: simplified data processing on large clusters. Commun. ACM. 51(1), 107 (2008)","DOI":"10.1145\/1327452.1327492"},{"key":"7245_CR18","unstructured":"Deshpande, M., Ray, D., Dixit, S., Agasti, A.: Shareinsights: an unified approach to full-stack data processing. In: Proceedings of the 2015 ACM SIGMOD International Conference on Management of Data, Melbourne, Victoria, Australia, May 31\u2013June 4, 2015, pp. 1925\u20131940 (2015)"},{"key":"7245_CR19","unstructured":"Doan, A., Domingos, P.M., Halevy, A.Y.: Reconciling schemas of disparate data sources: a machine-learning approach. In: Proceedings of the 2001 ACM SIGMOD international conference on Management of data, Santa Barbara, CA, USA, May 21\u201324, 2001, pp. 509\u2013520 (2001)"},{"key":"7245_CR20","doi-asserted-by":"crossref","unstructured":"Ferrucci, D.A.: Introduction to \u2019this is watson\u2019. IBM J. Res. Dev. 56(3.4), 4:1\u20134:11 (2012)","DOI":"10.1147\/JRD.2012.2184356"},{"key":"7245_CR21","doi-asserted-by":"crossref","unstructured":"Freitas, A., Curry, E.: Big data curation. In: Cavanillas, J.M., (ed.), New Horizons for a Data-Driven Economy, pp. 87\u2013118. Springer, Berlin (2016)","DOI":"10.1007\/978-3-319-21569-3_6"},{"key":"7245_CR22","unstructured":"Terrizzano, I. et al.: Data wrangling: the challenging journey from the wild to the lake. In: CIDR (2015)"},{"key":"7245_CR23","unstructured":"Kim, N.W., Jung, J., Ko, E.-Y., Han, S., Lee, C.W., Kim, J., Kim, J.: Budgetmap: engaging taxpayers in the issue-driven classification of a government budget. In: Proceedings of the 19th ACM Conference on Computer-Supported Cooperative Work & Social Computing, CSCW 2016, San Francisco, CA, USA, February 27\u2013March 2, 2016, pp. 1026\u20131037 (2016)"},{"key":"7245_CR24","unstructured":"Lee, K., Agrawal, A., Choudhary, A.: Real-time disease surveillance using twitter data: demonstration on flu and cancer. In: Proceedings of the 19th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, KDD \u201913, pages 1474\u20131477, New York, NY, USA (2013). ACM"},{"key":"7245_CR25","unstructured":"Lohr, S.: The age of big data. New York Times, 11 (2012)"},{"key":"7245_CR26","unstructured":"Nakov, P., Ritter, A., Rosenthal, S., Sebastiani, F., Stoyanov, V.: Semeval-2016 task 4: sentiment analysis in twitter. In: Proceedings of the 10th International Workshop on Semantic Evaluation, SemEval@NAACL-HLT 2016, San Diego, CA, USA, June 16\u201317, 2016, pp. 1\u201318 (2016)"},{"key":"7245_CR27","unstructured":"Pandey, N., Natarajan, S.: How social media can contribute during disaster events? case study of chennai floods 2015. In: 2016 International Conference on Advances in Computing, Communications and Informatics, ICACCI 2016, Jaipur, India, September 21\u201324, 2016, pp. 1352\u20131356 (2016)"},{"key":"7245_CR28","unstructured":"Paul Suganthan, G.C., Sun, C., Krishna\u00a0Gayatri, K., Zhang, H., Yang, F., Rampalli, N., Prasad, S., Arcaute, E., Krishnan, G., Deep, R., Raghavendra, V., Doan, A.: Why big data industrial systems need rules and what we can do about it. In: Proceedings of the 2015 ACM SIGMOD International Conference on Management of Data, Melbourne, Victoria, Australia, May 31\u2013June 4, 2015, pp. 265\u2013276 (2015)"},{"key":"7245_CR29","unstructured":"Pu, X., Jin, R., Wu, G., Han, D., Xue, G.-R.: Topic modeling in semantic space with keywords. In: Proceedings of the 24th ACM International Conference on Information and Knowledge Management, CIKM 2015, Melbourne, VIC, Australia, October 19\u201323, 2015, pp. 1141\u20131150 (2015)"},{"key":"7245_CR30","unstructured":"Ritter, A., Clark, S., Mausam, E., Oren: named entity recognition in tweets: an experimental study. In: Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing, EMNLP 2011, 27\u201331 July 2011, John McIntyre Conference Centre, Edinburgh, UK, A meeting of SIGDAT, a Special Interest Group of the ACL, pp. 1524\u20131534 (2011)"},{"key":"7245_CR31","doi-asserted-by":"crossref","unstructured":"Ruder, T.D., Hatch, G.M., Ampanozi, G., Thali, M.J., Fischer, N.: Suicide announcement on facebook. Crisis (2011)","DOI":"10.1027\/0227-5910\/a000086"},{"key":"7245_CR32","unstructured":"Russom, P., et al.: Big data analytics. TDWI best practices report, fourth quarter 19, 40 (2011)"},{"key":"7245_CR33","unstructured":"Sellam, T., M\u00fcller, E., Kersten, M.L.: Semi-automated exploration of data warehouses. In: Proceedings of the 24th ACM International Conference on Information and Knowledge Management, CIKM 2015, Melbourne, VIC, Australia, October 19\u201323, 2015, pp. 1321\u20131330 (2015)"},{"key":"7245_CR34","unstructured":"Stonebraker, M. et al.: Data curation at scale: the data tamer system. In: CIDR (2013)"},{"issue":"13","key":"7245_CR35","doi-asserted-by":"publisher","first-page":"1713","DOI":"10.14778\/2733004.2733069","volume":"7","author":"M Fabian","year":"2014","unstructured":"Fabian, M.: Suchanek and Gerhard Weikum. Knowledge bases in the age of big data analytics. Proc. VLDB Endow. 7(13), 1713\u20131714 (2014)","journal-title":"Proc. VLDB Endow."},{"key":"7245_CR36","doi-asserted-by":"crossref","unstructured":"Tabebordbar, A., Beheshti, A.: Adaptive rule monitoring system. In: 40th International Conference on Software Engineering (ICSE), International Workshop on Software Engineering for Cognitive Services (SE4COG) (2018)","DOI":"10.1145\/3195555.3195564"},{"key":"7245_CR37","first-page":"xxvii","volume":"11","author":"O Tene","year":"2012","unstructured":"Tene, O., Polonetsky, J.: Big data for all: Privacy and user control in the age of analytics. N. J. Tech. Intell. Prop. 11, xxvii (2012)","journal-title":"N. J. Tech. Intell. Prop."},{"key":"7245_CR38","doi-asserted-by":"crossref","unstructured":"Troncy, R.: Linking entities for enriching and structuring social media content. In: WWW, pp. 597\u2013597 (2016)","DOI":"10.1145\/2872518.2892109"},{"key":"7245_CR39","unstructured":"Karlgren, J., Bohman, M., Ekgren, A., Isheden, G., Kullmann, E., Nilsson, D.: Semantic topology. In: Proceedings of the 23rd ACM International Conference on Conference on Information and Knowledge Management, CIKM 2014, Shanghai, China, November 3\u20137, 2014, pp. 1939\u20131942 (2014)"},{"key":"7245_CR40","unstructured":"Wang, S., Tang, J., Aggarwal, C.C., Liu, H.: Linked document embedding for classification. In: Proceedings of the 25th ACM International Conference on Information and Knowledge Management, CIKM 2016, Indianapolis, IN, USA, October 24\u201328, 2016, pp. 115\u2013124 (2016)"},{"key":"7245_CR41","unstructured":"Zarras, A.V., Vassiliadis, P., Dinos, I.: Keep calm and wait for the spike! insights on the evolution of amazon services. In: Advanced Information Systems Engineering - 28th International Conference, CAiSE 2016, Ljubljana, Slovenia, June 13-17, 2016. Proceedings, pp. 444\u2013458 (2016)"}],"container-title":["Distributed and Parallel Databases"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10619-018-7245-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10619-018-7245-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10619-018-7245-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,22]],"date-time":"2019-08-22T19:08:29Z","timestamp":1566500909000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10619-018-7245-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,8,23]]},"references-count":41,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2019,9]]}},"alternative-id":["7245"],"URL":"https:\/\/doi.org\/10.1007\/s10619-018-7245-1","relation":{},"ISSN":["0926-8782","1573-7578"],"issn-type":[{"value":"0926-8782","type":"print"},{"value":"1573-7578","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,8,23]]},"assertion":[{"value":"23 August 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}