{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T07:41:01Z","timestamp":1781854861473,"version":"3.54.5"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T00:00:00Z","timestamp":1563148800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"},{"start":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T00:00:00Z","timestamp":1563148800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["BMC Bioinformatics"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1186\/s12859-019-2973-4","type":"journal-article","created":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T11:27:34Z","timestamp":1563190054000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["RAFTS3G: an efficient and versatile clustering software to analyses in large protein datasets"],"prefix":"10.1186","volume":"20","author":[{"given":"Bruno Thiago","family":"de Lima Nichio","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aryel Marlus Repula","family":"de Oliveira","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Camilla Reginatto","family":"de Pierri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Leticia Graziela Costa","family":"Santos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alexandre Quadros","family":"Lejambre","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ricardo Assun\u00e7\u00e3o","family":"Vialle","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nilson Ant\u00f4nio","family":"da Rocha Coimbra","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dieval","family":"Guizelini","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jeroniza Nunes","family":"Marchaukoski","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fabio","family":"de Oliveira Pedrosa","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5271-991X","authenticated-orcid":false,"given":"Roberto Tadeu","family":"Raittz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,7,15]]},"reference":[{"issue":"1","key":"2973_CR1","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1186\/s13059-015-0721-2","volume":"16","author":"DM Emms","year":"2015","unstructured":"Emms DM, Kelly S. OrthoFinder: solving fundamental biases in whole genome comparisons dramatically improves orthogroup inference accuracy. Genome Biol. 2015;16(1):157. \n                    https:\/\/doi.org\/10.1186\/s13059-015-0721-2\n                    \n                  .","journal-title":"Genome Biol"},{"issue":"17","key":"2973_CR2","doi-asserted-by":"publisher","first-page":"2965","DOI":"10.1093\/bioinformatics\/bty224","volume":"34","author":"V Schw\u00e4mmle","year":"2018","unstructured":"Schw\u00e4mmle V, Jensen ON. VSClust: feature-based variance-sensitive clustering of omics data. Bioinformatics. 2018;34(17):2965\u201372. \n                    https:\/\/doi.org\/10.1093\/bioinformatics\/bty224\n                    \n                  .","journal-title":"Bioinformatics"},{"issue":"9","key":"2973_CR3","doi-asserted-by":"publisher","first-page":"1338","DOI":"10.1093\/bioinformatics\/btw815","volume":"33","author":"J Adams","year":"2017","unstructured":"Adams J, Mansfield MJ, Richard DJ, Doxey AC. Lineage-specific mutational clustering in protein structures predicts evolutionary shifts in function. Bioinformatics. 2017;33(9):1338\u201345. \n                    https:\/\/doi.org\/10.1093\/bioinformatics\/btw815\n                    \n                  .","journal-title":"Bioinformatics"},{"issue":"18","key":"2973_CR4","doi-asserted-by":"publisher","first-page":"2890","DOI":"10.1093\/bioinformatics\/btx322","volume":"33","author":"N St\u00e4dler","year":"2017","unstructured":"St\u00e4dler N, Dondelinger F, Hill SM, Akbani R, Lu Y, Mills GB, Mukherjee S. Molecular heterogeneity at the network level: high-dimensional testing, clustering and a TCGA case study. Oxf J Bioinforma. 2017;33(18):2890\u20136. \n                    https:\/\/doi.org\/10.1093\/bioinformatics\/btx322\n                    \n                  .","journal-title":"Oxf J Bioinforma"},{"key":"2973_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1093\/database\/baw139","volume":"2016","author":"B Bursteinas","year":"2016","unstructured":"Bursteinas B, Britto R, Bely B, Auchincloss A, Rivoire C, Redaschi N, et al. Minimizing proteome redundancy in the UniProt knowledgebase. Database. 2016;2016:1\u20139. \n                    https:\/\/doi.org\/10.1093\/database\/baw139\n                    \n                  .","journal-title":"Database"},{"key":"2973_CR6","volume-title":"Protein Bioinformatics. Methods in Molecular Biology","author":"C Chen","year":"2017","unstructured":"Chen C, Huang H, Wu CH. Protein Bioinformatics Databases and Resources. In: Wu C, Arighi C, Ross K, editors. Protein Bioinformatics. Methods in Molecular Biology, vol. 1558. New York: Humana Press; 2017."},{"issue":"D1","key":"2973_CR7","doi-asserted-by":"publisher","first-page":"D506","DOI":"10.1093\/nar\/gky1049","volume":"47","author":"TU Consortium","year":"2019","unstructured":"Consortium TU. UniProt: a worldwide hub of protein knowledge. Nucleic Acids Res. 2019;47(D1):D506\u201315. \n                    https:\/\/doi.org\/10.1093\/nar\/gky1049\n                    \n                  .","journal-title":"Nucleic Acids Res"},{"issue":"6","key":"2973_CR8","doi-asserted-by":"publisher","first-page":"545","DOI":"10.1038\/nmeth.4299","volume":"14","author":"N Altman","year":"2017","unstructured":"Altman N, Krzywinski M. Points of significance: clustering. Nat Methods. 2017;14(6):545\u20136. \n                    https:\/\/doi.org\/10.1038\/nmeth.4299\n                    \n                  .","journal-title":"Nat Methods"},{"issue":"13","key":"2973_CR9","doi-asserted-by":"publisher","first-page":"1658","DOI":"10.1093\/bioinformatics\/btl158","volume":"22","author":"W Li","year":"2006","unstructured":"Li W, Godzik A. Cd-hit: a fast program for clustering and comparing large sets of protein or nucleotide sequences. Bioinformatics. 2006;22(13):1658\u20139.","journal-title":"Bioinformatics"},{"issue":"Database issue","key":"2973_CR10","doi-asserted-by":"publisher","first-page":"D115","DOI":"10.1093\/nar\/gkh131","volume":"32","author":"R Apweiler","year":"2004","unstructured":"Apweiler R, Bairoch A, Wu CH, Barker WC, Boeckmann B, Ferro S, Yeh L-SL. UniProt: the universal protein knowledgebase. Nucleic Acids Res. 2004;32(Database issue):D115\u20139.","journal-title":"Nucleic Acids Res"},{"key":"2973_CR11","series-title":"Proceedings - 2016 IEEE International Conference on Bioinformatics and Biomedicine, BIBM 2016, 703\u2013706","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM.2016.7822604","volume-title":"Evaluation of CD-HIT for constructing non-redundant databases","author":"Q Chen","year":"2017","unstructured":"Chen Q, Wan Y, Lei Y, Zobel J, Verspoor K. Evaluation of CD-HIT for constructing non-redundant databases, Proceedings - 2016 IEEE International Conference on Bioinformatics and Biomedicine, BIBM 2016, 703\u2013706; 2017. \n                    https:\/\/doi.org\/10.1109\/BIBM.2016.7822604\n                    \n                  ."},{"issue":"19","key":"2973_CR12","doi-asserted-by":"publisher","first-page":"2460","DOI":"10.1093\/bioinformatics\/btq461","volume":"26","author":"RC Edgar","year":"2010","unstructured":"Edgar RC. Search and clustering orders of magnitude faster than BLAST. Bioinformatics. 2010;26(19):2460\u20131.","journal-title":"Bioinformatics"},{"issue":"1","key":"2973_CR13","doi-asserted-by":"publisher","first-page":"e00003","DOI":"10.1128\/mSystems.00003-15","volume":"1","author":"E Kopylova","year":"2016","unstructured":"Kopylova E, Navas-Molina JA, Mercier C, Xu ZZ, Mah\u00e9 F, He Y, et al. Open-source sequence clustering methods improve the state of the art. MSystems. 2016;1(1):e00003\u201315. \n                    https:\/\/doi.org\/10.1128\/mSystems.00003-15\n                    \n                  .","journal-title":"MSystems"},{"issue":"August","key":"2973_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/srep32333","volume":"6","author":"J Chen","year":"2016","unstructured":"Chen J, Long R, Wang XL, Liu B, Chou KC. DRHP-PseRA: detecting remote homology proteins using profile-based pseudo protein sequence and rank aggregation. Sci Rep. 2016;6(August):1\u20137. \n                    https:\/\/doi.org\/10.1038\/srep32333\n                    \n                  .","journal-title":"Sci Rep"},{"issue":"6","key":"2973_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1093\/nar\/gkx1313","volume":"46","author":"A Azad","year":"2018","unstructured":"Azad A, Pavlopoulos GA, Ouzounis CA, Kyrpides NC, Bulu\u00e7 A. HipMCL: a high-performance parallel implementation of the Markov clustering algorithm for large-scale networks. Nucleic Acids Res. 2018;46(6):1\u201311. \n                    https:\/\/doi.org\/10.1093\/nar\/gkx1313\n                    \n                  .","journal-title":"Nucleic Acids Res"},{"key":"2973_CR16","doi-asserted-by":"publisher","first-page":"513","DOI":"10.1093\/bioinformatics\/btg005","volume":"19","author":"S Vinga","year":"2003","unstructured":"Vinga S, Almeida J. Alignment-free sequence comparison--a review. Bioinformatics. 2003;19:513\u201323. \n                    https:\/\/doi.org\/10.1093\/bioinformatics\/btg005\n                    \n                  .","journal-title":"Bioinformatics"},{"key":"2973_CR17","doi-asserted-by":"publisher","first-page":"e44","DOI":"10.1093\/nar\/gkr1261","volume":"40","author":"K Mahmood","year":"2012","unstructured":"Mahmood K, Webb GI, Song J, Whisstock JC, Konagurthu AS. Efficient large-scale protein sequence comparison and gene matching to identify orthologs and co- orthologs. Nucleic Acids Res. 2012;40:e44. \n                    https:\/\/doi.org\/10.1093\/nar\/gkr1261\n                    \n                  .","journal-title":"Nucleic Acids Res"},{"issue":"1","key":"2973_CR18","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1186\/s41044-016-0019-8","volume":"2","author":"E Tabari","year":"2017","unstructured":"Tabari E, Su Z. PorthoMCL: parallel orthology prediction using MCL for the realm of massive genome availability. Big Data Anal. 2017;2(1):4. \n                    https:\/\/doi.org\/10.1186\/s41044-016-0019-8\n                    \n                  .","journal-title":"Big Data Anal"},{"key":"2973_CR19","doi-asserted-by":"publisher","unstructured":"Zielezinski A, Vinga S, Almeida J, Karlowski WM. Alignment-free sequence comparison: benefits, applications, and tools. Genome Biol. 2017. \n                    https:\/\/doi.org\/10.1186\/s13059-017-1319-7\n                    \n                  .","DOI":"10.1186\/s13059-017-1319-7"},{"issue":"11","key":"2973_CR20","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1038\/nbt.3988","volume":"35","author":"M Steinegger","year":"2017","unstructured":"Steinegger M, S\u00f6ding J. MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets. Nat Biotechnol. 2017;35(11):2\u20134. \n                    https:\/\/doi.org\/10.1038\/nbt.3988\n                    \n                  .","journal-title":"Nat Biotechnol"},{"key":"2973_CR21","doi-asserted-by":"publisher","unstructured":"Steinegger M, S\u00f6ding J. Clustering huge protein sequence sets in linear time. Nat Commun. 2018;9(1). \n                    https:\/\/doi.org\/10.1038\/s41467-018-04964-5\n                    \n                  .","DOI":"10.1038\/s41467-018-04964-5"},{"issue":"1","key":"2973_CR22","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1146\/annurev-biodatasci-080917-013431","volume":"1","author":"Jie Ren","year":"2018","unstructured":"Ren J, Bai X, Lu YY, Tang K, Wang Y, Reinert G, Sun F. Alignment-free sequence analysis and applications. Annu Rev Biomed Data Sci. 2018;1:93-114. \n                    https:\/\/doi.org\/10.1146\/annurev-biodatasci-080917-013431\n                    \n                  .","journal-title":"Annual Review of Biomedical Data Science"},{"key":"2973_CR23","unstructured":"Srivastava A, Baranwal M, Salapaka S. On the persistence of clustering solutions and true number of clusters in a dataset. Retrieved from arXiv\u00a02018.\u00a0\n                    http:\/\/arxiv.org\/abs\/1811.00102\n                    \n                  ."},{"issue":"11","key":"2973_CR24","doi-asserted-by":"publisher","first-page":"1033","DOI":"10.1038\/nmeth.3583","volume":"12","author":"C Wiwie","year":"2015","unstructured":"Wiwie C, Baumbach J, R\u00f6ttger R. Comparing the performance of biomedical clustering methods. Nat Methods. 2015;12(11):1033\u20138. \n                    https:\/\/doi.org\/10.1038\/nmeth.3583\n                    \n                  .","journal-title":"Nat Methods"},{"issue":"OCT","key":"2973_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.3389\/fgene.2017.00165","volume":"8","author":"BTL Nichio","year":"2017","unstructured":"Nichio BTL, Marchaukoski JN, Raittz RT. New tools in orthology analysis: a brief review of promising perspectives. Front Genet. 2017;8(OCT):1\u201312. \n                    https:\/\/doi.org\/10.3389\/fgene.2017.00165\n                    \n                  .","journal-title":"Front Genet"},{"key":"2973_CR26","doi-asserted-by":"publisher","unstructured":"Pavlopoulos GA. How to cluster protein sequences: tools, tips and commands. MOJ Proteomics Bioinform. 2017;5(5). \n                    https:\/\/doi.org\/10.15406\/mojpb.2017.05.00174\n                    \n                  .","DOI":"10.15406\/mojpb.2017.05.00174"},{"key":"2973_CR27","doi-asserted-by":"publisher","unstructured":"Vialle RA, Pedrosa FO, Weiss VA, Guizelini D, Tibaes JH, Marchaukoski JN, Raittz RT. RAFTS3: rapid alignment-free tool for sequence similarity search. bioRxiv. 2016;55269. \n                    https:\/\/doi.org\/10.1101\/055269\n                    \n                  .","DOI":"10.1101\/055269"},{"key":"2973_CR28","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-59745-440-7","volume-title":"Bioinformatics for Systems Biology","year":"2009","unstructured":"Krawetz S. Bioinformatics for systems biology. Cap. 27 Clustering algorithms, vol. 9781597454407: Humana Press; 2009. ISBN 978-1-59745-440-7. \n                    https:\/\/doi.org\/10.1007\/978-1-59745-440-7\n                    \n                  ."},{"issue":"D1","key":"2973_CR29","doi-asserted-by":"publisher","first-page":"D23","DOI":"10.1093\/nar\/gky1069","volume":"47","author":"A Marchler-Bauer","year":"2018","unstructured":"Marchler-Bauer A, Schoch CL, Canese K, Schneider VA, Hefferon T, Bolton EE, Kimchi A. Database resources of the National Center for biotechnology information. Nucleic Acids Res. 2018;47(D1):D23\u20138. \n                    https:\/\/doi.org\/10.1093\/nar\/gky1069\n                    \n                  .","journal-title":"Nucleic Acids Res"},{"issue":"1","key":"2973_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/gb-2006-7-1-r8","volume":"7","author":"SD Brown","year":"2006","unstructured":"Brown SD, Gerlt JA, Seffernick JL, Babbitt P. A gold standard set of mechanistically diverse enzyme superfamilies. Genome Biol. 2006;7(1):1\u201315. \n                    https:\/\/doi.org\/10.1186\/gb-2006-7-1-r8\n                    \n                  .","journal-title":"Genome Biol"},{"issue":"1","key":"2973_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/s41598-018-29246-4","volume":"8","author":"M Xiong","year":"2018","unstructured":"Xiong M, Liu X, Hao M, Li Y, Shugart YY, Qiao C, et al. Nuclear norm clustering: a promising alternative method for clustering tasks. Sci Rep. 2018;8(1):1\u20137. \n                    https:\/\/doi.org\/10.1038\/s41598-018-29246-4\n                    \n                  .","journal-title":"Sci Rep"},{"key":"2973_CR32","doi-asserted-by":"publisher","unstructured":"Bernardes JS, Vieira FRJ, Costa LMM, Zaverucha G. Evaluation and improvements of clustering algorithms for detecting remote homologous protein families. BMC Bioinformatics. 2015;16(1). \n                    https:\/\/doi.org\/10.1186\/s12859-014-0445-4\n                    \n                  .","DOI":"10.1186\/s12859-014-0445-4"},{"issue":"D1","key":"2973_CR33","doi-asserted-by":"publisher","first-page":"D304","DOI":"10.1093\/nar\/gkt1240","volume":"42","author":"Naomi K. Fox","year":"2013","unstructured":"Fox NK, Brenner SE, Chandonia JM. SCOPe: structural classification of proteins - extended, integrating SCOP and ASTRAL data and classification of new structures. Nucleic Acids Res. 2014;42(D1). \n                    https:\/\/doi.org\/10.1093\/nar\/gkt1240\n                    \n                  .","journal-title":"Nucleic Acids Research"}],"container-title":["BMC Bioinformatics"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s12859-019-2973-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1186\/s12859-019-2973-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s12859-019-2973-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,7,13]],"date-time":"2020-07-13T19:07:35Z","timestamp":1594667255000},"score":1,"resource":{"primary":{"URL":"https:\/\/bmcbioinformatics.biomedcentral.com\/articles\/10.1186\/s12859-019-2973-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,7,15]]},"references-count":33,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["2973"],"URL":"https:\/\/doi.org\/10.1186\/s12859-019-2973-4","relation":{"has-preprint":[{"id-type":"doi","id":"10.1101\/407437","asserted-by":"object"}]},"ISSN":["1471-2105"],"issn-type":[{"value":"1471-2105","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,7,15]]},"assertion":[{"value":"16 December 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 June 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 July 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Not applicable.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare that they have no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"392"}}