{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,29]],"date-time":"2026-08-29T09:51:00Z","timestamp":1787997060109,"version":"build-2784847793"},"reference-count":10,"publisher":"Oxford University Press (OUP)","issue":"2","license":[{"start":{"date-parts":[[2018,6,15]],"date-time":"2018-06-15T00:00:00Z","timestamp":1529020800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000051","name":"National Human Genome Research Institute","doi-asserted-by":"publisher","award":["UM1 HG009443"],"award-info":[{"award-number":["UM1 HG009443"]}],"id":[{"id":"10.13039\/100000051","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,1,15]]},"abstract":"<jats:title>Abstract<\/jats:title>\n               <jats:sec>\n                  <jats:title>Motivation<\/jats:title>\n                  <jats:p>Long-read, single-molecule sequencing platforms hold great potential for isoform discovery and characterization of multi-exon transcripts. However, their high error rates are an obstacle to distinguishing novel transcript isoforms from sequencing artifacts. Therefore, we developed the package TranscriptClean to correct mismatches, microindels and noncanonical splice junctions in mapped transcripts using the reference genome while preserving known variants.<\/jats:p>\n               <\/jats:sec>\n               <jats:sec>\n                  <jats:title>Results<\/jats:title>\n                  <jats:p>Our method corrects nearly all mismatches and indels present in a publically available human PacBio Iso-seq dataset, and rescues 39% of noncanonical splice junctions.<\/jats:p>\n               <\/jats:sec>\n               <jats:sec>\n                  <jats:title>Availability and implementation<\/jats:title>\n                  <jats:p>All Python and R scripts used in this paper are available at https:\/\/github.com\/dewyman\/TranscriptClean.<\/jats:p>\n               <\/jats:sec>","DOI":"10.1093\/bioinformatics\/bty483","type":"journal-article","created":{"date-parts":[[2018,6,13]],"date-time":"2018-06-13T11:13:21Z","timestamp":1528888401000},"page":"340-342","source":"Crossref","is-referenced-by-count":74,"title":["TranscriptClean: variant-aware correction of indels, mismatches and splice junctions in long-read transcripts"],"prefix":"10.1093","volume":"35","author":[{"given":"Dana","family":"Wyman","sequence":"first","affiliation":[{"name":"Department of Developmental and Cell Biology, UC Irvine, Irvine, CA, USA"},{"name":"Center for Complex Biological Systems, UC Irvine, Irvine, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ali","family":"Mortazavi","sequence":"additional","affiliation":[{"name":"Department of Developmental and Cell Biology, UC Irvine, Irvine, CA, USA"},{"name":"Center for Complex Biological Systems, UC Irvine, Irvine, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"286","published-online":{"date-parts":[[2018,6,15]]},"reference":[{"key":"2023013107231289400_bty483-B1","doi-asserted-by":"crossref","first-page":"11706","DOI":"10.1038\/ncomms11706","article-title":"A survey of the sorghum transcriptome using single-molecule long reads","volume":"7","author":"Abdel-Ghany","year":"2016","journal-title":"Nat. Commun."},{"key":"2023013107231289400_bty483-B2","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1186\/s13059-016-0881-8","article-title":"A survey of best practices for RNA-seq data analysis","volume":"17","author":"Conesa","year":"2016","journal-title":"Genome Biol."},{"key":"2023013107231289400_bty483-B3","doi-asserted-by":"crossref","first-page":"15","DOI":"10.1093\/bioinformatics\/bts635","article-title":"STAR: ultrafast universal RNA-seq aligner","volume":"29","author":"Dobin","year":"2013","journal-title":"Bioinformatics (Oxford, England)"},{"key":"2023013107231289400_bty483-B4","doi-asserted-by":"crossref","first-page":"133","DOI":"10.1126\/science.1162986","article-title":"Real-time DNA sequencing from single polymerase molecules","volume":"323","author":"Eid","year":"2009","journal-title":"Science"},{"key":"2023013107231289400_bty483-B5","doi-asserted-by":"crossref","first-page":"e0132628","DOI":"10.1371\/journal.pone.0132628","article-title":"Widespread polycistronic transcripts in fungi revealed by single-molecule mRNA sequencing","volume":"10","author":"Gordon","year":"2015","journal-title":"PLoS One"},{"key":"2023013107231289400_bty483-B6","doi-asserted-by":"crossref","first-page":"108","DOI":"10.1109\/TNB.2017.2675981","article-title":"HapIso: an accurate method for the haplotype-specific isoforms reconstruction from long single-molecule reads","volume":"16","author":"Mangul","year":"2017","journal-title":"IEEE Trans. NanoBioscience"},{"key":"2023013107231289400_bty483-B7","doi-asserted-by":"crossref","first-page":"10564","DOI":"10.1093\/nar\/gku744","article-title":"A comprehensive survey of non-canonical splice sites in the human transcriptome","volume":"42","author":"Parada","year":"2014","journal-title":"Nucleic Acids Res."},{"key":"2023013107231289400_bty483-B8","doi-asserted-by":"crossref","first-page":"278","DOI":"10.1016\/j.gpb.2015.08.002","article-title":"PacBio sequencing and its applications","volume":"13","author":"Rhoads","year":"2015","journal-title":"Genomics Proteomics Bioinf"},{"key":"2023013107231289400_bty483-B9","doi-asserted-by":"crossref","first-page":"396","DOI":"10.1101\/gr.222976.117","article-title":"SQANTI: extensive characterization of long-read transcript sequences for quality control in full-length transcriptome identification and quantification","volume":"28","author":"Tardaguila","year":"2018","journal-title":"Genome Res"},{"key":"2023013107231289400_bty483-B10","doi-asserted-by":"crossref","first-page":"9869","DOI":"10.1073\/pnas.1400447111","article-title":"Defining a personal, allele-specific, and single-molecule long-read transcriptome","volume":"111","author":"Tilgner","year":"2014","journal-title":"Proc. Natl. Acad. Sci. USA"}],"container-title":["Bioinformatics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/academic.oup.com\/bioinformatics\/article-pdf\/35\/2\/340\/48962820\/bioinformatics_35_2_340.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/academic.oup.com\/bioinformatics\/article-pdf\/35\/2\/340\/48962820\/bioinformatics_35_2_340.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,31]],"date-time":"2023-01-31T10:07:32Z","timestamp":1675159652000},"score":1,"resource":{"primary":{"URL":"https:\/\/academic.oup.com\/bioinformatics\/article\/35\/2\/340\/5038460"}},"subtitle":[],"editor":[{"given":"Bonnie","family":"Berger","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"editor"}]}],"short-title":[],"issued":{"date-parts":[[2018,6,15]]},"references-count":10,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2019,1,15]]}},"URL":"https:\/\/doi.org\/10.1093\/bioinformatics\/bty483","relation":{},"ISSN":["1367-4803","1367-4811"],"issn-type":[{"value":"1367-4803","type":"print"},{"value":"1367-4811","type":"electronic"}],"subject":[],"published-other":{"date-parts":[[2019,1,15]]},"published":{"date-parts":[[2018,6,15]]}}}