{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T13:46:29Z","timestamp":1762955189285,"version":"3.45.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,24]]},"DOI":"10.1145\/3721201.3721413","type":"proceedings-article","created":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T13:35:21Z","timestamp":1762954521000},"page":"289-293","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Optimizing Veterinary Language Modeling with Mamba Architecture for Long-Sequence Efficiency"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-3593-9598","authenticated-orcid":false,"given":"Lakshmi Priya","family":"Ramisetty","sequence":"first","affiliation":[{"name":"Yeshiva University, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0074-0979","authenticated-orcid":false,"given":"Youshan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Yeshiva University, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"State space models as foundation models: A control theoretic overview. arXiv preprint arXiv:2403.16899","author":"Alonso Carmen Amo","year":"2024","unstructured":"Carmen Amo Alonso, Jerome Sieber, and Melanie N Zeilinger. 2024. State space models as foundation models: A control theoretic overview. arXiv preprint arXiv:2403.16899 (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150","author":"Beltagy Iz","year":"2020","unstructured":"Iz Beltagy, Matthew E Peters, and Arman Cohan. 2020. Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150 (2020)."},{"key":"e_1_3_2_1_3_1","volume-title":"Language models are few-shot learners. arXiv preprint arXiv:2005.14165","author":"Brown Tom B","year":"2020","unstructured":"Tom B Brown. 2020. Language models are few-shot learners. arXiv preprint arXiv:2005.14165 (2020)."},{"key":"e_1_3_2_1_4_1","volume-title":"Peter-John Noble Mantyl\u00e4, and Noura Al Moubayed","author":"Farrell Sean","year":"2023","unstructured":"Sean Farrell, Charlotte Appleton, Peter-John Noble Mantyl\u00e4, and Noura Al Moubayed. 2023. PetBERT: Automated ICD-11 syndromic disease coding for outbreak detection in first opinion veterinary electronic health records. Scientific Reports 13 (2023). https:\/\/api.semanticscholar.org\/CorpusID:264406461"},{"key":"e_1_3_2_1_5_1","volume-title":"Digital control of dynamic systems","author":"Franklin Gene F","unstructured":"Gene F Franklin, J David Powell, and Michael L Workman. 1997. Digital control of dynamic systems. Addison-Wesley Longman Publishing Co., Inc."},{"key":"e_1_3_2_1_6_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)."},{"key":"e_1_3_2_1_7_1","volume-title":"Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv:2111.00396","author":"Gu Albert","year":"2021","unstructured":"Albert Gu, Karan Goel, and Christopher R\u00e9. 2021. Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv:2111.00396 (2021)."},{"key":"e_1_3_2_1_8_1","first-page":"1474","article-title":"HiPPO: Recurrent Memory with Optimal Polynomial Projections","volume":"34","author":"Gu Albert","year":"2021","unstructured":"Albert Gu, Krishan Goel, and Christopher Re. 2021. HiPPO: Recurrent Memory with Optimal Polynomial Projections. In Advances in Neural Information Processing Systems, Vol. 34. 1474\u20131487. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/ba5294a2f5af54161f800eb291c0af20-Abstract.html","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458754"},{"key":"e_1_3_2_1_10_1","unstructured":"TinyAI Initiative. 2023. TinyLlama: Efficient Language Modeling with Compact Architectures. https:\/\/github.com\/TinyAI\/TinyLlama."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Ming-Wei Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei Chang Kenton and Lee Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of naacL-HLT, Vol. 1. Minneapolis, Minnesota, 2."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btz682"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"e_1_3_2_1_14_1","unstructured":"Pinxue Lin Sayed Raheel Hussain Tirupathi Kadari Varun Biyyala Sakshi Bennur and Jainam Bhansal. [n. d.]. VetMedGPT: Generative Pre-trained Transformer for Veteri-nary Medicine Healthcare. ([n. d.])."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.447"},{"key":"e_1_3_2_1_16_1","volume-title":"SGDR: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983","author":"Loshchilov Ilya","year":"2016","unstructured":"Ilya Loshchilov and Frank Hutter. 2016. SGDR: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)."},{"key":"e_1_3_2_1_17_1","volume-title":"Mixed precision training. arXiv preprint arXiv:1710.03740","author":"Micikevicius Paulius","year":"2018","unstructured":"Paulius Micikevicius, Sharan Narang, Jonah Alben, Gregory Diamos, Erich Garcia, Boris Ginsburg, Michael Houston, Oleksii Kuchaiev, Ganesh Venkatesh, and Huan Wu. 2018. Mixed precision training. arXiv preprint arXiv:1710.03740 (2018)."},{"key":"e_1_3_2_1_18_1","volume-title":"Mistral 7B: A Highly Efficient Dense Language Model. arXiv preprint arXiv:2310.08481","author":"Mistral Team","year":"2023","unstructured":"Team Mistral. 2023. Mistral 7B: A Highly Efficient Dense Language Model. arXiv preprint arXiv:2310.08481 (2023). https:\/\/arxiv.org\/abs\/2310.08481"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the ACM International Conference on High Performance Computing, Networking, Storage and Analysis (SC'20)","author":"Rasley Jeff","year":"2020","unstructured":"Jeff Rasley, Samyam Rajbhandari, Olatunji Ruwase, and Yuxiong He. 2020. Deep-speed: System optimizations enable training deep learning models with over 100 billion parameters. Proceedings of the ACM International Conference on High Performance Computing, Networking, Storage and Analysis (SC'20) (2020)."},{"key":"e_1_3_2_1_20_1","volume-title":"Falcon: An Open-Source Family of Language Models. arXiv preprint arXiv:2306.01116","author":"RefinedWeb TII","year":"2023","unstructured":"TII and RefinedWeb. 2023. Falcon: An Open-Source Family of Language Models. arXiv preprint arXiv:2306.01116 (2023). https:\/\/arxiv.org\/abs\/2306.01116"},{"key":"e_1_3_2_1_21_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Sergey Bashlykov Siddhartha Batra Akhilesh Bhargava Shruti Bhosale et al. 2023. LLaMA 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv:2307.09288 (2023). https:\/\/arxiv.org\/abs\/2307.09288"},{"key":"e_1_3_2_1_22_1","volume-title":"(Nips)","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention Is All You Need.(Nips), 2017. arXiv preprint arXiv:1706.03762 10 (2017), S0140525X16001837."},{"key":"e_1_3_2_1_23_1","volume-title":"Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768","author":"Wang Sinong","year":"2020","unstructured":"Sinong Wang, Belinda Z Li, Madian Khabsa, Han Fang, and Hao Ma. 2020. Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)."},{"key":"e_1_3_2_1_24_1","unstructured":"Maurice Weber Daniel Fu Quentin Anthony Yonatan Oren Shane Adams Anton Alexandrov Xiaozhong Lyu Huu Nguyen Xiaozhe Yao Virginia Adams et al. 2024. Redpajama: an open dataset for training large language models. arXiv preprint arXiv:2411.12372 (2024)."}],"event":{"name":"CHASE '25: ACM\/IEEE International Conference on Connected Health: Applications, Systems and Engineering Technologies","location":"Yeshiva University Museum New York NY USA","acronym":"CHASE '25","sponsor":["SIGBED ACM Special Interest Group on Embedded Systems","IEEE Computer Society"]},"container-title":["Proceedings of the ACM\/IEEE International Conference on Connected Health: Applications, Systems and Engineering Technologies"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3721201.3721413","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T13:37:24Z","timestamp":1762954644000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721201.3721413"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,24]]},"references-count":24,"alternative-id":["10.1145\/3721201.3721413","10.1145\/3721201"],"URL":"https:\/\/doi.org\/10.1145\/3721201.3721413","relation":{},"subject":[],"published":{"date-parts":[[2025,6,24]]},"assertion":[{"value":"2025-11-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}