{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T08:04:14Z","timestamp":1784189054160,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":24,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234165","type":"print"},{"value":"9789819234172","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3417-2_25","type":"book-chapter","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:11:05Z","timestamp":1784185865000},"page":"286-298","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["HVM: A HV-Attention Based Multi-modal Framework for Mathematical Expression Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0443-9917","authenticated-orcid":false,"given":"Linnan","family":"Jiang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4802-3191","authenticated-orcid":false,"given":"Fei","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4936-269X","authenticated-orcid":false,"given":"JiaoJiao","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1413-6024","authenticated-orcid":false,"given":"WeiXing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"25_CR1","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1145\/958220.958239","volume-title":"Proceedings of the 2003 ACM Symposium on Document Engineering","author":"M Suzuki","year":"2003","unstructured":"Suzuki, M., Tamari, F., Fukuda, R., Uchida, S., Kanahori, T.: Infty: an integrated OCR system for mathematical documents. In: Proceedings of the 2003 ACM Symposium on Document Engineering, pp. 95\u2013104 (2003)"},{"key":"25_CR2","volume-title":"Optical Character Recognition","author":"S Mori","year":"1999","unstructured":"Mori, S., Nishida, H., Yamada, H.: Optical Character Recognition. John Wiley & Sons, Inc. (1999)"},{"key":"25_CR3","first-page":"980","volume-title":"International Conference on Machine Learning","author":"Y Deng","year":"2017","unstructured":"Deng, Y., Kanervisto, A., Ling, J., Rush, A.M.: Image-to-markup generation with coarse-to-fine attention. In: International Conference on Machine Learning, pp. 980\u2013989. PMLR (2017)"},{"key":"25_CR4","first-page":"1","volume-title":"2022 International Conference on Multimedia Analysis and Pattern Recognition (MAPR)","author":"VX Vu","year":"2022","unstructured":"Vu, V.X., Bui, T.N., Le, T.L., Phong, B.H., Hoang, M.T.: Transformer-based method for mathematical expression recognition in document images. In: 2022 International Conference on Multimedia Analysis and Pattern Recognition (MAPR), pp. 1\u20136. IEEE (2022)"},{"key":"25_CR5","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Proces. Syst. 30, 5998\u20136008 (2017)"},{"key":"25_CR6","first-page":"784","volume-title":"Proceedings of the Fifteenth National Conference on Artificial Intelligence and Tenth Innovative Applications of Artificial Intelligence Conference, AAAI 98, IAAI 98, July 26\u201330, 1998, Madison, Wisconsin, USA, Jack Mostow and Chuck Rich, Eds","author":"EG Miller","year":"1998","unstructured":"Miller, E.G., Viola, P.A.: Ambiguity and constraint in mathematical expression recognition. In: Proceedings of the Fifteenth National Conference on Artificial Intelligence and Tenth Innovative Applications of Artificial Intelligence Conference, AAAI 98, IAAI 98, July 26\u201330, 1998, Madison, Wisconsin, USA, Jack Mostow and Chuck Rich, Eds, pp. 784\u2013791. AAAI Press \/ The MIT Press (1998)"},{"issue":"4","key":"25_CR7","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/s10032-011-0174-4","volume":"15","author":"R Zanibbi","year":"2012","unstructured":"Zanibbi, R., Blostein, D.: Recognition and retrieval of mathematical expressions. Int. J. Document Anal. Recognit. 15(4), 331\u2013357 (2012)","journal-title":"Int. J. Document Anal. Recognit."},{"key":"25_CR8","unstructured":"Singh, S.S.: \u201cTeaching machines to code: neural markup generation with visual attention,\u201d arXiv preprint arXiv:1802.05415, 2018."},{"key":"25_CR9","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1145\/3330393.3330410","volume-title":"Proceedings of the 2019 4th International Conference on Multimedia Systems and Signal Processing","author":"W Zhang","year":"2019","unstructured":"Zhang, W., Bai, Z.Q., Zhu, Y.S.: An improved approach based on cnn-rnns for mathematical expression recognition. In: Proceedings of the 2019 4th International Conference on Multimedia Systems and Signal Processing, pp. 57\u201361 (2019)"},{"key":"25_CR10","first-page":"1","volume-title":"2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA)","author":"AD Le","year":"2022","unstructured":"Le, A.D., Pham, V.L., Ly, V.L., Nguyen, N.Q., Nguyen, H.T., Tran, T.A.: A hybrid vision transformer approach for mathematical expression recognition. In: 2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA), pp. 1\u20137. IEEE (2022)"},{"key":"25_CR11","unstructured":"Blecher, L., Cucurull, G., Scialom, T., Stojnic, R.: Nougat: neural optical understanding for academic documents. arXiv preprint arXiv:2308.13418. (2023)"},{"key":"25_CR12","unstructured":"Wang, B., et al.: Unimernet: a universal network for real-world mathematical expression recognition. arXiv preprint arXiv:2404.15254. (2024)"},{"issue":"1","key":"25_CR13","doi-asserted-by":"publisher","first-page":"177","DOI":"10.3390\/math11010177","volume":"11","author":"ML Zhou","year":"2022","unstructured":"Zhou, M.L., Cai, M., Li, G., Li, M.: An end-to-end formula recognition method integrated attention mechanism. Mathematics. 11(1), 177 (2022)","journal-title":"Mathematics"},{"key":"25_CR14","first-page":"5583","volume-title":"International Conference on Machine Learning","author":"W Kim","year":"2021","unstructured":"Kim, W., Son, B., Kim, I.: Vilt: vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp. 5583\u20135594. PMLR (2021)"},{"key":"25_CR15","unstructured":"Lu, J.S., et al.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inf. Proces. Syst. 32,13\u201323 (2019)"},{"key":"25_CR16","first-page":"4904","volume-title":"International Conference on Machine Learning","author":"C Jia","year":"2021","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"25_CR17","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"25_CR18","first-page":"9694","volume":"34","author":"JN Li","year":"2021","unstructured":"Li, J.N., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C.M., Hoi, S.C.H.: Align before fuse: vision and language representation learning with momentum distillation. Adv. Neural Inf. Proces. Syst. 34, 9694\u20139705 (2021)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"25_CR19","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural Inf. Proces. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"25_CR20","first-page":"12888","volume-title":"International Conference on Machine Learning","author":"JN Li","year":"2022","unstructured":"Li, J.N., Li, D.X., Xiong, C.M., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"25_CR21","first-page":"19730","volume-title":"International Conference on Machine Learning","author":"JN Li","year":"2023","unstructured":"Li, J.N., Li, D.X., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"25_CR22","doi-asserted-by":"crossref","unstructured":"Gao, T.Y., Yao, X.C., Chen, D.Q.: Simcse: simple contrastive learning of sentence embeddings. arXiv preprint arXiv:2104.08821. (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.552"},{"key":"25_CR23","first-page":"30392","volume":"34","author":"TT Xiao","year":"2021","unstructured":"Xiao, T.T., Singh, M., Mintun, E., Darrell, T., Doll\u00e1r, P., Girshick, R.: Early convolutions help transformers see better. Adv. Neural Inf. Proces. Syst. 34, 30392\u201330400 (2021)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"25_CR24","first-page":"12175","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"JY Guo","year":"2022","unstructured":"Guo, J.Y., et al.: Cmt: convolutional neural networks meet vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12175\u201312185 (2022)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3417-2_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:11:10Z","timestamp":1784185870000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3417-2_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,17]]},"ISBN":["9789819234165","9789819234172"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3417-2_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,17]]},"assertion":[{"value":"17 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}