{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T14:07:28Z","timestamp":1749564448258,"version":"3.37.3"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T00:00:00Z","timestamp":1694995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T00:00:00Z","timestamp":1694995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"The BUPT innovation and entrepreneurship support program","award":["2022-YC-S002"],"award-info":[{"award-number":["2022-YC-S002"]}]},{"name":"The Beijing Key Laboratory of Work Safety and Intelligent Monitoring Foundation"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16626-w","type":"journal-article","created":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T04:01:41Z","timestamp":1695009701000},"page":"31527-31543","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Multi-modal hierarchical fusion network for fine-grained paper classification"],"prefix":"10.1007","volume":"83","author":[{"given":"Tan","family":"Yue","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiedong","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8362-6353","authenticated-orcid":false,"given":"Zonghai","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,18]]},"reference":[{"key":"16626_CR1","doi-asserted-by":"publisher","unstructured":"Chen T, Guestrin C (2016) Xgboost. Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining. https:\/\/doi.org\/10.1145\/2939672.2939785","DOI":"10.1145\/2939672.2939785"},{"key":"16626_CR2","unstructured":"Ke G, Meng Q, Finley T, Wang T, Chen W, Ma W, Ye Q, Liu T-Y (2017) Lightgbm: A highly efficient gradient boosting decision tree. In:Guyon I, Luxburg UV, Bengio S, Wallach H, Fergus R, Vishwanathan S, Garnett R (eds.) Advances in Neural Information Processing Systems, vol. 30"},{"key":"16626_CR3","doi-asserted-by":"publisher","unstructured":"Dhaliwal SS, Nahid A-A, Abbas R (2018) Effective intrusion detection system using xgboost. Inf 9(7). https:\/\/doi.org\/10.3390\/info9070149","DOI":"10.3390\/info9070149"},{"key":"16626_CR4","doi-asserted-by":"publisher","unstructured":"Yue T, Li Y, Hu Z (2021) Dwsa: An intelligent document structural analysis model for information extraction and data mining. Electron 10(19). https:\/\/doi.org\/10.3390\/electronics10192443","DOI":"10.3390\/electronics10192443"},{"key":"16626_CR5","doi-asserted-by":"publisher","first-page":"79887","DOI":"10.1109\/ACCESS.2019.2923293","volume":"7","author":"X Ma","year":"2019","unstructured":"Ma X, Wang R (2019) Personalized scientific paper recommendation based on heterogeneous graph representation. IEEE Access 7:79887\u201379894. https:\/\/doi.org\/10.1109\/ACCESS.2019.2923293","journal-title":"IEEE Access"},{"key":"16626_CR6","unstructured":"Adhikari A, Ram A, Tang R, Lin J (2019) DocBERT: BERT for Document Classification"},{"key":"16626_CR7","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/978-3-319-13296-9_4","volume-title":"New Horizons in Web Based Learning","author":"J Quan","year":"2014","unstructured":"Quan J, Li Q, Li M (2014) Computer science paper classification for csar. In: Cao Y, V\u00e4ljataga T, Tang JKT, Leung H, Laanpere M (eds) New Horizons in Web Based Learning. Springer, Cham, pp 34\u201343"},{"issue":"5","key":"16626_CR8","doi-asserted-by":"publisher","first-page":"5709","DOI":"10.3233\/JIFS-213022","volume":"43","author":"T Yue","year":"2022","unstructured":"Yue T, He Z, Li C, Hu Z, Li Y (2022) Lightweight fine-grained classification for scientific paper. J Intell Fuzzy Syst 43(5):5709\u20135719","journal-title":"J Intell Fuzzy Syst"},{"key":"16626_CR9","doi-asserted-by":"publisher","unstructured":"Shi C, Quan J, Li M (2013) Information extraction for computer science academic rankings system. In: 2013 International Conference on Cloud and Service Computing, pp. 69\u201376. https:\/\/doi.org\/10.1109\/CSC.2013.19","DOI":"10.1109\/CSC.2013.19"},{"key":"16626_CR10","doi-asserted-by":"publisher","unstructured":"Schifanella R, de Juan P, Tetreault J, Cao L (2016) Detecting sarcasm in multimodal social platforms. In: Proceedings of the 24th ACM International Conference on Multimedia. MM \u201916. Association for Computing Machinery, New York, NY, USA, pp. 1136\u20131145. https:\/\/doi.org\/10.1145\/2964284.2964321","DOI":"10.1145\/2964284.2964321"},{"key":"16626_CR11","unstructured":"Li LH, Yatskar M, Yin D, Hsieh C, Chang K (2019) Visualbert: A simple and performant baseline for vision and language. CoRR abs\/1908.03557 arXiv:1908.03557"},{"key":"16626_CR12","doi-asserted-by":"publisher","unstructured":"van Aken, B, Winter B, L\u00f6ser A, Gers FA (2020) Visbert: Hidden-state visualizations for transformers. https:\/\/doi.org\/10.48550\/ARXIV.2011.04507","DOI":"10.48550\/ARXIV.2011.04507"},{"key":"16626_CR13","doi-asserted-by":"publisher","unstructured":"Tan H, Bansal M (2019) LXMERT: Learning Cross-Modality Encoder Representations from Transformers. https:\/\/doi.org\/10.48550\/ARXIV.1908.07490. arXiv:1908.07490","DOI":"10.48550\/ARXIV.1908.07490"},{"key":"16626_CR14","doi-asserted-by":"publisher","unstructured":"Su W, Zhu X, Cao Y, Li B, Lu L, Wei F, Dai J (2019) VL-BERT: Pre-training of Generic Visual-Linguistic Representations. https:\/\/doi.org\/10.48550\/ARXIV.1908.08530. arXiv:1908.08530","DOI":"10.48550\/ARXIV.1908.08530"},{"key":"16626_CR15","doi-asserted-by":"crossref","unstructured":"Chen Y-C, Li L, Yu L, Kholy AE, Ahmed F, Gan Z, Cheng Y, Liu J (2020) Uniter: Universal image-text representation learning. In: ECCV","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"16626_CR16","doi-asserted-by":"crossref","unstructured":"Lan Z, Chen M, Goodman S, Gimpel K, Sharma P, Soricut R (2020) Albert: A lite bert for self-supervised learning of language representations. arXiv:1909.11942","DOI":"10.1109\/SLT48900.2021.9383575"},{"key":"16626_CR17","doi-asserted-by":"crossref","unstructured":"Cadene R, Ben-younes H, Cord M, Thome N (2019) Murel: Multimodal relational reasoning for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR.2019.00209"},{"issue":"05","key":"16626_CR18","doi-asserted-by":"publisher","first-page":"9749","DOI":"10.1609\/aaai.v34i05.6525","volume":"34","author":"J Zhu","year":"2020","unstructured":"Zhu J, Zhou Y, Zhang J, Li H, Zong C, Li C (2020) Multimodal summarization with guidance of multimodal reference. Proc AAAI Conf Art Intell 34(05):9749\u20139756. https:\/\/doi.org\/10.1609\/aaai.v34i05.6525","journal-title":"Proc AAAI Conf Art Intell"},{"issue":"2","key":"16626_CR19","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1109\/TMM.2015.2510329","volume":"18","author":"S Qian","year":"2016","unstructured":"Qian S, Zhang T, Xu C, Shao J (2016) Multi-modal event topic model for social event analysis. IEEE Trans Multimed 18(2):233\u2013246. https:\/\/doi.org\/10.1109\/TMM.2015.2510329","journal-title":"IEEE Trans Multimed"},{"issue":"8","key":"16626_CR20","doi-asserted-by":"publisher","first-page":"3748","DOI":"10.1109\/TIP.2016.2639438","volume":"26","author":"Y Xia","year":"2017","unstructured":"Xia Y, Zhang L, Liu Z, Nie L, Li X (2017) Weakly supervised multimodal kernel for categorizing aerial photographs. IEEE Trans Image Process 26(8):3748\u20133758. https:\/\/doi.org\/10.1109\/TIP.2016.2639438","journal-title":"IEEE Trans Image Process"},{"key":"16626_CR21","doi-asserted-by":"crossref","unstructured":"Zadeh A, Chen M, Poria S, Cambria E, Morency L-P (2017) Tensor fusion network for multimodal sentiment analysis. In: Empirical Methods in Natural Language Processing, EMNLP","DOI":"10.18653\/v1\/D17-1115"},{"key":"16626_CR22","first-page":"1575","volume":"35","author":"X Hu","year":"2021","unstructured":"Hu X, Yin X, Lin K, Zhang L, Gao J, Wang L, Liu Z (2021) Vivo: Visual vocabulary pre-training for novel object captioning. Proc AAAI Conf Art Intell 35:1575\u20131583","journal-title":"Proc AAAI Conf Art Intell"},{"key":"16626_CR23","doi-asserted-by":"crossref","unstructured":"Malik M, Tom\u00e1s D, Rosso P (2023) How challenging is multimodal irony detection? In: International Conference on Applications of Natural Language to Information Systems. pp. 18\u201332","DOI":"10.1007\/978-3-031-35320-8_2"},{"key":"16626_CR24","doi-asserted-by":"publisher","unstructured":"Lecun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521:436\u2013444. https:\/\/doi.org\/10.1038\/nature14539","DOI":"10.1038\/nature14539"},{"key":"16626_CR25","doi-asserted-by":"crossref","unstructured":"Peters ME, Neumann M, Iyyer M, Gardner M, Clark C, Lee K, Zettlemoyer L (2018) Deep contextualized word representations. arXiv:1802.05365","DOI":"10.18653\/v1\/N18-1202"},{"key":"16626_CR26","unstructured":"Brown TB, Mann B, Ryder N, Subbiah M, Kaplan J, Dhariwal P, Neelakantan A, Shyam P, Sastry G, Askell A, Agarwal S, Herbert-Voss A, Krueger G, Henighan T, Child R, Ramesh A, Ziegler DM, Wu J, Winter C, Hesse C, Chen M, Sigler E, Litwin M, Gray S, Chess B, Clark J, Berner C, McCandlish S, Radford, A, Sutskever I, Amodei D (2020) Language Models are Few-Shot Learners"},{"key":"16626_CR27","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805"},{"key":"16626_CR28","unstructured":"Touvron H, Lavril T, Izacard G, Martinet X, Lachaux M-A, Lacroix T, Rozi\u00e8re B, Goyal N, Hambro E, Azhar F, Rodriguez A, Joulin A, Grave E, Lample G (2023) LLaMA: Open and Efficient Foundation Language Models"},{"key":"16626_CR29","doi-asserted-by":"crossref","unstructured":"Gallo I, Calefati A, Nawaz S, Janjua MK (2018) Image and encoded text fusion for multi-modal classification. In: 2018 Digital Image Computing: Techniques and Applications (DICTA). IEEE, pp. 1\u20137","DOI":"10.1109\/DICTA.2018.8615789"},{"key":"16626_CR30","doi-asserted-by":"crossref","unstructured":"Gallo I, Calefati A, Nawaz S (2017) Multimodal classification fusion in real-world scenarios. In: 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), vol. 5. pp. 36\u201341. IEEE","DOI":"10.1109\/ICDAR.2017.326"},{"key":"16626_CR31","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25:1097\u20131105","journal-title":"Adv Neural Inf Process Syst"},{"key":"16626_CR32","doi-asserted-by":"crossref","unstructured":"Kim Y (2014) Convolutional Neural Networks for Sentence Classification","DOI":"10.3115\/v1\/D14-1181"},{"key":"16626_CR33","doi-asserted-by":"crossref","unstructured":"Morvant E, Habrard A, Ayache S (2014) Majority vote of diverse classifiers for late fusion. In: Joint IAPR International Workshops on Statistical Techniques in Pattern Recognition (SPR) and Structural and Syntactic Pattern Recognition (SSPR). Springer, pp. 153\u2013162","DOI":"10.1007\/978-3-662-44415-3_16"},{"key":"16626_CR34","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"16626_CR35","doi-asserted-by":"crossref","unstructured":"Sandler M, Howard A, Zhu M, Zhmoginov A, Chen L-C (2018) Mobilenetv2: Inverted residuals and linear bottlenecks. In:Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4510\u20134520","DOI":"10.1109\/CVPR.2018.00474"},{"key":"16626_CR36","first-page":"1383","volume":"2020","author":"H Pan","year":"2020","unstructured":"Pan H, Lin Z, Fu P, Qi Y, Wang W (2020) Modeling intra and inter-modality incongruity for multi-modal sarcasm detection. Findings of the Association for Computational Linguistics: EMNLP 2020:1383\u20131392","journal-title":"Findings of the Association for Computational Linguistics: EMNLP"},{"key":"16626_CR37","doi-asserted-by":"crossref","unstructured":"Liu Z, Mao H, Wu C-Y, Feichtenhofer C, Darrell T, Xie S (2022) A convnet for the 2020s. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR52688.2022.01167"},{"issue":"6","key":"16626_CR38","doi-asserted-by":"publisher","first-page":"7399","DOI":"10.1007\/s12652-022-04447-y","volume":"14","author":"D Tom\u00e1s","year":"2023","unstructured":"Tom\u00e1s D, Ortega-Bueno R, Zhang G, Rosso P, Schifanella R (2023) Transformer-based models for multimodal irony detection. J Ambient Intell Human Comput 14(6):7399\u20137410","journal-title":"J Ambient Intell Human Comput"},{"key":"16626_CR39","unstructured":"Kipf TN, Welling M (2017) Semi-supervised classification with graph convolutional networks. In: International Conference on Learning Representations (ICLR)"},{"key":"16626_CR40","unstructured":"Kingma DP, Ba J (2017) Adam: A Method for Stochastic Optimization"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16626-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16626-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16626-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,8]],"date-time":"2024-03-08T06:40:36Z","timestamp":1709880036000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16626-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,18]]},"references-count":40,"journal-issue":{"issue":"11","published-online":{"date-parts":[[2024,3]]}},"alternative-id":["16626"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16626-w","relation":{},"ISSN":["1573-7721"],"issn-type":[{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2023,9,18]]},"assertion":[{"value":"22 December 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 August 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"No conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}