{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T10:16:10Z","timestamp":1783160170613,"version":"3.54.6"},"reference-count":57,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2662025-XXPY006"],"award-info":[{"award-number":["2662025-XXPY006"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2024YFE0112200"],"award-info":[{"award-number":["2024YFE0112200"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132739","type":"journal-article","created":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T21:28:45Z","timestamp":1777930125000},"page":"132739","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["SCA-Net: Semantic text-enhanced context-aware multimodal framework for fish feeding assessment in aquaculture"],"prefix":"10.1016","volume":"326","author":[{"given":"Junkai","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feiyue","family":"Xue","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tongguan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2016-864X","authenticated-orcid":false,"given":"Chunfang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9194-0257","authenticated-orcid":false,"given":"Zongyao","family":"Sha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3236-6838","authenticated-orcid":false,"given":"Ying","family":"Sha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"2","key":"10.1016\/j.eswa.2026.132739_bib0001","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1111\/are.14907","article-title":"Application of computer vision in fish intelligent feeding system-a review","volume":"52","author":"An","year":"2021","journal-title":"Aquaculture Research"},{"key":"10.1016\/j.eswa.2026.132739_bib0002","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6836","article-title":"Vivit: A video vision transformer","author":"Arnab","year":"2021"},{"key":"10.1016\/j.eswa.2026.132739_bib0003","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132739_bib0004","series-title":"ICML","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"vol. 2","author":"Bertasius","year":"2021"},{"issue":"3","key":"10.1016\/j.eswa.2026.132739_bib0005","doi-asserted-by":"crossref","first-page":"805","DOI":"10.1007\/s12571-021-01246-9","article-title":"The contribution of fisheries and aquaculture to the global protein supply","volume":"14","author":"Boyd","year":"2022","journal-title":"Food Security"},{"issue":"2","key":"10.1016\/j.eswa.2026.132739_bib0006","doi-asserted-by":"crossref","first-page":"261","DOI":"10.1016\/j.inpa.2019.09.001","article-title":"Feed intake prediction model for group fish using the MEA-BP neural network in intensive aquaculture","volume":"7","author":"Chen","year":"2020","journal-title":"Information Processing in Agriculture"},{"issue":"6","key":"10.1016\/j.eswa.2026.132739_bib0007","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","article-title":"WavLm: Large-scale self-supervised pre-training for full stack speech processing","volume":"16","author":"Chen","year":"2022","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"10.1016\/j.eswa.2026.132739_bib0008","doi-asserted-by":"crossref","unstructured":"Cho, K., Van Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., & Bengio, Y. (2014). Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv: 1406.1078.","DOI":"10.3115\/v1\/D14-1179"},{"key":"10.1016\/j.eswa.2026.132739_bib0009","doi-asserted-by":"crossref","first-page":"9485","DOI":"10.1109\/TASE.2024.3507098","article-title":"Multimodal fish feeding intensity assessment in aquaculture","volume":"22","author":"Cui","year":"2024","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"1","key":"10.1016\/j.eswa.2026.132739_bib0010","doi-asserted-by":"crossref","DOI":"10.1111\/raq.13001","article-title":"Fish tracking, counting, and behaviour analysis in digital aquaculture: A comprehensive survey","volume":"17","author":"Cui","year":"2025","journal-title":"Reviews in Aquaculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0011","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132739_bib0012","doi-asserted-by":"crossref","first-page":"135","DOI":"10.1016\/j.biosystemseng.2024.08.001","article-title":"Harnessing multimodal data fusion to advance accurate identification of fish feeding intensity","volume":"246","author":"Du","year":"2024","journal-title":"Biosystems Engineering"},{"key":"10.1016\/j.eswa.2026.132739_bib0013","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2023.108310","article-title":"Feature fusion strategy and improved ghostnet for accurate recognition of fish feeding behavior","volume":"214","author":"Du","year":"2023","journal-title":"Computers and Electronics in Agriculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"203","article-title":"X3d: Expanding architectures for efficient video recognition","author":"Feichtenhofer","year":"2020"},{"issue":"1","key":"10.1016\/j.eswa.2026.132739_bib0015","doi-asserted-by":"crossref","first-page":"107","DOI":"10.1080\/23308249.2019.1678111","article-title":"A global blue revolution: Aquaculture growth across regions, species, and countries","volume":"28","author":"Garlock","year":"2020","journal-title":"Reviews in Fisheries Science & Aquaculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0016","doi-asserted-by":"crossref","unstructured":"Gong, Y., Chung, Y.-A., & Glass, J. (2021). AST: Audio spectrogram transformer. arXiv: 2104.01778.","DOI":"10.21437\/Interspeech.2021-698"},{"key":"10.1016\/j.eswa.2026.132739_bib0017","series-title":"Supervised sequence labelling with recurrent neural networks","first-page":"37","article-title":"Long short-term memory","author":"Graves","year":"2012"},{"key":"10.1016\/j.eswa.2026.132739_bib0018","unstructured":"He, P., Gao, J., & Chen, W. (2021). Debertav3: Improving deberta using electra-style pre-training with gradient-disentangled embedding sharing. arXiv: 2111.09543."},{"key":"10.1016\/j.eswa.2026.132739_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"1314","article-title":"Searching for mobilenetv3","author":"Howard","year":"2019"},{"key":"10.1016\/j.eswa.2026.132739_bib0020","unstructured":"Howard, A. G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., & Adam, H. (2017). MobileNets: Efficient convolutional neural networks for mobile vision applications. arXiv: 1704.04861."},{"key":"10.1016\/j.eswa.2026.132739_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2021.115051","article-title":"Real-time nondestructive fish behavior detecting in mixed polyculture system using deep-learning and low-cost devices","volume":"178","author":"Hu","year":"2021","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132739_bib0022","unstructured":"Kim, Y. (2014). Convolutional neural networks for sentence classification. arXiv: 1408.5882."},{"key":"10.1016\/j.eswa.2026.132739_bib0023","doi-asserted-by":"crossref","first-page":"2880","DOI":"10.1109\/TASLP.2020.3030497","article-title":"PANNs: Large-scale pretrained audio neural networks for audio pattern recognition","volume":"28","author":"Kong","year":"2020","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"1","key":"10.1016\/j.eswa.2026.132739_bib0024","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1111\/raq.12842","article-title":"Recent advances in acoustic technology for aquaculture: A review","volume":"16","author":"Li","year":"2024","journal-title":"Reviews in Aquaculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.aquaculture.2020.735508","article-title":"Automatic recognition methods of fish feeding behavior in aquaculture: A review","volume":"528","author":"Li","year":"2020","journal-title":"Aquaculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0026","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2024.109367","article-title":"A review of aquaculture: From single modality analysis to multimodality fusion","volume":"226","author":"Li","year":"2024","journal-title":"Computers and Electronics in Agriculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0027","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). Roberta: A robustly optimized bert pretraining approach. arXiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.132739_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11976","article-title":"A convnet for the 2020s","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132739_bib0029","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3202","article-title":"Video swin transformer","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132739_bib0030","unstructured":"Mehta, S., & Rastegari, M. (2021). MobileViT: Light-weight, general-purpose, and mobile-friendly vision transformer. arXiv: 2110.02178."},{"issue":"3","key":"10.1016\/j.eswa.2026.132739_bib0031","doi-asserted-by":"crossref","first-page":"506","DOI":"10.1016\/j.physbeh.2005.11.012","article-title":"Behavioral indicators of stress-coping style in rainbow trout: Do males and females react differently to novelty?","volume":"87","author":"\u00d8verli","year":"2006","journal-title":"Physiology & Behavior"},{"key":"10.1016\/j.eswa.2026.132739_bib0032","series-title":"European conference on computer vision","first-page":"78","article-title":"MobileNetV4: Universal models for the mobile ecosystem","author":"Qin","year":"2024"},{"key":"10.1016\/j.eswa.2026.132739_bib0033","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132739_bib0034","doi-asserted-by":"crossref","first-page":"582","DOI":"10.1016\/j.aquaculture.2016.07.037","article-title":"The oxygen threshold for maximal feed intake of atlantic salmon post-smolts is highly temperature-dependent","volume":"464","author":"Remen","year":"2016","journal-title":"Aquaculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0035","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4510","article-title":"MobileNetV2: Inverted residuals and linear bottlenecks","author":"Sandler","year":"2018"},{"issue":"11","key":"10.1016\/j.eswa.2026.132739_bib0036","doi-asserted-by":"crossref","first-page":"2673","DOI":"10.1109\/78.650093","article-title":"Bidirectional recurrent neural networks","volume":"45","author":"Schuster","year":"1997","journal-title":"IEEE Transactions on Signal Processing"},{"key":"10.1016\/j.eswa.2026.132739_bib0037","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"618","article-title":"Grad-CAM: Visual explanations from deep networks via gradient-based localization","author":"Selvaraju","year":"2017"},{"key":"10.1016\/j.eswa.2026.132739_bib0038","doi-asserted-by":"crossref","first-page":"159719","DOI":"10.1109\/ACCESS.2024.3478831","article-title":"A comprehensive design of hybrid residual (2+ 1)-dimensional CNN and dense networks with multi-modal sensor for fish appetite detection","volume":"12","author":"Syafalni","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.eswa.2026.132739_bib0039","series-title":"International conference on machine learning","first-page":"10096","article-title":"EfficientNetV2: Smaller models and faster training","author":"Tan","year":"2021"},{"key":"10.1016\/j.eswa.2026.132739_bib0040","unstructured":"Q. Team et al. (2024). Qwen2 technical report. arXiv: 2407.10671."},{"key":"10.1016\/j.eswa.2026.132739_bib0041","doi-asserted-by":"crossref","first-page":"10078","DOI":"10.52202\/068431-0732","article-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training","volume":"35","author":"Tong","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132739_bib0042","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"6450","article-title":"A closer look at spatiotemporal convolutions for action recognition","author":"Tran","year":"2018"},{"key":"10.1016\/j.eswa.2026.132739_bib0043","series-title":"Proceedings of the conference on association for computational linguistics meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"Vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.eswa.2026.132739_bib0044","unstructured":"Tschannen, M., Gritsenko, A., Wang, X., Naeem, M. F., Alabdulmohsin, I., Parthasarathy, N., Evans, T., Beyer, L., Xia, Y., Mustafa, B. et al. (2025). SigLIP 2: Multilingual vision-language encoders with improved semantic understanding, localization, and dense features. arXiv: 2502.14786."},{"key":"10.1016\/j.eswa.2026.132739_bib0045","doi-asserted-by":"crossref","DOI":"10.1016\/j.aquaeng.2021.102178","article-title":"Evaluating fish feeding intensity in aquaculture with convolutional neural networks","volume":"94","author":"Ubina","year":"2021","journal-title":"Aquacultural Engineering"},{"key":"10.1016\/j.eswa.2026.132739_bib0046","first-page":"6000","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132739_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15909","article-title":"RepViT: Revisiting mobile cnn from vit perspective","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132739_bib0048","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"21180","article-title":"DLF: Disentangled-language-focused multimodal sentiment analysis","volume":"Vol. 39","author":"Wang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132739_bib0049","first-page":"5776","article-title":"MiniLM: Deep self-attention distillation for task-agnostic compression of pre-trained transformers","volume":"33","author":"Wang","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132739_bib0050","doi-asserted-by":"crossref","unstructured":"Xiao, Y., Huang, L., Zhang, S., Bi, C., You, X., He, S., & Guan, J. (2025). Feeding behavior quantification and recognition for intelligent fish farming application: A review. Applied animal behaviour science,106588.","DOI":"10.1016\/j.applanim.2025.106588"},{"key":"10.1016\/j.eswa.2026.132739_bib0051","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"305","article-title":"Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification","author":"Xie","year":"2018"},{"key":"10.1016\/j.eswa.2026.132739_bib0052","doi-asserted-by":"crossref","unstructured":"Yamagishi, Y., Kikuchi, T., Hanaoka, S., Yoshikawa, T., & Abe, O. (2025). ModernBERT is more efficient than conventional BERT for chest CT findings classification in Japanese radiology reports. arXiv: 2503.05060.","DOI":"10.1038\/s41598-026-44292-z"},{"issue":"3","key":"10.1016\/j.eswa.2026.132739_bib0053","first-page":"285","article-title":"Fish feeding behavior recognition using adaptive dmca-umt algorithm","volume":"32","author":"Yang","year":"2023","journal-title":"Journal of Beijing Institute of Technology"},{"key":"10.1016\/j.eswa.2026.132739_bib0054","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2022.107580","article-title":"Fish school feeding behavior quantification using acoustic signal and improved swin transformer","volume":"204","author":"Zeng","year":"2023","journal-title":"Computers and Electronics in Agriculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0055","doi-asserted-by":"crossref","first-page":"233","DOI":"10.1016\/j.compag.2017.02.013","article-title":"Near-infrared imaging to quantify the feeding behavior of fish in aquaculture","volume":"135","author":"Zhou","year":"2017","journal-title":"Computers and Electronics in Agriculture"},{"key":"10.1016\/j.eswa.2026.132739_bib0056","series-title":"Proceedings of the 54th annual meeting of the association for computational linguistics (Volume 2: Short papers)","first-page":"207","article-title":"Attention-based bidirectional long short-term memory networks for relation classification","author":"Zhou","year":"2016"},{"key":"10.1016\/j.eswa.2026.132739_bib0057","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"1800","article-title":"GLoMo: Global-local modal fusion for multimodal sentiment analysis","author":"Zhuang","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016520?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426016520?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T09:25:28Z","timestamp":1783157128000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426016520"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":57,"alternative-id":["S0957417426016520"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132739","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"SCA-Net: Semantic text-enhanced context-aware multimodal framework for fish feeding assessment in aquaculture","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132739","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132739"}}