{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T20:15:43Z","timestamp":1784837743845,"version":"3.55.0"},"reference-count":59,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T00:00:00Z","timestamp":1783382400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/100005825","name":"National Institute of Food and Agriculture","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100005825","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006033","name":"US Poultry and Egg Association","doi-asserted-by":"publisher","award":["F-114"],"award-info":[{"award-number":["F-114"]}],"id":[{"id":"10.13039\/100006033","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computers and Electronics in Agriculture"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.compag.2026.112159","type":"journal-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:54:45Z","timestamp":1784181285000},"page":"112159","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Spatiotemporal video encoder integrated with metadata fusion for action recognition and behavior analysis"],"prefix":"10.1016","volume":"253","author":[{"given":"William O.","family":"Holden","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7624-8051","authenticated-orcid":false,"given":"Guoming","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ebenezer","family":"Agyemang-Duah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ricky A.","family":"Poku","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christopher","family":"Adenyo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Richard","family":"Osei-Amponsah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Samuel E.","family":"Aggrey","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.compag.2026.112159_b0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2025.111305","article-title":"Spatiotemporal video encoders and zero-shot segmentation for 3D action recognition and behavior analysis of broiler chickens associated with different welfare indicators and body weight","volume":"241","author":"Asali","year":"2026","journal-title":"Comput. Electron. Agric."},{"key":"10.1016\/j.compag.2026.112159_b0010","doi-asserted-by":"crossref","first-page":"12431","DOI":"10.1109\/ACCESS.2020.3047818","article-title":"A review on computer vision technology for monitoring poultry farm\u2014application, hardware, and software","volume":"9","author":"Aziz","year":"2021","journal-title":"IEEE Access"},{"key":"10.1016\/j.compag.2026.112159_b0015","unstructured":"Barnum, G., Talukder, S., Yue, Y., 2020. On the benefits of early fusion in multimodal representation learning. arXiv preprint arXiv:2011.07191."},{"key":"10.1016\/j.compag.2026.112159_b0020","doi-asserted-by":"crossref","first-page":"126601","DOI":"10.1109\/ACCESS.2023.3331092","article-title":"Animal behavior for chicken identification and monitoring the health condition using computer vision: a systematic review","volume":"11","author":"Bhuiyan","year":"2023","journal-title":"IEEE Access"},{"key":"10.1016\/j.compag.2026.112159_b0025","unstructured":"Bochkovskiy, A., Wang, C.-Y., Liao, H.-Y.M., 2020. Yolov4: optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934."},{"key":"10.1016\/j.compag.2026.112159_b0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.psj.2025.105126","article-title":"Identifying mating events of group-housed broiler breeders via bio-inspired deep learning models","volume":"104","author":"Bodempudi","year":"2025","journal-title":"Poult. Sci."},{"key":"10.1016\/j.compag.2026.112159_b0035","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6299","article-title":"Quo vadis, action recognition? a new model and the kinetics dataset","author":"Carreira","year":"2017"},{"key":"10.1016\/j.compag.2026.112159_b0040","series-title":"2024 Third International Conference on Distributed Computing and Electrical Circuits and Electronics (ICDCECE). IEEE","first-page":"1","article-title":"A spatio-temporl deepfake video detection method based on TimeSformer-CNN","author":"Chen","year":"2024"},{"key":"10.1016\/j.compag.2026.112159_b0045","unstructured":"CVHub520, 2024. X-AnyLabeling. Accessed on December 2025. Aviable at https:\/\/github.com\/CVHub520\/X-AnyLabeling."},{"key":"10.1016\/j.compag.2026.112159_b0050","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123397","article-title":"Multi-object behavior recognition based on object detection for dense crowds","volume":"248","author":"Dang","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.compag.2026.112159_b0055","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"2625","article-title":"Long-term recurrent convolutional networks for visual recognition and description","author":"Donahue","year":"2015"},{"key":"10.1016\/j.compag.2026.112159_b0060","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., 2020. X3d: Expanding architectures for efficient video recognition, Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 203\u2013213.","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"10.1016\/j.compag.2026.112159_b0065","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K., 2019. Slowfast networks for video recognition, Proceedings of the IEEE\/CVF international conference on computer vision, pp. 6202\u20136211.","DOI":"10.1109\/ICCV.2019.00630"},{"key":"10.1016\/j.compag.2026.112159_b0070","doi-asserted-by":"crossref","first-page":"381","DOI":"10.1145\/358669.358692","article-title":"Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography","volume":"24","author":"Fischler","year":"1981","journal-title":"CACM"},{"key":"10.1016\/j.compag.2026.112159_b0075","doi-asserted-by":"crossref","first-page":"829","DOI":"10.1162\/neco_a_01273","article-title":"A survey on deep learning for multimodal data fusion","volume":"32","author":"Gao","year":"2020","journal-title":"Neural Computation"},{"key":"10.1016\/j.compag.2026.112159_b0080","doi-asserted-by":"crossref","unstructured":"Gaw, N., Yousefi, S., Gahrooei, M.R., 2022. Multimodal data fusion for systems improvement: A review. Handbook of Scholarly Publications from the Air Force Institute of Technology (AFIT), Volume 1, 2000-2020, 101\u2013136.","DOI":"10.1201\/9781003220978-7"},{"key":"10.1016\/j.compag.2026.112159_b0085","doi-asserted-by":"crossref","first-page":"3390","DOI":"10.3390\/ani12233390","article-title":"Monitoring behaviors of broiler chickens at different ages with deep learning","volume":"12","author":"Guo","year":"2022","journal-title":"Animals"},{"key":"10.1016\/j.compag.2026.112159_b0090","doi-asserted-by":"crossref","first-page":"2141","DOI":"10.3390\/agriculture12122141","article-title":"Research on laying hens feeding behavior detection and model visualization based on convolutional neural network","volume":"12","author":"Hao","year":"2022","journal-title":"Agriculture"},{"key":"10.1016\/j.compag.2026.112159_b0095","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6546","article-title":"Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet?","author":"Hara","year":"2018"},{"key":"10.1016\/j.compag.2026.112159_b0100","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.compag.2026.112159_b0105","unstructured":"Joo, K.H., Duan, S., Weimer, S.L., Teli, M.N., 2022. Birds' Eye View: Measuring Behavior and Posture of Chickens as a Metric for Their Well-Being. arXiv preprint arXiv:2205.00069."},{"key":"10.1016\/j.compag.2026.112159_b0110","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13289","article-title":"MMTM: Multimodal transfer module for CNN fusion","author":"Joze","year":"2020"},{"key":"10.1016\/j.compag.2026.112159_b0115","first-page":"36","article-title":"Mobile applications empowering smallholder farmers: a review of the impact on agricultural development","volume":"8","author":"Kamal","year":"2023","journal-title":"Int. J. Social Anal."},{"key":"10.1016\/j.compag.2026.112159_b0120","unstructured":"Khanam, R., Hussain, M., 2024. Yolov11: An overview of the key architectural enhancements. arXiv preprint arXiv:2410.17725."},{"key":"10.1016\/j.compag.2026.112159_b0125","doi-asserted-by":"crossref","first-page":"2381","DOI":"10.3390\/s20082381","article-title":"A spatiotemporal convolutional network for multi-behavior recognition of pigs","volume":"20","author":"Li","year":"2020","journal-title":"Sensors"},{"key":"10.1016\/j.compag.2026.112159_b0130","doi-asserted-by":"crossref","first-page":"1492","DOI":"10.3390\/s21041492","article-title":"Practices and applications of convolutional neural network-based computer vision systems in animal farming: a review","volume":"21","author":"Li","year":"2021","journal-title":"Sensors"},{"key":"10.1016\/j.compag.2026.112159_b0135","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2020.105333","article-title":"Assessment of layer pullet drinking behaviors under selectable light colors using convolutional neural network","volume":"172","author":"Li","year":"2020","journal-title":"Comput. Electron Agric."},{"key":"10.1016\/j.compag.2026.112159_b0140","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2020.105596","article-title":"Analysis of feeding and drinking behaviors of group-reared broilers via image processing","volume":"175","author":"Li","year":"2020","journal-title":"Comput. Electron Agric."},{"key":"10.1016\/j.compag.2026.112159_b0145","unstructured":"Lugaresi, C., Tang, J., Nash, H., McClanahan, C., Uboweja, E., Hays, M., Zhang, F., Chang, C.-L., Yong, M.G., Lee, J., 2019. Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172."},{"key":"10.1016\/j.compag.2026.112159_b0150","unstructured":"Martius, G., 2020. VidStab. Accessed on December 2025. Aviable at https:\/\/github.com\/georgmartius\/vid.stab."},{"key":"10.1016\/j.compag.2026.112159_b0155","doi-asserted-by":"crossref","first-page":"1281","DOI":"10.1038\/s41593-018-0209-y","article-title":"DeepLabCut: markerless pose estimation of user-defined body parts with deep learning","volume":"21","author":"Mathis","year":"2018","journal-title":"Nature Neurosci."},{"key":"10.1016\/j.compag.2026.112159_b0160","doi-asserted-by":"crossref","first-page":"403","DOI":"10.1007\/s11119-019-09675-5","article-title":"Smartphone adoption and use in agriculture: empirical evidence from Germany","volume":"21","author":"Michels","year":"2020","journal-title":"Precision Agric."},{"key":"10.1016\/j.compag.2026.112159_b0165","doi-asserted-by":"crossref","DOI":"10.1016\/j.psj.2025.104954","article-title":"Automatic analysis of high, medium, and low activities of broilers with heat stress operations via image processing and machine learning","volume":"104","author":"Oso","year":"2025","journal-title":"Poult. Sci."},{"key":"10.1016\/j.compag.2026.112159_b0170","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1038\/s41592-018-0234-5","article-title":"Fast animal pose estimation using deep neural networks","volume":"16","author":"Pereira","year":"2019","journal-title":"Nat. Methods"},{"key":"10.1016\/j.compag.2026.112159_b0175","article-title":"BroilerTrack: automatic multi-camera multi-broiler tracking","volume":"12","author":"Phan","year":"2025","journal-title":"Smart Agric. Technol."},{"key":"10.1016\/j.compag.2026.112159_b0180","unstructured":"Ren, S., He, K., Girshick, R., Sun, J., 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Adv. Neural Inf. Process. Syst. 28."},{"key":"10.1016\/j.compag.2026.112159_b0185","first-page":"2564","article-title":"ORB: an efficient alternative to SIFT or SURF, 2011 International conference on computer vision","author":"Rublee","year":"2011","journal-title":"Ieee"},{"key":"10.1016\/j.compag.2026.112159_b0190","series-title":"Automated detection of broiler chicken behaviors through the integration of 3D accelerometer sensor and machine learning techniques","author":"Sakib","year":"2024"},{"key":"10.1016\/j.compag.2026.112159_b0195","first-page":"1","article-title":"An instrument indication acquisition algorithm based on lightweight deep convolutional neural network and hybrid attention fine-grained features","volume":"73","author":"Shen","year":"2024","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.compag.2026.112159_b0200","article-title":"Lightweight semantic feature extraction model with direction awareness for aerial traffic object detection","volume":"1\u201318","author":"Shen","year":"2025","journal-title":"IEEE Trans. Intelli. Transp. Syst."},{"key":"10.1016\/j.compag.2026.112159_b0205","first-page":"1","article-title":"Finger vein recognition algorithm based on lightweight deep convolutional neural network","volume":"71","author":"Shen","year":"2022","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.compag.2026.112159_b0210","doi-asserted-by":"crossref","first-page":"24330","DOI":"10.1109\/TITS.2022.3203715","article-title":"An anchor-free lightweight deep convolutional network for vehicle detection in aerial images","volume":"23","author":"Shen","year":"2022","journal-title":"IEEE Trans. Intelli. Transp. Syst."},{"key":"10.1016\/j.compag.2026.112159_b0215","unstructured":"Simonyan, K., Zisserman, A., 2014. Two-stream convolutional networks for action recognition in videos. Adv. Neural Inf. Process. Syst. 27."},{"key":"10.1016\/j.compag.2026.112159_b0220","first-page":"0280","article-title":"Thin mobilenet: an enhanced mobilenet architecture, 2019 IEEE 10th annual ubiquitous computing, electronics & mobile communication conference (UEMCON)","author":"Sinha","year":"2019","journal-title":"IEEE"},{"key":"10.1016\/j.compag.2026.112159_b0225","doi-asserted-by":"crossref","first-page":"1181","DOI":"10.3390\/ani12091181","article-title":"Visual sensor placement optimization with 3D animation for cattle health monitoring in a confined operation","volume":"12","author":"Sourav","year":"2022","journal-title":"Animals"},{"key":"10.1016\/j.compag.2026.112159_b0230","unstructured":"Spannbauer, A., 2021. Python Video Stabilization. Accessed on December 2025. Aviable at https:\/\/github.com\/AdamSpannbauer\/python_video_stab."},{"key":"10.1016\/j.compag.2026.112159_b0235","doi-asserted-by":"crossref","first-page":"342","DOI":"10.1038\/nature02226","article-title":"Chicken welfare is influenced more by housing conditions than by stocking density","volume":"427","author":"Stamp Dawkins","year":"2004","journal-title":"Nature"},{"key":"10.1016\/j.compag.2026.112159_b0240","unstructured":"Thakur, A., Papakipos, Z., Clauss, C., Hollinger, C., Boivin, V., Lowe, B., Schoentgen, M., Bouckenooghe, R., 2022. abhiTronix\/vidgear: VidGear Stable v0.2.5. Zenodo."},{"key":"10.1016\/j.compag.2026.112159_b0245","unstructured":"Tian, Y., Ye, Q., Doermann, D., 2025. Yolov12: Attention-centric real-time object detectors. arXiv preprint arXiv:2502.12524."},{"key":"10.1016\/j.compag.2026.112159_b0250","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"4489","article-title":"Learning spatiotemporal features with 3d convolutional networks","author":"Tran","year":"2015"},{"key":"10.1016\/j.compag.2026.112159_b0255","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pcbi.1011462","article-title":"OpenCap: human movement dynamics from smartphone videos","volume":"19","author":"Uhlrich","year":"2023","journal-title":"PLOS Comput. Biol."},{"key":"10.1016\/j.compag.2026.112159_b0260","unstructured":"Ultralytics, 2023. YOLOv8. Accessed on December 2025. Aviable at https:\/\/github.com\/autogyro\/yolo-V8."},{"key":"10.1016\/j.compag.2026.112159_b0265","unstructured":"Ultralytics, 2024. YOLOv11. Accessed on December 2025. Aviable at https:\/\/github.com\/yt7589\/yolov11."},{"key":"10.1016\/j.compag.2026.112159_b0270","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7464","article-title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","author":"Wang","year":"2023"},{"key":"10.1016\/j.compag.2026.112159_b0275","article-title":"I3d-lstm: a new model for human action recognition, IOP conference series: materials science and engineering","author":"Wang","year":"2019","journal-title":"IOP Publishing"},{"key":"10.1016\/j.compag.2026.112159_b0280","volume":"arXiv","author":"Yaseen","year":"2024","journal-title":"What Is YOLOv8: an in-Depth Exploration of the Internal Features of the next-Generation Object Detector"},{"key":"10.1016\/j.compag.2026.112159_b0285","doi-asserted-by":"crossref","first-page":"1085","DOI":"10.3390\/s20041085","article-title":"Automated video behavior recognition of pigs using two-stream convolutional networks","volume":"20","author":"Zhang","year":"2020","journal-title":"Sensors"},{"key":"10.1016\/j.compag.2026.112159_b0290","first-page":"1","article-title":"Bytetrack: multi-object tracking by associating every detection box","author":"Zhang","year":"2022","journal-title":"European Conference on Computer Vision. Springer"},{"key":"10.1016\/j.compag.2026.112159_b0295","doi-asserted-by":"crossref","unstructured":"Zhao, R., Hua, F., Wei, B., Li, C., Ma, Y., Wong, E.S.W., Liu, F., 2024. A Review of abnormal crowd behavior recognition technology based on computer vision, Appl. Sci., p. 9758.","DOI":"10.20944\/preprints202409.1879.v1"}],"container-title":["Computers and Electronics in Agriculture"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0168169926007544?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0168169926007544?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T19:49:52Z","timestamp":1784836192000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0168169926007544"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":59,"alternative-id":["S0168169926007544"],"URL":"https:\/\/doi.org\/10.1016\/j.compag.2026.112159","relation":{},"ISSN":["0168-1699"],"issn-type":[{"value":"0168-1699","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Spatiotemporal video encoder integrated with metadata fusion for action recognition and behavior analysis","name":"articletitle","label":"Article Title"},{"value":"Computers and Electronics in Agriculture","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.compag.2026.112159","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"112159"}}