{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T22:14:57Z","timestamp":1783203297931,"version":"3.54.6"},"reference-count":77,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52074302"],"award-info":[{"award-number":["52074302"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Advanced Engineering Informatics"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.aei.2026.104834","type":"journal-article","created":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T10:16:50Z","timestamp":1780654610000},"page":"104834","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["Risk-aware trajectory forecasting via multi-modal knowledge distillation"],"prefix":"10.1016","volume":"76","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-7453-2628","authenticated-orcid":false,"given":"Xin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Leyao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangyang","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8003-5478","authenticated-orcid":false,"given":"Ruipeng","family":"Tong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.aei.2026.104834_b1","series-title":"Leveraging future relationship reasoning for vehicle trajectory prediction","author":"Park","year":"2023"},{"key":"10.1016\/j.aei.2026.104834_b2","series-title":"European Conference on Computer Vision","first-page":"511","article-title":"Socialvae: Human trajectory prediction using timewise latents","author":"Xu","year":"2022"},{"key":"10.1016\/j.aei.2026.104834_b3","series-title":"Pedestrian trajectory prediction with missing data: Datasets, imputation, and benchmarking","author":"Chib","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b4","doi-asserted-by":"crossref","unstructured":"Y. Xu, Y. Fu, Adapting to length shift: Flexilength network for trajectory prediction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 15226\u201315237.","DOI":"10.1109\/CVPR52733.2024.01442"},{"key":"10.1016\/j.aei.2026.104834_b5","series-title":"Progressive pretext task learning for human trajectory prediction","author":"Lin","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b6","doi-asserted-by":"crossref","unstructured":"C. Xu, R.T. Tan, Y. Tan, S. Chen, Y.G. Wang, X. Wang, Y. Wang, Eqmotion: Equivariant multi-agent motion prediction with invariant interaction reasoning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 1410\u20131420.","DOI":"10.1109\/CVPR52729.2023.00142"},{"key":"10.1016\/j.aei.2026.104834_b7","doi-asserted-by":"crossref","unstructured":"T. Gu, G. Chen, J. Li, C. Lin, Y. Rao, J. Zhou, J. Lu, Stochastic trajectory prediction via motion indeterminacy diffusion, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 17113\u201317122.","DOI":"10.1109\/CVPR52688.2022.01660"},{"key":"10.1016\/j.aei.2026.104834_b8","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVIII 16","first-page":"683","article-title":"Trajectron++: Dynamically-feasible trajectory forecasting with heterogeneous data","author":"Salzmann","year":"2020"},{"key":"10.1016\/j.aei.2026.104834_b9","series-title":"Multi-transmotion: Pre-trained model for human motion prediction","author":"Gao","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b10","series-title":"European Conference on Computer Vision","first-page":"56","article-title":"Learning semantic latent directions for accurate and controllable human motion prediction","author":"Xu","year":"2025"},{"key":"10.1016\/j.aei.2026.104834_b11","doi-asserted-by":"crossref","unstructured":"X. Gu, G. Song, I. Gilitschenski, M. Pavone, B. Ivanovic, Producing and leveraging online map uncertainty in trajectory prediction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14521\u201314530.","DOI":"10.1109\/CVPR52733.2024.01376"},{"key":"10.1016\/j.aei.2026.104834_b12","unstructured":"Z. Lan, Y. Jiang, Y. Mu, C. Chen, S.E. Li, Sept: Towards efficient scene representation learning for motion prediction, in: The Twelfth International Conference on Learning Representations, 2023, p. 2309."},{"key":"10.1016\/j.aei.2026.104834_b13","doi-asserted-by":"crossref","unstructured":"H. Zhang, Y. Xu, H. Lu, T. Shimizu, Y. Fu, OOSTraj: Out-of-Sight Trajectory Prediction With Vision-Positioning Denoising, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14802\u201314811.","DOI":"10.1109\/CVPR52733.2024.01402"},{"key":"10.1016\/j.aei.2026.104834_b14","doi-asserted-by":"crossref","unstructured":"D. Park, J. Jeong, S.-H. Yoon, J. Jeong, K.-J. Yoon, T4P: Test-Time Training of Trajectory Prediction via Masked Autoencoder and Actor-specific Token Memory, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 15065\u201315076.","DOI":"10.1109\/CVPR52733.2024.01427"},{"key":"10.1016\/j.aei.2026.104834_b15","doi-asserted-by":"crossref","unstructured":"J.-S. Ham, D.H. Kim, N. Jung, J. Moon, CIPF: Crossing Intention Prediction Network Based on Feature Fusion Modules for Improving Pedestrian Safety, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 3665\u20133674.","DOI":"10.1109\/CVPRW59228.2023.00374"},{"key":"10.1016\/j.aei.2026.104834_b16","doi-asserted-by":"crossref","unstructured":"Z. Zhou, J. Wang, Y.-H. Li, Y.-K. Huang, Query-centric trajectory prediction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 17863\u201317873.","DOI":"10.1109\/CVPR52729.2023.01713"},{"key":"10.1016\/j.aei.2026.104834_b17","doi-asserted-by":"crossref","unstructured":"M. Liu, H. Cheng, L. Chen, H. Broszio, J. Li, R. Zhao, M. Sester, M.Y. Yang, Laformer: Trajectory prediction for autonomous driving with lane-aware scene constraints, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 2039\u20132049.","DOI":"10.1109\/CVPRW63382.2024.00209"},{"key":"10.1016\/j.aei.2026.104834_b18","doi-asserted-by":"crossref","unstructured":"J. Cheng, X. Mei, M. Liu, Forecast-mae: Self-supervised pre-training for motion forecasting with masked autoencoders, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 8679\u20138689.","DOI":"10.1109\/ICCV51070.2023.00797"},{"key":"10.1016\/j.aei.2026.104834_b19","doi-asserted-by":"crossref","unstructured":"C. Pan, B. Yaman, T. Nesti, A. Mallik, A.G. Allievi, S. Velipasalar, L. Ren, VLP: Vision Language Planning for Autonomous Driving, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14760\u201314769.","DOI":"10.1109\/CVPR52733.2024.01398"},{"key":"10.1016\/j.aei.2026.104834_b20","series-title":"Reasoning multi-agent behavioral topology for interactive autonomous driving","author":"Liu","year":"2024"},{"issue":"11","key":"10.1016\/j.aei.2026.104834_b21","doi-asserted-by":"crossref","first-page":"7090","DOI":"10.1109\/LRA.2023.3312035","article-title":"Robots that can see: Leveraging human pose for trajectory prediction","volume":"8","author":"Salzmann","year":"2023","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.aei.2026.104834_b22","article-title":"SiT dataset: socially interactive pedestrian trajectory dataset for social navigation robots","volume":"36","author":"Bae","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.aei.2026.104834_b23","series-title":"Social-transmotion: Promptable human trajectory prediction","author":"Saadatnejad","year":"2023"},{"issue":"7","key":"10.1016\/j.aei.2026.104834_b24","doi-asserted-by":"crossref","first-page":"1985","DOI":"10.1109\/TCSVT.2018.2857489","article-title":"Trajectory-based surveillance analysis: A survey","volume":"29","author":"Ahmed","year":"2018","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.aei.2026.104834_b25","doi-asserted-by":"crossref","unstructured":"H. Lin, C. Wei, L. He, Y. Guo, Y. Zhao, S. Li, L. Fang, GigaTraj: Predicting Long-term Trajectories of Hundreds of Pedestrians in Gigapixel Complex Scenes, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 19331\u201319340.","DOI":"10.1109\/CVPR52733.2024.01829"},{"key":"10.1016\/j.aei.2026.104834_b26","doi-asserted-by":"crossref","unstructured":"A.D. Pazho, G.A. Noghre, V. Katariya, H. Tabkhi, VT-Former: An Exploratory Study on Vehicle Trajectory Prediction for Highway Surveillance through Graph Isomorphism and Transformer, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 5651\u20135662.","DOI":"10.1109\/CVPRW63382.2024.00574"},{"key":"10.1016\/j.aei.2026.104834_b27","series-title":"Building and better understanding vision-language models: insights and future directions","author":"Lauren\u00e7on","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b28","series-title":"Evlm: An efficient vision-language model for visual understanding","author":"Chen","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b29","series-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015"},{"issue":"6","key":"10.1016\/j.aei.2026.104834_b30","doi-asserted-by":"crossref","first-page":"6748","DOI":"10.1109\/TPAMI.2021.3070543","article-title":"Jrdb: A dataset and benchmark of egocentric robot visual perception of humans in built environments","volume":"45","author":"Martin-Martin","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.aei.2026.104834_b31","series-title":"2009 IEEE 12th International Conference on Computer Vision","first-page":"261","article-title":"You\u2019ll never walk alone: Modeling social behavior for multi-target tracking","author":"Pellegrini","year":"2009"},{"issue":"3","key":"10.1016\/j.aei.2026.104834_b32","doi-asserted-by":"crossref","first-page":"655","DOI":"10.1111\/j.1467-8659.2007.01089.x","article-title":"Crowds by example","volume":"26","author":"Lerner","year":"2007","journal-title":"Comput. Graph. Forum"},{"key":"10.1016\/j.aei.2026.104834_b33","doi-asserted-by":"crossref","unstructured":"Z. Zhou, L. Ye, J. Wang, K. Wu, K. Lu, Hivt: Hierarchical vector transformer for multi-agent motion prediction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 8823\u20138833.","DOI":"10.1109\/CVPR52688.2022.00862"},{"key":"10.1016\/j.aei.2026.104834_b34","series-title":"MART: MultiscAle relational transformer networks for multi-agent trajectory prediction","author":"Lee","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b35","series-title":"VisionTrap: Vision-augmented trajectory prediction guided by textual descriptions","author":"Moon","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b36","doi-asserted-by":"crossref","unstructured":"J.-C.W. Wang, H.-H. Shuai, W.-H. Cheng, TrajPrompt: Aligning Color Trajectory with Vision-Language Representations, in: European Conference on Computer Vision, 2024, pp. 275\u2013292.","DOI":"10.1007\/978-3-031-72940-9_16"},{"key":"10.1016\/j.aei.2026.104834_b37","doi-asserted-by":"crossref","unstructured":"Y. Ma, C. Cui, X. Cao, W. Ye, P. Liu, J. Lu, A. Abdelraouf, R. Gupta, K. Han, A. Bera, et al., Lampilot: An open benchmark dataset for autonomous driving with language model programs, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 15141\u201315151.","DOI":"10.1109\/CVPR52733.2024.01434"},{"key":"10.1016\/j.aei.2026.104834_b38","article-title":"Receive, reason, and react: Drive as you say, with large language models in autonomous vehicles","author":"Cui","year":"2024","journal-title":"IEEE Intell. Transp. Syst. Mag."},{"key":"10.1016\/j.aei.2026.104834_b39","series-title":"European Conference on Computer Vision","first-page":"22","article-title":"Asynchronous large language model enhanced planner for autonomous driving","author":"Chen","year":"2025"},{"key":"10.1016\/j.aei.2026.104834_b40","series-title":"Drivevlm: The convergence of autonomous driving and large vision-language models","author":"Tian","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b41","series-title":"Rag-driver: Generalisable driving explanations with retrieval-augmented in-context learning in multi-modal large language model","author":"Yuan","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b42","doi-asserted-by":"crossref","unstructured":"N. Deo, M.M. Trivedi, Convolutional Social Pooling for Vehicle Trajectory Prediction, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, CVPRW, 2018, pp. 1468\u20131476.","DOI":"10.1109\/CVPRW.2018.00196"},{"key":"10.1016\/j.aei.2026.104834_b43","doi-asserted-by":"crossref","unstructured":"N. Rhinehart, R. McAllister, K. Kitani, S. Levine, PRECOG: PREdiction Conditioned On Goals in Visual Multi-Agent Settings, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2019, pp. 2821\u20132831.","DOI":"10.1109\/ICCV.2019.00291"},{"key":"10.1016\/j.aei.2026.104834_b44","doi-asserted-by":"crossref","unstructured":"S. Suo, S. Regalado, S. Casas, R. Urtasun, TrafficSim: Learning to Simulate Realistic Multi-Agent Behaviors, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2021, pp. 10400\u201310409.","DOI":"10.1109\/CVPR46437.2021.01026"},{"key":"10.1016\/j.aei.2026.104834_b45","doi-asserted-by":"crossref","unstructured":"J. Wang, A. Pun, J. Tu, S. Manivasagam, A. Sadat, S. Casas, M. Ren, R. Urtasun, AdvSim: Generating Safety-Critical Scenarios for Self-Driving Vehicles, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2021, pp. 9909\u20139918.","DOI":"10.1109\/CVPR46437.2021.00978"},{"key":"10.1016\/j.aei.2026.104834_b46","doi-asserted-by":"crossref","unstructured":"A. Rasouli, I. Kotseruba, T. Kunic, J.K. Tsotsos, PIE: A Large-Scale Dataset and Models for Pedestrian Intention Estimation and Trajectory Prediction, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2019, pp. 6262\u20136271.","DOI":"10.1109\/ICCV.2019.00636"},{"key":"10.1016\/j.aei.2026.104834_b47","doi-asserted-by":"crossref","unstructured":"I. Kotseruba, A. Rasouli, J.K. Tsotsos, Benchmark for Evaluating Pedestrian Action Prediction, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, WACV, 2021, pp. 1258\u20131268.","DOI":"10.1109\/WACV48630.2021.00130"},{"key":"10.1016\/j.aei.2026.104834_b48","doi-asserted-by":"crossref","unstructured":"A. Rasouli, I. Kotseruba, J.K. Tsotsos, Are They Going to Cross? A Benchmark Dataset and Baseline for Pedestrian Crosswalk Behavior, in: Proceedings of the IEEE International Conference on Computer Vision Workshops, ICCVW, 2017, pp. 206\u2013213.","DOI":"10.1109\/ICCVW.2017.33"},{"issue":"3","key":"10.1016\/j.aei.2026.104834_b49","doi-asserted-by":"crossref","first-page":"2331","DOI":"10.1109\/TITS.2021.3074829","article-title":"Pedestrian crossing intention prediction at red-light using pose estimation","volume":"23","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.aei.2026.104834_b50","doi-asserted-by":"crossref","unstructured":"H. Xu, P. Peng, G. Tan, Y. Li, X. Xu, Y. Tian, DMR: Decomposed Multi-Modality Representations for Frames and Events Fusion in Visual Reinforcement Learning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26508\u201326518.","DOI":"10.1109\/CVPR52733.2024.02503"},{"key":"10.1016\/j.aei.2026.104834_b51","doi-asserted-by":"crossref","unstructured":"S. Zhou, W. Liu, C. Hu, S. Zhou, C. Ma, UniDistill: A Universal Cross-Modality Knowledge Distillation Framework for 3D Object Detection in Bird\u2019s-Eye View, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 5116\u20135125.","DOI":"10.1109\/CVPR52729.2023.00495"},{"key":"10.1016\/j.aei.2026.104834_b52","doi-asserted-by":"crossref","unstructured":"X. Lin, S. Wang, R. Cai, Y. Liu, Y. Fu, W. Tang, Z. Yu, A. Kot, Suppress and Rebalance: Towards Generalized Multi-Modal Face Anti-Spoofing, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 211\u2013221.","DOI":"10.1109\/CVPR52733.2024.00028"},{"key":"10.1016\/j.aei.2026.104834_b53","doi-asserted-by":"crossref","unstructured":"Y. Li, Y. Wang, Z. Cui, Decoupled multimodal distilling for emotion recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 6631\u20136640.","DOI":"10.1109\/CVPR52729.2023.00641"},{"key":"10.1016\/j.aei.2026.104834_b54","doi-asserted-by":"crossref","unstructured":"G. Radevski, D. Grujicic, M. Blaschko, M.-F. Moens, T. Tuytelaars, Multimodal distillation for egocentric action recognition, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 5213\u20135224.","DOI":"10.1109\/ICCV51070.2023.00481"},{"key":"10.1016\/j.aei.2026.104834_b55","series-title":"European Conference on Computer Vision","first-page":"19","article-title":"LabelDistill: Label-guided cross-modal knowledge distillation for camera-based 3D object detection","author":"Kim","year":"2025"},{"key":"10.1016\/j.aei.2026.104834_b56","doi-asserted-by":"crossref","unstructured":"M. Klingner, S. Borse, V.R. Kumar, B. Rezaei, V. Narayanan, S. Yogamani, F. Porikli, X3kd: Knowledge distillation across modalities, tasks and stages for multi-camera 3d object detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 13343\u201313353.","DOI":"10.1109\/CVPR52729.2023.01282"},{"key":"10.1016\/j.aei.2026.104834_b57","doi-asserted-by":"crossref","unstructured":"Q. Wang, L. Zhan, P. Thompson, J. Zhou, Multimodal learning with incomplete modalities by knowledge distillation, in: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, 2020, pp. 1828\u20131838.","DOI":"10.1145\/3394486.3403234"},{"key":"10.1016\/j.aei.2026.104834_b58","series-title":"Enabling multimodal generation on clip via vision-language knowledge distillation","author":"Dai","year":"2022"},{"key":"10.1016\/j.aei.2026.104834_b59","series-title":"Computer Vision \u2013 ECCV 2024","isbn-type":"print","doi-asserted-by":"crossref","first-page":"368","DOI":"10.1007\/978-3-031-72992-8_21","article-title":"LiDAR-based all-weather 3D object detection via prompting and distilling 4D radar","author":"Chae","year":"2025","ISBN":"https:\/\/id.crossref.org\/isbn\/9783031729928"},{"key":"10.1016\/j.aei.2026.104834_b60","doi-asserted-by":"crossref","unstructured":"M. Li, D. Yang, X. Zhao, S. Wang, Y. Wang, K. Yang, M. Sun, D. Kou, Z. Qian, L. Zhang, Correlation-Decoupled Knowledge Distillation for Multimodal Sentiment Analysis with Incomplete Modalities, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 12458\u201312468.","DOI":"10.1109\/CVPR52733.2024.01184"},{"key":"10.1016\/j.aei.2026.104834_b61","doi-asserted-by":"crossref","first-page":"33485","DOI":"10.52202\/075280-1455","article-title":"Egodistill: Egocentric head motion distillation for efficient video understanding","volume":"36","author":"Tan","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.aei.2026.104834_b62","doi-asserted-by":"crossref","unstructured":"T. Zhang, H. Guo, Q. Jiao, Q. Zhang, J. Han, Efficient rgb-t tracking via cross-modality distillation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 5404\u20135413.","DOI":"10.1109\/CVPR52729.2023.00523"},{"key":"10.1016\/j.aei.2026.104834_b63","unstructured":"A. Nath, P. Kabra, I. Gupta, P. Singh, et al., MS-TIP: Imputation Aware Pedestrian Trajectory Prediction, in: Forty-First International Conference on Machine Learning, 2024, pp. 8389\u20138402."},{"key":"10.1016\/j.aei.2026.104834_b64","doi-asserted-by":"crossref","unstructured":"I. Bae, Y.-J. Park, H.-G. Jeon, SingularTrajectory: Universal Trajectory Predictor Using Diffusion Model, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 17890\u201317901.","DOI":"10.1109\/CVPR52733.2024.01694"},{"key":"10.1016\/j.aei.2026.104834_b65","doi-asserted-by":"crossref","first-page":"14400","DOI":"10.52202\/075280-0633","article-title":"Bcdiff: Bidirectional consistent diffusion for instantaneous trajectory prediction","volume":"36","author":"Li","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.aei.2026.104834_b66","doi-asserted-by":"crossref","unstructured":"R. Li, C. Li, Y. Li, H. Li, Y. Chen, Y. Yuan, G. Wang, ITPNet: Towards Instantaneous Trajectory Prediction for Autonomous Driving, in: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2024, pp. 1643\u20131654.","DOI":"10.1145\/3637528.3671681"},{"key":"10.1016\/j.aei.2026.104834_b67","doi-asserted-by":"crossref","unstructured":"A. Monti, A. Porrello, S. Calderara, P. Coscia, L. Ballan, R. Cucchiara, How many observations are enough? knowledge distillation for trajectory forecasting, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 6553\u20136562.","DOI":"10.1109\/CVPR52688.2022.00644"},{"key":"10.1016\/j.aei.2026.104834_b68","series-title":"Tinybert: Distilling bert for natural language understanding","author":"Jiao","year":"2019"},{"key":"10.1016\/j.aei.2026.104834_b69","series-title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","author":"Xu","year":"2024"},{"key":"10.1016\/j.aei.2026.104834_b70","doi-asserted-by":"crossref","unstructured":"T.-Y. Lin, P. Goyal, R. Girshick, K. He, P. Doll\u00e1r, Focal loss for dense object detection, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"10.1016\/j.aei.2026.104834_b71","doi-asserted-by":"crossref","unstructured":"S. Jahangard, Z. Cai, S. Wen, H. Rezatofighi, JRDB-Social: A Multifaceted Robotic Dataset for Understanding of Context and Dynamics of Human Interactions Within Social Groups, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 22087\u201322097.","DOI":"10.1109\/CVPR52733.2024.02085"},{"key":"10.1016\/j.aei.2026.104834_b72","doi-asserted-by":"crossref","unstructured":"Y. Sun, W. Liu, Q. Bao, Y. Fu, T. Mei, M.J. Black, Putting people in their place: Monocular regression of 3d people in depth, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 13243\u201313252.","DOI":"10.1109\/CVPR52688.2022.01289"},{"key":"10.1016\/j.aei.2026.104834_b73","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.aei.2026.104834_b74","doi-asserted-by":"crossref","unstructured":"W. Mao, C. Xu, Q. Zhu, S. Chen, Y. Wang, Leapfrog diffusion model for stochastic trajectory prediction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 5517\u20135526.","DOI":"10.1109\/CVPR52729.2023.00534"},{"key":"10.1016\/j.aei.2026.104834_b75","unstructured":"A. Shrikumar, P. Greenside, A. Kundaje, Learning Important Features Through Propagating Activation Differences, in: International Conference on Machine Learning, 2017, pp. 3145\u20133153."},{"key":"10.1016\/j.aei.2026.104834_b76","series-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","first-page":"4190","article-title":"Quantifying attention flow in transformers","author":"Abnar","year":"2020"},{"key":"10.1016\/j.aei.2026.104834_b77","unstructured":"M. Sundararajan, A. Taly, Q. Yan, Axiomatic Attribution for Deep Networks, in: International Conference on Machine Learning, 2017, pp. 3319\u20133328."}],"container-title":["Advanced Engineering Informatics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1474034626005264?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1474034626005264?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T21:15:42Z","timestamp":1783199742000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1474034626005264"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":77,"alternative-id":["S1474034626005264"],"URL":"https:\/\/doi.org\/10.1016\/j.aei.2026.104834","relation":{},"ISSN":["1474-0346"],"issn-type":[{"value":"1474-0346","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Risk-aware trajectory forecasting via multi-modal knowledge distillation","name":"articletitle","label":"Article Title"},{"value":"Advanced Engineering Informatics","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.aei.2026.104834","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104834"}}