{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T16:25:47Z","timestamp":1784823947375,"version":"3.55.0"},"reference-count":71,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Max Planck Institute for Informatics"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,5]]},"DOI":"10.1109\/tpami.2024.3352811","type":"journal-article","created":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T18:55:36Z","timestamp":1705085736000},"page":"3955-3971","source":"Crossref","is-referenced-by-count":160,"title":["MTR++: Multi-Agent Motion Prediction With Symmetric Scene Modeling and Guided Intention Querying"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2558-181X","authenticated-orcid":false,"given":"Shaoshuai","family":"Shi","sequence":"first","affiliation":[{"name":"Max Planck Institute for Informatics, Saarland Informatics Campus, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7058-6957","authenticated-orcid":false,"given":"Li","family":"Jiang","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Informatics, Saarland Informatics Campus, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5440-9678","authenticated-orcid":false,"given":"Dengxin","family":"Dai","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Informatics, Saarland Informatics Campus, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9683-5237","authenticated-orcid":false,"given":"Bernt","family":"Schiele","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Informatics, Saarland Informatics Campus, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Argoverse 2: Motion forecasting competition","year":"2022"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.110"},{"key":"ref3","article-title":"BEiT: BERT pre-training of image transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bao"},{"key":"ref4","article-title":"PRANK: Motion prediction based on ranking","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Biktairov"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196697"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58592-1_37"},{"key":"ref8","first-page":"947","article-title":"IntentNet: Learning to predict intention from raw sensor data","volume-title":"Proc. 2nd Conf. Robot Learn.","author":"Casas"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01417"},{"key":"ref10","first-page":"86","article-title":"MultiPath: Multiple probabilistic anchor trajectory hypotheses for behavior prediction","volume-title":"Proc. Conf. Robot Learn.","author":"Chai"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_39"},{"key":"ref12","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093332"},{"key":"ref14","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00957"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00683"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01154"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/itsc48978.2021.9564944"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC48978.2021.9564944"},{"key":"ref20","article-title":"THOMAS: Trajectory heatmap output with learned multi-agent sampling","author":"Gilles","year":"2021"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01502"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00240"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00865"},{"key":"ref24","first-page":"910","article-title":"Towards capturing the temporal dynamics for trajectory prediction: A coarse-to-fine approach","volume-title":"Proc. 6th Conf. Robot Learn.","author":"Jia"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3062309"},{"key":"ref26","first-page":"1434","article-title":"Multi-agent trajectory prediction by combining egocentric and allocentric views","volume-title":"Proc. 5th Conf. Robot Learn.","author":"Jia"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2023.3298301"},{"key":"ref28","first-page":"1","article-title":"MotionCNN: A strong baseline for motion prediction in autonomous driving","volume-title":"Proc. Conf. Comput. Vis. Pattern Recognit. Workshop","author":"Konev"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00831"},{"key":"ref30","article-title":"EvolveGraph: Multi-agent trajectory prediction with dynamic relational reasoning","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_32"},{"key":"ref32","article-title":"DAB-DETR: Dynamic anchor boxes are better queries for DETR","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Liu"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00749"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_45"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00717"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9197340"},{"key":"ref38","first-page":"1","article-title":"Multi-modal interactive agent trajectory prediction using heterogeneous edge-enhanced graph attention network","volume-title":"Proc. Conf. Comput. Vis. Pattern Recognit. Workshop","author":"Mo"},{"key":"ref39","article-title":"Scene transformer: A unified architecture for predicting future trajectories of multiple agents","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ngiam"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_17"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01408"},{"key":"ref42","first-page":"77","article-title":"PointNet: Deep learning on point sets for 3D classification and segmentation","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","author":"Qi"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_47"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00291"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_40"},{"key":"ref46","first-page":"6531","article-title":"Motion transformer with global intention localization and local movement refinement","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Shi"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00643"},{"key":"ref48","article-title":"Multiple futures prediction","volume-title":"Proc. 33rd Int. Conf. Neural Inf. Process. Syst.","author":"Tang"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561967"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812107"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01299"},{"key":"ref53","first-page":"6036","article-title":"Multi-person 3D motion prediction with multi-range transformers","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"ref55","article-title":"TENET: Transformer encoding network for effective temporal flow on motion prediction","author":"Wang","year":"2022"},{"key":"ref56","article-title":"Waymo open dataset interaction prediction challenge 2021","year":"2021"},{"key":"ref57","article-title":"Waymo open dataset motion prediction challenge 2022","year":"2022"},{"key":"ref58","article-title":"Waymo open dataset motion prediction challenge 2023","year":"2023"},{"key":"ref59","article-title":"Argoverse 2: Next generation datasets for self-driving perception and forecasting","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wilson"},{"key":"ref60","first-page":"1","article-title":"Air2 for interaction prediction","volume-title":"Proc. Conf. Comput. Vis. Pattern Recognit. Workshop","author":"Wu"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3057326"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00835"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01116"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19812-0_38"},{"key":"ref65","article-title":"DINO: DETR with improved denoising anchor boxes for end-to-end object detection","author":"Zhang","year":"2022"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5473"},{"key":"ref67","first-page":"895","article-title":"TNT: Target-driven trajectory prediction","volume-title":"Proc. Conf. Robot Learn.","author":"Zhao"},{"key":"ref68","first-page":"1","article-title":"ReCoAt: A deep learning framework with attention mechanism for multi-modal motion prediction","volume-title":"Proc. Conf. Comput. Vis. Pattern Recognit. Workshop","author":"Lv"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00862"},{"key":"ref70","article-title":"Deformable DETR: Deformable transformers for end-to-end object detection","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhu"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967811"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10490207\/10398503.pdf?arnumber=10398503","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,9]],"date-time":"2024-04-09T19:30:17Z","timestamp":1712691017000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10398503\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5]]},"references-count":71,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3352811","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5]]}}}